PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""[STEP] Normalizer"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from sklearn.preprocessing import Normalizer
|
|
6
|
+
from ...actionable import Actionable
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...data_type import DataType
|
|
9
|
+
from ...dataset import Dataset
|
|
10
|
+
from ...decorators.all import is_step
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@is_step('normalize')
|
|
14
|
+
class ActNormalizer(Actionable):
|
|
15
|
+
"""[STEP] Normalizer"""
|
|
16
|
+
|
|
17
|
+
name: str = "Normalizer"
|
|
18
|
+
_description: str = textwrap.dedent('''\
|
|
19
|
+
Normalizer scales each sample so its L1 or L2 norm equals one,
|
|
20
|
+
keeping per-sample magnitudes comparable.''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
Normalizer rescales each row independently by dividing its values
|
|
23
|
+
by the L1 or L2 norm. This preserves the direction of each sample
|
|
24
|
+
while making their magnitudes comparable, which is helpful when
|
|
25
|
+
features represent frequencies, counts, or embeddings.''')
|
|
26
|
+
_usage: str = "Use when you need per-sample L1/L2 normalization for row magnitude comparability, especially for counts or embeddings. Applicable to numeric features where each row should be unit norm. Avoid when feature-wise scaling is needed; consider ActMinMaxScaler or ActRobustScaler."
|
|
27
|
+
|
|
28
|
+
def __init__(self):
|
|
29
|
+
self.columns: list[str] = None
|
|
30
|
+
self.normalizer: Normalizer = None
|
|
31
|
+
|
|
32
|
+
self.configuration = {
|
|
33
|
+
'norm': {
|
|
34
|
+
'description': 'Normalization type to apply to each sample.',
|
|
35
|
+
'default': 'l2',
|
|
36
|
+
'categorical': ['l1', 'l2']
|
|
37
|
+
},
|
|
38
|
+
'copy': {
|
|
39
|
+
'description': 'Set to False to perform normalization in-place when possible.',
|
|
40
|
+
'default': True,
|
|
41
|
+
'categorical': [True, False]
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
46
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
47
|
+
if self.columns and not dataset.X.empty:
|
|
48
|
+
values = dataset.X[self.columns]
|
|
49
|
+
self.normalizer = Normalizer(**self.passthrough_parameters())
|
|
50
|
+
self.normalizer.fit(values)
|
|
51
|
+
else:
|
|
52
|
+
self.normalizer = None
|
|
53
|
+
return self
|
|
54
|
+
|
|
55
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
56
|
+
"""Apply normalization
|
|
57
|
+
|
|
58
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
59
|
+
:return: Transformed dataset
|
|
60
|
+
"""
|
|
61
|
+
if self.normalizer and self.columns:
|
|
62
|
+
columns = [column for column in self.columns if column in X.columns]
|
|
63
|
+
if not columns:
|
|
64
|
+
return X
|
|
65
|
+
X[columns] = self.normalizer.transform(X[columns])
|
|
66
|
+
return X
|
|
67
|
+
|
|
68
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
69
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
70
|
+
return bool(columns) and not dataset.X.empty
|
|
71
|
+
|
|
72
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
73
|
+
if candidate is None:
|
|
74
|
+
return 0.0
|
|
75
|
+
columns = candidate.dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
76
|
+
if not columns or candidate.dataset.X.empty:
|
|
77
|
+
return 0.0
|
|
78
|
+
values = candidate.dataset.X[columns]
|
|
79
|
+
if values.empty:
|
|
80
|
+
return 0.0
|
|
81
|
+
matrix = values.to_numpy(dtype=float, copy=True)
|
|
82
|
+
if np.isnan(matrix).any():
|
|
83
|
+
matrix = np.nan_to_num(matrix, nan=0.0)
|
|
84
|
+
if self.get_config('norm') == 'l1':
|
|
85
|
+
norms = np.sum(np.abs(matrix), axis=1)
|
|
86
|
+
else:
|
|
87
|
+
norms = np.linalg.norm(matrix, axis=1)
|
|
88
|
+
if norms.size == 0:
|
|
89
|
+
return 0.0
|
|
90
|
+
nonzero = norms > 0
|
|
91
|
+
if not nonzero.any():
|
|
92
|
+
return 0.0
|
|
93
|
+
norms = norms[nonzero]
|
|
94
|
+
mean_deviation = float(np.mean(np.abs(norms - 1.0)))
|
|
95
|
+
return min(1.0, mean_deviation)
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""[STEP] Robust Scaler"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.preprocessing import RobustScaler
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...candidate import Candidate
|
|
7
|
+
from ...data_type import DataType
|
|
8
|
+
from ...dataset import Dataset
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('normalize')
|
|
13
|
+
class ActRobustScaler(Actionable):
|
|
14
|
+
"""[STEP] Robust Scaler"""
|
|
15
|
+
|
|
16
|
+
name: str = "Robust Scaler"
|
|
17
|
+
_usage: str = "Use when numeric features have outliers or skew and you want robust scaling vs ActMinMaxScaler or ActMaxAbsScaler. Applicable to continuous numeric columns. Avoid when you need unit-norm vectors or the data has no outliers."
|
|
18
|
+
_description: str = textwrap.dedent('''\
|
|
19
|
+
RobustScaler centers and scales numeric data using the median and IQR
|
|
20
|
+
to reduce the impact of outliers.''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
RobustScaler is a scaling technique that uses the median to center each
|
|
23
|
+
feature and the interquartile range (IQR) to scale it.
|
|
24
|
+
Because these statistics are resilient to extreme values, the transformation
|
|
25
|
+
is well suited for data sets that contain outliers.''')
|
|
26
|
+
|
|
27
|
+
def __init__(self):
|
|
28
|
+
self.columns: list[str] = None
|
|
29
|
+
self.scaler: RobustScaler = None
|
|
30
|
+
|
|
31
|
+
self.configuration = {
|
|
32
|
+
'with_centering': {
|
|
33
|
+
'description': 'Center data before scaling.',
|
|
34
|
+
'default': True,
|
|
35
|
+
'categorical': [True, False]
|
|
36
|
+
},
|
|
37
|
+
'with_scaling': {
|
|
38
|
+
'description': 'Scale data to the IQR.',
|
|
39
|
+
'default': True,
|
|
40
|
+
'categorical': [True, False]
|
|
41
|
+
},
|
|
42
|
+
'quantile_range_low': {
|
|
43
|
+
'description': 'Lower quantile used to compute the IQR.',
|
|
44
|
+
'default': 25.0,
|
|
45
|
+
'range': [0.0, 50.0]
|
|
46
|
+
},
|
|
47
|
+
'quantile_range_high': {
|
|
48
|
+
'description': 'Upper quantile used to compute the IQR.',
|
|
49
|
+
'default': 75.0,
|
|
50
|
+
'range': [50.0, 100.0]
|
|
51
|
+
},
|
|
52
|
+
'unit_variance': {
|
|
53
|
+
'description': 'Scale data so that scaled features have unit variance.',
|
|
54
|
+
'default': False,
|
|
55
|
+
'categorical': [True, False]
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
def _build_scaler(self) -> RobustScaler:
|
|
60
|
+
params = self.passthrough_parameters()
|
|
61
|
+
low = float(params.pop('quantile_range_low'))
|
|
62
|
+
high = float(params.pop('quantile_range_high'))
|
|
63
|
+
if low >= high:
|
|
64
|
+
low, high = 25.0, 75.0
|
|
65
|
+
params['quantile_range'] = (low, high)
|
|
66
|
+
return RobustScaler(**params)
|
|
67
|
+
|
|
68
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
69
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
70
|
+
if self.columns:
|
|
71
|
+
values = dataset.X[self.columns]
|
|
72
|
+
self.scaler = self._build_scaler()
|
|
73
|
+
self.scaler.fit(values)
|
|
74
|
+
else:
|
|
75
|
+
self.scaler = None
|
|
76
|
+
return self
|
|
77
|
+
|
|
78
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
79
|
+
"""Apply robust scaler
|
|
80
|
+
|
|
81
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
82
|
+
:return: Transformed dataset
|
|
83
|
+
"""
|
|
84
|
+
if self.scaler and self.columns:
|
|
85
|
+
columns = [column for column in self.columns if column in X.columns]
|
|
86
|
+
if not columns:
|
|
87
|
+
return X
|
|
88
|
+
X[columns] = self.scaler.transform(X[columns])
|
|
89
|
+
return X
|
|
90
|
+
|
|
91
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
92
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
93
|
+
return bool(columns) and not dataset.X.empty
|
|
94
|
+
|
|
95
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
96
|
+
if candidate is None:
|
|
97
|
+
return 0.0
|
|
98
|
+
columns = candidate.dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
99
|
+
if not columns or candidate.dataset.X.empty:
|
|
100
|
+
return 0.0
|
|
101
|
+
values = candidate.dataset.X[columns]
|
|
102
|
+
q1 = values.quantile(0.25)
|
|
103
|
+
q3 = values.quantile(0.75)
|
|
104
|
+
iqr = q3 - q1
|
|
105
|
+
if (iqr == 0).all():
|
|
106
|
+
return 0.1
|
|
107
|
+
lower = q1 - 1.5 * iqr
|
|
108
|
+
upper = q3 + 1.5 * iqr
|
|
109
|
+
outliers = ((values < lower) | (values > upper)).sum().sum()
|
|
110
|
+
total = values.size or 1
|
|
111
|
+
return min(1.0, outliers / total * 5.0)
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""[STEP] Standard Scaler"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from sklearn.preprocessing import StandardScaler
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...data_type import DataType
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
@is_step('normalize')
|
|
12
|
+
class ActStandardScaler(Actionable):
|
|
13
|
+
"""[STEP] Standard Scaler"""
|
|
14
|
+
|
|
15
|
+
name: str = "Standard Scaler"
|
|
16
|
+
_description: str = textwrap.dedent('''\
|
|
17
|
+
StandardScaler helps computers understand complex data by transforming
|
|
18
|
+
it into numbers centered around zero with a standard deviation of one.''')
|
|
19
|
+
_description_long: str = textwrap.dedent('''\
|
|
20
|
+
StandardScaler is a machine learning technique used to standardize numerical
|
|
21
|
+
features by removing the mean and scaling to unit variance.
|
|
22
|
+
It works by calculating the mean and standard deviation for each feature in the training data,
|
|
23
|
+
then transforming all values such that they have a mean of zero and a variance of one.
|
|
24
|
+
This transformation helps to center data and is particularly useful in algorithms that assume
|
|
25
|
+
normality of features, like many machine learning models.''')
|
|
26
|
+
|
|
27
|
+
def __init__(self):
|
|
28
|
+
self.columns: list[str] = None
|
|
29
|
+
self.scaler: StandardScaler = None
|
|
30
|
+
|
|
31
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
32
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
33
|
+
if self.columns:
|
|
34
|
+
values = dataset.X[self.columns]
|
|
35
|
+
self.scaler = StandardScaler()
|
|
36
|
+
self.scaler.fit(values)
|
|
37
|
+
else:
|
|
38
|
+
self.scaler = None
|
|
39
|
+
return self
|
|
40
|
+
|
|
41
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
42
|
+
"""Apply standard scaler
|
|
43
|
+
|
|
44
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
45
|
+
:return: Transformed dataset
|
|
46
|
+
"""
|
|
47
|
+
if self.scaler and self.columns:
|
|
48
|
+
columns = [column for column in self.columns if column in X.columns]
|
|
49
|
+
if not columns:
|
|
50
|
+
return X
|
|
51
|
+
X[columns] = self.scaler.transform(X[columns])
|
|
52
|
+
return X
|
|
53
|
+
|
|
54
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
55
|
+
return 0.5
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""Feature-name adaptation shared by the XGBoost predictors."""
|
|
2
|
+
import pandas as pd
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
_FEATURE_NAME_ESCAPES = str.maketrans({"%": "%25", "[": "%5B", "]": "%5D", "<": "%3C"})
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def xgboost_features(X):
|
|
9
|
+
"""Escape names without collisions, preserving data and native name validation."""
|
|
10
|
+
if not isinstance(X, pd.DataFrame):
|
|
11
|
+
return X
|
|
12
|
+
|
|
13
|
+
renamed = X.copy(deep=False)
|
|
14
|
+
# Escaping '%' also distinguishes a literal escape sequence from its source.
|
|
15
|
+
renamed.rename(columns=lambda name: str(name).translate(_FEATURE_NAME_ESCAPES), inplace=True)
|
|
16
|
+
return renamed
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Classifier Predictors Actionables
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from .act_svm_svc import ActSVMSVC
|
|
6
|
+
from .act_randomforest import ActRandomForest
|
|
7
|
+
from .act_xgboost import ActXGBoost
|
|
8
|
+
from .act_catboost_classifier import ActCatBoost
|
|
9
|
+
from .act_gaussian_nb import ActGaussianNb
|
|
10
|
+
from .act_knn import ActKNN
|
|
11
|
+
from .act_logistic_regression import ActLogisticRegression
|
|
12
|
+
from .act_bernoulli_nb import ActBernoulliNb
|
|
13
|
+
from .act_extra_trees_classifier import ActExtraTreesClassifier
|
|
14
|
+
from .act_linear_discriminant_analysis import ActLinearDiscriminantAnalysis
|
|
15
|
+
from .act_mlp_classifier import ActMLPClassifier
|
|
16
|
+
from .act_multinomial_nb import ActMultinomialNB
|
|
17
|
+
from .act_quadratic_discriminant_analysis import ActQuadraticDiscriminantAnalysis
|
|
18
|
+
from .act_decision_tree_classifier import ActDecisionTreeClassifier
|
|
19
|
+
from .act_hist_gradient_boosting_classifier import ActHistGradientBoostingClassifier
|
|
20
|
+
from .act_sgd_classifier import ActSGDClassifier
|
|
21
|
+
from .act_ridge_classifier import ActRidgeClassifier
|
|
22
|
+
from .act_linear_svc import ActLinearSVC
|
|
23
|
+
from .act_passive_aggressive_classifier import ActPassiveAggressiveClassifier
|
|
24
|
+
from .act_bagging_classifier import ActBaggingClassifier
|
|
25
|
+
from .act_complement_nb import ActComplementNB
|
|
26
|
+
from .act_light_gbm_classifier import ActLightGBMClassifier
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""[STEP] Bagging Classifier"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
from sklearn.ensemble import BaggingClassifier
|
|
5
|
+
from ....predictor import Predictor
|
|
6
|
+
from ....dataset import Dataset
|
|
7
|
+
from ....candidate import Candidate
|
|
8
|
+
from ....data_type import DataType
|
|
9
|
+
from ....decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('predictor', 'tabular', 'classifier')
|
|
13
|
+
class ActBaggingClassifier(Predictor):
|
|
14
|
+
"""[STEP] Bagging Classifier"""
|
|
15
|
+
|
|
16
|
+
name: str = "Bagging Classifier"
|
|
17
|
+
_description: str = textwrap.dedent('''\
|
|
18
|
+
BaggingClassifier is an ensemble method that combines multiple
|
|
19
|
+
base learners trained on bootstrapped samples to reduce variance.''')
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
BaggingClassifier (Bootstrap Aggregating) fits several base estimators on
|
|
22
|
+
random subsets of the training data and optionally on random subsets of
|
|
23
|
+
features. Predictions are aggregated by majority vote, yielding a more
|
|
24
|
+
stable classifier that is less sensitive to noise.''')
|
|
25
|
+
_usage: str = "Use when you need variance reduction on numeric features and want a simple ensemble, versus ActExtraTreesClassifier. Applicable to tabular binary/multiclass/multilabel numeric data. Avoid when you need a single interpretable tree or can use ActDecisionTreeClassifier."
|
|
26
|
+
refs: list[dict[str, Any]] = [
|
|
27
|
+
{
|
|
28
|
+
'year': 1996,
|
|
29
|
+
'name': 'Bagging Predictors',
|
|
30
|
+
'authors': [
|
|
31
|
+
'Leo Breiman'
|
|
32
|
+
],
|
|
33
|
+
'doi': 'https://doi.org/10.1023/A:1018054314350',
|
|
34
|
+
'publisher': 'Machine Learning Vol. 24 page 123--140'
|
|
35
|
+
}
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
def __init__(self):
|
|
39
|
+
self.configuration = {
|
|
40
|
+
'n_estimators': {
|
|
41
|
+
'description': 'Number of base estimators in the ensemble.',
|
|
42
|
+
'default': 50,
|
|
43
|
+
'range': [5, 500]
|
|
44
|
+
},
|
|
45
|
+
'max_samples': {
|
|
46
|
+
'description': 'Fraction of samples to draw for each base estimator.',
|
|
47
|
+
'default': 1.0,
|
|
48
|
+
'range': [0.1, 1.0]
|
|
49
|
+
},
|
|
50
|
+
'max_features': {
|
|
51
|
+
'description': 'Fraction of features to draw for each base estimator.',
|
|
52
|
+
'default': 1.0,
|
|
53
|
+
'range': [0.1, 1.0]
|
|
54
|
+
},
|
|
55
|
+
'bootstrap': {
|
|
56
|
+
'description': 'Whether samples are drawn with replacement.',
|
|
57
|
+
'default': True,
|
|
58
|
+
'categorical': [True, False]
|
|
59
|
+
},
|
|
60
|
+
'bootstrap_features': {
|
|
61
|
+
'description': 'Whether features are drawn with replacement.',
|
|
62
|
+
'default': False,
|
|
63
|
+
'categorical': [True, False]
|
|
64
|
+
},
|
|
65
|
+
'oob_score': {
|
|
66
|
+
'description': textwrap.dedent('''\
|
|
67
|
+
Whether to use out-of-bag samples to estimate generalization.'''),
|
|
68
|
+
'default': False,
|
|
69
|
+
'categorical': [True, False]
|
|
70
|
+
},
|
|
71
|
+
'random_state': {
|
|
72
|
+
'description': 'Random state for reproducibility.',
|
|
73
|
+
'default': 42
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
self.model: BaggingClassifier = None
|
|
77
|
+
self.columns: list[str] = []
|
|
78
|
+
|
|
79
|
+
def _select_features(self, X):
|
|
80
|
+
if self.columns and hasattr(X, 'columns'):
|
|
81
|
+
return X[self.columns]
|
|
82
|
+
return X
|
|
83
|
+
|
|
84
|
+
def fit(self, dataset: Dataset):
|
|
85
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
86
|
+
if not self.columns:
|
|
87
|
+
raise ValueError(
|
|
88
|
+
"BaggingClassifier requires at least one numeric feature."
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
params = self.passthrough_parameters()
|
|
92
|
+
if params.get('oob_score') and not params.get('bootstrap', True):
|
|
93
|
+
raise ValueError("oob_score=True requires bootstrap=True.")
|
|
94
|
+
self.model = BaggingClassifier(**params)
|
|
95
|
+
self.model.fit(self._select_features(dataset.X), dataset.y)
|
|
96
|
+
return self
|
|
97
|
+
|
|
98
|
+
def predict(self, X):
|
|
99
|
+
return super().predict(self._select_features(X))
|
|
100
|
+
|
|
101
|
+
def predict_proba(self, X):
|
|
102
|
+
return super().predict_proba(self._select_features(X))
|
|
103
|
+
|
|
104
|
+
def score(self, X, y=None, *args, **kwargs):
|
|
105
|
+
return self.model.score(self._select_features(X), y, *args, **kwargs)
|
|
106
|
+
|
|
107
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
108
|
+
return dataset.type_of_target in \
|
|
109
|
+
['binary', 'multiclass', 'multilabel-indicator'] \
|
|
110
|
+
and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
|
|
111
|
+
|
|
112
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
113
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""
|
|
2
|
+
[STEP] Bernoulli NB
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import textwrap
|
|
6
|
+
from sklearn.naive_bayes import BernoulliNB
|
|
7
|
+
from ....predictor import Predictor
|
|
8
|
+
from ....dataset import Dataset
|
|
9
|
+
from ....candidate import Candidate
|
|
10
|
+
from ....decorators.all import is_step
|
|
11
|
+
|
|
12
|
+
@is_step('predictor', 'tabular', 'classifier')
|
|
13
|
+
class ActBernoulliNb(Predictor):
|
|
14
|
+
"""
|
|
15
|
+
[STEP] Bernoulli NB
|
|
16
|
+
"""
|
|
17
|
+
name = "Bernoulli NB"
|
|
18
|
+
_description = textwrap.dedent('''\
|
|
19
|
+
BernoulliNB is a tool that helps computers predict categories
|
|
20
|
+
by analyzing binary features, even if the input isn't strictly binary.''')
|
|
21
|
+
_description_long = textwrap.dedent('''\
|
|
22
|
+
BernoulliNB is a type of Naive Bayes classifier specifically
|
|
23
|
+
designed for binary features. While it's primarily meant for binary inputs,
|
|
24
|
+
scikit-learn implements it in a way that can handle non-binary data.''')
|
|
25
|
+
_usage = "Use when you have mostly binary or presence/absence features and want a fast baseline vs ActComplementNB. Applicable to sparse tabular or text-like data with binary indicators. Avoid when features are continuous or you need nonlinear interactions; try ActGaussianNb or ActCatBoost."
|
|
26
|
+
refs = [
|
|
27
|
+
{
|
|
28
|
+
'year': 1998,
|
|
29
|
+
'name': 'A Comparison of Event Models for Naive Bayes Text Classification',
|
|
30
|
+
'authors': [
|
|
31
|
+
'Andrew McCallum',
|
|
32
|
+
'Kamal Nigam'
|
|
33
|
+
],
|
|
34
|
+
'doi': "https://www.semanticscholar.org/paper/ \
|
|
35
|
+
A-comparison-of-event-models-for-naive-bayes-text-McCallum-Nigam/ \
|
|
36
|
+
04ce064505b1635583fa0d9cc07cac7e9ea993cc",
|
|
37
|
+
'publisher': (
|
|
38
|
+
'AAAI-98 workshop on learning for text categorization, '
|
|
39
|
+
'752, page 41--48. (1998)'
|
|
40
|
+
)
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
'year': 2006,
|
|
44
|
+
'name': 'Spam Filtering with Naive Bayes - Which Naive Bayes?',
|
|
45
|
+
'authors': [
|
|
46
|
+
'Vangelis Metsis',
|
|
47
|
+
'Ion Androutsopoulos',
|
|
48
|
+
'Georgios Paliouras'
|
|
49
|
+
],
|
|
50
|
+
'doi': "https://www.semanticscholar.org/paper/ \
|
|
51
|
+
Spam-Filtering-with-Naive-Bayes-Which-Naive-Bayes-Metsis-Androutsopoulos/ \
|
|
52
|
+
7f5ce28afc0c2eafd4a6ef711e399bee4056c3b8",
|
|
53
|
+
'publisher': (
|
|
54
|
+
'The Third Conference on Email and Anti-Spam 2006 (CEAS)'
|
|
55
|
+
)
|
|
56
|
+
}
|
|
57
|
+
]
|
|
58
|
+
def __init__(self):
|
|
59
|
+
self.configuration = {
|
|
60
|
+
'alpha': {
|
|
61
|
+
'description': textwrap.dedent('''\
|
|
62
|
+
Additive (Laplace/Lidstone) smoothing parameter (set
|
|
63
|
+
alpha=0 and force_alpha=True, for no smoothing).'''),
|
|
64
|
+
'default': 1.0,
|
|
65
|
+
'range': [0.01, 100.0]
|
|
66
|
+
},
|
|
67
|
+
'fit_prior': {
|
|
68
|
+
'description': textwrap.dedent('''\
|
|
69
|
+
Whether to learn class prior probabilities or not. If
|
|
70
|
+
false, a uniform prior will be used.'''),
|
|
71
|
+
'default': True,
|
|
72
|
+
'categorical': [True, False]
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
self.model: BernoulliNB = None
|
|
76
|
+
|
|
77
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
78
|
+
self.model = BernoulliNB(**self.passthrough_parameters())
|
|
79
|
+
|
|
80
|
+
self.model.fit(dataset.X, dataset.y)
|
|
81
|
+
|
|
82
|
+
return self
|
|
83
|
+
|
|
84
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
85
|
+
return dataset.type_of_target in \
|
|
86
|
+
['binary', 'multiclass', 'multilabel-indicator']
|
|
87
|
+
|
|
88
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
89
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""[STEP] CatBoost"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
from catboost import CatBoostClassifier, CatBoostError
|
|
5
|
+
from sklearn.preprocessing import LabelEncoder
|
|
6
|
+
from ....predictor import Predictor
|
|
7
|
+
from ....dataset import Dataset
|
|
8
|
+
from ....candidate import Candidate
|
|
9
|
+
from ....decorators.all import is_step
|
|
10
|
+
from ....logger import Logger
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@is_step('predictor', 'tabular', 'classifier', 'minimal_predictor')
|
|
14
|
+
class ActCatBoost(Predictor):
|
|
15
|
+
"""[STEP] CatBoost Classifier"""
|
|
16
|
+
|
|
17
|
+
name: str = "CatBoost Classifier"
|
|
18
|
+
_usage: str = "Use when you want high-accuracy tabular classification with categorical features, often stronger than ActDecisionTreeClassifier or ActExtraTreesClassifier. Applicable to binary or multiclass tabular data. Avoid when data is tiny, compute is tight, or you prefer ActGaussianNb."
|
|
19
|
+
_description: str = textwrap.dedent('''\
|
|
20
|
+
CatBoostClassifier is a powerful tool that helps computers make accurate
|
|
21
|
+
predictions by learning from both positive and negative examples simultaneously.''')
|
|
22
|
+
_description_long: str = textwrap.dedent('''\
|
|
23
|
+
CatBoostClassifier is a gradient boosting algorithm specifically
|
|
24
|
+
designed for classification tasks.''')
|
|
25
|
+
refs: list[dict[str, Any]] = [
|
|
26
|
+
{
|
|
27
|
+
'year': 2017,
|
|
28
|
+
'name': 'CatBoost: unbiased boosting with categorical features',
|
|
29
|
+
'authors': [
|
|
30
|
+
'Liudmila Prokhorenkova',
|
|
31
|
+
'Gleb Gusev',
|
|
32
|
+
'Aleksandr Vorobev',
|
|
33
|
+
'Anna Veronika Dorogush',
|
|
34
|
+
'Andrey Gulin'
|
|
35
|
+
],
|
|
36
|
+
'doi': 'https://doi.org/10.48550/arXiv.1706.09516',
|
|
37
|
+
'publisher': 'Advances in Neural Information Processing Systems 31 (NeurIPS 2018)'
|
|
38
|
+
}
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
def __init__(self):
|
|
42
|
+
self.configuration = {
|
|
43
|
+
'iterations': {
|
|
44
|
+
'description': 'The maximum number of trees that can be built.',
|
|
45
|
+
'default': 1000,
|
|
46
|
+
'range': [100, 10000]
|
|
47
|
+
},
|
|
48
|
+
'learning_rate': {
|
|
49
|
+
'description': 'The learning rate.',
|
|
50
|
+
'default': 0.03,
|
|
51
|
+
'range': [0.001, 1.0]
|
|
52
|
+
},
|
|
53
|
+
'depth': {
|
|
54
|
+
'description': 'Depth of the tree.',
|
|
55
|
+
'default': 6,
|
|
56
|
+
'range': [1, 16]
|
|
57
|
+
},
|
|
58
|
+
'l2_leaf_reg': {
|
|
59
|
+
'description': 'Coefficient at the L2 regularization term of the cost function.',
|
|
60
|
+
'default': 3,
|
|
61
|
+
'range': [0, 10]
|
|
62
|
+
},
|
|
63
|
+
'border_count': {
|
|
64
|
+
'description': 'The number of splits for numerical features.',
|
|
65
|
+
'default': 254,
|
|
66
|
+
'range': [1, 255]
|
|
67
|
+
},
|
|
68
|
+
'loss_function': {
|
|
69
|
+
'description': 'The metric to use in training.',
|
|
70
|
+
'default': 'Logloss',
|
|
71
|
+
'categorical': ['Logloss', 'CrossEntropy', 'MultiClass', 'MultiClassOneVsAll']
|
|
72
|
+
},
|
|
73
|
+
'eval_metric': {
|
|
74
|
+
'description': 'The metric to be used for validation data.',
|
|
75
|
+
'default': 'AUC',
|
|
76
|
+
'categorical': ['AUC', 'Accuracy', 'Logloss']
|
|
77
|
+
},
|
|
78
|
+
'bootstrap_type': {
|
|
79
|
+
'description': 'The method for sampling the weights of objects.',
|
|
80
|
+
'default': 'Bayesian',
|
|
81
|
+
'categorical': ['Bayesian', 'Bernoulli', 'MVS']
|
|
82
|
+
},
|
|
83
|
+
'leaf_estimation_iterations': {
|
|
84
|
+
'description': 'The number of iterations for leaf estimation.',
|
|
85
|
+
'default': 10,
|
|
86
|
+
'range': [1, 50]
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
self.model: CatBoostClassifier = None
|
|
91
|
+
self.label_encoder: LabelEncoder = LabelEncoder()
|
|
92
|
+
|
|
93
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
94
|
+
if dataset.type_of_target == 'binary':
|
|
95
|
+
self.configuration['loss_function']['categorical'] = ['Logloss', 'CrossEntropy']
|
|
96
|
+
else:
|
|
97
|
+
self.configuration['eval_metric']['categorical'] = ['AUC', 'Accuracy']
|
|
98
|
+
self.configuration['loss_function']['categorical'] = [
|
|
99
|
+
'MultiClass',
|
|
100
|
+
'MultiClassOneVsAll'
|
|
101
|
+
]
|
|
102
|
+
self.check_configuration()
|
|
103
|
+
|
|
104
|
+
self.model = CatBoostClassifier(verbose=0, **self.passthrough_parameters())
|
|
105
|
+
self.label_encoder.fit(dataset.y)
|
|
106
|
+
encoded_target = self.label_encoder.transform(dataset.y)
|
|
107
|
+
try:
|
|
108
|
+
self.model.fit(dataset.X, encoded_target)
|
|
109
|
+
except CatBoostError as exc:
|
|
110
|
+
self._log_failure(dataset, exc)
|
|
111
|
+
raise ValueError(f"CatBoostClassifier training failed: {exc}") from exc
|
|
112
|
+
except Exception as exc: # pragma: no cover - defensive
|
|
113
|
+
self._log_failure(dataset, exc)
|
|
114
|
+
raise
|
|
115
|
+
return self
|
|
116
|
+
|
|
117
|
+
def _log_failure(self, dataset: Dataset, exc: Exception) -> None:
|
|
118
|
+
"""Log enriched debug info when CatBoost crashes."""
|
|
119
|
+
shape = getattr(dataset.X, "shape", None)
|
|
120
|
+
message = (
|
|
121
|
+
"[CatBoostClassifier] crash detected "
|
|
122
|
+
f"(shape={shape}, target_len={len(dataset.y)}, "
|
|
123
|
+
f"params={self.passthrough_parameters()}): {exc}"
|
|
124
|
+
)
|
|
125
|
+
logger = Logger()
|
|
126
|
+
if logger.verbose <= 3 and logger.verbose != -1:
|
|
127
|
+
logger.console.log(message)
|
|
128
|
+
logger.error(message)
|
|
129
|
+
|
|
130
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
131
|
+
return dataset.type_of_target in \
|
|
132
|
+
['binary', 'multiclass', 'multilabel-indicator']
|
|
133
|
+
|
|
134
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
135
|
+
return 0.5 # neutral
|