PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,95 @@
1
+ """[STEP] Normalizer"""
2
+ import textwrap
3
+ import numpy as np
4
+ import pandas as pd
5
+ from sklearn.preprocessing import Normalizer
6
+ from ...actionable import Actionable
7
+ from ...candidate import Candidate
8
+ from ...data_type import DataType
9
+ from ...dataset import Dataset
10
+ from ...decorators.all import is_step
11
+
12
+
13
+ @is_step('normalize')
14
+ class ActNormalizer(Actionable):
15
+ """[STEP] Normalizer"""
16
+
17
+ name: str = "Normalizer"
18
+ _description: str = textwrap.dedent('''\
19
+ Normalizer scales each sample so its L1 or L2 norm equals one,
20
+ keeping per-sample magnitudes comparable.''')
21
+ _description_long: str = textwrap.dedent('''\
22
+ Normalizer rescales each row independently by dividing its values
23
+ by the L1 or L2 norm. This preserves the direction of each sample
24
+ while making their magnitudes comparable, which is helpful when
25
+ features represent frequencies, counts, or embeddings.''')
26
+ _usage: str = "Use when you need per-sample L1/L2 normalization for row magnitude comparability, especially for counts or embeddings. Applicable to numeric features where each row should be unit norm. Avoid when feature-wise scaling is needed; consider ActMinMaxScaler or ActRobustScaler."
27
+
28
+ def __init__(self):
29
+ self.columns: list[str] = None
30
+ self.normalizer: Normalizer = None
31
+
32
+ self.configuration = {
33
+ 'norm': {
34
+ 'description': 'Normalization type to apply to each sample.',
35
+ 'default': 'l2',
36
+ 'categorical': ['l1', 'l2']
37
+ },
38
+ 'copy': {
39
+ 'description': 'Set to False to perform normalization in-place when possible.',
40
+ 'default': True,
41
+ 'categorical': [True, False]
42
+ }
43
+ }
44
+
45
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
46
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
47
+ if self.columns and not dataset.X.empty:
48
+ values = dataset.X[self.columns]
49
+ self.normalizer = Normalizer(**self.passthrough_parameters())
50
+ self.normalizer.fit(values)
51
+ else:
52
+ self.normalizer = None
53
+ return self
54
+
55
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
56
+ """Apply normalization
57
+
58
+ :param pd.DataFrame X: DataFrame to transform
59
+ :return: Transformed dataset
60
+ """
61
+ if self.normalizer and self.columns:
62
+ columns = [column for column in self.columns if column in X.columns]
63
+ if not columns:
64
+ return X
65
+ X[columns] = self.normalizer.transform(X[columns])
66
+ return X
67
+
68
+ def suitable(self, dataset: Dataset) -> bool:
69
+ columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
70
+ return bool(columns) and not dataset.X.empty
71
+
72
+ def priorize(self, candidate: Candidate = None) -> float:
73
+ if candidate is None:
74
+ return 0.0
75
+ columns = candidate.dataset.get_columns_names_by_type(DataType.NUMERIC)
76
+ if not columns or candidate.dataset.X.empty:
77
+ return 0.0
78
+ values = candidate.dataset.X[columns]
79
+ if values.empty:
80
+ return 0.0
81
+ matrix = values.to_numpy(dtype=float, copy=True)
82
+ if np.isnan(matrix).any():
83
+ matrix = np.nan_to_num(matrix, nan=0.0)
84
+ if self.get_config('norm') == 'l1':
85
+ norms = np.sum(np.abs(matrix), axis=1)
86
+ else:
87
+ norms = np.linalg.norm(matrix, axis=1)
88
+ if norms.size == 0:
89
+ return 0.0
90
+ nonzero = norms > 0
91
+ if not nonzero.any():
92
+ return 0.0
93
+ norms = norms[nonzero]
94
+ mean_deviation = float(np.mean(np.abs(norms - 1.0)))
95
+ return min(1.0, mean_deviation)
@@ -0,0 +1,111 @@
1
+ """[STEP] Robust Scaler"""
2
+ import textwrap
3
+ import pandas as pd
4
+ from sklearn.preprocessing import RobustScaler
5
+ from ...actionable import Actionable
6
+ from ...candidate import Candidate
7
+ from ...data_type import DataType
8
+ from ...dataset import Dataset
9
+ from ...decorators.all import is_step
10
+
11
+
12
+ @is_step('normalize')
13
+ class ActRobustScaler(Actionable):
14
+ """[STEP] Robust Scaler"""
15
+
16
+ name: str = "Robust Scaler"
17
+ _usage: str = "Use when numeric features have outliers or skew and you want robust scaling vs ActMinMaxScaler or ActMaxAbsScaler. Applicable to continuous numeric columns. Avoid when you need unit-norm vectors or the data has no outliers."
18
+ _description: str = textwrap.dedent('''\
19
+ RobustScaler centers and scales numeric data using the median and IQR
20
+ to reduce the impact of outliers.''')
21
+ _description_long: str = textwrap.dedent('''\
22
+ RobustScaler is a scaling technique that uses the median to center each
23
+ feature and the interquartile range (IQR) to scale it.
24
+ Because these statistics are resilient to extreme values, the transformation
25
+ is well suited for data sets that contain outliers.''')
26
+
27
+ def __init__(self):
28
+ self.columns: list[str] = None
29
+ self.scaler: RobustScaler = None
30
+
31
+ self.configuration = {
32
+ 'with_centering': {
33
+ 'description': 'Center data before scaling.',
34
+ 'default': True,
35
+ 'categorical': [True, False]
36
+ },
37
+ 'with_scaling': {
38
+ 'description': 'Scale data to the IQR.',
39
+ 'default': True,
40
+ 'categorical': [True, False]
41
+ },
42
+ 'quantile_range_low': {
43
+ 'description': 'Lower quantile used to compute the IQR.',
44
+ 'default': 25.0,
45
+ 'range': [0.0, 50.0]
46
+ },
47
+ 'quantile_range_high': {
48
+ 'description': 'Upper quantile used to compute the IQR.',
49
+ 'default': 75.0,
50
+ 'range': [50.0, 100.0]
51
+ },
52
+ 'unit_variance': {
53
+ 'description': 'Scale data so that scaled features have unit variance.',
54
+ 'default': False,
55
+ 'categorical': [True, False]
56
+ }
57
+ }
58
+
59
+ def _build_scaler(self) -> RobustScaler:
60
+ params = self.passthrough_parameters()
61
+ low = float(params.pop('quantile_range_low'))
62
+ high = float(params.pop('quantile_range_high'))
63
+ if low >= high:
64
+ low, high = 25.0, 75.0
65
+ params['quantile_range'] = (low, high)
66
+ return RobustScaler(**params)
67
+
68
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
69
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
70
+ if self.columns:
71
+ values = dataset.X[self.columns]
72
+ self.scaler = self._build_scaler()
73
+ self.scaler.fit(values)
74
+ else:
75
+ self.scaler = None
76
+ return self
77
+
78
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
79
+ """Apply robust scaler
80
+
81
+ :param pd.DataFrame X: DataFrame to transform
82
+ :return: Transformed dataset
83
+ """
84
+ if self.scaler and self.columns:
85
+ columns = [column for column in self.columns if column in X.columns]
86
+ if not columns:
87
+ return X
88
+ X[columns] = self.scaler.transform(X[columns])
89
+ return X
90
+
91
+ def suitable(self, dataset: Dataset) -> bool:
92
+ columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
93
+ return bool(columns) and not dataset.X.empty
94
+
95
+ def priorize(self, candidate: Candidate = None) -> float:
96
+ if candidate is None:
97
+ return 0.0
98
+ columns = candidate.dataset.get_columns_names_by_type(DataType.NUMERIC)
99
+ if not columns or candidate.dataset.X.empty:
100
+ return 0.0
101
+ values = candidate.dataset.X[columns]
102
+ q1 = values.quantile(0.25)
103
+ q3 = values.quantile(0.75)
104
+ iqr = q3 - q1
105
+ if (iqr == 0).all():
106
+ return 0.1
107
+ lower = q1 - 1.5 * iqr
108
+ upper = q3 + 1.5 * iqr
109
+ outliers = ((values < lower) | (values > upper)).sum().sum()
110
+ total = values.size or 1
111
+ return min(1.0, outliers / total * 5.0)
@@ -0,0 +1,55 @@
1
+ """[STEP] Standard Scaler"""
2
+ import textwrap
3
+ from sklearn.preprocessing import StandardScaler
4
+ import pandas as pd
5
+ from ...actionable import Actionable
6
+ from ...dataset import Dataset
7
+ from ...candidate import Candidate
8
+ from ...data_type import DataType
9
+ from ...decorators.all import is_step
10
+
11
+ @is_step('normalize')
12
+ class ActStandardScaler(Actionable):
13
+ """[STEP] Standard Scaler"""
14
+
15
+ name: str = "Standard Scaler"
16
+ _description: str = textwrap.dedent('''\
17
+ StandardScaler helps computers understand complex data by transforming
18
+ it into numbers centered around zero with a standard deviation of one.''')
19
+ _description_long: str = textwrap.dedent('''\
20
+ StandardScaler is a machine learning technique used to standardize numerical
21
+ features by removing the mean and scaling to unit variance.
22
+ It works by calculating the mean and standard deviation for each feature in the training data,
23
+ then transforming all values such that they have a mean of zero and a variance of one.
24
+ This transformation helps to center data and is particularly useful in algorithms that assume
25
+ normality of features, like many machine learning models.''')
26
+
27
+ def __init__(self):
28
+ self.columns: list[str] = None
29
+ self.scaler: StandardScaler = None
30
+
31
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
32
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
33
+ if self.columns:
34
+ values = dataset.X[self.columns]
35
+ self.scaler = StandardScaler()
36
+ self.scaler.fit(values)
37
+ else:
38
+ self.scaler = None
39
+ return self
40
+
41
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
42
+ """Apply standard scaler
43
+
44
+ :param pd.DataFrame X: DataFrame to transform
45
+ :return: Transformed dataset
46
+ """
47
+ if self.scaler and self.columns:
48
+ columns = [column for column in self.columns if column in X.columns]
49
+ if not columns:
50
+ return X
51
+ X[columns] = self.scaler.transform(X[columns])
52
+ return X
53
+
54
+ def priorize(self, candidate: Candidate = None) -> float:
55
+ return 0.5
@@ -0,0 +1,6 @@
1
+ """
2
+ Predictors Actionables
3
+ """
4
+ from .classifier import *
5
+ from .regressor import *
6
+ from .survival import *
@@ -0,0 +1,16 @@
1
+ """Feature-name adaptation shared by the XGBoost predictors."""
2
+ import pandas as pd
3
+
4
+
5
+ _FEATURE_NAME_ESCAPES = str.maketrans({"%": "%25", "[": "%5B", "]": "%5D", "<": "%3C"})
6
+
7
+
8
+ def xgboost_features(X):
9
+ """Escape names without collisions, preserving data and native name validation."""
10
+ if not isinstance(X, pd.DataFrame):
11
+ return X
12
+
13
+ renamed = X.copy(deep=False)
14
+ # Escaping '%' also distinguishes a literal escape sequence from its source.
15
+ renamed.rename(columns=lambda name: str(name).translate(_FEATURE_NAME_ESCAPES), inplace=True)
16
+ return renamed
@@ -0,0 +1,26 @@
1
+ """
2
+ Classifier Predictors Actionables
3
+ """
4
+
5
+ from .act_svm_svc import ActSVMSVC
6
+ from .act_randomforest import ActRandomForest
7
+ from .act_xgboost import ActXGBoost
8
+ from .act_catboost_classifier import ActCatBoost
9
+ from .act_gaussian_nb import ActGaussianNb
10
+ from .act_knn import ActKNN
11
+ from .act_logistic_regression import ActLogisticRegression
12
+ from .act_bernoulli_nb import ActBernoulliNb
13
+ from .act_extra_trees_classifier import ActExtraTreesClassifier
14
+ from .act_linear_discriminant_analysis import ActLinearDiscriminantAnalysis
15
+ from .act_mlp_classifier import ActMLPClassifier
16
+ from .act_multinomial_nb import ActMultinomialNB
17
+ from .act_quadratic_discriminant_analysis import ActQuadraticDiscriminantAnalysis
18
+ from .act_decision_tree_classifier import ActDecisionTreeClassifier
19
+ from .act_hist_gradient_boosting_classifier import ActHistGradientBoostingClassifier
20
+ from .act_sgd_classifier import ActSGDClassifier
21
+ from .act_ridge_classifier import ActRidgeClassifier
22
+ from .act_linear_svc import ActLinearSVC
23
+ from .act_passive_aggressive_classifier import ActPassiveAggressiveClassifier
24
+ from .act_bagging_classifier import ActBaggingClassifier
25
+ from .act_complement_nb import ActComplementNB
26
+ from .act_light_gbm_classifier import ActLightGBMClassifier
@@ -0,0 +1,113 @@
1
+ """[STEP] Bagging Classifier"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.ensemble import BaggingClassifier
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....data_type import DataType
9
+ from ....decorators.all import is_step
10
+
11
+
12
+ @is_step('predictor', 'tabular', 'classifier')
13
+ class ActBaggingClassifier(Predictor):
14
+ """[STEP] Bagging Classifier"""
15
+
16
+ name: str = "Bagging Classifier"
17
+ _description: str = textwrap.dedent('''\
18
+ BaggingClassifier is an ensemble method that combines multiple
19
+ base learners trained on bootstrapped samples to reduce variance.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ BaggingClassifier (Bootstrap Aggregating) fits several base estimators on
22
+ random subsets of the training data and optionally on random subsets of
23
+ features. Predictions are aggregated by majority vote, yielding a more
24
+ stable classifier that is less sensitive to noise.''')
25
+ _usage: str = "Use when you need variance reduction on numeric features and want a simple ensemble, versus ActExtraTreesClassifier. Applicable to tabular binary/multiclass/multilabel numeric data. Avoid when you need a single interpretable tree or can use ActDecisionTreeClassifier."
26
+ refs: list[dict[str, Any]] = [
27
+ {
28
+ 'year': 1996,
29
+ 'name': 'Bagging Predictors',
30
+ 'authors': [
31
+ 'Leo Breiman'
32
+ ],
33
+ 'doi': 'https://doi.org/10.1023/A:1018054314350',
34
+ 'publisher': 'Machine Learning Vol. 24 page 123--140'
35
+ }
36
+ ]
37
+
38
+ def __init__(self):
39
+ self.configuration = {
40
+ 'n_estimators': {
41
+ 'description': 'Number of base estimators in the ensemble.',
42
+ 'default': 50,
43
+ 'range': [5, 500]
44
+ },
45
+ 'max_samples': {
46
+ 'description': 'Fraction of samples to draw for each base estimator.',
47
+ 'default': 1.0,
48
+ 'range': [0.1, 1.0]
49
+ },
50
+ 'max_features': {
51
+ 'description': 'Fraction of features to draw for each base estimator.',
52
+ 'default': 1.0,
53
+ 'range': [0.1, 1.0]
54
+ },
55
+ 'bootstrap': {
56
+ 'description': 'Whether samples are drawn with replacement.',
57
+ 'default': True,
58
+ 'categorical': [True, False]
59
+ },
60
+ 'bootstrap_features': {
61
+ 'description': 'Whether features are drawn with replacement.',
62
+ 'default': False,
63
+ 'categorical': [True, False]
64
+ },
65
+ 'oob_score': {
66
+ 'description': textwrap.dedent('''\
67
+ Whether to use out-of-bag samples to estimate generalization.'''),
68
+ 'default': False,
69
+ 'categorical': [True, False]
70
+ },
71
+ 'random_state': {
72
+ 'description': 'Random state for reproducibility.',
73
+ 'default': 42
74
+ }
75
+ }
76
+ self.model: BaggingClassifier = None
77
+ self.columns: list[str] = []
78
+
79
+ def _select_features(self, X):
80
+ if self.columns and hasattr(X, 'columns'):
81
+ return X[self.columns]
82
+ return X
83
+
84
+ def fit(self, dataset: Dataset):
85
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
86
+ if not self.columns:
87
+ raise ValueError(
88
+ "BaggingClassifier requires at least one numeric feature."
89
+ )
90
+
91
+ params = self.passthrough_parameters()
92
+ if params.get('oob_score') and not params.get('bootstrap', True):
93
+ raise ValueError("oob_score=True requires bootstrap=True.")
94
+ self.model = BaggingClassifier(**params)
95
+ self.model.fit(self._select_features(dataset.X), dataset.y)
96
+ return self
97
+
98
+ def predict(self, X):
99
+ return super().predict(self._select_features(X))
100
+
101
+ def predict_proba(self, X):
102
+ return super().predict_proba(self._select_features(X))
103
+
104
+ def score(self, X, y=None, *args, **kwargs):
105
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
106
+
107
+ def suitable(self, dataset: Dataset) -> bool:
108
+ return dataset.type_of_target in \
109
+ ['binary', 'multiclass', 'multilabel-indicator'] \
110
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
111
+
112
+ def priorize(self, candidate: Candidate = None) -> float:
113
+ return 0.5 # neutral
@@ -0,0 +1,89 @@
1
+ """
2
+ [STEP] Bernoulli NB
3
+ """
4
+
5
+ import textwrap
6
+ from sklearn.naive_bayes import BernoulliNB
7
+ from ....predictor import Predictor
8
+ from ....dataset import Dataset
9
+ from ....candidate import Candidate
10
+ from ....decorators.all import is_step
11
+
12
+ @is_step('predictor', 'tabular', 'classifier')
13
+ class ActBernoulliNb(Predictor):
14
+ """
15
+ [STEP] Bernoulli NB
16
+ """
17
+ name = "Bernoulli NB"
18
+ _description = textwrap.dedent('''\
19
+ BernoulliNB is a tool that helps computers predict categories
20
+ by analyzing binary features, even if the input isn't strictly binary.''')
21
+ _description_long = textwrap.dedent('''\
22
+ BernoulliNB is a type of Naive Bayes classifier specifically
23
+ designed for binary features. While it's primarily meant for binary inputs,
24
+ scikit-learn implements it in a way that can handle non-binary data.''')
25
+ _usage = "Use when you have mostly binary or presence/absence features and want a fast baseline vs ActComplementNB. Applicable to sparse tabular or text-like data with binary indicators. Avoid when features are continuous or you need nonlinear interactions; try ActGaussianNb or ActCatBoost."
26
+ refs = [
27
+ {
28
+ 'year': 1998,
29
+ 'name': 'A Comparison of Event Models for Naive Bayes Text Classification',
30
+ 'authors': [
31
+ 'Andrew McCallum',
32
+ 'Kamal Nigam'
33
+ ],
34
+ 'doi': "https://www.semanticscholar.org/paper/ \
35
+ A-comparison-of-event-models-for-naive-bayes-text-McCallum-Nigam/ \
36
+ 04ce064505b1635583fa0d9cc07cac7e9ea993cc",
37
+ 'publisher': (
38
+ 'AAAI-98 workshop on learning for text categorization, '
39
+ '752, page 41--48. (1998)'
40
+ )
41
+ },
42
+ {
43
+ 'year': 2006,
44
+ 'name': 'Spam Filtering with Naive Bayes - Which Naive Bayes?',
45
+ 'authors': [
46
+ 'Vangelis Metsis',
47
+ 'Ion Androutsopoulos',
48
+ 'Georgios Paliouras'
49
+ ],
50
+ 'doi': "https://www.semanticscholar.org/paper/ \
51
+ Spam-Filtering-with-Naive-Bayes-Which-Naive-Bayes-Metsis-Androutsopoulos/ \
52
+ 7f5ce28afc0c2eafd4a6ef711e399bee4056c3b8",
53
+ 'publisher': (
54
+ 'The Third Conference on Email and Anti-Spam 2006 (CEAS)'
55
+ )
56
+ }
57
+ ]
58
+ def __init__(self):
59
+ self.configuration = {
60
+ 'alpha': {
61
+ 'description': textwrap.dedent('''\
62
+ Additive (Laplace/Lidstone) smoothing parameter (set
63
+ alpha=0 and force_alpha=True, for no smoothing).'''),
64
+ 'default': 1.0,
65
+ 'range': [0.01, 100.0]
66
+ },
67
+ 'fit_prior': {
68
+ 'description': textwrap.dedent('''\
69
+ Whether to learn class prior probabilities or not. If
70
+ false, a uniform prior will be used.'''),
71
+ 'default': True,
72
+ 'categorical': [True, False]
73
+ }
74
+ }
75
+ self.model: BernoulliNB = None
76
+
77
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
78
+ self.model = BernoulliNB(**self.passthrough_parameters())
79
+
80
+ self.model.fit(dataset.X, dataset.y)
81
+
82
+ return self
83
+
84
+ def suitable(self, dataset: Dataset) -> bool:
85
+ return dataset.type_of_target in \
86
+ ['binary', 'multiclass', 'multilabel-indicator']
87
+
88
+ def priorize(self, candidate: Candidate = None) -> float:
89
+ return 0.5 # neutral
@@ -0,0 +1,135 @@
1
+ """[STEP] CatBoost"""
2
+ import textwrap
3
+ from typing import Any
4
+ from catboost import CatBoostClassifier, CatBoostError
5
+ from sklearn.preprocessing import LabelEncoder
6
+ from ....predictor import Predictor
7
+ from ....dataset import Dataset
8
+ from ....candidate import Candidate
9
+ from ....decorators.all import is_step
10
+ from ....logger import Logger
11
+
12
+
13
+ @is_step('predictor', 'tabular', 'classifier', 'minimal_predictor')
14
+ class ActCatBoost(Predictor):
15
+ """[STEP] CatBoost Classifier"""
16
+
17
+ name: str = "CatBoost Classifier"
18
+ _usage: str = "Use when you want high-accuracy tabular classification with categorical features, often stronger than ActDecisionTreeClassifier or ActExtraTreesClassifier. Applicable to binary or multiclass tabular data. Avoid when data is tiny, compute is tight, or you prefer ActGaussianNb."
19
+ _description: str = textwrap.dedent('''\
20
+ CatBoostClassifier is a powerful tool that helps computers make accurate
21
+ predictions by learning from both positive and negative examples simultaneously.''')
22
+ _description_long: str = textwrap.dedent('''\
23
+ CatBoostClassifier is a gradient boosting algorithm specifically
24
+ designed for classification tasks.''')
25
+ refs: list[dict[str, Any]] = [
26
+ {
27
+ 'year': 2017,
28
+ 'name': 'CatBoost: unbiased boosting with categorical features',
29
+ 'authors': [
30
+ 'Liudmila Prokhorenkova',
31
+ 'Gleb Gusev',
32
+ 'Aleksandr Vorobev',
33
+ 'Anna Veronika Dorogush',
34
+ 'Andrey Gulin'
35
+ ],
36
+ 'doi': 'https://doi.org/10.48550/arXiv.1706.09516',
37
+ 'publisher': 'Advances in Neural Information Processing Systems 31 (NeurIPS 2018)'
38
+ }
39
+ ]
40
+
41
+ def __init__(self):
42
+ self.configuration = {
43
+ 'iterations': {
44
+ 'description': 'The maximum number of trees that can be built.',
45
+ 'default': 1000,
46
+ 'range': [100, 10000]
47
+ },
48
+ 'learning_rate': {
49
+ 'description': 'The learning rate.',
50
+ 'default': 0.03,
51
+ 'range': [0.001, 1.0]
52
+ },
53
+ 'depth': {
54
+ 'description': 'Depth of the tree.',
55
+ 'default': 6,
56
+ 'range': [1, 16]
57
+ },
58
+ 'l2_leaf_reg': {
59
+ 'description': 'Coefficient at the L2 regularization term of the cost function.',
60
+ 'default': 3,
61
+ 'range': [0, 10]
62
+ },
63
+ 'border_count': {
64
+ 'description': 'The number of splits for numerical features.',
65
+ 'default': 254,
66
+ 'range': [1, 255]
67
+ },
68
+ 'loss_function': {
69
+ 'description': 'The metric to use in training.',
70
+ 'default': 'Logloss',
71
+ 'categorical': ['Logloss', 'CrossEntropy', 'MultiClass', 'MultiClassOneVsAll']
72
+ },
73
+ 'eval_metric': {
74
+ 'description': 'The metric to be used for validation data.',
75
+ 'default': 'AUC',
76
+ 'categorical': ['AUC', 'Accuracy', 'Logloss']
77
+ },
78
+ 'bootstrap_type': {
79
+ 'description': 'The method for sampling the weights of objects.',
80
+ 'default': 'Bayesian',
81
+ 'categorical': ['Bayesian', 'Bernoulli', 'MVS']
82
+ },
83
+ 'leaf_estimation_iterations': {
84
+ 'description': 'The number of iterations for leaf estimation.',
85
+ 'default': 10,
86
+ 'range': [1, 50]
87
+ }
88
+ }
89
+
90
+ self.model: CatBoostClassifier = None
91
+ self.label_encoder: LabelEncoder = LabelEncoder()
92
+
93
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
94
+ if dataset.type_of_target == 'binary':
95
+ self.configuration['loss_function']['categorical'] = ['Logloss', 'CrossEntropy']
96
+ else:
97
+ self.configuration['eval_metric']['categorical'] = ['AUC', 'Accuracy']
98
+ self.configuration['loss_function']['categorical'] = [
99
+ 'MultiClass',
100
+ 'MultiClassOneVsAll'
101
+ ]
102
+ self.check_configuration()
103
+
104
+ self.model = CatBoostClassifier(verbose=0, **self.passthrough_parameters())
105
+ self.label_encoder.fit(dataset.y)
106
+ encoded_target = self.label_encoder.transform(dataset.y)
107
+ try:
108
+ self.model.fit(dataset.X, encoded_target)
109
+ except CatBoostError as exc:
110
+ self._log_failure(dataset, exc)
111
+ raise ValueError(f"CatBoostClassifier training failed: {exc}") from exc
112
+ except Exception as exc: # pragma: no cover - defensive
113
+ self._log_failure(dataset, exc)
114
+ raise
115
+ return self
116
+
117
+ def _log_failure(self, dataset: Dataset, exc: Exception) -> None:
118
+ """Log enriched debug info when CatBoost crashes."""
119
+ shape = getattr(dataset.X, "shape", None)
120
+ message = (
121
+ "[CatBoostClassifier] crash detected "
122
+ f"(shape={shape}, target_len={len(dataset.y)}, "
123
+ f"params={self.passthrough_parameters()}): {exc}"
124
+ )
125
+ logger = Logger()
126
+ if logger.verbose <= 3 and logger.verbose != -1:
127
+ logger.console.log(message)
128
+ logger.error(message)
129
+
130
+ def suitable(self, dataset: Dataset) -> bool:
131
+ return dataset.type_of_target in \
132
+ ['binary', 'multiclass', 'multilabel-indicator']
133
+
134
+ def priorize(self, candidate: Candidate = None) -> float:
135
+ return 0.5 # neutral