PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,77 @@
1
+ """
2
+ [STEP] Decompose features with PCA
3
+ """
4
+ import textwrap
5
+ import pandas as pd
6
+ from sklearn.decomposition import PCA
7
+ from ...actionable import Actionable
8
+ from ...dataset import Dataset
9
+ from ...candidate import Candidate
10
+ from ...decorators.all import is_step
11
+
12
+
13
+ def _is_numeric_matrix(values: pd.DataFrame) -> bool:
14
+ if values.empty:
15
+ return False
16
+ for column in values.columns:
17
+ if not pd.api.types.is_numeric_dtype(values[column]):
18
+ return False
19
+ return not values.isna().any().any()
20
+
21
+
22
+ @is_step('features_preprocessing')
23
+ class ActPCA(Actionable):
24
+ """[STEP] Reduce dimensions with PCA"""
25
+
26
+ name: str = "PCA"
27
+ _description: str = "Apply PCA for dimensionality reduction over a list of columns"
28
+ _usage: str = "Use when you need fast linear dimensionality reduction for numeric features; consider ActKernelPCA or ActFastICA for nonlinear or independent components. Applicable to scaled numeric matrices. Avoid when features are categorical or you must keep original feature meaning."
29
+ _description_long: str = textwrap.dedent('''\
30
+ PCA, or Principal Component Analysis, is a dimensionality reduction technique.
31
+ It transforms the data into a set of linearly uncorrelated components, capturing
32
+ the maximum variance in the data with each successive component.
33
+ This method is unsupervised, meaning it does not require labeled data,
34
+ and is particularly useful for simplifying datasets while retaining
35
+ as much of the underlying structure as possible.
36
+ ''')
37
+
38
+ def __init__(self):
39
+ self.configuration = {
40
+ 'n_components': {
41
+ 'description': 'Number of components to keep.',
42
+ 'default': 0.999,
43
+ 'range': [0.5, 0.999]
44
+ },
45
+ 'random_state': {
46
+ 'description': 'Random State',
47
+ 'default': 42
48
+ }
49
+ }
50
+
51
+ self.optimizable = True
52
+ self.preprocessor = None
53
+
54
+ def fit(self, dataset: Dataset) -> Actionable:
55
+ self.preprocessor = None
56
+ if not _is_numeric_matrix(dataset.X):
57
+ return self
58
+
59
+ self.preprocessor = PCA(**self.passthrough_parameters())
60
+ self.preprocessor.fit(dataset.X)
61
+ return self
62
+
63
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
64
+ """Apply PCA
65
+
66
+ :param pd.DataFrame X: DataFrame to transform
67
+ :return: Transformed dataset
68
+ """
69
+ if self.preprocessor is None:
70
+ return X
71
+ return pd.DataFrame(self.preprocessor.transform(X))
72
+
73
+ def suitable(self, dataset: Dataset) -> bool:
74
+ return _is_numeric_matrix(dataset.X)
75
+
76
+ def priorize(self, candidate: Candidate = None) -> float:
77
+ return 0.5
@@ -0,0 +1,86 @@
1
+ """Experimental polynomial expansion, available only through an explicit import.
2
+
3
+ Unbounded output dimensionality can exhaust memory during automatic exploration.
4
+ Kept outside the default preprocessing stage; see docs/component_status.rst.
5
+ """
6
+ import textwrap
7
+ import pandas as pd
8
+ from sklearn.preprocessing import PolynomialFeatures
9
+ from ...actionable import Actionable
10
+ from ...dataset import Dataset
11
+ from ...candidate import Candidate
12
+ from ...decorators.all import is_step
13
+
14
+ def _is_numeric_matrix(values: pd.DataFrame) -> bool:
15
+ if values.empty:
16
+ return False
17
+ for column in values.columns:
18
+ if not pd.api.types.is_numeric_dtype(values[column]):
19
+ return False
20
+ return not values.isna().any().any()
21
+
22
+
23
+ @is_step('experimental')
24
+ class ActPolynomialFeatures(Actionable):
25
+ """[STEP] Preprocess with PolynomialFeatures"""
26
+
27
+ name: str = "Preprocess with PolynomialFeatures"
28
+ _usage: str = "Use when you want explicit polynomial interactions for linear models; consider ActKernelPCA for projection-based nonlinearity. Applicable to numeric tabular features with moderate dimensionality. Avoid when feature count will explode or when ActKBinsDiscretizer is a better match."
29
+ _description: str = textwrap.dedent('''\
30
+ PolynomialFeatures creates new features by combining existing
31
+ features mathematically. It squares, cubes, and multiplies features to
32
+ create more complex patterns.''')
33
+ _description_long: str = textwrap.dedent('''\
34
+ PolynomialFeatures is a preprocessing technique that
35
+ generates new features based on polynomial relationships between existing
36
+ features. This helps capture non-linear relationships in the data that may
37
+ not be apparent from the original features alone. PolynomialFeatures is
38
+ particularly useful when you suspect the underlying relationship in your data
39
+ might not be straightforward or linear.''')
40
+
41
+ def __init__(self):
42
+ self.configuration = {
43
+ 'include_bias': {
44
+ 'description': 'If True (default), then include a bias \
45
+ column, the feature in which all polynomial powers are zero',
46
+ 'default': True
47
+ },
48
+ 'interaction_only': {
49
+ 'description': 'If True, only interaction features are produced',
50
+ 'default': False
51
+ },
52
+ 'degree': {
53
+ 'description': 'Degree of the polynomial kernel.',
54
+ 'default': 3,
55
+ 'range': [2, 5]
56
+ }
57
+ }
58
+
59
+ self.optimizable: bool = True
60
+ self.preprocessor: bool = None
61
+
62
+ def fit(self, dataset: Dataset) -> Actionable:
63
+ self.preprocessor = None
64
+ if not _is_numeric_matrix(dataset.X):
65
+ return self
66
+
67
+ self.preprocessor = PolynomialFeatures(**self.passthrough_parameters())
68
+ self.preprocessor.fit(dataset.X)
69
+
70
+ return self
71
+
72
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
73
+ """Apply PolynomialFeatures
74
+
75
+ :param pd.DataFrame X: DataFrame to transform
76
+ :return: Transformed dataset
77
+ """
78
+ if self.preprocessor is None:
79
+ return X
80
+ return pd.DataFrame(self.preprocessor.transform(X))
81
+
82
+ def priorize(self, candidate: Candidate = None) -> float:
83
+ return 0.5
84
+
85
+ def suitable(self, dataset: Dataset) -> bool:
86
+ return _is_numeric_matrix(dataset.X)
@@ -0,0 +1,106 @@
1
+ """[STEP] Preprocess with PowerTransformer"""
2
+ from typing import Any
3
+ import textwrap
4
+ import pandas as pd
5
+ from sklearn.preprocessing import PowerTransformer
6
+ from ...actionable import Actionable
7
+ from ...dataset import Dataset
8
+ from ...candidate import Candidate
9
+ from ...decorators.all import is_step
10
+
11
+
12
+ def _is_numeric_matrix(values: pd.DataFrame) -> bool:
13
+ if values.empty:
14
+ return False
15
+ for column in values.columns:
16
+ if not pd.api.types.is_numeric_dtype(values[column]):
17
+ return False
18
+ return not values.isna().any().any()
19
+
20
+
21
+ @is_step('features_preprocessing')
22
+ class ActPowerTransformer(Actionable):
23
+ """[STEP] Preprocess with PowerTransformer"""
24
+
25
+ name: str = "Preprocess with PowerTransformer"
26
+ _description: str = textwrap.dedent('''\
27
+ PowerTransformer changes data to make it more "bell-curve" shaped.
28
+ It uses special math tricks to flatten out irregular distributions and make the data
29
+ behave more like a normal distribution.''')
30
+ _description_long: str = textwrap.dedent('''\
31
+ PowerTransformer is a preprocessing technique that applies a power
32
+ transformation to make data more Gaussian-like. PowerTransformer is useful when you want to apply
33
+ machine learning models that assume normal distribution,
34
+ even if your original data doesn't meet this assumption.
35
+ It helps make your data more compatible with many common ML algorithms.''')
36
+ _usage: str = "Use when numeric features are skewed and need Gaussian-like scaling, instead of ActKernelPCA or ActFastICA. Applicable to continuous numeric data with unimodal, non-normal distributions. Avoid when data are categorical/one-hot, already near-normal, or highly multimodal."
37
+ refs: list[dict[str, Any]] = [
38
+ {
39
+ 'year': 1964,
40
+ 'name': 'An Analysis of Transformations',
41
+ 'authors': [
42
+ 'G. E. P. Box',
43
+ 'D. R. Cox'
44
+ ],
45
+ 'doi': 'https://doi.org/10.1111/j.2517-6161.1964.tb00553.x',
46
+ 'publisher': 'Journal of the Royal Statistical Society: Series B (Methodological), \
47
+ Vol.26, No.2 page 211--243'
48
+ },
49
+ {
50
+ 'year': 2000,
51
+ 'name': 'A New Family of Power Transformations to Improve Normality or Symmetry',
52
+ 'authors': [
53
+ 'In-Kwon Yeo',
54
+ 'Richard A. Johnson'
55
+ ],
56
+ 'doi': 'https://doi.org/10.1093/biomet/87.4.954',
57
+ 'publisher': 'Oxford University Press, Biometrika Vol.87 No.4 page 954--959'
58
+ },
59
+ ]
60
+
61
+ def __init__(self):
62
+ self.configuration = {
63
+ 'method': {
64
+ 'description': 'The power transform method.',
65
+ 'default': 'yeo-johnson',
66
+ 'categorical': ['yeo-johnson', 'box-cox']
67
+ },
68
+ 'standardize': {
69
+ 'description': 'Set to True to apply zero-mean, \
70
+ unit-variance normalization to the transformed output.',
71
+ 'default': True
72
+ }
73
+ }
74
+
75
+ self.optimizable: bool = True
76
+ self.preprocessor: bool = None
77
+
78
+ def fit(self, dataset: Dataset) -> Actionable:
79
+ self.preprocessor = None
80
+ if not _is_numeric_matrix(dataset.X):
81
+ return self
82
+ self.preprocessor = PowerTransformer(**self.passthrough_parameters())
83
+ self.preprocessor.fit(dataset.X)
84
+
85
+ return self
86
+
87
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
88
+ """Apply PowerTransformer
89
+
90
+ :param pd.DataFrame X: DataFrame to transform
91
+ :return: Transformed dataset
92
+ """
93
+
94
+ if self.preprocessor is None:
95
+ return X
96
+ return pd.DataFrame(self.preprocessor.transform(X))
97
+
98
+ def priorize(self, candidate: Candidate = None) -> float:
99
+ return 0.5
100
+
101
+ def suitable(self, dataset: Dataset) -> bool:
102
+ if not _is_numeric_matrix(dataset.X):
103
+ return False
104
+ if self.get_config('method') == 'box-cox' and not (dataset.X < 0).any().any():
105
+ self.configure('method', 'yeo-johnson') # pylint: disable=too-many-function-args
106
+ return True
@@ -0,0 +1,114 @@
1
+ """[STEP] Preprocess with QuantileTransformer"""
2
+ import textwrap
3
+ import pandas as pd
4
+ from sklearn.preprocessing import QuantileTransformer
5
+ from ...actionable import Actionable
6
+ from ...dataset import Dataset
7
+ from ...candidate import Candidate
8
+ from ...data_type import DataType
9
+ from ...decorators.all import is_step
10
+
11
+
12
+ @is_step('features_preprocessing')
13
+ class ActQuantileTransformer(Actionable):
14
+ """[STEP] Preprocess with QuantileTransformer"""
15
+
16
+ name: str = "Preprocess with QuantileTransformer"
17
+ _description: str = textwrap.dedent('''\
18
+ QuantileTransformer remaps numeric features to a uniform or normal distribution,
19
+ reducing skew and making feature scales more comparable.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ QuantileTransformer estimates the empirical cumulative distribution for each
22
+ numeric feature and maps values to a chosen target distribution.
23
+ This non-linear transformation can reduce the impact of outliers and
24
+ produce more Gaussian-like features for models that benefit from it.''')
25
+ _usage: str = "Use when numeric features are skewed or unevenly scaled; compare ActKBinsDiscretizer. Applicable to continuous numeric columns before models preferring near-normal inputs. Avoid when original units or interpretability must stay, or data is very sparse; compare ActKernelPCA."
26
+
27
+ def __init__(self):
28
+ self.columns: list[str] = None
29
+ self.preprocessor: QuantileTransformer = None
30
+
31
+ self.configuration = {
32
+ 'n_quantiles': {
33
+ 'description': 'Number of quantiles to estimate.',
34
+ 'default': 1000,
35
+ 'range': [10, 1000]
36
+ },
37
+ 'output_distribution': {
38
+ 'description': 'Target distribution for the transformed data.',
39
+ 'default': 'normal',
40
+ 'categorical': ['uniform', 'normal']
41
+ },
42
+ 'subsample': {
43
+ 'description': 'Maximum number of samples used to estimate quantiles.',
44
+ 'default': 100000,
45
+ 'range': [1000, 200000]
46
+ },
47
+ 'random_state': {
48
+ 'description': 'Random state used when subsampling.',
49
+ 'default': 42
50
+ },
51
+ 'copy': {
52
+ 'description': 'Set to False to perform transformation in-place when possible.',
53
+ 'default': True,
54
+ 'categorical': [True, False]
55
+ }
56
+ }
57
+
58
+ self.optimizable: bool = True
59
+
60
+ def _build_transformer(self, n_samples: int) -> QuantileTransformer:
61
+ params = self.passthrough_parameters()
62
+ n_samples = max(1, int(n_samples))
63
+ n_quantiles = int(params.pop('n_quantiles'))
64
+ subsample = int(params.pop('subsample'))
65
+ n_quantiles = max(1, min(n_quantiles, n_samples))
66
+ subsample = max(1, min(subsample, n_samples))
67
+ params['n_quantiles'] = n_quantiles
68
+ params['subsample'] = subsample
69
+ return QuantileTransformer(**params)
70
+
71
+ def fit(self, dataset: Dataset) -> Actionable:
72
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
73
+ if self.columns and not dataset.X.empty:
74
+ values = dataset.X[self.columns]
75
+ if values.isna().any().any():
76
+ self.preprocessor = None
77
+ return self
78
+ self.preprocessor = self._build_transformer(values.shape[0])
79
+ self.preprocessor.fit(values)
80
+ else:
81
+ self.preprocessor = None
82
+ return self
83
+
84
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
85
+ """Apply QuantileTransformer
86
+
87
+ :param pd.DataFrame X: DataFrame to transform
88
+ :return: Transformed dataset
89
+ """
90
+ if self.preprocessor and self.columns:
91
+ X[self.columns] = self.preprocessor.transform(X[self.columns])
92
+ return X
93
+
94
+ def suitable(self, dataset: Dataset) -> bool:
95
+ columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
96
+ if not columns or dataset.X.empty:
97
+ return False
98
+ values = dataset.X[columns]
99
+ return not values.isna().any().any()
100
+
101
+ def priorize(self, candidate: Candidate = None) -> float:
102
+ if candidate is None:
103
+ return 0.0
104
+ columns = candidate.dataset.get_columns_names_by_type(DataType.NUMERIC)
105
+ if not columns or candidate.dataset.X.empty:
106
+ return 0.0
107
+ values = candidate.dataset.X[columns]
108
+ if values.empty:
109
+ return 0.0
110
+ skewness = values.skew().abs().fillna(0.0)
111
+ if skewness.empty:
112
+ return 0.0
113
+ mean_skew = float(skewness.mean())
114
+ return min(1.0, mean_skew / 2.0)
@@ -0,0 +1,88 @@
1
+ """[STEP] Decompose features with RBFSampler"""
2
+ from typing import Any
3
+ import textwrap
4
+ import pandas as pd
5
+ from sklearn.kernel_approximation import RBFSampler
6
+ from ...actionable import Actionable
7
+ from ...dataset import Dataset
8
+ from ...candidate import Candidate
9
+ from ...decorators.all import is_step
10
+
11
+ def _is_numeric_matrix(values: pd.DataFrame) -> bool:
12
+ if values.empty:
13
+ return False
14
+ for column in values.columns:
15
+ if not pd.api.types.is_numeric_dtype(values[column]):
16
+ return False
17
+ return not values.isna().any().any()
18
+
19
+
20
+ @is_step('features_preprocessing')
21
+ class ActRBFSampler(Actionable):
22
+ """[STEP] Approximate with RBFSampler"""
23
+
24
+ name: str = "Approximate with RBFSampler"
25
+ _usage: str = "Use when you want a fast nonlinear kernel approximation for numeric features, as a lighter alternative to ActKernelPCA. Applicable to dense tabular data where scaling is reasonable. Avoid when data are categorical heavy, very sparse, or when you need exact kernel features."
26
+ _description: str = textwrap.dedent('''\
27
+ RBFSampler is a tool that helps computers understand complex relationships
28
+ between things by turning them into simpler numbers.''')
29
+ _description_long: str = textwrap.dedent('''\
30
+ RBFSampler is a machine learning technique that transforms data into
31
+ a higher-dimensional space where it's easier for algorithms to find patterns.
32
+ It works by creating random projections of the original data onto a new set of axes.
33
+ This allows it to approximate the effects of a radial basis function kernel, which is a
34
+ mathematical way of measuring similarity between data points.''')
35
+ refs: list[dict[str, Any]] =[
36
+ {
37
+ 'year': 2008,
38
+ 'name': 'Weighted Sums of Random Kitchen Sinks: Replacing minimization with \
39
+ randomization in learning',
40
+ 'authors': [
41
+ 'Ali Rahimi',
42
+ 'Benjamin Recht'
43
+ ],
44
+ 'doi': None,
45
+ 'publisher': 'Advances in Neural Information Processing Systems 21 page 1313--1320'
46
+ }
47
+ ]
48
+
49
+ def __init__(self):
50
+ self.configuration = {
51
+ 'n_components': {
52
+ 'description': 'Number of components to keep.',
53
+ 'default': 100,
54
+ 'range': [50, 10000]
55
+ },
56
+ 'random_state': {
57
+ 'description': 'Random State',
58
+ 'default': 42
59
+ }
60
+ }
61
+
62
+ self.optimizable: bool = True
63
+ self.preprocessor: bool = None
64
+
65
+ def fit(self, dataset: Dataset) -> Actionable:
66
+ self.preprocessor = None
67
+ if not _is_numeric_matrix(dataset.X):
68
+ return self
69
+ self.preprocessor = RBFSampler(**self.passthrough_parameters())
70
+ self.preprocessor.fit(dataset.X)
71
+
72
+ return self
73
+
74
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
75
+ """Apply RBFSampler
76
+
77
+ :param pd.DataFrame X: DataFrame to transform
78
+ :return: Transformed dataset
79
+ """
80
+ if self.preprocessor is None:
81
+ return X
82
+ return pd.DataFrame(self.preprocessor.transform(X))
83
+
84
+ def priorize(self, candidate: Candidate = None) -> float:
85
+ return 0.5
86
+
87
+ def suitable(self, dataset: Dataset) -> bool:
88
+ return _is_numeric_matrix(dataset.X)
@@ -0,0 +1,112 @@
1
+ """[STEP] Decompose features with SelectPercentile"""
2
+
3
+ import textwrap
4
+ from collections.abc import Callable
5
+
6
+ import pandas as pd
7
+ from sklearn.feature_selection import SelectPercentile, chi2, f_classif
8
+ from ...actionable import Actionable
9
+ from ...dataset import Dataset
10
+ from ...candidate import Candidate
11
+ from ...decorators.all import is_step
12
+
13
+
14
+ def _is_numeric_matrix(values: pd.DataFrame) -> bool:
15
+ if values.empty:
16
+ return False
17
+ for column in values.columns:
18
+ if not pd.api.types.is_numeric_dtype(values[column]):
19
+ return False
20
+ return not values.isna().any().any()
21
+
22
+
23
+ def _resolve_score_func(value) -> Callable | None:
24
+ if callable(value):
25
+ return value
26
+ if isinstance(value, str):
27
+ lowered = value.strip().lower()
28
+ if lowered == 'chi2':
29
+ return chi2
30
+ if lowered == 'f_classif':
31
+ return f_classif
32
+ return None
33
+
34
+
35
+ @is_step('features_preprocessing')
36
+ class ActSelectPercentile(Actionable):
37
+ """[STEP] Preprocess with SelectPercentile"""
38
+
39
+ name: str = "Preprocess with SelectPercentile"
40
+ _usage: str = "Use when you need fast univariate feature selection by percentile, rather than ActKernelPCA. Applicable to non-negative features with classification targets (chi2 or f_classif). Avoid when you want feature engineering via clustering like ActKMeansFeatures."
41
+ _description: str = textwrap.dedent('''\
42
+ SelectPercentile is a tool that helps choose important features from a
43
+ group of variables by looking at how well each one predicts the outcome.''')
44
+ _description_long: str = textwrap.dedent('''\
45
+ SelectPercentile is a feature selection technique used in machine
46
+ learning. It works by assigning scores to each feature based on how well it predicts
47
+ the outcome. Then, it selects only the top-scoring percentage of features.
48
+ This helps reduce the number of variables while keeping the most informative ones.''')
49
+
50
+ def __init__(self):
51
+ self.configuration = {
52
+ 'score_func': {
53
+ 'description': 'function taking two arrays X and y, \
54
+ and returning a pair of arrays',
55
+ 'default': chi2,
56
+ 'categorical': [chi2, f_classif]
57
+ },
58
+ 'percentile': {
59
+ 'description': 'Percent of features to keep.',
60
+ 'default': 50.0,
61
+ 'range': [1.0, 99.0]
62
+ }
63
+ }
64
+
65
+ self.optimizable: bool = True
66
+ self.preprocessor: SelectPercentile | None = None
67
+
68
+ def fit(self, dataset: Dataset) -> Actionable:
69
+ self.preprocessor = None
70
+ if dataset.y is None or not _is_numeric_matrix(dataset.X):
71
+ return self
72
+
73
+ score_func = _resolve_score_func(self.get_config('score_func'))
74
+ if score_func is None:
75
+ return self
76
+
77
+ params = self.passthrough_parameters()
78
+ params['score_func'] = score_func
79
+ self.preprocessor = SelectPercentile(**params)
80
+ self.preprocessor.fit(dataset.X, dataset.y)
81
+
82
+ return self
83
+
84
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
85
+ """Apply SelectPercentile
86
+
87
+ :param pd.DataFrame X: DataFrame to transform
88
+ :return: Transformed dataset
89
+ """
90
+ if self.preprocessor is None:
91
+ return X
92
+ return pd.DataFrame(self.preprocessor.transform(X))
93
+
94
+ def suitable(self, dataset: Dataset) -> bool:
95
+ # Negative values are not supported
96
+ if dataset.y is None or dataset.type_of_target is None:
97
+ return False
98
+ if not _is_numeric_matrix(dataset.X):
99
+ return False
100
+ score_func = _resolve_score_func(self.get_config('score_func'))
101
+ if score_func is None:
102
+ return False
103
+ if dataset.type_of_target not in [
104
+ 'binary', 'multiclass', 'multilabel-indicator'
105
+ ]:
106
+ return False
107
+ if score_func == chi2:
108
+ return not (dataset.X < 0).any().any()
109
+ return True
110
+
111
+ def priorize(self, candidate: Candidate = None) -> float:
112
+ return 0.5