PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,296 @@
1
+ """[STEP] Add KMeans distance/cluster features."""
2
+ import textwrap
3
+ import pandas as pd
4
+ from sklearn.cluster import KMeans
5
+ from ...actionable import Actionable
6
+ from ...candidate import Candidate
7
+ from ...data_type import DataType
8
+ from ...dataset import Dataset
9
+ from ...decorators.all import is_step
10
+
11
+
12
+ @is_step('features_preprocessing')
13
+ class ActKMeansFeatures(Actionable):
14
+ """[STEP] Add KMeans distance/cluster features."""
15
+
16
+ name: str = "KMeans Features"
17
+ _description: str = "Add KMeans distance and cluster assignment features"
18
+ _usage: str = "Use when numeric features may form clusters and you want distance/cluster signals (vs ActKernelPCA or ActFeatureAgglomeration). Applicable to numeric tables with enough rows for k clusters. Avoid when data are tiny, mostly categorical, or clustering adds noise."
19
+ _description_long: str = textwrap.dedent('''\
20
+ KMeans groups numeric observations into k clusters by minimizing the
21
+ within-cluster variance. This step fits KMeans on numeric features and
22
+ appends distance-to-centroid features and, optionally, the assigned
23
+ cluster label. These additional features can surface non-linear structure
24
+ in the numeric feature space for downstream models.
25
+ ''')
26
+
27
+ def __init__(self) -> None:
28
+ self.configuration = {
29
+ 'n_clusters': {
30
+ 'description': 'Number of clusters to form.',
31
+ 'default': 8,
32
+ 'range': [2, 200]
33
+ },
34
+ 'init': {
35
+ 'description': 'Initialization method.',
36
+ 'default': 'k-means++',
37
+ 'categorical': ['k-means++', 'random']
38
+ },
39
+ 'n_init': {
40
+ 'description': 'Number of time the k-means algorithm will be run.',
41
+ 'default': 10,
42
+ 'range': [1, 20]
43
+ },
44
+ 'max_iter': {
45
+ 'description': 'Maximum number of iterations per run.',
46
+ 'default': 300,
47
+ 'range': [50, 1000]
48
+ },
49
+ 'tol': {
50
+ 'description': 'Relative tolerance with regards to inertia.',
51
+ 'default': 0.0001,
52
+ 'range': [1e-05, 0.01]
53
+ },
54
+ 'algorithm': {
55
+ 'description': 'KMeans algorithm variant.',
56
+ 'default': 'lloyd',
57
+ 'categorical': ['lloyd', 'elkan']
58
+ },
59
+ 'random_state': {
60
+ 'description': 'Random state for reproducibility.',
61
+ 'default': 42
62
+ },
63
+ 'add_distances': {
64
+ 'description': 'Whether to add distance-to-centroid features.',
65
+ 'default': True,
66
+ 'passthrough': False
67
+ },
68
+ 'distance_mode': {
69
+ 'description': 'Distance features to add: all or closest.',
70
+ 'default': 'all',
71
+ 'categorical': ['all', 'closest'],
72
+ 'passthrough': False
73
+ },
74
+ 'add_cluster_label': {
75
+ 'description': 'Whether to add the cluster assignment feature.',
76
+ 'default': True,
77
+ 'passthrough': False
78
+ },
79
+ 'feature_prefix': {
80
+ 'description': 'Prefix for generated KMeans features.',
81
+ 'default': 'kmeans',
82
+ 'passthrough': False
83
+ }
84
+ }
85
+
86
+ self.columns: list[str] = []
87
+ self.active_columns: list[str] = []
88
+ self.preprocessor: KMeans | None = None
89
+ self.distance_feature_names: list[str] = []
90
+ self.label_column: str | None = None
91
+ self.feature_prefix: str = 'kmeans'
92
+ self.add_distances: bool = True
93
+ self.add_cluster_label: bool = True
94
+ self.distance_mode: str = 'all'
95
+ self.impute_values: pd.Series | None = None
96
+ self.optimizable: bool = True
97
+
98
+ @staticmethod
99
+ def _coerce_int(value: object, default: int) -> int:
100
+ try:
101
+ return int(value)
102
+ except (TypeError, ValueError):
103
+ return default
104
+
105
+ @staticmethod
106
+ def _coerce_bool(value: object, default: bool) -> bool:
107
+ if isinstance(value, bool):
108
+ return value
109
+ if isinstance(value, str):
110
+ lowered = value.strip().lower()
111
+ if lowered in ('1', 'true', 'yes', 'y'):
112
+ return True
113
+ if lowered in ('0', 'false', 'no', 'n'):
114
+ return False
115
+ return default
116
+
117
+ @staticmethod
118
+ def _coerce_str(value: object, default: str) -> str:
119
+ if value is None:
120
+ return default
121
+ text = str(value).strip()
122
+ return text or default
123
+
124
+ @staticmethod
125
+ def _resolve_distance_mode(value: object) -> str:
126
+ if value is None:
127
+ return 'all'
128
+ text = str(value).strip().lower()
129
+ if text in ('closest', 'min', 'nearest'):
130
+ return 'closest'
131
+ return 'all'
132
+
133
+ @staticmethod
134
+ def _select_active_columns(values: pd.DataFrame) -> list[str]:
135
+ if values.empty:
136
+ return []
137
+ unique_counts = values.nunique(dropna=True)
138
+ return [col for col in values.columns if unique_counts.get(col, 0) > 1]
139
+
140
+ @staticmethod
141
+ def _unique_name(name: str, reserved: set[str]) -> str:
142
+ if name not in reserved:
143
+ return name
144
+ idx = 1
145
+ candidate = f"{name}_{idx}"
146
+ while candidate in reserved:
147
+ idx += 1
148
+ candidate = f"{name}_{idx}"
149
+ return candidate
150
+
151
+ def _resolve_n_clusters(self, n_samples: int) -> int | None:
152
+ if n_samples < 2:
153
+ return None
154
+ n_clusters = self._coerce_int(self.get_config('n_clusters'), 8)
155
+ n_clusters = max(2, n_clusters)
156
+ return min(n_clusters, n_samples)
157
+
158
+ def _build_kmeans(self, n_clusters: int) -> KMeans:
159
+ params = self.passthrough_parameters()
160
+ params['n_clusters'] = int(n_clusters)
161
+ params['n_init'] = self._coerce_int(params.get('n_init'), 10)
162
+ params['max_iter'] = self._coerce_int(params.get('max_iter'), 300)
163
+ params['tol'] = float(params.get('tol', 0.0001))
164
+ try:
165
+ return KMeans(**params)
166
+ except TypeError:
167
+ params.pop('algorithm', None)
168
+ return KMeans(**params)
169
+
170
+ def fit(self, dataset: Dataset) -> Actionable:
171
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
172
+ self.active_columns = []
173
+ self.preprocessor = None
174
+ self.distance_feature_names = []
175
+ self.label_column = None
176
+ self.impute_values = None
177
+ self.feature_prefix = self._coerce_str(self.get_config('feature_prefix'), 'kmeans')
178
+ self.add_distances = self._coerce_bool(self.get_config('add_distances'), True)
179
+ self.add_cluster_label = self._coerce_bool(self.get_config('add_cluster_label'), True)
180
+ self.distance_mode = self._resolve_distance_mode(self.get_config('distance_mode'))
181
+ self.explanations = []
182
+
183
+ if not self.columns or dataset.X.empty:
184
+ return self
185
+
186
+ values = dataset.X[self.columns]
187
+ self.active_columns = self._select_active_columns(values)
188
+ if not self.active_columns:
189
+ return self
190
+
191
+ if not self.add_distances and not self.add_cluster_label:
192
+ return self
193
+
194
+ active_values = values[self.active_columns]
195
+ n_clusters = self._resolve_n_clusters(active_values.shape[0])
196
+ if n_clusters is None:
197
+ return self
198
+
199
+ self.configure('n_clusters', n_clusters) # pylint: disable=too-many-function-args
200
+ self.impute_values = active_values.median(numeric_only=True)
201
+ active_values = active_values.fillna(self.impute_values)
202
+
203
+ self.preprocessor = self._build_kmeans(n_clusters)
204
+ self.preprocessor.fit(active_values)
205
+
206
+ reserved = set(dataset.X.columns)
207
+ if self.add_distances:
208
+ if self.distance_mode == 'closest':
209
+ name = self._unique_name(f"{self.feature_prefix}_cluster_dist", reserved)
210
+ self.distance_feature_names = [name]
211
+ reserved.add(name)
212
+ else:
213
+ self.distance_feature_names = []
214
+ for idx in range(n_clusters):
215
+ name = self._unique_name(f"{self.feature_prefix}_cluster_{idx}_dist", reserved)
216
+ self.distance_feature_names.append(name)
217
+ reserved.add(name)
218
+
219
+ if self.add_cluster_label:
220
+ self.label_column = self._unique_name(f"{self.feature_prefix}_cluster", reserved)
221
+ reserved.add(self.label_column)
222
+
223
+ if self.add_distances and self.distance_feature_names:
224
+ self.explanations.append(
225
+ f"Added {len(self.distance_feature_names)} KMeans distance features."
226
+ )
227
+ if self.add_cluster_label and self.label_column:
228
+ self.explanations.append(
229
+ f"Added KMeans cluster label feature `{self.label_column}`."
230
+ )
231
+
232
+ return self
233
+
234
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
235
+ """Apply KMeans distance/cluster features.
236
+
237
+ :param pd.DataFrame X: DataFrame to transform
238
+ :return: Transformed dataset
239
+ """
240
+ if self.preprocessor is None or not self.active_columns:
241
+ return X
242
+
243
+ if any(column not in X.columns for column in self.active_columns):
244
+ return X
245
+
246
+ values = X[self.active_columns]
247
+ if self.impute_values is not None:
248
+ values = values.fillna(self.impute_values)
249
+
250
+ if self.add_distances and self.distance_feature_names:
251
+ distances = self.preprocessor.transform(values)
252
+ if self.distance_mode == 'closest':
253
+ distances = distances.min(axis=1).reshape(-1, 1)
254
+ distance_df = pd.DataFrame(
255
+ distances,
256
+ columns=self.distance_feature_names,
257
+ index=X.index
258
+ )
259
+ X = pd.concat([X, distance_df], axis=1)
260
+
261
+ if self.add_cluster_label and self.label_column:
262
+ X[self.label_column] = self.preprocessor.predict(values)
263
+
264
+ return X
265
+
266
+ def suitable(self, dataset: Dataset) -> bool:
267
+ if dataset.X.empty:
268
+ return False
269
+ if not self._coerce_bool(self.get_config('add_distances'), True) \
270
+ and not self._coerce_bool(self.get_config('add_cluster_label'), True):
271
+ return False
272
+ columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
273
+ if not columns:
274
+ return False
275
+ values = dataset.X[columns]
276
+ active = self._select_active_columns(values)
277
+ if not active:
278
+ return False
279
+ return self._resolve_n_clusters(values.shape[0]) is not None
280
+
281
+ def priorize(self, candidate: Candidate = None) -> float:
282
+ if candidate is None:
283
+ return 0.0
284
+ dataset = candidate.dataset
285
+ if not self.suitable(dataset):
286
+ return 0.0
287
+ columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
288
+ if not columns:
289
+ return 0.0
290
+ values = dataset.X[columns]
291
+ active = self._select_active_columns(values)
292
+ if not active:
293
+ return 0.0
294
+ n_samples = max(1, values.shape[0])
295
+ ratio = len(active) / n_samples
296
+ return min(1.0, 0.2 + ratio)
@@ -0,0 +1,143 @@
1
+ """[STEP] Decompose features with KernelPCA"""
2
+ import textwrap
3
+ import pandas as pd
4
+ from sklearn.decomposition import KernelPCA
5
+ from ...actionable import Actionable
6
+ from ...dataset import Dataset
7
+ from ...candidate import Candidate
8
+ from ...decorators.all import is_step
9
+
10
+
11
+ def _is_numeric_matrix(values: pd.DataFrame) -> bool:
12
+ if values.empty:
13
+ return False
14
+ for column in values.columns:
15
+ if not pd.api.types.is_numeric_dtype(values[column]):
16
+ return False
17
+ return not values.isna().any().any()
18
+
19
+
20
+ @is_step('features_preprocessing')
21
+ class ActKernelPCA(Actionable):
22
+ """[STEP] Apply KernelPCA for dimensionality reduction"""
23
+ name = "KernelPCA"
24
+ _description = "Perform Kernel Principal Component Analysis (KernelPCA) on a dataset"
25
+ _description_long = textwrap.dedent('''\
26
+ KernelPCA is a dimensionality reduction technique that extends Principal Component Analysis (PCA)
27
+ using kernel methods. It projects data into a higher-dimensional space before performing PCA,
28
+ enabling it to capture complex, non-linear structures in the data. KernelPCA is useful for reducing
29
+ dimensionality while preserving intricate patterns and relationships within the data.
30
+ ''')
31
+ _usage = "Use when non-linear structure matters and linear reductions are insufficient; consider ActFastICA if you want independent components. Applicable to dense numeric, scaled features. Avoid when data is very large, sparse, or interpretability is required."
32
+
33
+ refs = [
34
+ {
35
+ 'year': 1997,
36
+ 'name': 'Kernel principal component analysis',
37
+ 'authors': [
38
+ 'Bernhard Schölkopf',
39
+ 'Alexander Smola',
40
+ 'Klaus-Robert Müller'
41
+ ],
42
+ 'doi': 'https://doi.org/10.1007/BFb0020217',
43
+ 'publisher': 'Springer, Berlin, Heidelberg'
44
+ },
45
+ {
46
+ 'year': 2003,
47
+ 'name': 'Learning to find pre-images',
48
+ 'authors': [
49
+ 'Jason Weston',
50
+ 'Bernhard Schölkopf',
51
+ 'Gökhan Bakir'
52
+ ],
53
+ 'doi': 'https://proceedings.neurips.cc/paper_files/paper/2003/file/ \
54
+ ac1ad983e08ad3304a97e147f522747e-Paper.pdf',
55
+ 'publisher': 'Advances in neural information processing systems 16 (2004) page 449--456'
56
+ },
57
+ {
58
+ 'year': 2009,
59
+ 'name': 'Finding structure with randomness: Probabilistic algorithms for constructing \
60
+ approximate matrix decompositions',
61
+ 'authors': [
62
+ 'Nathan Halko',
63
+ 'Per-Gunnar Martinsson',
64
+ 'Joel A. Tropp'
65
+ ],
66
+ 'doi': 'https://doi.org/10.48550/arXiv.0909.4061',
67
+ 'publisher': 'SIAM Rev., Survey and Review section, Vol.53, No.2 page 217--288'
68
+ },
69
+ {
70
+ 'year': 2011,
71
+ 'name': 'A randomized algorithm for the decomposition of matrices',
72
+ 'authors': [
73
+ 'Per-Gunnar Martinsson',
74
+ 'Vladimir Rokhlin',
75
+ 'Mark Tygert'
76
+ ],
77
+ 'doi': 'https://doi.org/10.1016/j.acha.2010.02.003',
78
+ 'publisher': 'Applied and Computational Harmonic Analysis, Vol.30, No.1 page 47--68'
79
+ }
80
+ ]
81
+ def __init__(self):
82
+ self.configuration = {
83
+ 'kernel': {
84
+ 'description': 'Kernel used for PCA.',
85
+ 'default': 'rbf',
86
+ 'categorical': ['poly', 'rbf', 'sigmoid', 'cosine']
87
+ },
88
+ 'n_components': {
89
+ 'description': 'Number of components to keep.',
90
+ 'default': 100,
91
+ 'range': [10, 2000]
92
+ },
93
+ 'coef0': {
94
+ 'description': textwrap.dedent('''\
95
+ Independent term in poly and sigmoid kernels. Ignored by
96
+ other kernels.'''),
97
+ 'default': 1.0,
98
+ 'range': [-1.0, 1.0]
99
+ },
100
+ 'degree': {
101
+ 'description': 'Degree for poly kernels. Ignored by other kernels.',
102
+ 'default': 3,
103
+ 'range': [2, 5]
104
+ },
105
+ 'random_state': {
106
+ 'description': 'Random State',
107
+ 'default': 42
108
+ }
109
+ }
110
+ self.optimizable: bool = True
111
+ self.preprocessor: bool = None
112
+
113
+
114
+ def fit(self, dataset: Dataset) -> Actionable:
115
+ self.preprocessor = None
116
+ if not _is_numeric_matrix(dataset.X):
117
+ return self
118
+ try:
119
+ self.preprocessor = KernelPCA(**self.passthrough_parameters())
120
+ self.preprocessor.fit(dataset.X)
121
+ except ValueError:
122
+ higher_gamma = 1 / dataset.X.shape[1] + 0.05
123
+ self.preprocessor = KernelPCA(gamma=higher_gamma, **self.passthrough_parameters())
124
+ self.preprocessor.fit(dataset.X)
125
+
126
+ return self
127
+
128
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
129
+ """Apply KernelPCA
130
+
131
+ :param pd.DataFrame X: DataFrame to transform
132
+ :return: Transformed dataset
133
+ """
134
+
135
+ if self.preprocessor is None:
136
+ return X
137
+ return pd.DataFrame(self.preprocessor.transform(X))
138
+
139
+ def suitable(self, dataset: Dataset) -> bool:
140
+ return _is_numeric_matrix(dataset.X)
141
+
142
+ def priorize(self, candidate: Candidate = None) -> float:
143
+ return 0.5
@@ -0,0 +1,122 @@
1
+ """[STEP] Apply log1p to skewed numeric features."""
2
+ import textwrap
3
+ import numpy as np
4
+ import pandas as pd
5
+ from ...actionable import Actionable
6
+ from ...candidate import Candidate
7
+ from ...dataset import Dataset
8
+ from ...data_type import DataType
9
+ from ...decorators.all import is_step
10
+
11
+
12
+ @is_step('features_preprocessing')
13
+ class ActLogTransformer(Actionable):
14
+ """[STEP] Apply log1p to skewed numeric features."""
15
+
16
+ name: str = "Log1p Transformer"
17
+ _description: str = "Apply log1p to highly skewed numeric columns"
18
+ _usage: str = "Use when numeric features are highly right-skewed and >= -1, often before ActKBinsDiscretizer or ActKernelPCA. Applicable to continuous numeric columns with long-tailed distributions. Avoid when values are <= -1 or already log/scale transformed."
19
+ _description_long: str = textwrap.dedent('''\
20
+ This step detects numeric features with strong positive skew
21
+ and applies a log1p (log(1+x)) transformation to compress
22
+ extreme values. The transformation is automatic and only
23
+ applied to columns that meet the skewness threshold and have
24
+ values above the configured minimum.
25
+ ''')
26
+
27
+ def __init__(self):
28
+ self.columns: list[str] = []
29
+
30
+ self.configuration = {
31
+ 'skew_threshold': {
32
+ 'description': 'Minimum skewness required to apply log1p.',
33
+ 'default': 1.5,
34
+ 'range': [0.5, 5.0]
35
+ },
36
+ 'min_value': {
37
+ 'description': 'Minimum allowed value for applying log1p.',
38
+ 'default': 0.0,
39
+ 'range': [-0.99, 1.0]
40
+ }
41
+ }
42
+
43
+ self.optimizable: bool = True
44
+
45
+ @staticmethod
46
+ def _coerce_float(value: object, default: float) -> float:
47
+ try:
48
+ return float(value)
49
+ except (TypeError, ValueError):
50
+ return default
51
+
52
+ def _resolve_skew_threshold(self) -> float:
53
+ threshold = self._coerce_float(self.get_config('skew_threshold'), 1.5)
54
+ if threshold < 0:
55
+ threshold = 0.0
56
+ return threshold
57
+
58
+ def _resolve_min_value(self) -> float:
59
+ min_value = self._coerce_float(self.get_config('min_value'), 0.0)
60
+ if min_value <= -1.0:
61
+ min_value = -0.999
62
+ return min_value
63
+
64
+ def _select_columns(self, values: pd.DataFrame) -> list[str]:
65
+ if values.empty:
66
+ return []
67
+ skewness = values.skew().fillna(0.0)
68
+ min_values = values.min(skipna=True)
69
+ threshold = self._resolve_skew_threshold()
70
+ min_value = self._resolve_min_value()
71
+ eligible = (skewness >= threshold) & (min_values >= min_value)
72
+ return [col for col in values.columns if bool(eligible.get(col, False))]
73
+
74
+ def fit(self, dataset: Dataset) -> Actionable:
75
+ self.columns = []
76
+ columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
77
+ if not columns or dataset.X.empty:
78
+ return self
79
+ values = dataset.X[columns]
80
+ self.columns = self._select_columns(values)
81
+ return self
82
+
83
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
84
+ """Apply log1p
85
+
86
+ :param pd.DataFrame X: DataFrame to transform
87
+ :return: Transformed dataset
88
+ """
89
+ if not self.columns:
90
+ return X
91
+ columns = [col for col in self.columns if col in X.columns]
92
+ if not columns:
93
+ return X
94
+ min_value = self._resolve_min_value()
95
+ values = X[columns].clip(lower=min_value)
96
+ X[columns] = np.log1p(values)
97
+ return X
98
+
99
+ def suitable(self, dataset: Dataset) -> bool:
100
+ columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
101
+ if not columns or dataset.X.empty:
102
+ return False
103
+ values = dataset.X[columns]
104
+ return bool(self._select_columns(values))
105
+
106
+ def priorize(self, candidate: Candidate = None) -> float:
107
+ if candidate is None:
108
+ return 0.0
109
+ dataset = candidate.dataset
110
+ columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
111
+ if not columns or dataset.X.empty:
112
+ return 0.0
113
+ values = dataset.X[columns]
114
+ selected = self._select_columns(values)
115
+ if not selected:
116
+ return 0.0
117
+ skewness = values[selected].skew().fillna(0.0)
118
+ threshold = self._resolve_skew_threshold()
119
+ if threshold <= 0:
120
+ return 1.0
121
+ mean_skew = float(skewness.mean()) if not skewness.empty else 0.0
122
+ return min(1.0, mean_skew / (threshold * 2.0))
@@ -0,0 +1,100 @@
1
+ """[STEP] Decompose features with Nystroem"""
2
+ import textwrap
3
+ import pandas as pd
4
+ from sklearn.kernel_approximation import Nystroem
5
+ from ...actionable import Actionable
6
+ from ...dataset import Dataset
7
+ from ...candidate import Candidate
8
+ from ...decorators.all import is_step
9
+
10
+
11
+ def _is_numeric_matrix(values: pd.DataFrame) -> bool:
12
+ if values.empty:
13
+ return False
14
+ for column in values.columns:
15
+ if not pd.api.types.is_numeric_dtype(values[column]):
16
+ return False
17
+ return not values.isna().any().any()
18
+
19
+
20
+ @is_step('features_preprocessing')
21
+ class ActNystroem(Actionable):
22
+ """[STEP] Apply Nystroem method for dimensionality reduction"""
23
+
24
+ name: str = "Nystroem"
25
+ _description: str = "Apply the Nystroem method for dimensionality reduction \
26
+ over a list of columns"
27
+ _usage: str = "Use when you need a fast nonlinear kernel map approximation for large numeric data, as a lighter option than ActKernelPCA. Applicable to scaled continuous features with many rows. Avoid when you need independent components or tiny data where ActFastICA or ActKernelPCA is fine."
28
+ _description_long: str = textwrap.dedent('''\
29
+ The Nystroem method is a technique used for approximating kernel methods,
30
+ which helps in reducing the computational cost of kernel-based algorithms.
31
+ It approximates a kernel map using a subset of the data, making it suitable
32
+ for large datasets. This approach enables dimensionality reduction by creating
33
+ a low-rank approximation of the original kernel matrix.
34
+ ''')
35
+
36
+ def __init__(self):
37
+ self.configuration = {
38
+ 'kernel': {
39
+ 'description': 'Kernel map to be approximated.',
40
+ 'default': 'rbf',
41
+ 'categorical': ["poly", "rbf", "sigmoid", "cosine", "chi2"]
42
+ },
43
+ 'n_components': {
44
+ 'description': 'Number of components to keep.',
45
+ 'default': 100,
46
+ 'range': [50, 10000]
47
+ },
48
+ 'coef0': {
49
+ 'description': 'Zero coefficient for polynomial and sigmoid kernels.',
50
+ 'default': 0.0,
51
+ 'range': [-1.0, 1.0]
52
+ },
53
+ 'degree': {
54
+ 'description': 'Degree of the polynomial kernel.',
55
+ 'default': 3,
56
+ 'range': [2, 5]
57
+ },
58
+ 'gamma': {
59
+ 'description': textwrap.dedent('''\
60
+ Gamma parameter for the RBF, laplacian, polynomial,
61
+ exponential chi2 and sigmoid kernels.'''),
62
+ 'default': 0.1,
63
+ 'range': [3.06e-05, 8.0]
64
+ },
65
+ 'random_state': {
66
+ 'description': 'Random State',
67
+ 'default': 42
68
+ }
69
+ }
70
+ self.optimizable: bool = True
71
+ self.preprocessor: bool = None
72
+
73
+ def fit(self, dataset: Dataset) -> Actionable:
74
+ self.preprocessor = None
75
+ if not _is_numeric_matrix(dataset.X):
76
+ return self
77
+ self.preprocessor = Nystroem(**self.passthrough_parameters())
78
+ self.preprocessor.fit(dataset.X)
79
+ return self
80
+
81
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
82
+ """Apply Nystroem
83
+
84
+ :param pd.DataFrame X: DataFrame to transform
85
+ :return: Transformed dataset
86
+ """
87
+ if self.preprocessor is None:
88
+ return X
89
+ return pd.DataFrame(self.preprocessor.transform(X))
90
+
91
+ def priorize(self, candidate: Candidate = None) -> float:
92
+ return 0.5
93
+
94
+
95
+ def suitable(self, dataset: Dataset) -> bool:
96
+ if not _is_numeric_matrix(dataset.X):
97
+ return False
98
+ if self.get_config('kernel') == 'chi2' and not (dataset.X < 0).any().any():
99
+ self.configure('kernel', 'rbf') # pylint: disable=too-many-function-args
100
+ return True