PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,186 @@
1
+ """[STEP] Vectorize large text with HashingVectorizer."""
2
+ import textwrap
3
+ from typing import Any
4
+
5
+ import pandas as pd
6
+ from sklearn.feature_extraction.text import HashingVectorizer
7
+
8
+ from ...actionable import Actionable
9
+ from ...candidate import Candidate
10
+ from ...data_type import DataType
11
+ from ...dataset import Dataset
12
+ from ...decorators.all import is_step
13
+
14
+
15
+ @is_step('cleaning')
16
+ class ActHashingVectorizer(Actionable):
17
+ """[STEP] Vectorize large text with HashingVectorizer."""
18
+
19
+ name: str = 'Hashing Vectorizer'
20
+ _description: str = textwrap.dedent('''\
21
+ Vectorize text columns with the hashing trick for large vocabularies.''')
22
+ _description_long: str = textwrap.dedent('''\
23
+ Convert long text columns into fixed-size hashed feature vectors without
24
+ building an explicit vocabulary. This keeps memory usage bounded even for
25
+ very large vocabularies, at the cost of possible hash collisions.''')
26
+ _usage: str = "Use when text columns have huge vocabularies and you need fixed-size features, especially vs ActCountVectorizer for memory bounds. Applicable to free-text columns that you want numeric n-grams from. Avoid when token interpretability or collision-free features are required."
27
+
28
+ def __init__(self) -> None:
29
+ self.columns: list[str] = []
30
+ self.vectorizer: HashingVectorizer | None = None
31
+ self.configuration = {
32
+ 'n_features': {
33
+ 'description': 'Number of hash bins (power of two recommended).',
34
+ 'default': 4096
35
+ },
36
+ 'ngram_min': {
37
+ 'description': 'Minimum n-gram size to include.',
38
+ 'default': 1
39
+ },
40
+ 'ngram_max': {
41
+ 'description': 'Maximum n-gram size to include.',
42
+ 'default': 2
43
+ },
44
+ 'alternate_sign': {
45
+ 'description': 'Use alternating signs to reduce hash collisions.',
46
+ 'default': False
47
+ },
48
+ 'binary': {
49
+ 'description': 'If True, store binary occurrences instead of counts.',
50
+ 'default': False
51
+ },
52
+ 'norm': {
53
+ 'description': 'Vector normalization ("l1", "l2", or None).',
54
+ 'default': 'l2'
55
+ },
56
+ 'lowercase': {
57
+ 'description': 'Lowercase text before hashing.',
58
+ 'default': True
59
+ }
60
+ }
61
+
62
+ def fit(self, dataset: Dataset) -> Actionable:
63
+ self.columns = []
64
+ self.vectorizer = None
65
+ self.explanations = []
66
+
67
+ columns = dataset.get_columns_names_by_type([DataType.TEXT])
68
+ if not columns or dataset.X.empty:
69
+ return self
70
+
71
+ params = self._build_vectorizer_params()
72
+ self.vectorizer = HashingVectorizer(**params)
73
+ self.columns = [column for column in columns if column in dataset.X.columns]
74
+
75
+ n_features = params['n_features']
76
+ for column in self.columns:
77
+ self.explanations.append(
78
+ f'Encoded text column **`{column}`** into **{n_features}** hashed features.'
79
+ )
80
+
81
+ if not self.columns:
82
+ self.explanations.append(
83
+ 'Hashing vectorizer skipped: no usable text columns.'
84
+ )
85
+
86
+ return self
87
+
88
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
89
+ """Apply HashingVectorizer to text columns.
90
+
91
+ :param pd.DataFrame X: DataFrame to transform.
92
+ :return: Transformed dataset.
93
+ """
94
+ if not self.columns or self.vectorizer is None:
95
+ return X
96
+
97
+ X = X.reset_index(drop=True)
98
+
99
+ for name in self.columns:
100
+ if name not in X.columns:
101
+ continue
102
+
103
+ values = X[name].fillna('').astype(str)
104
+ transformed = self.vectorizer.transform(values)
105
+ n_features = transformed.shape[1]
106
+ new_names = [f"{name}_hash_{i}" for i in range(n_features)]
107
+ vector_df = pd.DataFrame(transformed.toarray(), columns=new_names)
108
+
109
+ X = pd.concat([X, vector_df], axis=1).drop([name], axis=1)
110
+
111
+ return X
112
+
113
+ def suitable(self, dataset: Dataset) -> bool:
114
+ if dataset.X.empty:
115
+ return False
116
+ return bool(dataset.get_columns_names_by_type([DataType.TEXT]))
117
+
118
+ def priorize(self, candidate: Candidate = None) -> float:
119
+ if candidate is None or candidate.dataset.X.empty:
120
+ return 0.0
121
+
122
+ columns = candidate.dataset.get_columns_names_by_type([DataType.TEXT])
123
+ if not columns:
124
+ return 0.0
125
+
126
+ total_columns = candidate.dataset.X.shape[1] or 1
127
+ return min(1.0, len(columns) / total_columns)
128
+
129
+ def _build_vectorizer_params(self) -> dict[str, Any]:
130
+ n_features = self._coerce_positive_int(self.get_config('n_features'), 4096)
131
+ ngram_min = self._coerce_int(self.get_config('ngram_min'), 1)
132
+ ngram_max = self._coerce_int(self.get_config('ngram_max'), max(ngram_min, 1))
133
+ ngram_min = max(1, ngram_min)
134
+ ngram_max = max(ngram_min, ngram_max)
135
+
136
+ return {
137
+ 'n_features': n_features,
138
+ 'ngram_range': (ngram_min, ngram_max),
139
+ 'alternate_sign': self._coerce_bool(self.get_config('alternate_sign'), False),
140
+ 'binary': self._coerce_bool(self.get_config('binary'), False),
141
+ 'norm': self._coerce_norm(self.get_config('norm')),
142
+ 'lowercase': self._coerce_bool(self.get_config('lowercase'), True)
143
+ }
144
+
145
+ @staticmethod
146
+ def _coerce_int(value: Any, default: int) -> int:
147
+ try:
148
+ return int(value)
149
+ except (TypeError, ValueError):
150
+ return default
151
+
152
+ @staticmethod
153
+ def _coerce_positive_int(value: Any, default: int) -> int:
154
+ try:
155
+ numeric = int(value)
156
+ except (TypeError, ValueError):
157
+ return default
158
+ if numeric <= 0:
159
+ return default
160
+ return numeric
161
+
162
+ @staticmethod
163
+ def _coerce_bool(value: Any, default: bool) -> bool:
164
+ if isinstance(value, bool):
165
+ return value
166
+ if isinstance(value, str):
167
+ normalized = value.strip().lower()
168
+ if normalized in {'true', '1', 'yes', 'y'}:
169
+ return True
170
+ if normalized in {'false', '0', 'no', 'n'}:
171
+ return False
172
+ if value is None:
173
+ return default
174
+ return bool(value)
175
+
176
+ @staticmethod
177
+ def _coerce_norm(value: Any) -> str | None:
178
+ if value is None:
179
+ return None
180
+ if isinstance(value, str):
181
+ normalized = value.strip().lower()
182
+ if normalized in {'none', ''}:
183
+ return None
184
+ if normalized in {'l1', 'l2'}:
185
+ return normalized
186
+ return 'l2'
@@ -0,0 +1,152 @@
1
+ """[STEP] KNN imputer for numeric columns."""
2
+ import textwrap
3
+
4
+ import pandas as pd
5
+ from sklearn.impute import KNNImputer
6
+
7
+ from ...actionable import Actionable
8
+ from ...candidate import Candidate
9
+ from ...data_type import DataType
10
+ from ...dataset import Dataset
11
+ from ...decorators.all import is_step
12
+
13
+
14
+ @is_step('cleaning')
15
+ class ActKNNImputer(Actionable):
16
+ """[STEP] Impute missing numeric values using KNN."""
17
+
18
+ name: str = 'Impute missing values (KNN)'
19
+ _usage: str = 'Use when numeric features have gaps and you want to preserve rows vs ActDropNumericalColumn. Applicable to numeric columns with enough non-missing rows; use ActCategoricalImputer for categoricals. Avoid when missingness is extreme, dataset is tiny, or you plan to drop the column.'
20
+ _description: str = textwrap.dedent('''\
21
+ Impute missing numeric values using a k-nearest neighbors strategy.''')
22
+ _description_long: str = textwrap.dedent('''\
23
+ Uses scikit-learn KNNImputer to fill missing values in numeric columns
24
+ by averaging the k nearest neighbors in feature space. Non-numeric
25
+ columns are left untouched.''')
26
+
27
+ def __init__(self):
28
+ self.columns: list[str] = []
29
+ self.knn_columns: list[str] = []
30
+ self.imputer: KNNImputer | None = None
31
+ self._all_nan_cols: list[str] = []
32
+ self._fallback_values: dict[str, float] = {}
33
+ self._nan_stats: dict[str, tuple[int, int, float]] = {}
34
+
35
+ self.configuration = {
36
+ 'n_neighbors': {
37
+ 'description': 'Number of neighbors used for imputing missing values.',
38
+ 'default': 5,
39
+ 'range': [1, 50],
40
+ 'passthrough': False
41
+ },
42
+ 'weights': {
43
+ 'description': 'Weight function used in prediction.',
44
+ 'default': 'uniform',
45
+ 'categorical': ['uniform', 'distance']
46
+ },
47
+ 'metric': {
48
+ 'description': 'Distance metric to use for missing-aware KNN.',
49
+ 'default': 'nan_euclidean',
50
+ 'categorical': ['nan_euclidean']
51
+ }
52
+ }
53
+
54
+ def fit(self, dataset: Dataset) -> Actionable:
55
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
56
+ self.knn_columns = []
57
+ self.imputer = None
58
+ self._all_nan_cols = []
59
+ self._fallback_values = {}
60
+ self._nan_stats = {}
61
+ self.explanations = []
62
+
63
+ if not self.columns:
64
+ return self
65
+
66
+ X_num = dataset.X[self.columns]
67
+ total_rows = len(X_num)
68
+ if total_rows == 0:
69
+ self.explanations = ["Skipped KNN imputation: dataset has 0 rows."]
70
+ return self
71
+
72
+ missing_counts = X_num.isna().sum()
73
+ means = X_num.mean()
74
+ for column in self.columns:
75
+ missing = int(missing_counts[column])
76
+ pct = (missing / total_rows * 100.0) if total_rows else 0.0
77
+ self._nan_stats[column] = (missing, total_rows, pct)
78
+
79
+ mean_value = means[column]
80
+ if pd.isna(mean_value):
81
+ mean_value = 0.0
82
+ self._fallback_values[column] = float(mean_value)
83
+
84
+ self._all_nan_cols = [
85
+ column for column in self.columns if missing_counts[column] == total_rows
86
+ ]
87
+ self.knn_columns = [
88
+ column for column in self.columns if column not in self._all_nan_cols
89
+ ]
90
+
91
+ if self.knn_columns and total_rows >= 2:
92
+ n_neighbors = min(self.get_config('n_neighbors'), total_rows - 1)
93
+ n_neighbors = max(1, int(n_neighbors))
94
+ params = self.passthrough_parameters()
95
+ params['n_neighbors'] = n_neighbors
96
+ self.imputer = KNNImputer(**params)
97
+ self.imputer.fit(X_num[self.knn_columns])
98
+
99
+ if self.imputer is not None:
100
+ self.explanations = [
101
+ f"Imputed missing values of column **`{c}`** using **KNN** "
102
+ f"(**{n}** / **{t}**; **{pct:.2f}%** missing in train data)."
103
+ for c, (n, t, pct) in self._nan_stats.items()
104
+ if n > 0 and c in self.knn_columns
105
+ ]
106
+ if self._all_nan_cols:
107
+ self.explanations.append(
108
+ "Filled all-NaN numeric columns with 0.0: " +
109
+ ", ".join(f"`{c}`" for c in self._all_nan_cols) +
110
+ "."
111
+ )
112
+ if self.imputer is None and not self.explanations:
113
+ if any(n > 0 for n, _, _ in self._nan_stats.values()):
114
+ self.explanations.append(
115
+ "Skipped KNN imputation; filled numeric columns with their mean "
116
+ "(0.0 when undefined)."
117
+ )
118
+
119
+ return self
120
+
121
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
122
+ if not self.columns:
123
+ return X
124
+
125
+ if self.imputer is not None and self.knn_columns:
126
+ if all(column in X.columns for column in self.knn_columns):
127
+ X_knn = X[self.knn_columns].copy()
128
+ X_knn = X_knn.apply(pd.to_numeric, errors='coerce')
129
+ imputed = self.imputer.transform(X_knn)
130
+ X.loc[:, self.knn_columns] = imputed
131
+
132
+ for column, fill_value in self._fallback_values.items():
133
+ if column in X.columns:
134
+ X[column] = X[column].fillna(fill_value).infer_objects(copy=False)
135
+
136
+ return X
137
+
138
+ def suitable(self, dataset: Dataset) -> bool:
139
+ columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
140
+ if not columns or dataset.X.empty:
141
+ return False
142
+ return bool(dataset.X[columns].isna().any().any())
143
+
144
+ def priorize(self, candidate: Candidate = None) -> float:
145
+ if candidate is None or candidate.dataset.X.empty:
146
+ return 0.0
147
+ columns = candidate.dataset.get_columns_names_by_type(DataType.NUMERIC)
148
+ if not columns:
149
+ return 0.0
150
+ missing = candidate.dataset.X[columns].isna().sum().sum()
151
+ total = candidate.dataset.X[columns].size or 1
152
+ return min(1.0, missing / total)
@@ -0,0 +1,79 @@
1
+ """[STEP] Fill missing values with mean"""
2
+ import textwrap
3
+ import pandas as pd
4
+ import numpy as np
5
+ from ...actionable import Actionable
6
+ from ...dataset import Dataset
7
+ from ...candidate import Candidate
8
+ from ...decorators.all import is_step
9
+ from ...data_type import DataType
10
+
11
+
12
+ @is_step('cleaning', 'baseline_cleaning')
13
+ class ActMeanColumn(Actionable):
14
+ """[STEP] Fill missing values with the mean."""
15
+
16
+ name: str = 'Fill missing values'
17
+ _usage: str = "Use when numeric columns have some missing values and you want a quick baseline imputation over ActDropNumericalColumn. Applicable to numerical data with moderate missingness. Avoid when missingness is high or systematic, or when ActDropNumericalColumn is safer."
18
+ _description: str = textwrap.dedent('''\
19
+ Fill missing values with the mean of non-missing values
20
+ when the proportion of empty rows is lower than {empty_threshold:.0%}.''')
21
+ _description_long: str = textwrap.dedent('''\
22
+ Fill a column missings values with the mean of the columns
23
+ when the proportion of empty rows is lower than {empty_threshold}.
24
+ Work only for numerical columns.''')
25
+ can_be_disabled: bool = False
26
+
27
+ def __init__(self):
28
+ self.columns: list[str] = None
29
+ self.configuration:dict = {
30
+ 'empty_threshold': {
31
+ 'description': textwrap.dedent('''\
32
+ Column with less or equal proportion of empty row will be
33
+ fill with mean value. 1 will always fill void values'''),
34
+ 'default': 1 # TODO Review when adding new kind of imputer
35
+ }
36
+ }
37
+
38
+ def fit(self, dataset: Dataset) -> Actionable:
39
+ self.columns = []
40
+ explain = []
41
+
42
+ for column in dataset.get_columns_names_by_type(DataType.NUMERIC):
43
+ values = dataset.X[column]
44
+ nan_values_count = values.isnull().sum()
45
+
46
+ mean = values.mean()
47
+ if np.isnan(mean):
48
+ mean = 0
49
+
50
+ self.columns.append((column, mean))
51
+ explain.append((
52
+ nan_values_count,
53
+ len(values),
54
+ nan_values_count / len(values) * 100,
55
+ ))
56
+
57
+ self.explanations = [
58
+ f"""Filled missing values of column **`{c}`** with **{mean:.2f}**
59
+ (**{v[0]}** out of **{v[1]}** values (**{v[2]:.2f}**%)
60
+ were missing in train data)."""
61
+ for (c, mean), v in zip(self.columns, explain)
62
+ if v[0] > 0 # hide processings that affected no values
63
+ ]
64
+
65
+ return self
66
+
67
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
68
+ """Fill NA values with the mean.
69
+
70
+ :param pd.DataFrame x: DataFrame to transform.
71
+ :return: Transformed dataset.
72
+ """
73
+ for name, mean in self.columns:
74
+ X[name] = X[name].fillna(mean).infer_objects(copy=False)
75
+
76
+ return X
77
+
78
+ def priorize(self, candidate: Candidate = None) -> float:
79
+ return 1 - (candidate.dataset.X.isnull().sum().min() / len(candidate.dataset.X))