PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,337 @@
1
+ """[STEP] Drop or hash high-cardinality categorical columns."""
2
+ import textwrap
3
+ from typing import Any
4
+
5
+ import pandas as pd
6
+
7
+ from ...actionable import Actionable
8
+ from ...candidate import Candidate
9
+ from ...data_type import DataType
10
+ from ...dataset import Dataset
11
+ from ...decorators.all import is_step
12
+
13
+
14
+ @is_step('cleaning')
15
+ class ActDropHighCardinalityCategorical(Actionable):
16
+ """[STEP] Drop or hash high-cardinality categorical columns."""
17
+
18
+ name: str = 'Handle high-cardinality categorical columns'
19
+ _description: str = textwrap.dedent('''\
20
+ Drop or hash categorical columns whose cardinality meets or exceeds {max_unique}
21
+ or {unique_ratio_threshold:.0%} of non-missing rows.''')
22
+ _description_long: str = textwrap.dedent('''\
23
+ High-cardinality categorical columns can create large sparse encodings.
24
+ Columns with too many distinct values are either removed or replaced with
25
+ hashed integer codes (modulo {hash_bins}) depending on strategy {strategy}.
26
+ Columns with fewer than {min_non_null} non-missing rows are ignored.''')
27
+ _usage: str = "Use when categorical columns are extremely high-cardinality and you want drop/hash instead of ActCountVectorizer. Applicable to categorical features with high unique-to-row ratios. Avoid when categories are low-cardinality or predictive, or ActCategoricalImputer is enough."
28
+
29
+ def __init__(self) -> None:
30
+ self.configuration = {
31
+ 'strategy': {
32
+ 'description': 'How to handle high-cardinality columns.',
33
+ 'default': 'drop',
34
+ 'categorical': ['drop', 'hash']
35
+ },
36
+ 'max_unique': {
37
+ 'description': 'Maximum number of distinct values before handling.',
38
+ 'default': 50
39
+ },
40
+ 'unique_ratio_threshold': {
41
+ 'description': 'Minimum unique/non-missing ratio to mark as high-cardinality.',
42
+ 'default': 0.5,
43
+ 'range': [0.0, 1.0]
44
+ },
45
+ 'min_non_null': {
46
+ 'description': 'Minimum number of non-missing rows to evaluate.',
47
+ 'default': 10
48
+ },
49
+ 'hash_bins': {
50
+ 'description': 'Number of hash bins when strategy="hash".',
51
+ 'default': 64
52
+ },
53
+ 'hash_key': {
54
+ 'description': 'Stable hash key used for hashing (16 chars recommended).',
55
+ 'default': 'iaml_hashing_key'
56
+ },
57
+ 'missing_value': {
58
+ 'description': 'Numeric value used for missing categories when hashing.',
59
+ 'default': -1
60
+ }
61
+ }
62
+ self.columns_to_drop: list[str] = []
63
+ self.columns_to_hash: list[str] = []
64
+ self.cardinality_stats: dict[str, dict[str, float]] = {}
65
+ self.strategy: str = 'drop'
66
+ self.hash_bins: int = 0
67
+ self.hash_key: str | bytes | None = None
68
+ self.missing_value: int = -1
69
+
70
+ def fit(self, dataset: Dataset) -> Actionable:
71
+ self.columns_to_drop = []
72
+ self.columns_to_hash = []
73
+ self.cardinality_stats = {}
74
+ self.explanations = []
75
+
76
+ if dataset.X.empty:
77
+ return self
78
+
79
+ columns = self._select_columns(dataset)
80
+ if not columns:
81
+ return self
82
+
83
+ self.strategy = self._coerce_strategy(self.get_config('strategy'))
84
+ max_unique = self._coerce_int(self.get_config('max_unique'), 0)
85
+ ratio_threshold = self._coerce_ratio(self.get_config('unique_ratio_threshold'), 0.0)
86
+ min_non_null = self._coerce_int(self.get_config('min_non_null'), 0)
87
+
88
+ if max_unique <= 0 and ratio_threshold <= 0:
89
+ return self
90
+
91
+ stats = self._cardinality_stats(dataset.X[columns])
92
+ high_columns = self._high_cardinality_columns(
93
+ columns,
94
+ stats,
95
+ max_unique,
96
+ ratio_threshold,
97
+ min_non_null
98
+ )
99
+
100
+ if not high_columns:
101
+ return self
102
+
103
+ if self.strategy == 'hash':
104
+ self.hash_bins = self._coerce_bins(self.get_config('hash_bins'), 64)
105
+ self.hash_key = self._coerce_hash_key(self.get_config('hash_key'))
106
+ self.missing_value = self._coerce_int(self.get_config('missing_value'), -1)
107
+ self.columns_to_hash = high_columns
108
+ else:
109
+ self.columns_to_drop = high_columns
110
+
111
+ self.cardinality_stats = self._stats_to_dict(stats, high_columns)
112
+ self._build_explanations(stats, high_columns)
113
+
114
+ return self
115
+
116
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
117
+ if self.columns_to_drop:
118
+ drop_columns = [col for col in self.columns_to_drop if col in X.columns]
119
+ if drop_columns:
120
+ X = X.drop(columns=drop_columns)
121
+
122
+ if not self.columns_to_hash:
123
+ return X
124
+
125
+ hash_bins = self.hash_bins or self._coerce_bins(self.get_config('hash_bins'), 64)
126
+ hash_key = self.hash_key if self.hash_key is not None else \
127
+ self._coerce_hash_key(self.get_config('hash_key'))
128
+ missing_value = self.missing_value
129
+
130
+ for column in self.columns_to_hash:
131
+ if column not in X.columns:
132
+ continue
133
+ X[column] = self._hash_series(X[column], hash_bins, hash_key, missing_value)
134
+
135
+ return X
136
+
137
+ def suitable(self, dataset: Dataset) -> bool:
138
+ if dataset.X.empty:
139
+ return False
140
+
141
+ columns = self._select_columns(dataset)
142
+ if not columns:
143
+ return False
144
+
145
+ max_unique = self._coerce_int(self.get_config('max_unique'), 0)
146
+ ratio_threshold = self._coerce_ratio(self.get_config('unique_ratio_threshold'), 0.0)
147
+ min_non_null = self._coerce_int(self.get_config('min_non_null'), 0)
148
+
149
+ if max_unique <= 0 and ratio_threshold <= 0:
150
+ return False
151
+
152
+ stats = self._cardinality_stats(dataset.X[columns])
153
+ high_columns = self._high_cardinality_columns(
154
+ columns,
155
+ stats,
156
+ max_unique,
157
+ ratio_threshold,
158
+ min_non_null
159
+ )
160
+ return bool(high_columns)
161
+
162
+ def priorize(self, candidate: Candidate = None) -> float:
163
+ if candidate is None or candidate.dataset.X.empty:
164
+ return 0.0
165
+
166
+ columns = self._select_columns(candidate.dataset)
167
+ if not columns:
168
+ return 0.0
169
+
170
+ max_unique = self._coerce_int(self.get_config('max_unique'), 0)
171
+ ratio_threshold = self._coerce_ratio(self.get_config('unique_ratio_threshold'), 0.0)
172
+ min_non_null = self._coerce_int(self.get_config('min_non_null'), 0)
173
+
174
+ if max_unique <= 0 and ratio_threshold <= 0:
175
+ return 0.0
176
+
177
+ stats = self._cardinality_stats(candidate.dataset.X[columns])
178
+ high_columns = self._high_cardinality_columns(
179
+ columns,
180
+ stats,
181
+ max_unique,
182
+ ratio_threshold,
183
+ min_non_null
184
+ )
185
+ if not high_columns:
186
+ return 0.0
187
+
188
+ ratio = len(high_columns) / max(1, len(columns))
189
+ return min(1.5, 0.5 + ratio)
190
+
191
+ def _select_columns(self, dataset: Dataset) -> list[str]:
192
+ columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
193
+ category_columns = list(dataset.X.select_dtypes(include=['category']).columns)
194
+ seen: set[str] = set()
195
+ ordered: list[str] = []
196
+ for column in columns + category_columns:
197
+ if column in dataset.X.columns and column not in seen:
198
+ ordered.append(column)
199
+ seen.add(column)
200
+ return ordered
201
+
202
+ def _cardinality_stats(self, X: pd.DataFrame) -> pd.DataFrame:
203
+ non_null = X.notna().sum()
204
+ unique = X.nunique(dropna=True)
205
+ ratio = unique / non_null.replace(0, pd.NA)
206
+ ratio = ratio.fillna(0.0)
207
+ return pd.DataFrame({'unique': unique, 'non_null': non_null, 'ratio': ratio})
208
+
209
+ def _high_cardinality_columns(
210
+ self,
211
+ columns: list[str],
212
+ stats: pd.DataFrame,
213
+ max_unique: int,
214
+ ratio_threshold: float,
215
+ min_non_null: int
216
+ ) -> list[str]:
217
+ eligible = stats['non_null'] >= min_non_null
218
+ conditions = pd.Series(False, index=stats.index)
219
+ if max_unique > 0:
220
+ conditions |= stats['unique'] >= max_unique
221
+ if ratio_threshold > 0:
222
+ conditions |= stats['ratio'] >= ratio_threshold
223
+ high = eligible & conditions
224
+ return [column for column in columns if column in high.index and bool(high[column])]
225
+
226
+ def _build_explanations(self, stats: pd.DataFrame, columns: list[str]) -> None:
227
+ for column in columns:
228
+ values = stats.loc[column]
229
+ unique = int(values['unique'])
230
+ non_null = int(values['non_null'])
231
+ ratio = float(values['ratio']) if non_null else 0.0
232
+ if self.strategy == 'hash':
233
+ self.explanations.append(
234
+ f"Hashed `{column}` into {self.hash_bins} bins "
235
+ f"(unique {unique}/{non_null}, ratio {ratio:.2%})."
236
+ )
237
+ else:
238
+ self.explanations.append(
239
+ f"Dropped `{column}` with {unique} unique values "
240
+ f"({ratio:.2%} of {non_null} non-missing)."
241
+ )
242
+
243
+ @staticmethod
244
+ def _stats_to_dict(stats: pd.DataFrame, columns: list[str]) -> dict[str, dict[str, float]]:
245
+ result: dict[str, dict[str, float]] = {}
246
+ for column in columns:
247
+ values = stats.loc[column]
248
+ result[column] = {
249
+ 'unique': float(values['unique']),
250
+ 'non_null': float(values['non_null']),
251
+ 'ratio': float(values['ratio'])
252
+ }
253
+ return result
254
+
255
+ @staticmethod
256
+ def _hash_series(
257
+ series: pd.Series,
258
+ bins: int,
259
+ hash_key: str | bytes | None,
260
+ missing_value: int
261
+ ) -> pd.Series:
262
+ if bins <= 0:
263
+ return pd.Series(missing_value, index=series.index, dtype='int64')
264
+
265
+ mask = series.isna()
266
+ values = series.astype('object')
267
+ hashed = ActDropHighCardinalityCategorical._hash_values(values, hash_key)
268
+ hashed = (hashed % bins).astype('int64')
269
+
270
+ if missing_value is not None and mask.any():
271
+ hashed = hashed.where(~mask, int(missing_value))
272
+
273
+ return hashed
274
+
275
+ @staticmethod
276
+ def _hash_values(values: pd.Series, hash_key: str | bytes | None) -> pd.Series:
277
+ if hash_key is None:
278
+ return pd.util.hash_pandas_object(values, index=False)
279
+ try:
280
+ return pd.util.hash_pandas_object(values, index=False, hash_key=hash_key)
281
+ except TypeError:
282
+ return pd.util.hash_pandas_object(values, index=False)
283
+
284
+ @staticmethod
285
+ def _coerce_int(value: Any, default: int) -> int:
286
+ try:
287
+ return int(value)
288
+ except (TypeError, ValueError):
289
+ return default
290
+
291
+ @staticmethod
292
+ def _coerce_ratio(value: Any, default: float) -> float:
293
+ try:
294
+ ratio = float(value)
295
+ except (TypeError, ValueError):
296
+ return default
297
+ return min(1.0, max(0.0, ratio))
298
+
299
+ @staticmethod
300
+ def _coerce_bins(value: Any, default: int) -> int:
301
+ try:
302
+ bins = int(value)
303
+ except (TypeError, ValueError):
304
+ return default
305
+ if bins < 2:
306
+ return default
307
+ return bins
308
+
309
+ @staticmethod
310
+ def _coerce_strategy(value: Any) -> str:
311
+ if isinstance(value, str):
312
+ normalized = value.strip().lower()
313
+ if normalized in {'drop', 'hash'}:
314
+ return normalized
315
+ return 'drop'
316
+
317
+ @staticmethod
318
+ def _coerce_hash_key(value: Any) -> str | bytes | None:
319
+ if value is None:
320
+ return None
321
+ if isinstance(value, (bytes, bytearray)):
322
+ key = bytes(value)
323
+ if not key:
324
+ return None
325
+ if len(key) < 16:
326
+ key = key.ljust(16, b'0')
327
+ elif len(key) > 16:
328
+ key = key[:16]
329
+ return key
330
+ key = str(value)
331
+ if not key:
332
+ return None
333
+ if len(key) < 16:
334
+ key = key.ljust(16, '0')
335
+ elif len(key) > 16:
336
+ key = key[:16]
337
+ return key
@@ -0,0 +1,75 @@
1
+ """[STEP] Drop Numerical Column"""
2
+ import textwrap
3
+ import pandas as pd
4
+ from ...actionable import Actionable
5
+ from ...dataset import Dataset
6
+ from ...data_type import DataType
7
+ from ...candidate import Candidate
8
+ from ...decorators.all import is_step
9
+
10
+
11
+ @is_step('cleaning')
12
+ class ActDropNumericalColumn(Actionable):
13
+ """[STEP] Drop Numerical Column"""
14
+
15
+ name: str = 'Remove numerical columns'
16
+ _description: str = textwrap.dedent('''\
17
+ Remove numerical columns where the proportion of empty rows
18
+ in the dataset is higher than {empty_threshold}.''')
19
+ _description_long: str = textwrap.dedent('''\
20
+ Remove numerical columns from the dataset where the proportion of empty
21
+ rows in the dataset is higher than {empty_threshold:.0%}. This ensure that every columns will
22
+ be relevant for the model to train on.''')
23
+ _usage: str = "Use when numeric columns are mostly empty and dropping is acceptable; compare ActDropCategoricalColumn for non-numeric drops. Applicable to datasets with numeric fields and high missingness. Avoid when you should impute or the numeric signal is critical."
24
+
25
+ def __init__(self):
26
+ self.columns_to_drop: list[str] = None
27
+ self.configuration = {
28
+ 'empty_threshold': {
29
+ 'description': textwrap.dedent('''\
30
+ Column with more or equal proportion of empty row will
31
+ dropped. 1 will drop all columns'''),
32
+ 'default': 0.5
33
+ }
34
+ }
35
+
36
+ def fit(self, dataset: Dataset) -> Actionable:
37
+ self.columns_to_drop = []
38
+ explain = []
39
+
40
+ for column in dataset.get_columns_names_by_type(DataType.NUMERIC):
41
+ values = dataset.X[column]
42
+ nan_values_count = values.isnull().sum()
43
+
44
+ if nan_values_count / len(values) >= self.get_config('empty_threshold'):
45
+ self.columns_to_drop.append(column)
46
+ explain.append((nan_values_count, len(values)))
47
+
48
+ self.explanations = [
49
+ f"""Dropped column **`{c}`** because **{v[0]}** values out of
50
+ **{v[1]}** (**{(v[0] / v[1] * 100):.2f}%**) are empty."""
51
+ for c, v in zip(self.columns_to_drop, explain)
52
+ ]
53
+
54
+ return self
55
+
56
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
57
+ """Drop columns.
58
+
59
+ :param pd.DataFrame X: DataFrame to transform.
60
+ :return: Transformed DataFrame.
61
+ """
62
+ return X.drop(self.columns_to_drop, axis=1)
63
+
64
+ def priorize(self, candidate: Candidate = None) -> float:
65
+ return 0
66
+
67
+ def suitable(self, dataset: Dataset) -> bool:
68
+ for column in dataset.get_columns_names_by_type(DataType.NUMERIC):
69
+ values = dataset.X[column]
70
+ nan_values_count = values.isnull().sum()
71
+
72
+ if nan_values_count / len(values) >= self.get_config('empty_threshold'):
73
+ return True
74
+
75
+ return False
@@ -0,0 +1,51 @@
1
+ """[STEP] Drop Textual Column"""
2
+ import textwrap
3
+ import pandas as pd
4
+ from ...actionable import Actionable
5
+ from ...dataset import Dataset
6
+ from ...data_type import DataType
7
+ from ...candidate import Candidate
8
+ from ...decorators.all import is_step
9
+
10
+
11
+ @is_step('cleaning', 'baseline_cleaning')
12
+ class ActDropTextualColumn(Actionable):
13
+ """[STEP] Drop Textual Column"""
14
+
15
+ name: str = 'Remove textual columns'
16
+ _description: str = 'Remove all columns containing textual data from the dataset'
17
+ _usage: str = 'Use when text columns are irrelevant for modeling or downstream steps. Applicable to datasets with free-text or short-text fields detected as text/object. Avoid when text should be vectorized or cleaned; consider ActCountVectorizer or ActCategoricalImputer.'
18
+ _description_long: str = textwrap.dedent('''\
19
+ Remove all columns containing textual data from the dataset.
20
+ This step is used to clean the dataset in order to perform other actions later on
21
+ that can't be applied to textual columns.''')
22
+ can_be_disabled: bool = False
23
+
24
+ def __init__(self):
25
+ self.columns_to_drop: list[str] = None
26
+
27
+ def fit(self, dataset: Dataset) -> Actionable:
28
+ self.columns_to_drop = list(set(
29
+ dataset.get_columns_names_by_type([DataType.TEXT, DataType.SHORT_TEXT]) + \
30
+ list(dataset.X.select_dtypes(include='object').columns)
31
+ ))
32
+
33
+ self.explanations = [
34
+ f'Dropped column **`{c}`**.' for c in self.columns_to_drop
35
+ ]
36
+
37
+ return self
38
+
39
+ def transform(self, X) -> pd.DataFrame:
40
+ """Drop columns.
41
+
42
+ :param pd.DataFrame X: DataFrame to transform.
43
+ :return: Transformed DataFrame.
44
+ """
45
+ return X.drop(self.columns_to_drop, axis=1)
46
+
47
+ def priorize(self, candidate: Candidate = None) -> float:
48
+ return 0
49
+
50
+ def suitable(self, dataset: Dataset) -> bool:
51
+ return bool(dataset.get_columns_names_by_type([DataType.TEXT, DataType.SHORT_TEXT]))
@@ -0,0 +1,56 @@
1
+ """Experimental target encoder, available only through an explicit module import.
2
+
3
+ This prototype transforms y rather than X, recomputes the mapping on each call,
4
+ and cannot reverse predictions. It is incompatible with the pipeline transformer
5
+ contract and must remain outside automatic cleaning. See docs/component_status.rst.
6
+ """
7
+
8
+ import textwrap
9
+ import numpy as np
10
+ from ...actionable import Actionable
11
+ from ...dataset import Dataset
12
+ from ...candidate import Candidate
13
+ from ...decorators.all import is_step
14
+
15
+
16
+ @is_step('experimental')
17
+ class ActCategoryStringToNumeric(Actionable):
18
+ """Encode categorical target column to numeric"""
19
+
20
+ name: str = 'Textual Category To Numeric Value'
21
+ _description: str ='Encode categorical text data column to numeric value'
22
+ _usage: str = 'Use when target labels are categorical strings and models require numeric y; prefer ActDropCategoricalColumn only if target is unusable. Applicable to single target columns of low-to-moderate cardinality. Avoid when target is already numeric or when ActDropHighCardinalityCategorical is more appropriate.'
23
+ _description_long: str = textwrap.dedent('''\
24
+ Retrieve all unique values from a column, then transform those values to a numeric type.
25
+ Exemple: If a column contain 3 uniques values like "coffee", "tea" and "water",
26
+ then all the coffee values will be transformed to 0, tea to 1 and water to 2.''')
27
+
28
+ def __init__(self):
29
+ self.column_to_encode: list[str] = None
30
+
31
+ def fit(self, dataset: Dataset) -> Actionable:
32
+ self.column_to_encode = [dataset.y]
33
+ self.explanations = [
34
+ f'Encoded column **`{np.unique(dataset.y)}`**. \
35
+ Mapping of categorical values to numerical values:\n- `{value}`: {i}'
36
+ for i, value in enumerate(np.unique(dataset.y)) ]
37
+
38
+ return self
39
+
40
+ def transform(self, y: np.array) -> np.array:
41
+ """Convert categorical target columns of candidate's dataset to numeric.
42
+
43
+ :param np.array y: Target column to transform.
44
+ :return: Transformed target column.
45
+ """
46
+ if np.issubdtype(y.dtype, object) or np.issubdtype(y.dtype, np.bool_):
47
+ categories = np.unique(y)
48
+ encoded_target = np.zeros_like(y, dtype=int)
49
+ for i, category in enumerate(categories):
50
+ encoded_target[y == category] = i
51
+ return encoded_target
52
+
53
+ return y
54
+
55
+ def priorize(self, candidate: Candidate = None) -> float:
56
+ return 0.4
@@ -0,0 +1,127 @@
1
+ """[STEP] Encode categorical features by relative frequency."""
2
+ import textwrap
3
+ from typing import Any
4
+
5
+ import pandas as pd
6
+
7
+ from ...actionable import Actionable
8
+ from ...candidate import Candidate
9
+ from ...data_type import DataType
10
+ from ...dataset import Dataset
11
+ from ...decorators.all import is_step
12
+
13
+
14
+ @is_step('cleaning')
15
+ class ActFrequencyEncoder(Actionable):
16
+ """[STEP] Encode categorical features by relative frequency."""
17
+
18
+ name: str = 'Frequency encoding'
19
+ _usage: str = "Use when categorical features need numeric encoding and frequencies are stable; compare ActDropHighCardinalityCategorical. Applicable to categorical/string features with moderate cardinality. Avoid when categories are very sparse, drifting, or too few rows to estimate."
20
+ _description: str = textwrap.dedent('''\
21
+ Encode categorical columns using their relative frequency.''')
22
+ _description_long: str = textwrap.dedent('''\
23
+ Replace each category with its relative frequency observed in the training data.
24
+ Frequencies are computed on non-missing values and unseen or missing values are
25
+ mapped to {unknown_value}.''')
26
+
27
+ def __init__(self) -> None:
28
+ self.columns: list[str] = []
29
+ self.encodings: dict[str, dict[Any, float]] = {}
30
+ self.fallback_values: dict[str, float] = {}
31
+ self.unknown_value: float = 0.0
32
+
33
+ self.configuration = {
34
+ 'unknown_value': {
35
+ 'description': 'Value used for unseen or missing categories.',
36
+ 'default': 0.0
37
+ }
38
+ }
39
+
40
+ def fit(self, dataset: Dataset) -> Actionable:
41
+ self.columns = self._select_columns(dataset)
42
+ self.encodings = {}
43
+ self.fallback_values = {}
44
+ self.explanations = []
45
+
46
+ if not self.columns or dataset.X.empty:
47
+ return self
48
+
49
+ self.unknown_value = self._coerce_float(self.get_config('unknown_value'), 0.0)
50
+
51
+ for column in self.columns:
52
+ series = dataset.X[column]
53
+ counts = series.value_counts(dropna=True)
54
+ total = int(counts.sum())
55
+ if total <= 0 or counts.empty:
56
+ self.fallback_values[column] = self.unknown_value
57
+ continue
58
+
59
+ frequencies = (counts / total).to_dict()
60
+ self.encodings[column] = frequencies
61
+ self.fallback_values[column] = self.unknown_value
62
+ self.explanations.append(
63
+ f"Encoded `{column}` using {len(frequencies)} categories."
64
+ )
65
+
66
+ return self
67
+
68
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
69
+ if not self.columns:
70
+ return X
71
+
72
+ for column in self.columns:
73
+ if column not in X.columns:
74
+ continue
75
+
76
+ mapping = self.encodings.get(column, {})
77
+ fallback = self.fallback_values.get(column, self.unknown_value)
78
+ fallback = self._coerce_float(fallback, 0.0)
79
+
80
+ encoded = X[column].map(mapping)
81
+ X[column] = encoded.fillna(fallback).astype(float)
82
+
83
+ return X
84
+
85
+ def suitable(self, dataset: Dataset) -> bool:
86
+ columns = self._select_columns(dataset)
87
+ if not columns or dataset.X.empty:
88
+ return False
89
+ return True
90
+
91
+ def priorize(self, candidate: Candidate = None) -> float:
92
+ if candidate is None or candidate.dataset.X.empty:
93
+ return 0.0
94
+
95
+ columns = self._select_columns(candidate.dataset)
96
+ if not columns:
97
+ return 0.0
98
+
99
+ total_rows = len(candidate.dataset.X)
100
+ if total_rows <= 0:
101
+ return 0.0
102
+
103
+ unique_counts = candidate.dataset.X[columns].nunique(dropna=True)
104
+ avg_cardinality = float((unique_counts / total_rows).mean())
105
+
106
+ total_columns = candidate.dataset.X.shape[1] or 1
107
+ cat_ratio = len(columns) / total_columns
108
+
109
+ return min(1.0, max(cat_ratio, avg_cardinality))
110
+
111
+ def _select_columns(self, dataset: Dataset) -> list[str]:
112
+ columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
113
+ category_columns = list(dataset.X.select_dtypes(include=['category']).columns)
114
+ seen = set()
115
+ ordered = []
116
+ for column in columns + category_columns:
117
+ if column in dataset.X.columns and column not in seen:
118
+ ordered.append(column)
119
+ seen.add(column)
120
+ return ordered
121
+
122
+ @staticmethod
123
+ def _coerce_float(value: Any, default: float) -> float:
124
+ try:
125
+ return float(value)
126
+ except (TypeError, ValueError):
127
+ return default