PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,94 @@
1
+ """[STEP] Drop Columns with High Missing Values"""
2
+ import textwrap
3
+ from typing import Any
4
+ import pandas as pd
5
+ from ...actionable import Actionable
6
+ from ...dataset import Dataset
7
+ from ...data_type import DataType
8
+ from ...candidate import Candidate
9
+ from ...decorators.all import is_step
10
+
11
+
12
+ @is_step('features_precleaning')
13
+ class ActDropHighMissingColumns(Actionable):
14
+ """[STEP] Drop Columns with High Missing Values"""
15
+
16
+ name: str = 'Drop columns with high missing values'
17
+ _description: str = textwrap.dedent('''\
18
+ Drop columns where the missing ratio is above {missing_threshold:.0%}.''')
19
+ _description_long: str = textwrap.dedent('''\
20
+ In datasets, some columns can be mostly empty.
21
+ This step removes columns whose missing-value ratio exceeds a configured threshold,
22
+ keeping the dataset focused on informative features.''')
23
+ _usage: str = 'Use when many features are mostly missing and should be removed rather than imputed. Applicable to tabular data with real NaN or after ActSentinelToNaN reveals missingness. Avoid when sparsity is meaningful or columns are ID-like, consider ActDropIdLikeColumns.'
24
+ refs: list[dict[str, Any]] = []
25
+
26
+ def __init__(self):
27
+ self.configuration = {
28
+ 'missing_threshold': {
29
+ 'description': textwrap.dedent('''\
30
+ Column with a missing ratio greater than or equal to this value will be
31
+ dropped. 1 will drop only columns that are fully missing.'''),
32
+ 'default': 0.5,
33
+ 'range': [0.0, 1.0]
34
+ }
35
+ }
36
+ self.columns_to_drop: list[str] = []
37
+
38
+ def fit(self, dataset: Dataset) -> Actionable:
39
+ columns = self.__candidate_columns(dataset)
40
+ if not columns:
41
+ self.columns_to_drop = []
42
+ self.explanations = []
43
+ return self
44
+
45
+ missing_ratio = dataset.X[columns].isna().mean()
46
+ threshold = self.get_config('missing_threshold')
47
+ self.columns_to_drop = missing_ratio[missing_ratio >= threshold].index.tolist()
48
+
49
+ self.explanations = [
50
+ f'Dropped column **`{column}`** because missing ratio is '
51
+ f'**{missing_ratio[column]:.2%}** (threshold {threshold:.2%}).'
52
+ for column in self.columns_to_drop
53
+ ]
54
+
55
+ return self
56
+
57
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
58
+ """Drop columns.
59
+
60
+ :param pd.DataFrame X: DataFrame to transform.
61
+ :return: Transformed DataFrame.
62
+ """
63
+ if not self.columns_to_drop:
64
+ return X
65
+
66
+ drop_columns = [column for column in self.columns_to_drop if column in X.columns]
67
+ if not drop_columns:
68
+ return X
69
+ return X.drop(columns=drop_columns)
70
+
71
+ def priorize(self, candidate: Candidate = None) -> float:
72
+ if candidate is None:
73
+ return 1.0
74
+ columns = self.__candidate_columns(candidate.dataset)
75
+ if not columns:
76
+ return 0.0
77
+ missing_ratio = candidate.dataset.X[columns].isna().mean()
78
+ threshold = self.get_config('missing_threshold')
79
+ ratio = (missing_ratio >= threshold).sum() / max(1, len(columns))
80
+ return min(1.5, 0.5 + ratio)
81
+
82
+ def suitable(self, dataset: Dataset) -> bool:
83
+ columns = self.__candidate_columns(dataset)
84
+ if not columns:
85
+ return False
86
+ missing_ratio = dataset.X[columns].isna().mean()
87
+ return (missing_ratio >= self.get_config('missing_threshold')).any()
88
+
89
+ def __candidate_columns(self, dataset: Dataset) -> list[str]:
90
+ columns = dataset.get_columns_names_by_type(list(DataType))
91
+ if len(columns) != dataset.X.shape[1]:
92
+ missing = [column for column in dataset.X.columns if column not in columns]
93
+ columns.extend(missing)
94
+ return columns
@@ -0,0 +1,294 @@
1
+ """[STEP] Drop id-like columns"""
2
+ import re
3
+ import textwrap
4
+ from typing import Any
5
+ import numpy as np
6
+ import pandas as pd
7
+ from pandas.api.types import (
8
+ is_categorical_dtype,
9
+ is_datetime64_any_dtype,
10
+ is_numeric_dtype,
11
+ is_object_dtype,
12
+ is_string_dtype,
13
+ )
14
+ from ...actionable import Actionable
15
+ from ...candidate import Candidate
16
+ from ...data_type import DataType
17
+ from ...dataset import Dataset
18
+ from ...decorators.all import is_step
19
+
20
+
21
+ _UUID_PATTERN = (
22
+ r'^(?:[0-9a-f]{32}|'
23
+ r'[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})$'
24
+ )
25
+ _HEX_PATTERN = r'^[0-9a-f]{16,}$'
26
+ _TOKEN_PATTERN = r'^[A-Za-z0-9_-]+$'
27
+ _DIGITS_PATTERN = r'^\d+$'
28
+ _CAMEL_RE = re.compile(r'([a-z])([A-Z])')
29
+ _SPLIT_RE = re.compile(r'[^a-z0-9]+')
30
+
31
+
32
+ @is_step('features_precleaning')
33
+ class ActDropIdLikeColumns(Actionable):
34
+ """[STEP] Drop id-like columns"""
35
+
36
+ name: str = 'Drop id-like columns'
37
+ _description: str = textwrap.dedent('''\
38
+ Drop columns that are quasi-unique and look like identifiers.''')
39
+ _description_long: str = textwrap.dedent('''\
40
+ Identifier columns such as IDs, UUIDs, hashes, and keys are often quasi-unique
41
+ and do not carry predictive signal. This step removes columns that are nearly
42
+ unique and match identifier-like name or value patterns.''')
43
+ _usage: str = 'Use when columns are quasi-unique identifiers with little signal, rather than ActDropDuplicateRows. Applicable to mixed tabular data with ID/UUID/hash-like fields. Avoid when IDs encode meaning or when missingness cleanup via ActDropHighMissingColumns is the issue.'
44
+ refs: list[dict[str, Any]] = []
45
+
46
+ def __init__(self) -> None:
47
+ self.configuration = {
48
+ 'unique_ratio_threshold': {
49
+ 'description': textwrap.dedent('''\
50
+ Minimum ratio of unique (non-null) values for a column to be
51
+ considered quasi-unique.'''),
52
+ 'default': 0.98,
53
+ 'range': [0.0, 1.0]
54
+ },
55
+ 'min_unique': {
56
+ 'description': 'Minimum number of unique values to consider a column.',
57
+ 'default': 20
58
+ },
59
+ 'min_non_null': {
60
+ 'description': 'Minimum number of non-null values to evaluate a column.',
61
+ 'default': 10
62
+ },
63
+ 'name_tokens': {
64
+ 'description': 'Tokens indicating identifier-like column names.',
65
+ 'default': [
66
+ 'id', 'uuid', 'guid', 'identifier', 'key', 'code', 'ref',
67
+ 'reference', 'hash', 'token', 'serial', 'sequence', 'seq', 'index'
68
+ ]
69
+ },
70
+ 'sample_size': {
71
+ 'description': textwrap.dedent('''\
72
+ Sample size used for pattern detection on text columns.
73
+ Set to -1 to scan the full column.'''),
74
+ 'default': 500
75
+ },
76
+ 'uuid_ratio_threshold': {
77
+ 'description': 'Minimum ratio of UUID-like values.',
78
+ 'default': 0.8,
79
+ 'range': [0.0, 1.0]
80
+ },
81
+ 'hex_ratio_threshold': {
82
+ 'description': 'Minimum ratio of hex-like values.',
83
+ 'default': 0.9,
84
+ 'range': [0.0, 1.0]
85
+ },
86
+ 'digits_ratio_threshold': {
87
+ 'description': 'Minimum ratio of digit-only values.',
88
+ 'default': 0.9,
89
+ 'range': [0.0, 1.0]
90
+ },
91
+ 'digits_min_length': {
92
+ 'description': 'Minimum length for digit-only values to count as IDs.',
93
+ 'default': 4
94
+ },
95
+ 'token_ratio_threshold': {
96
+ 'description': 'Minimum ratio of mixed alphanumeric tokens.',
97
+ 'default': 0.9,
98
+ 'range': [0.0, 1.0]
99
+ },
100
+ 'token_min_length': {
101
+ 'description': 'Minimum length for mixed alphanumeric tokens.',
102
+ 'default': 6
103
+ },
104
+ 'integer_ratio_threshold': {
105
+ 'description': 'Minimum ratio of integer-like values for numeric checks.',
106
+ 'default': 0.95,
107
+ 'range': [0.0, 1.0]
108
+ }
109
+ }
110
+ self.columns_to_drop: list[str] = []
111
+ self.drop_reasons: dict[str, str] = {}
112
+
113
+ def fit(self, dataset: Dataset) -> Actionable:
114
+ matches = self.__find_id_like_columns(dataset)
115
+ self.columns_to_drop = [match['column'] for match in matches]
116
+ self.drop_reasons = {
117
+ match['column']: match['reason']
118
+ for match in matches
119
+ }
120
+ self.explanations = [
121
+ f'Dropped column **`{match["column"]}`** because it is quasi-unique '
122
+ f'({match["unique_ratio"]:.2%}) and {match["reason"]}.'
123
+ for match in matches
124
+ ]
125
+ return self
126
+
127
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
128
+ """Drop id-like columns.
129
+
130
+ :param pd.DataFrame X: DataFrame to transform.
131
+ :return: Transformed DataFrame.
132
+ """
133
+ if not self.columns_to_drop:
134
+ return X
135
+ drop_columns = [column for column in self.columns_to_drop if column in X.columns]
136
+ if not drop_columns:
137
+ return X
138
+ return X.drop(columns=drop_columns)
139
+
140
+ def priorize(self, candidate: Candidate = None) -> float:
141
+ if candidate is None or candidate.dataset.X.empty:
142
+ return 0.0
143
+ matches = self.__find_id_like_columns(candidate.dataset)
144
+ if not matches:
145
+ return 0.0
146
+ ratio = len(matches) / max(1, candidate.dataset.X.shape[1])
147
+ return min(1.5, 0.5 + ratio)
148
+
149
+ def suitable(self, dataset: Dataset) -> bool:
150
+ if dataset.X.empty:
151
+ return False
152
+ return bool(self.__find_id_like_columns(dataset))
153
+
154
+ def __find_id_like_columns(self, dataset: Dataset) -> list[dict[str, Any]]:
155
+ columns = self.__candidate_columns(dataset)
156
+ if not columns or dataset.X.empty:
157
+ return []
158
+
159
+ unique_threshold = self.get_config('unique_ratio_threshold')
160
+ min_unique = self.get_config('min_unique')
161
+ min_non_null = self.get_config('min_non_null')
162
+
163
+ matches: list[dict[str, Any]] = []
164
+ for column in columns:
165
+ series = dataset.X[column]
166
+ non_null = int(series.notna().sum())
167
+ if non_null < min_non_null:
168
+ continue
169
+ unique_count = int(series.nunique(dropna=True))
170
+ if unique_count < min_unique:
171
+ continue
172
+ unique_ratio = unique_count / non_null if non_null else 0.0
173
+ if unique_ratio < unique_threshold:
174
+ continue
175
+
176
+ reason = self.__id_like_reason(column, series)
177
+ if reason:
178
+ matches.append({
179
+ 'column': column,
180
+ 'unique_ratio': unique_ratio,
181
+ 'unique_count': unique_count,
182
+ 'non_null': non_null,
183
+ 'reason': reason
184
+ })
185
+
186
+ return matches
187
+
188
+ def __candidate_columns(self, dataset: Dataset) -> list[str]:
189
+ columns = dataset.get_columns_names_by_type(list(DataType))
190
+ if len(columns) != dataset.X.shape[1]:
191
+ missing = [column for column in dataset.X.columns if column not in columns]
192
+ columns.extend(missing)
193
+ return columns
194
+
195
+ def __id_like_reason(self, column: object, series: pd.Series) -> str | None:
196
+ reasons: list[str] = []
197
+ if self.__name_looks_like_id(column):
198
+ reasons.append('column name looks like an identifier')
199
+
200
+ if is_datetime64_any_dtype(series):
201
+ return '; '.join(reasons) if reasons else None
202
+
203
+ if is_numeric_dtype(series):
204
+ numeric_reason = self.__numeric_reason(series)
205
+ if numeric_reason:
206
+ reasons.append(numeric_reason)
207
+ elif is_categorical_dtype(series) or is_string_dtype(series) or is_object_dtype(series):
208
+ text_reason = self.__text_reason(series)
209
+ if text_reason:
210
+ reasons.append(text_reason)
211
+
212
+ if reasons:
213
+ return '; '.join(reasons)
214
+ return None
215
+
216
+ def __numeric_reason(self, series: pd.Series) -> str | None:
217
+ values = pd.to_numeric(series, errors='coerce').dropna()
218
+ if values.empty:
219
+ return None
220
+
221
+ integer_ratio = self.__integer_ratio(values)
222
+ if integer_ratio < self.get_config('integer_ratio_threshold'):
223
+ return None
224
+
225
+ unique_count = values.nunique(dropna=True)
226
+ if unique_count <= 1:
227
+ return None
228
+
229
+ value_min = values.min()
230
+ value_max = values.max()
231
+ if value_max - value_min == unique_count - 1:
232
+ return 'values form a consecutive integer range'
233
+
234
+ return None
235
+
236
+ def __text_reason(self, series: pd.Series) -> str | None:
237
+ sample = self.__sample_text(series)
238
+ if sample.empty:
239
+ return None
240
+
241
+ lower = sample.str.lower()
242
+ uuid_ratio = lower.str.fullmatch(_UUID_PATTERN).mean()
243
+ if uuid_ratio >= self.get_config('uuid_ratio_threshold'):
244
+ return f'values look like UUIDs ({uuid_ratio:.0%})'
245
+
246
+ hex_ratio = lower.str.fullmatch(_HEX_PATTERN).mean()
247
+ if hex_ratio >= self.get_config('hex_ratio_threshold'):
248
+ return f'values look like hex hashes ({hex_ratio:.0%})'
249
+
250
+ lengths = sample.str.len()
251
+ digits_mask = lower.str.fullmatch(_DIGITS_PATTERN)
252
+ digits_ratio = (digits_mask & (lengths >= self.get_config('digits_min_length'))).mean()
253
+ if digits_ratio >= self.get_config('digits_ratio_threshold'):
254
+ return f'values are digit-only tokens ({digits_ratio:.0%})'
255
+
256
+ token_mask = sample.str.fullmatch(_TOKEN_PATTERN)
257
+ mixed_mask = (
258
+ token_mask
259
+ & (lengths >= self.get_config('token_min_length'))
260
+ & sample.str.contains(r'[A-Za-z]', regex=True)
261
+ & sample.str.contains(r'\d', regex=True)
262
+ )
263
+ mixed_ratio = mixed_mask.mean()
264
+ if mixed_ratio >= self.get_config('token_ratio_threshold'):
265
+ return f'values are mixed alphanumeric tokens ({mixed_ratio:.0%})'
266
+
267
+ return None
268
+
269
+ def __sample_text(self, series: pd.Series) -> pd.Series:
270
+ values = series.dropna()
271
+ if values.empty:
272
+ return values.astype(str)
273
+
274
+ sample_size = self.get_config('sample_size')
275
+ if sample_size is not None and sample_size > 0 and len(values) > sample_size:
276
+ values = values.sample(n=sample_size, random_state=0)
277
+
278
+ return values.astype(str)
279
+
280
+ def __name_looks_like_id(self, name: object) -> bool:
281
+ if name is None:
282
+ return False
283
+ raw = str(name)
284
+ normalized = _CAMEL_RE.sub(r'\1_\2', raw).lower()
285
+ tokens = [token for token in _SPLIT_RE.split(normalized) if token]
286
+ id_tokens = {token.lower() for token in (self.get_config('name_tokens') or [])}
287
+ return any(token in id_tokens for token in tokens)
288
+
289
+ def __integer_ratio(self, values: pd.Series) -> float:
290
+ numeric = pd.to_numeric(values, errors='coerce').dropna().to_numpy()
291
+ if numeric.size == 0:
292
+ return 0.0
293
+ frac = np.mod(numeric, 1)
294
+ return float(np.isclose(frac, 0).mean())
@@ -0,0 +1,157 @@
1
+ """[STEP] Normalize column names"""
2
+ import re
3
+ import textwrap
4
+ import pandas as pd
5
+ from ...actionable import Actionable
6
+ from ...dataset import Dataset
7
+ from ...candidate import Candidate
8
+ from ...decorators.all import is_step
9
+
10
+
11
+ _NON_ALNUM_RE = re.compile(r'[^0-9a-zA-Z_]+')
12
+ _MULTI_UNDERSCORE_RE = re.compile(r'_+')
13
+
14
+
15
+ @is_step('features_precleaning')
16
+ class ActNormalizeColumnNames(Actionable):
17
+ """[STEP] Normalize column names"""
18
+
19
+ name: str = 'Normalize column names'
20
+ _description: str = 'Standardize column names with lowercase and underscores'
21
+ _usage: str = 'Use when column names are messy or inconsistent; run before ActCoerceNumericStrings or ActDateConverter. Applicable to raw tables with human-entered headers, spaces, symbols, or duplicates. Avoid when names already standardized or must remain exact for downstream joins.'
22
+ _description_long: str = textwrap.dedent('''\
23
+ Standardize column names so they are lowercase and use underscores.
24
+ This reduces naming collisions and keeps downstream feature selection consistent.
25
+ ''')
26
+
27
+ refs = []
28
+
29
+ def __init__(self):
30
+ self.configuration = {
31
+ 'lowercase': {
32
+ 'description': 'Convert column names to lowercase',
33
+ 'default': True
34
+ },
35
+ 'strip': {
36
+ 'description': 'Trim leading and trailing spaces',
37
+ 'default': True
38
+ },
39
+ 'replace_spaces': {
40
+ 'description': 'Replace spaces with underscores',
41
+ 'default': True
42
+ },
43
+ 'replace_non_alnum': {
44
+ 'description': 'Replace non alphanumeric characters with underscores',
45
+ 'default': True
46
+ },
47
+ 'collapse_underscores': {
48
+ 'description': 'Collapse consecutive underscores',
49
+ 'default': True
50
+ },
51
+ 'strip_underscores': {
52
+ 'description': 'Trim leading and trailing underscores',
53
+ 'default': True
54
+ },
55
+ 'deduplicate': {
56
+ 'description': 'Ensure column names are unique after normalization',
57
+ 'default': True
58
+ },
59
+ 'dedupe_sep': {
60
+ 'description': 'Separator used when deduplicating column names',
61
+ 'default': '_'
62
+ },
63
+ 'empty_fallback': {
64
+ 'description': 'Fallback base name for empty column names',
65
+ 'default': 'column'
66
+ },
67
+ }
68
+
69
+ self.columns: list = None
70
+ self.normalized_columns: list[str] = None
71
+
72
+ def fit(self, dataset: Dataset) -> Actionable:
73
+ self.columns = dataset.features
74
+ normalized = self._normalize_columns(self.columns)
75
+ if self.get_config('deduplicate'):
76
+ normalized = self._deduplicate(normalized)
77
+
78
+ self.normalized_columns = normalized
79
+ self.explanations = [
80
+ f'Normalize column name **`{old}`** -> **`{new}`**.'
81
+ for old, new in zip(self.columns, self.normalized_columns)
82
+ if old != new
83
+ ]
84
+
85
+ return self
86
+
87
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
88
+ """Normalize column names.
89
+
90
+ :param pd.DataFrame X: DataFrame to transform.
91
+ :return: Transformed dataset.
92
+ """
93
+ if not self.normalized_columns or list(X.columns) == self.normalized_columns:
94
+ return X
95
+
96
+ X.columns = self.normalized_columns
97
+ return X
98
+
99
+ def suitable(self, dataset: Dataset) -> bool:
100
+ columns = dataset.features
101
+ normalized = self._normalize_columns(columns)
102
+ if self.get_config('deduplicate'):
103
+ normalized = self._deduplicate(normalized)
104
+ return columns != normalized
105
+
106
+ def priorize(self, candidate: Candidate = None) -> float:
107
+ return 1.0
108
+
109
+ def _normalize_columns(self, columns: list) -> list[str]:
110
+ return [self._normalize_name(column) for column in columns]
111
+
112
+ def _normalize_name(self, name: object) -> str:
113
+ value = '' if name is None else str(name)
114
+
115
+ if self.get_config('strip'):
116
+ value = value.strip()
117
+ if self.get_config('lowercase'):
118
+ value = value.lower()
119
+ if self.get_config('replace_spaces'):
120
+ value = re.sub(r'\s+', '_', value)
121
+ if self.get_config('replace_non_alnum'):
122
+ value = _NON_ALNUM_RE.sub('_', value)
123
+ if self.get_config('collapse_underscores'):
124
+ value = _MULTI_UNDERSCORE_RE.sub('_', value)
125
+ if self.get_config('strip_underscores'):
126
+ value = value.strip('_')
127
+
128
+ if value == '':
129
+ value = self.get_config('empty_fallback')
130
+
131
+ return value
132
+
133
+ def _deduplicate(self, names: list[str]) -> list[str]:
134
+ used = set()
135
+ counts = {}
136
+ sep = self.get_config('dedupe_sep')
137
+ deduped = []
138
+
139
+ for name in names:
140
+ base = name or self.get_config('empty_fallback')
141
+ idx = counts.get(base, 0)
142
+
143
+ if base in used:
144
+ idx = max(idx, 1)
145
+ candidate = f"{base}{sep}{idx}"
146
+ while candidate in used:
147
+ idx += 1
148
+ candidate = f"{base}{sep}{idx}"
149
+ name = candidate
150
+ else:
151
+ name = base
152
+
153
+ used.add(name)
154
+ counts[base] = idx + 1
155
+ deduped.append(name)
156
+
157
+ return deduped