PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,270 @@
1
+ """[STEP] Replace sentinel values with NaN"""
2
+ import textwrap
3
+ from typing import Any
4
+ import numpy as np
5
+ import pandas as pd
6
+ from pandas.api.types import is_numeric_dtype, is_object_dtype, is_string_dtype
7
+ from ...actionable import Actionable
8
+ from ...candidate import Candidate
9
+ from ...data_type import DataType
10
+ from ...dataset import Dataset
11
+ from ...decorators.all import is_step
12
+
13
+
14
+ @is_step('features_precleaning')
15
+ class ActSentinelToNaN(Actionable):
16
+ """[STEP] Replace sentinel values with NaN"""
17
+
18
+ name: str = 'Replace sentinel values'
19
+ _description: str = textwrap.dedent('''\
20
+ Replace configured sentinel values (e.g., -999, "NA", "unknown") with NaN.''')
21
+ _description_long: str = textwrap.dedent('''\
22
+ Sentinel values are placeholders for missing data.
23
+ This step replaces common numeric and text sentinels with NaN so downstream
24
+ steps can treat missing values consistently.''')
25
+ _usage: str = "Use when numeric or text columns contain sentinel placeholders (e.g., -999, 'NA') and you want them treated as missing; run before ActCoerceNumericStrings. Applicable to datasets with explicit sentinel codes or empty-string markers across numeric and categorical fields. Avoid when sentinel values are meaningful domain codes or you should remove sparse fields via ActDropHighMissingColumns."
26
+ refs: list[dict[str, Any]] = []
27
+
28
+ def __init__(self) -> None:
29
+ self.configuration = {
30
+ 'numeric_sentinels': {
31
+ 'description': textwrap.dedent('''\
32
+ Numeric sentinel values to replace with NaN.'''),
33
+ 'default': [-999, -9999, -99999]
34
+ },
35
+ 'text_sentinels': {
36
+ 'description': textwrap.dedent('''\
37
+ Text sentinel values to replace with NaN.'''),
38
+ 'default': ['NA', 'N/A', 'NULL', 'NONE', 'UNKNOWN', 'MISSING', 'NAN']
39
+ },
40
+ 'case_insensitive': {
41
+ 'description': 'Match text sentinels ignoring case.',
42
+ 'default': True
43
+ },
44
+ 'strip_whitespace': {
45
+ 'description': 'Trim whitespace before matching text sentinels.',
46
+ 'default': True
47
+ },
48
+ 'include_empty_string': {
49
+ 'description': 'Treat empty strings as missing values.',
50
+ 'default': True
51
+ },
52
+ 'numeric_in_text': {
53
+ 'description': 'Also match numeric sentinels stored as text.',
54
+ 'default': True
55
+ }
56
+ }
57
+ self.numeric_columns: list[str] = []
58
+ self.text_columns: list[str] = []
59
+ self.numeric_sentinels: list[float] = []
60
+ self.text_sentinels: list[str] = []
61
+
62
+ def fit(self, dataset: Dataset) -> Actionable:
63
+ self.numeric_columns, self.text_columns = self.__candidate_columns(dataset)
64
+ self.numeric_sentinels, self.text_sentinels = self.__normalized_sentinels()
65
+
66
+ self.explanations = []
67
+ if dataset.X.empty:
68
+ return self
69
+
70
+ if self.numeric_sentinels:
71
+ for column in self.numeric_columns:
72
+ count = int(dataset.X[column].isin(self.numeric_sentinels).sum())
73
+ if count:
74
+ self.explanations.append(
75
+ f'Replaced {count} numeric sentinel values in **`{column}`**.'
76
+ )
77
+
78
+ if self.text_sentinels:
79
+ for column in self.text_columns:
80
+ mask = self.__text_sentinel_mask(dataset.X[column], self.text_sentinels)
81
+ count = int(mask.sum())
82
+ if count:
83
+ self.explanations.append(
84
+ f'Replaced {count} text sentinel values in **`{column}`**.'
85
+ )
86
+
87
+ return self
88
+
89
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
90
+ """Replace sentinel values with NaN.
91
+
92
+ :param pd.DataFrame X: DataFrame to transform.
93
+ :return: Transformed DataFrame.
94
+ """
95
+ if X.empty:
96
+ return X
97
+
98
+ if self.numeric_sentinels and self.numeric_columns:
99
+ columns = [column for column in self.numeric_columns if column in X.columns]
100
+ if columns:
101
+ X[columns] = X[columns].replace(self.numeric_sentinels, np.nan)
102
+
103
+ if self.text_sentinels and self.text_columns:
104
+ for column in self.text_columns:
105
+ if column not in X.columns:
106
+ continue
107
+ mask = self.__text_sentinel_mask(X[column], self.text_sentinels)
108
+ if mask.any():
109
+ X[column] = X[column].mask(mask, np.nan)
110
+
111
+ return X
112
+
113
+ def priorize(self, candidate: Candidate = None) -> float:
114
+ if candidate is None or candidate.dataset.X.empty:
115
+ return 0.0
116
+
117
+ numeric_sentinels, text_sentinels = self.__normalized_sentinels()
118
+ if not numeric_sentinels and not text_sentinels:
119
+ return 0.0
120
+
121
+ numeric_columns, text_columns = self.__candidate_columns(candidate.dataset)
122
+ total = candidate.dataset.X.size or 1
123
+ count = self.__count_sentinels(
124
+ candidate.dataset,
125
+ numeric_columns,
126
+ text_columns,
127
+ numeric_sentinels,
128
+ text_sentinels
129
+ )
130
+ return min(1.5, 0.5 + count / total)
131
+
132
+ def suitable(self, dataset: Dataset) -> bool:
133
+ if dataset.X.empty:
134
+ return False
135
+
136
+ numeric_sentinels, text_sentinels = self.__normalized_sentinels()
137
+ if not numeric_sentinels and not text_sentinels:
138
+ return False
139
+
140
+ numeric_columns, text_columns = self.__candidate_columns(dataset)
141
+ if numeric_sentinels:
142
+ for column in numeric_columns:
143
+ if dataset.X[column].isin(numeric_sentinels).any():
144
+ return True
145
+
146
+ if text_sentinels:
147
+ for column in text_columns:
148
+ if self.__text_sentinel_mask(dataset.X[column], text_sentinels).any():
149
+ return True
150
+
151
+ return False
152
+
153
+ def __candidate_columns(self, dataset: Dataset) -> tuple[list[str], list[str]]:
154
+ numeric_candidates = set(dataset.get_columns_names_by_type(DataType.NUMERIC))
155
+ text_candidates = set(dataset.get_columns_names_by_type(
156
+ [DataType.CATEGORICAL, DataType.TEXT, DataType.SHORT_TEXT]
157
+ ))
158
+
159
+ numeric_columns = []
160
+ text_columns = []
161
+
162
+ for column in dataset.X.columns:
163
+ if column in numeric_candidates:
164
+ numeric_columns.append(column)
165
+ continue
166
+ if column in text_candidates:
167
+ text_columns.append(column)
168
+ continue
169
+
170
+ series = dataset.X[column]
171
+ if is_numeric_dtype(series):
172
+ numeric_columns.append(column)
173
+ elif is_string_dtype(series) or is_object_dtype(series):
174
+ text_columns.append(column)
175
+
176
+ return numeric_columns, text_columns
177
+
178
+ def __normalized_sentinels(self) -> tuple[list[float], list[str]]:
179
+ numeric_sentinels = self.__normalize_numeric_sentinels()
180
+ text_sentinels = self.__normalize_text_sentinels(numeric_sentinels)
181
+ return numeric_sentinels, text_sentinels
182
+
183
+ def __normalize_numeric_sentinels(self) -> list[float]:
184
+ values = self.get_config('numeric_sentinels') or []
185
+ cleaned: list[float] = []
186
+ seen: set[float] = set()
187
+
188
+ for value in values:
189
+ if value is None or isinstance(value, bool):
190
+ continue
191
+ try:
192
+ num = float(value)
193
+ except (TypeError, ValueError):
194
+ continue
195
+ if np.isnan(num) or num in seen:
196
+ continue
197
+ cleaned.append(num)
198
+ seen.add(num)
199
+
200
+ return cleaned
201
+
202
+ def __normalize_text_sentinels(self, numeric_sentinels: list[float]) -> list[str]:
203
+ values = list(self.get_config('text_sentinels') or [])
204
+ if self.get_config('include_empty_string'):
205
+ values.append('')
206
+ if self.get_config('numeric_in_text'):
207
+ values.extend(self.__numeric_sentinels_as_text(numeric_sentinels))
208
+
209
+ cleaned: list[str] = []
210
+ seen: set[str] = set()
211
+
212
+ for value in values:
213
+ if value is None:
214
+ continue
215
+ text = str(value)
216
+ if self.get_config('strip_whitespace'):
217
+ text = text.strip()
218
+ if self.get_config('case_insensitive'):
219
+ text = text.lower()
220
+ if text == '' and not self.get_config('include_empty_string'):
221
+ continue
222
+ if text not in seen:
223
+ cleaned.append(text)
224
+ seen.add(text)
225
+
226
+ return cleaned
227
+
228
+ def __numeric_sentinels_as_text(self, numeric_sentinels: list[float]) -> list[str]:
229
+ values: list[str] = []
230
+ for value in numeric_sentinels:
231
+ if np.isnan(value):
232
+ continue
233
+ if float(value).is_integer():
234
+ int_value = int(value)
235
+ values.append(str(int_value))
236
+ values.append(str(float(int_value)))
237
+ else:
238
+ values.append(str(value))
239
+ return values
240
+
241
+ def __text_sentinel_mask(self, series: pd.Series, sentinels: list[str]) -> pd.Series:
242
+ if not sentinels or series.empty:
243
+ return pd.Series(False, index=series.index)
244
+
245
+ values = series.astype('string')
246
+ if self.get_config('strip_whitespace'):
247
+ values = values.str.strip()
248
+ if self.get_config('case_insensitive'):
249
+ values = values.str.lower()
250
+ mask = values.isin(sentinels)
251
+ return mask.fillna(False)
252
+
253
+ def __count_sentinels(self,
254
+ dataset: Dataset,
255
+ numeric_columns: list[str],
256
+ text_columns: list[str],
257
+ numeric_sentinels: list[float],
258
+ text_sentinels: list[str]
259
+ ) -> int:
260
+ count = 0
261
+ if numeric_sentinels:
262
+ for column in numeric_columns:
263
+ count += int(dataset.X[column].isin(numeric_sentinels).sum())
264
+ if text_sentinels:
265
+ for column in text_columns:
266
+ count += int(self.__text_sentinel_mask(
267
+ dataset.X[column],
268
+ text_sentinels
269
+ ).sum())
270
+ return count
@@ -0,0 +1,79 @@
1
+ """[STEP] Trim spaces on each columns"""
2
+ import textwrap
3
+ from pandas.api.types import is_object_dtype
4
+ import pandas as pd
5
+ from ...actionable import Actionable
6
+ from ...dataset import Dataset
7
+ from ...candidate import Candidate
8
+ from ...decorators.all import is_step
9
+
10
+ @is_step('features_precleaning')
11
+ class ActTrimSpaces(Actionable):
12
+ """
13
+ [STEP] Trim spaces on each columns
14
+ """
15
+ name = "Trim columns spaces"
16
+ _description = textwrap.dedent('''\
17
+ This step trim spaces on each columns.
18
+ It helps preventing errors on the dataset when casting columns with spaces''')
19
+ _description_long = textwrap.dedent('''\
20
+ In datasets, columns with spaces can be an issue.
21
+ This step remove spaces in front and back of columns values.
22
+ This ensures that columns can be casted correctly with having spaces throwing
23
+ an error.''')
24
+ _usage = "Use when columns or string values have leading/trailing spaces; use before ActCoerceNumericStrings or ActNormalizeColumnNames. Applicable to object/string columns and column names. Avoid when spaces are meaningful or data is already clean."
25
+
26
+ refs = []
27
+
28
+ def __init__(self):
29
+ self.configuration = {
30
+ 'left_trim': {
31
+ 'description': "Trim all columns leading spaces",
32
+ 'default': True
33
+ },
34
+ 'right_trim': {
35
+ 'description': "Trim all columns ending spaces",
36
+ 'default': True
37
+ },
38
+ }
39
+
40
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
41
+ return self
42
+
43
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
44
+ """Apply the dropping of rows with at least 40% empty columns.
45
+
46
+ :param pd.DataFrame X: The dataframe to clean.
47
+ :return: The cleaned dataframe.
48
+ """
49
+ if self.get_config('left_trim'):
50
+ X = X.rename(columns=lambda x: x.lstrip())
51
+ for col in X.columns:
52
+ if isinstance(X[col].dtype, str) or is_object_dtype(X[col]):
53
+ X[col] = X[col].apply(
54
+ lambda x: x.lstrip() if isinstance(x, str) else x,
55
+ convert_dtype=False)
56
+
57
+ if self.get_config('right_trim'):
58
+ X = X.rename(columns=lambda x: x.rstrip())
59
+ for col in X.columns:
60
+ if isinstance(X[col].dtype, str) or is_object_dtype(X[col]):
61
+ X[col] = X[col].astype(object).apply(
62
+ lambda x: x.rstrip() if isinstance(x, str) else x
63
+ )
64
+
65
+ return X
66
+
67
+ def priorize(self, candidate: Candidate = None) -> float:
68
+ return 1.5
69
+
70
+ def suitable(self, dataset: Dataset) -> bool:
71
+ cond = (
72
+ dataset.X.apply(
73
+ lambda x: (
74
+ x.astype(str).str.strip() if (isinstance(x, str) or is_object_dtype(x)) else x
75
+ ) != x
76
+ ).stack().any()
77
+ or (dataset.X.columns.str.strip() != dataset.X.columns).any()
78
+ )
79
+ return cond
@@ -0,0 +1,18 @@
1
+ """
2
+ Usually last step before predictor, try features decomposition or aggregation.
3
+ Example : PCA
4
+ """
5
+ from .act_pca import ActPCA
6
+ from .act_kernel_pca import ActKernelPCA
7
+ from .act_feature_agglomeration import ActFeatureAgglomeration
8
+ from .act_nystroem import ActNystroem
9
+ from .act_rbf_sampler import ActRBFSampler
10
+ from .act_select_percentile import ActSelectPercentile
11
+ from .act_power_transformer import ActPowerTransformer
12
+ from .act_quantile_transformer import ActQuantileTransformer
13
+ from .act_truncated_svd import ActTruncatedSVD
14
+ from .act_fast_ica import ActFastICA
15
+ from .act_sparse_random_projection import ActSparseRandomProjection
16
+ from .act_k_bins_discretizer import ActKBinsDiscretizer
17
+ from .act_log_transformer import ActLogTransformer
18
+ from .act_k_means_features import ActKMeansFeatures
@@ -0,0 +1,212 @@
1
+ """Manual cyclical date encoding, available only through an explicit module import.
2
+
3
+ Date conversion and cleaning precede automatic feature preprocessing, so this
4
+ component needs an explicit position while datetime columns are still available.
5
+ See docs/component_status.rst.
6
+ """
7
+ import textwrap
8
+ import numpy as np
9
+ import pandas as pd
10
+ from ...actionable import Actionable
11
+ from ...candidate import Candidate
12
+ from ...dataset import Dataset
13
+ from ...data_type import DataType
14
+ from ...decorators.all import is_step
15
+
16
+
17
+ @is_step('experimental')
18
+ class ActCyclicalDateEncoding(Actionable):
19
+ """[STEP] Encode date columns with cyclical sine/cosine features."""
20
+
21
+ name: str = "Cyclical Date Encoding"
22
+ _description: str = "Encode date columns into sine/cosine features to capture cyclicity"
23
+ _usage: str = "Use when datetime columns have cyclic parts (month/weekday/hour) and models need numeric features. Applicable to datetime columns with clear periodicity. Avoid when dates are non-cyclic or already encoded; consider ActKBinsDiscretizer or ActLogTransformer instead."
24
+ _description_long: str = textwrap.dedent('''\
25
+ Cyclical encoding turns calendar components (month, weekday, hour, etc.)
26
+ into sine and cosine values. This preserves the circular nature of time,
27
+ so end and start points on a cycle stay close in feature space while
28
+ providing numeric values suitable for ML models.
29
+ ''')
30
+
31
+ _COMPONENTS: dict[str, tuple[str, int]] = {
32
+ 'month': ('month', 12),
33
+ 'dayofweek': ('dayofweek', 7),
34
+ 'day': ('day', 31),
35
+ 'hour': ('hour', 24),
36
+ 'minute': ('minute', 60),
37
+ 'second': ('second', 60),
38
+ }
39
+ _ALIASES: dict[str, str] = {
40
+ 'weekday': 'dayofweek',
41
+ 'dow': 'dayofweek',
42
+ 'dayofmonth': 'day',
43
+ 'dom': 'day',
44
+ }
45
+
46
+ def __init__(self) -> None:
47
+ self.configuration = {
48
+ 'components': {
49
+ 'description': 'Date components to encode (month, dayofweek, day, hour, minute, second).',
50
+ 'default': ('month', 'dayofweek', 'day', 'hour')
51
+ },
52
+ 'drop_original': {
53
+ 'description': 'Drop original date columns after encoding.',
54
+ 'default': True
55
+ }
56
+ }
57
+ self.columns: list[str] = []
58
+ self.encoding_plan: dict[str, list[tuple[str, str, int, str, str]]] = {}
59
+ self.drop_original: bool = True
60
+
61
+ @staticmethod
62
+ def _is_datetime(series: pd.Series) -> bool:
63
+ return pd.api.types.is_datetime64_any_dtype(series)
64
+
65
+ @staticmethod
66
+ def _coerce_bool(value: object, default: bool) -> bool:
67
+ if isinstance(value, bool):
68
+ return value
69
+ if isinstance(value, str):
70
+ lowered = value.strip().lower()
71
+ if lowered in ('1', 'true', 'yes', 'y'):
72
+ return True
73
+ if lowered in ('0', 'false', 'no', 'n'):
74
+ return False
75
+ return default
76
+
77
+ @staticmethod
78
+ def _unique_name(name: str, reserved: set[str]) -> str:
79
+ if name not in reserved:
80
+ return name
81
+ idx = 1
82
+ candidate = f"{name}_{idx}"
83
+ while candidate in reserved:
84
+ idx += 1
85
+ candidate = f"{name}_{idx}"
86
+ return candidate
87
+
88
+ def _resolve_components(self) -> list[str]:
89
+ raw = self.get_config('components')
90
+ if raw is None:
91
+ raw_components = list(self._COMPONENTS.keys())
92
+ elif isinstance(raw, str):
93
+ raw_components = [raw]
94
+ else:
95
+ try:
96
+ raw_components = list(raw)
97
+ except TypeError:
98
+ raw_components = [str(raw)]
99
+
100
+ components: list[str] = []
101
+ seen: set[str] = set()
102
+ for component in raw_components:
103
+ if component is None:
104
+ continue
105
+ component = str(component).strip().lower()
106
+ if not component:
107
+ continue
108
+ component = self._ALIASES.get(component, component)
109
+ if component not in self._COMPONENTS:
110
+ continue
111
+ if component in seen:
112
+ continue
113
+ seen.add(component)
114
+ components.append(component)
115
+
116
+ if not components:
117
+ components = list(self._COMPONENTS.keys())
118
+
119
+ return components
120
+
121
+ def _build_plan(self, dataset: Dataset) -> dict[str, list[tuple[str, str, int, str, str]]]:
122
+ columns = dataset.get_columns_names_by_type(DataType.DATE)
123
+ if not columns or dataset.X.empty:
124
+ return {}
125
+
126
+ components = self._resolve_components()
127
+ if not components:
128
+ return {}
129
+
130
+ reserved = set(dataset.X.columns)
131
+ plan: dict[str, list[tuple[str, str, int, str, str]]] = {}
132
+
133
+ for column in columns:
134
+ if column not in dataset.X.columns:
135
+ continue
136
+ series = dataset.X[column]
137
+ if not self._is_datetime(series):
138
+ continue
139
+
140
+ column_plan: list[tuple[str, str, int, str, str]] = []
141
+ for component in components:
142
+ accessor, period = self._COMPONENTS[component]
143
+ values = getattr(series.dt, accessor)
144
+ if values.nunique(dropna=True) <= 1:
145
+ continue
146
+ base = f"{column}_{component}"
147
+ sin_name = self._unique_name(f"{base}_sin", reserved)
148
+ reserved.add(sin_name)
149
+ cos_name = self._unique_name(f"{base}_cos", reserved)
150
+ reserved.add(cos_name)
151
+ column_plan.append((component, accessor, period, sin_name, cos_name))
152
+
153
+ if column_plan:
154
+ plan[column] = column_plan
155
+
156
+ return plan
157
+
158
+ def fit(self, dataset: Dataset) -> Actionable:
159
+ self.encoding_plan = self._build_plan(dataset)
160
+ self.columns = list(self.encoding_plan.keys())
161
+ self.drop_original = self._coerce_bool(self.get_config('drop_original'), True)
162
+ self.explanations = []
163
+
164
+ for column, parts in self.encoding_plan.items():
165
+ components = ", ".join([part[0] for part in parts])
166
+ self.explanations.append(
167
+ f'Encoded date column **`{column}`** into cyclical features ({components}).'
168
+ )
169
+
170
+ return self
171
+
172
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
173
+ """Apply cyclical encoding to date columns.
174
+
175
+ :param pd.DataFrame X: DataFrame to transform
176
+ :return: Transformed dataset
177
+ """
178
+ if not self.encoding_plan:
179
+ return X
180
+
181
+ for column, parts in self.encoding_plan.items():
182
+ if column not in X.columns:
183
+ continue
184
+ series = X[column]
185
+ if not self._is_datetime(series):
186
+ series = pd.to_datetime(series, errors='coerce')
187
+
188
+ for _, accessor, period, sin_name, cos_name in parts:
189
+ values = getattr(series.dt, accessor).astype(float)
190
+ angles = (2.0 * np.pi * values) / float(period)
191
+ X[sin_name] = np.sin(angles)
192
+ X[cos_name] = np.cos(angles)
193
+
194
+ if self.drop_original and self.columns:
195
+ to_drop = [column for column in self.columns if column in X.columns]
196
+ if to_drop:
197
+ X = X.drop(columns=to_drop)
198
+
199
+ return X
200
+
201
+ def suitable(self, dataset: Dataset) -> bool:
202
+ if dataset.X.empty:
203
+ return False
204
+ return bool(self._build_plan(dataset))
205
+
206
+ def priorize(self, candidate: Candidate = None) -> float:
207
+ if candidate is None:
208
+ return 0.0
209
+ dataset = candidate.dataset
210
+ if dataset.X.empty:
211
+ return 0.0
212
+ return 0.5 if self._build_plan(dataset) else 0.0