PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,182 @@
1
+ """[STEP] SMOTETomek"""
2
+ import inspect
3
+ import textwrap
4
+ from typing import Any
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+ from imblearn.combine import SMOTETomek
9
+ from imblearn.over_sampling import SMOTE
10
+ from imblearn.under_sampling import TomekLinks
11
+
12
+ from ...actionable import Actionable
13
+ from ...candidate import Candidate
14
+ from ...data_type import DataType
15
+ from ...dataset import Dataset
16
+ from ...decorators.all import is_step
17
+
18
+
19
+ @is_step('imbalance')
20
+ class ActSMOTETomek(Actionable):
21
+ """[STEP] SMOTETomek"""
22
+
23
+ name: str = "SMOTE Tomek"
24
+ _description: str = textwrap.dedent('''\
25
+ SMOTETomek balances data by creating synthetic minority samples
26
+ and removing Tomek links from overlapping classes.''')
27
+ _description_long: str = textwrap.dedent('''\
28
+ SMOTETomek combines SMOTE oversampling with Tomek links cleaning.
29
+ It generates synthetic minority samples, then removes nearest neighbor
30
+ pairs from different classes to reduce overlap and noise.''')
31
+ _usage: str = "Use when imbalanced numeric data has overlap and want SMOTE plus Tomek cleanup vs ActSMOTE. Applicable to binary or multiclass, all-numeric features. Avoid when categorical/text/date fields exist, minority has <2 samples, or you want pure undersampling like ActRandomUnderSampler."
32
+ refs: list[dict[str, Any]] = [
33
+ {
34
+ 'year': 2002,
35
+ 'name': 'SMOTE: Synthetic Minority Over-sampling Technique',
36
+ 'authors': [
37
+ 'Nitesh V. Chawla',
38
+ 'Kevin W. Bowyer',
39
+ 'Lawrence O. Hall',
40
+ 'W. Philip Kegelmeyer'
41
+ ],
42
+ 'doi': 'https://doi.org/10.1613/jair.953',
43
+ 'publisher': 'Journal of Artificial Intelligence Research Vol.16 page 321--357'
44
+ },
45
+ {
46
+ 'year': 1976,
47
+ 'name': 'Two Modifications of CNN',
48
+ 'authors': [
49
+ 'Ivan Tomek'
50
+ ],
51
+ 'publisher': 'IEEE Transactions on Systems, Man, and Cybernetics Vol.6 No.11 '
52
+ 'page 769--772'
53
+ }
54
+ ]
55
+
56
+ def __init__(self):
57
+ self.configuration = {
58
+ 'sampling_strategy': {
59
+ 'description': 'Sampling strategy to balance classes.',
60
+ 'default': 'auto',
61
+ 'categorical': ['minority', 'auto']
62
+ },
63
+ 'k_neighbors': {
64
+ 'description': 'Number of nearest neighbors used to create synthetic samples.',
65
+ 'default': 5,
66
+ 'range': [1, 20]
67
+ },
68
+ 'random_state': {
69
+ 'description': 'Random seed used for reproducibility.',
70
+ 'default': 42
71
+ }
72
+ }
73
+ self.resampler: SMOTETomek | None = None
74
+ self.categorical_columns: list[str] = []
75
+ self.numeric_columns: list[str] = []
76
+ self._effective_k_neighbors: int | None = None
77
+
78
+ def fit(self, dataset: Dataset) -> Actionable:
79
+ self.resampler = None
80
+ self.categorical_columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
81
+ self.numeric_columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
82
+ self._effective_k_neighbors = None
83
+
84
+ if dataset.X.empty or dataset.type_of_target not in ['binary', 'multiclass']:
85
+ return self
86
+
87
+ if dataset.y is None or len(dataset.y) == 0:
88
+ return self
89
+
90
+ unsupported = dataset.get_columns_names_by_type(
91
+ [DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
92
+ )
93
+ if unsupported:
94
+ return self
95
+
96
+ if self.categorical_columns:
97
+ return self
98
+
99
+ if not self.numeric_columns:
100
+ return self
101
+
102
+ _, counts = np.unique(dataset.y, return_counts=True)
103
+ if len(counts) < 2:
104
+ return self
105
+
106
+ min_count = int(counts.min())
107
+ if min_count <= 1:
108
+ return self
109
+
110
+ max_k = min_count - 1
111
+ k_neighbors = min(int(self.get_config('k_neighbors')), max_k)
112
+ k_neighbors = max(1, k_neighbors)
113
+ self._effective_k_neighbors = k_neighbors
114
+
115
+ smote_params = {
116
+ 'k_neighbors': k_neighbors
117
+ }
118
+ smote_sig = inspect.signature(SMOTE).parameters
119
+ if 'sampling_strategy' in smote_sig:
120
+ smote_params['sampling_strategy'] = self.get_config('sampling_strategy')
121
+ if 'random_state' in smote_sig:
122
+ smote_params['random_state'] = self.get_config('random_state')
123
+ smote = SMOTE(**smote_params)
124
+
125
+ tomek = TomekLinks()
126
+
127
+ smotetomek_params: dict[str, Any] = {}
128
+ smotetomek_sig = inspect.signature(SMOTETomek).parameters
129
+ if 'smote' in smotetomek_sig:
130
+ smotetomek_params['smote'] = smote
131
+ if 'tomek' in smotetomek_sig:
132
+ smotetomek_params['tomek'] = tomek
133
+ if 'sampling_strategy' in smotetomek_sig and 'smote' not in smotetomek_params:
134
+ smotetomek_params['sampling_strategy'] = self.get_config('sampling_strategy')
135
+ if 'random_state' in smotetomek_sig and 'smote' not in smotetomek_params:
136
+ smotetomek_params['random_state'] = self.get_config('random_state')
137
+
138
+ self.resampler = SMOTETomek(**smotetomek_params)
139
+ self.resampler.fit(dataset.X, dataset.y)
140
+ return self
141
+
142
+ def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
143
+ """Apply SMOTETomek.
144
+
145
+ :param pd.DataFrame X: Features to resample
146
+ :param pd.DataFrame y: Labels to resample
147
+ :return: Resampled X and y
148
+ """
149
+ if self.resampler is None:
150
+ return X, y
151
+ return self.resampler.fit_resample(X, y)
152
+
153
+ def priorize(self, candidate: Candidate = None) -> float:
154
+ if candidate is None or candidate.dataset.y is None:
155
+ return 0.0
156
+ y = candidate.dataset.y
157
+ if len(y) == 0:
158
+ return 0.0
159
+ _, counts = np.unique(y, return_counts=True)
160
+ if len(counts) < 2:
161
+ return 0.0
162
+ imbalance = 1.0 - (counts.min() / counts.max())
163
+ return float(min(1.0, max(0.0, imbalance)))
164
+
165
+ def suitable(self, dataset: Dataset) -> bool:
166
+ if dataset.type_of_target not in ['binary', 'multiclass']:
167
+ return False
168
+ if dataset.X.empty or dataset.y is None or len(dataset.y) == 0:
169
+ return False
170
+ if dataset.get_columns_names_by_type(DataType.CATEGORICAL):
171
+ return False
172
+ if not dataset.get_columns_names_by_type(DataType.NUMERIC):
173
+ return False
174
+ unsupported = dataset.get_columns_names_by_type(
175
+ [DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
176
+ )
177
+ if unsupported:
178
+ return False
179
+ _, counts = np.unique(dataset.y, return_counts=True)
180
+ if len(counts) < 2:
181
+ return False
182
+ return counts.min() > 1
@@ -0,0 +1,193 @@
1
+ """[STEP] SMOTEENN"""
2
+ import inspect
3
+ import textwrap
4
+ from typing import Any
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+ from imblearn.combine import SMOTEENN
9
+ from imblearn.over_sampling import SMOTE
10
+ from imblearn.under_sampling import EditedNearestNeighbours
11
+
12
+ from ...actionable import Actionable
13
+ from ...candidate import Candidate
14
+ from ...data_type import DataType
15
+ from ...dataset import Dataset
16
+ from ...decorators.all import is_step
17
+
18
+
19
+ @is_step('imbalance')
20
+ class ActSMOTEENN(Actionable):
21
+ """[STEP] SMOTEENN"""
22
+
23
+ name: str = "SMOTE ENN"
24
+ _description: str = textwrap.dedent('''\
25
+ SMOTEENN balances data by creating synthetic minority samples and
26
+ removing ambiguous samples with Edited Nearest Neighbors.''')
27
+ _description_long: str = textwrap.dedent('''\
28
+ SMOTEENN combines SMOTE oversampling with Edited Nearest Neighbors
29
+ cleaning. It adds synthetic minority samples, then removes samples
30
+ that disagree with their neighbors to reduce noise and class overlap.''')
31
+ _usage: str = "Use when imbalance with noisy borders needs SMOTE plus cleaning vs ActSMOTE. Applicable to numeric-only binary or multiclass data with >=2 samples per class. Avoid when categorical/text features, very small minorities, or you want pure under-sampling like ActNearMiss."
32
+ refs: list[dict[str, Any]] = [
33
+ {
34
+ 'year': 2004,
35
+ 'name': 'A Study of the Behavior of Several Methods for Balancing '
36
+ 'Machine Learning Training Data',
37
+ 'authors': [
38
+ 'Gustavo E. A. P. A. Batista',
39
+ 'Ronaldo C. Prati',
40
+ 'Maria Carolina Monard'
41
+ ],
42
+ 'doi': 'https://doi.org/10.1145/1007730.1007735',
43
+ 'publisher': 'ACM SIGKDD Explorations Newsletter Vol.6 No.1 page 20--29'
44
+ }
45
+ ]
46
+
47
+ def __init__(self):
48
+ self.configuration = {
49
+ 'sampling_strategy': {
50
+ 'description': 'Sampling strategy to balance classes.',
51
+ 'default': 'auto',
52
+ 'categorical': ['minority', 'auto']
53
+ },
54
+ 'k_neighbors': {
55
+ 'description': 'Number of nearest neighbors used to create synthetic samples.',
56
+ 'default': 5,
57
+ 'range': [1, 20]
58
+ },
59
+ 'n_neighbors': {
60
+ 'description': 'Number of neighbors used by Edited Nearest Neighbors.',
61
+ 'default': 3,
62
+ 'range': [1, 20]
63
+ },
64
+ 'kind_sel': {
65
+ 'description': 'Rule used by Edited Nearest Neighbors to select samples.',
66
+ 'default': 'all',
67
+ 'categorical': ['all', 'mode'],
68
+ 'passthrough': False
69
+ },
70
+ 'random_state': {
71
+ 'description': 'Random seed used for reproducibility.',
72
+ 'default': 42
73
+ }
74
+ }
75
+ self.resampler: SMOTEENN | None = None
76
+ self.categorical_columns: list[str] = []
77
+ self.numeric_columns: list[str] = []
78
+ self._effective_k_neighbors: int | None = None
79
+ self._effective_n_neighbors: int | None = None
80
+
81
+ def fit(self, dataset: Dataset) -> Actionable:
82
+ self.resampler = None
83
+ self.categorical_columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
84
+ self.numeric_columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
85
+ self._effective_k_neighbors = None
86
+ self._effective_n_neighbors = None
87
+
88
+ if dataset.X.empty or dataset.type_of_target not in ['binary', 'multiclass']:
89
+ return self
90
+
91
+ if dataset.y is None or len(dataset.y) == 0:
92
+ return self
93
+
94
+ unsupported = dataset.get_columns_names_by_type(
95
+ [DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
96
+ )
97
+ if unsupported:
98
+ return self
99
+
100
+ if self.categorical_columns:
101
+ return self
102
+
103
+ if not self.numeric_columns:
104
+ return self
105
+
106
+ _, counts = np.unique(dataset.y, return_counts=True)
107
+ if len(counts) < 2:
108
+ return self
109
+
110
+ min_count = int(counts.min())
111
+ if min_count <= 1:
112
+ return self
113
+
114
+ max_k = min_count - 1
115
+ k_neighbors = min(int(self.get_config('k_neighbors')), max_k)
116
+ k_neighbors = max(1, k_neighbors)
117
+ self._effective_k_neighbors = k_neighbors
118
+
119
+ total_count = int(len(dataset.y))
120
+ max_neighbors = max(1, total_count - 1)
121
+ n_neighbors = min(int(self.get_config('n_neighbors')), max_neighbors)
122
+ n_neighbors = max(1, n_neighbors)
123
+ self._effective_n_neighbors = n_neighbors
124
+
125
+ smote = SMOTE(
126
+ sampling_strategy=self.get_config('sampling_strategy'),
127
+ k_neighbors=k_neighbors,
128
+ random_state=self.get_config('random_state')
129
+ )
130
+
131
+ enn_params = {
132
+ 'n_neighbors': n_neighbors
133
+ }
134
+ if 'kind_sel' in inspect.signature(EditedNearestNeighbours).parameters:
135
+ enn_params['kind_sel'] = self.get_config('kind_sel')
136
+ enn = EditedNearestNeighbours(**enn_params)
137
+
138
+ smoteenn_params = {}
139
+ smoteenn_sig = inspect.signature(SMOTEENN).parameters
140
+ if 'smote' in smoteenn_sig:
141
+ smoteenn_params['smote'] = smote
142
+ if 'enn' in smoteenn_sig:
143
+ smoteenn_params['enn'] = enn
144
+ if 'sampling_strategy' in smoteenn_sig and 'smote' not in smoteenn_params:
145
+ smoteenn_params['sampling_strategy'] = self.get_config('sampling_strategy')
146
+ if 'random_state' in smoteenn_sig and 'smote' not in smoteenn_params:
147
+ smoteenn_params['random_state'] = self.get_config('random_state')
148
+
149
+ self.resampler = SMOTEENN(**smoteenn_params)
150
+ self.resampler.fit(dataset.X, dataset.y)
151
+ return self
152
+
153
+ def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
154
+ """Apply SMOTEENN.
155
+
156
+ :param pd.DataFrame X: Features to resample
157
+ :param pd.DataFrame y: Labels to resample
158
+ :return: Resampled X and y
159
+ """
160
+ if self.resampler is None:
161
+ return X, y
162
+ return self.resampler.fit_resample(X, y)
163
+
164
+ def priorize(self, candidate: Candidate = None) -> float:
165
+ if candidate is None or candidate.dataset.y is None:
166
+ return 0.0
167
+ y = candidate.dataset.y
168
+ if len(y) == 0:
169
+ return 0.0
170
+ _, counts = np.unique(y, return_counts=True)
171
+ if len(counts) < 2:
172
+ return 0.0
173
+ imbalance = 1.0 - (counts.min() / counts.max())
174
+ return float(min(1.0, max(0.0, imbalance)))
175
+
176
+ def suitable(self, dataset: Dataset) -> bool:
177
+ if dataset.type_of_target not in ['binary', 'multiclass']:
178
+ return False
179
+ if dataset.X.empty or dataset.y is None or len(dataset.y) == 0:
180
+ return False
181
+ if dataset.get_columns_names_by_type(DataType.CATEGORICAL):
182
+ return False
183
+ if not dataset.get_columns_names_by_type(DataType.NUMERIC):
184
+ return False
185
+ unsupported = dataset.get_columns_names_by_type(
186
+ [DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
187
+ )
188
+ if unsupported:
189
+ return False
190
+ _, counts = np.unique(dataset.y, return_counts=True)
191
+ if len(counts) < 2:
192
+ return False
193
+ return counts.min() > 1
@@ -0,0 +1,138 @@
1
+ """[STEP] Tomek Links"""
2
+ import inspect
3
+ import textwrap
4
+ from typing import Any
5
+
6
+ import numpy as np
7
+ import pandas as pd
8
+ from imblearn.under_sampling import TomekLinks
9
+
10
+ from ...actionable import Actionable
11
+ from ...candidate import Candidate
12
+ from ...data_type import DataType
13
+ from ...dataset import Dataset
14
+ from ...decorators.all import is_step
15
+
16
+
17
+ @is_step('imbalance')
18
+ class ActTomekLinks(Actionable):
19
+ """[STEP] Tomek Links"""
20
+
21
+ name: str = "Tomek Links"
22
+ _description: str = textwrap.dedent('''\
23
+ TomekLinks cleans class boundaries by removing samples that form
24
+ nearest-neighbor pairs across classes.''')
25
+ _description_long: str = textwrap.dedent('''\
26
+ TomekLinks identifies pairs of samples from different classes that are
27
+ each other's nearest neighbors (Tomek links). Removing the majority
28
+ samples in these pairs reduces overlap and cleans noisy boundaries.''')
29
+ _usage: str = "Use when you want light boundary cleaning instead of heavier ActNearMiss. Applicable to numeric-only binary or multiclass data. Avoid when data includes categorical/text/date or you need to add samples (ActSMOTE)."
30
+ refs: list[dict[str, Any]] = [
31
+ {
32
+ 'year': 1976,
33
+ 'name': 'Two Modifications of CNN',
34
+ 'authors': [
35
+ 'Ivan Tomek'
36
+ ],
37
+ 'publisher': 'IEEE Transactions on Systems, Man, and Cybernetics Vol.6 No.11 '
38
+ 'page 769--772'
39
+ }
40
+ ]
41
+
42
+ def __init__(self):
43
+ self.configuration = {
44
+ 'sampling_strategy': {
45
+ 'description': 'Sampling strategy to remove Tomek link pairs.',
46
+ 'default': 'auto',
47
+ 'categorical': ['auto', 'majority', 'all']
48
+ }
49
+ }
50
+ self.resampler: TomekLinks | None = None
51
+ self.categorical_columns: list[str] = []
52
+ self.numeric_columns: list[str] = []
53
+ self.imbalance_ratio: float = 0.0
54
+
55
+ def fit(self, dataset: Dataset) -> Actionable:
56
+ self.resampler = None
57
+ self.categorical_columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
58
+ self.numeric_columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
59
+ self.imbalance_ratio = 0.0
60
+
61
+ if dataset.X.empty or dataset.type_of_target not in ['binary', 'multiclass']:
62
+ return self
63
+
64
+ if dataset.y is None or len(dataset.y) == 0:
65
+ return self
66
+
67
+ unsupported = dataset.get_columns_names_by_type(
68
+ [DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
69
+ )
70
+ if unsupported:
71
+ return self
72
+
73
+ if self.categorical_columns:
74
+ return self
75
+
76
+ if not self.numeric_columns:
77
+ return self
78
+
79
+ _, counts = np.unique(dataset.y, return_counts=True)
80
+ if len(counts) < 2:
81
+ return self
82
+
83
+ max_count = int(counts.max())
84
+ min_count = int(counts.min())
85
+ if max_count <= 0 or min_count <= 0:
86
+ return self
87
+
88
+ self.imbalance_ratio = 1.0 - (min_count / max_count)
89
+
90
+ params = self.passthrough_parameters()
91
+ sig_params = inspect.signature(TomekLinks).parameters
92
+ params = {key: value for key, value in params.items() if key in sig_params}
93
+
94
+ self.resampler = TomekLinks(**params)
95
+ self.resampler.fit(dataset.X, dataset.y)
96
+ return self
97
+
98
+ def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
99
+ """Apply Tomek Links.
100
+
101
+ :param pd.DataFrame X: Features to resample
102
+ :param pd.DataFrame y: Labels to resample
103
+ :return: Resampled X and y
104
+ """
105
+ if self.resampler is None:
106
+ return X, y
107
+ return self.resampler.fit_resample(X, y)
108
+
109
+ def priorize(self, candidate: Candidate = None) -> float:
110
+ if candidate is None or candidate.dataset.y is None:
111
+ return 0.0
112
+ y = candidate.dataset.y
113
+ if len(y) == 0:
114
+ return 0.0
115
+ _, counts = np.unique(y, return_counts=True)
116
+ if len(counts) < 2:
117
+ return 0.0
118
+ imbalance = 1.0 - (counts.min() / counts.max())
119
+ return float(min(1.0, max(0.0, imbalance)))
120
+
121
+ def suitable(self, dataset: Dataset) -> bool:
122
+ if dataset.type_of_target not in ['binary', 'multiclass']:
123
+ return False
124
+ if dataset.X.empty or dataset.y is None or len(dataset.y) == 0:
125
+ return False
126
+ if dataset.get_columns_names_by_type(DataType.CATEGORICAL):
127
+ return False
128
+ if not dataset.get_columns_names_by_type(DataType.NUMERIC):
129
+ return False
130
+ unsupported = dataset.get_columns_names_by_type(
131
+ [DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
132
+ )
133
+ if unsupported:
134
+ return False
135
+ _, counts = np.unique(dataset.y, return_counts=True)
136
+ if len(counts) < 2:
137
+ return False
138
+ return True
@@ -0,0 +1,6 @@
1
+ """Normalize and scaler Actionables"""
2
+ from .act_minmax_scaler import ActMinMaxScaler
3
+ from .act_standard_scaler import ActStandardScaler
4
+ from .act_robust_scaler import ActRobustScaler
5
+ from .act_max_abs_scaler import ActMaxAbsScaler
6
+ from .act_normalizer import ActNormalizer
@@ -0,0 +1,78 @@
1
+ """[STEP] Max Abs Scaler"""
2
+ import textwrap
3
+ from sklearn.preprocessing import MaxAbsScaler
4
+ import pandas as pd
5
+ from ...actionable import Actionable
6
+ from ...dataset import Dataset
7
+ from ...candidate import Candidate
8
+ from ...decorators.all import is_step
9
+ from ...data_type import DataType
10
+
11
+
12
+ @is_step('normalize')
13
+ class ActMaxAbsScaler(Actionable):
14
+ """[STEP] Max Abs Scaler"""
15
+
16
+ name: str = "Max Abs Scaler"
17
+ _description: str = textwrap.dedent('''\
18
+ MaxAbsScaler rescales numeric data by dividing by the maximum absolute value,
19
+ keeping values within a [-1, 1] range.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ MaxAbsScaler scales each numeric feature by its maximum absolute value
22
+ observed in the training data. This keeps values within [-1, 1] while
23
+ preserving sparsity because it does not center the data.
24
+ It is a good fit for sparse datasets where zeros should remain zeros.''')
25
+ _usage: str = "Use when numeric features are sparse and you want scale to [-1, 1] without centering; compare ActMinMaxScaler. Applicable to numeric data with many zeros or sparse matrices. Avoid when you need centering or heavy outlier handling; consider ActNormalizer or ActRobustScaler."
26
+
27
+ def __init__(self):
28
+ self.columns: list[str] = None
29
+ self.scaler: MaxAbsScaler = None
30
+
31
+ self.configuration = {
32
+ 'copy': {
33
+ 'description': 'Set to False to perform scaling in-place when possible.',
34
+ 'default': True,
35
+ 'categorical': [True, False]
36
+ }
37
+ }
38
+
39
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
40
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
41
+ if self.columns and not dataset.X.empty:
42
+ values = dataset.X[self.columns]
43
+ self.scaler = MaxAbsScaler(**self.passthrough_parameters())
44
+ self.scaler.fit(values)
45
+ else:
46
+ self.scaler = None
47
+ return self
48
+
49
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
50
+ """Apply max abs scaler
51
+
52
+ :param pd.DataFrame X: DataFrame to transform
53
+ :return: Transformed dataset
54
+ """
55
+ if self.scaler and self.columns:
56
+ columns = [column for column in self.columns if column in X.columns]
57
+ if not columns:
58
+ return X
59
+ X[columns] = self.scaler.transform(X[columns])
60
+ return X
61
+
62
+ def suitable(self, dataset: Dataset) -> bool:
63
+ columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
64
+ return bool(columns) and not dataset.X.empty
65
+
66
+ def priorize(self, candidate: Candidate = None) -> float:
67
+ if candidate is None:
68
+ return 0.0
69
+ columns = candidate.dataset.get_columns_names_by_type(DataType.NUMERIC)
70
+ if not columns or candidate.dataset.X.empty:
71
+ return 0.0
72
+ values = candidate.dataset.X[columns]
73
+ total = values.size - values.isna().sum().sum()
74
+ if total <= 0:
75
+ return 0.0
76
+ zeros = (values == 0).sum().sum()
77
+ zero_ratio = zeros / total
78
+ return min(1.0, zero_ratio * 1.5)
@@ -0,0 +1,56 @@
1
+ """[STEP] Min Max Scaler"""
2
+ import textwrap
3
+ from sklearn.preprocessing import MinMaxScaler
4
+ import pandas as pd
5
+ from ...actionable import Actionable
6
+ from ...dataset import Dataset
7
+ from ...candidate import Candidate
8
+ from ...decorators.all import is_step
9
+ from ...data_type import DataType
10
+
11
+ @is_step('normalize')
12
+ class ActMinMaxScaler(Actionable):
13
+ """[STEP] Min Max Scaler"""
14
+
15
+ name: str = "Min Max Scaler"
16
+ _usage: str = "Use when you need bounded scaling of numeric features for scale-sensitive models; unlike ActNormalizer, keeps feature ranges. Applicable to continuous numeric columns with stable min/max. Avoid when outliers or range drift dominate; consider ActRobustScaler."
17
+ _description: str = textwrap.dedent('''\
18
+ MinMaxScaler is a tool that helps computers understand complex relationships
19
+ between things by turning them into simpler numbers within a fixed range.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ MinMaxScaler is a machine learning technique used to scale numerical
22
+ features to a fixed range, typically between zero and one.
23
+ It works by finding the minimum and maximum values for each feature in the training data,
24
+ then scaling all values to fall between those extremes.
25
+ This transformation helps ensure that all features are on the same scale,
26
+ which can improve the performance of many machine learning algorithms.''')
27
+
28
+ def __init__(self):
29
+ self.columns: list[str] = None
30
+ self.scaler: MinMaxScaler = None
31
+
32
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
33
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
34
+ if self.columns:
35
+ values = dataset.X[self.columns]
36
+ self.scaler = MinMaxScaler()
37
+ self.scaler.fit(values)
38
+ else:
39
+ self.scaler = None
40
+ return self
41
+
42
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
43
+ """Apply min max scaler
44
+
45
+ :param pd.DataFrame X: DataFrame to transform
46
+ :return: Transformed dataset
47
+ """
48
+ if self.scaler and self.columns:
49
+ columns = [column for column in self.columns if column in X.columns]
50
+ if not columns:
51
+ return X
52
+ X[columns] = self.scaler.transform(X[columns])
53
+ return X
54
+
55
+ def priorize(self, candidate: Candidate = None) -> float:
56
+ return 0.5