PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
iaml/__init__.py ADDED
@@ -0,0 +1,56 @@
1
+ """Integrated AutoML for Medical Labs (IAML).
2
+
3
+ IAML helps clinical research teams build, evaluate and inspect machine learning
4
+ pipelines for tabular classification, regression and survival analysis. Modular
5
+ preprocessing and modeling steps support automated search, while pipeline
6
+ descriptions, evaluation metrics and SHAP explanations support review of the
7
+ resulting models and reporting of research methods.
8
+ """
9
+ from .iaml import IAML
10
+ from .core_dispatcher import CoreDispatcher
11
+ from .step import *
12
+ from .metastep import MetaStep
13
+ from .actionable import Actionable
14
+ from .candidate import Candidate
15
+ from .dataset import Dataset
16
+ from .data_type import DataType
17
+ from .metric_plot import MetricPlot
18
+ from .metric import Metric
19
+ from .plot import Plot, StatisticPlot
20
+ from .statistic import Statistic
21
+ from .cache import Cache
22
+ from .void_step import VoidStep
23
+
24
+ from .meta_ordered_step import MetaOrderedStep
25
+ from .meta_explorer_step import MetaExplorerStep
26
+ from .meta_partial_explorer_step import MetaPartialExplorerStep
27
+
28
+ # Default Actionables
29
+ from .actionables import *
30
+
31
+ # Wrappers
32
+ from .wrapper import *
33
+
34
+ # Metrics
35
+ from .metrics import *
36
+
37
+ # Statistics
38
+ from .statistics import *
39
+
40
+ # Plots
41
+ from .plots import *
42
+
43
+ # Stack
44
+ from .stack import Stack
45
+
46
+ # Optimizer
47
+ from .optimizers import *
48
+
49
+ # Type of target
50
+ from .type_of_target import type_of_target
51
+
52
+ # Ensure star imports expose all public names, even if __all__ is set elsewhere.
53
+ __all__ = [
54
+ name for name in globals()
55
+ if not name.startswith("_") and name != "__all__"
56
+ ]
iaml/actionable.py ADDED
@@ -0,0 +1,11 @@
1
+ """Classic kind of Step that transform, resample or predict from Candidate"""
2
+
3
+ # -> Must be a wildcard import to help IAML to know all available the steps
4
+ from .step import * # pylint: disable=unused-wildcard-import,wildcard-import
5
+ from .decorators.all import is_step
6
+
7
+
8
+ @is_step('actionable')
9
+ class Actionable(Step):
10
+ """Classic kind of Step that transform, resample or predict from Candidate"""
11
+ _usage = "Use when you need a concrete transform/resample/predict step over a Candidate. Applicable to pipeline steps that operate directly on Candidate data. Avoid when you need meta orchestration like MetaStep or wrappers like StepWrapper."
@@ -0,0 +1,21 @@
1
+ """
2
+ All Step with transform, resample or predict function
3
+ """
4
+ from . import (
5
+ boosting,
6
+ predictors,
7
+ features_selection,
8
+ cleaning,
9
+ normalize,
10
+ features_precleaning,
11
+ features_preprocessing,
12
+ imbalance,
13
+ )
14
+ from .boosting import *
15
+ from .predictors import *
16
+ from .features_selection import *
17
+ from .cleaning import *
18
+ from .normalize import *
19
+ from .features_precleaning import *
20
+ from .features_preprocessing import *
21
+ from .imbalance import *
@@ -0,0 +1,4 @@
1
+ """
2
+ All boosting Actionables
3
+ """
4
+ from .act_adaboost import ActAdaBoost
@@ -0,0 +1,59 @@
1
+ """Apply AdaBoost on models"""
2
+ from typing import Any
3
+ from sklearn.ensemble import AdaBoostClassifier
4
+
5
+ from ...actionable import Actionable
6
+ from ...decorators.all import runner
7
+ from ...candidate import Candidate
8
+
9
+
10
+ class ActAdaBoost(Actionable):
11
+ """Apply Adaboost on models
12
+
13
+ Configuration:
14
+ * `random_state`: Random seed (defaults to 42).
15
+ * `n_estimator`: Number of estimators (defaults to 2000).
16
+ """
17
+
18
+ name: str = "AdaBoost Classifier"
19
+ refs: list[dict[str, Any]] = [
20
+ {
21
+ 'year': 1995,
22
+ 'name': (
23
+ 'A desicion-theoretic generalization of on-line learning'
24
+ 'and an application to boosting'
25
+ ),
26
+ 'authors': [
27
+ 'Yoav Freund',
28
+ 'Robert E. Schapire'
29
+ ],
30
+ 'doi': 'https://doi.org/10.1007/3-540-59119-2_166',
31
+ 'publisher': 'Springer, Berlin, Heidelberg'
32
+ }
33
+ ]
34
+
35
+ def __init__(self):
36
+ self.configuration = {
37
+ 'random_state': {
38
+ 'description': 'random_state',
39
+ 'default': 42
40
+ },
41
+ 'n_estimator': {
42
+ 'description': 'Number of estimators',
43
+ 'default': 2000
44
+ },
45
+ }
46
+
47
+ @runner
48
+ def run(self, candidate: Candidate) -> Candidate:
49
+ model = AdaBoostClassifier(
50
+ candidate.model,
51
+ n_estimators=self.get_config('n_estimator'),
52
+ random_state=self.get_config('random_state'))
53
+
54
+ model.fit(candidate.dataset.X, candidate.dataset.y)
55
+
56
+ return candidate.to_output(None, None, model)
57
+
58
+ def priorize(self, candidate: Candidate = None) -> float:
59
+ return 0.5
@@ -0,0 +1,26 @@
1
+ """
2
+ All cleaning actionables
3
+ """
4
+ from .act_mean_column import ActMeanColumn
5
+ from .act_drop_numerical_column import ActDropNumericalColumn
6
+ from .act_drop_textual_column import ActDropTextualColumn
7
+ from .act_drop_categorical_column import ActDropCategoricalColumn
8
+ from .act_onehot import ActOnehot
9
+ from .act_tf_idf import ActTfIdf
10
+ from .act_drop_date_column import ActDropDateColumn
11
+ from .act_split_date import ActSplitDate
12
+ from .act_word2vec import ActWord2Vec
13
+ from .act_mice import ActMICEForestImputer
14
+ from .act_simple_imputer import ActSimpleImputer
15
+ from .act_knn_imputer import ActKNNImputer
16
+ from .act_categorical_imputer import ActCategoricalImputer
17
+ from .act_rare_category_grouper import ActRareCategoryGrouper
18
+ from .act_frequency_encoder import ActFrequencyEncoder
19
+ from .act_target_encoder import ActTargetEncoder
20
+ from .act_ordinal_encoder import ActOrdinalEncoder
21
+ from .act_count_vectorizer import ActCountVectorizer
22
+ from .act_hashing_vectorizer import ActHashingVectorizer
23
+ from .act_text_normalizer import ActTextNormalizer
24
+ from .act_missing_indicator import ActMissingIndicator
25
+ from .act_missing_count_feature import ActMissingCountFeature
26
+ from .act_drop_high_cardinality_categorical import ActDropHighCardinalityCategorical
@@ -0,0 +1,124 @@
1
+ """[STEP] Impute missing categorical values."""
2
+ import textwrap
3
+ from typing import Any
4
+
5
+ import pandas as pd
6
+
7
+ from ...actionable import Actionable
8
+ from ...candidate import Candidate
9
+ from ...data_type import DataType
10
+ from ...dataset import Dataset
11
+ from ...decorators.all import is_step
12
+
13
+
14
+ @is_step('cleaning')
15
+ class ActCategoricalImputer(Actionable):
16
+ """[STEP] Impute missing categorical values."""
17
+
18
+ name: str = 'Impute missing categorical values'
19
+ _usage: str = 'Use when categorical columns have missing labels you want to keep (vs ActDropCategoricalColumn). Applicable to categorical/category dtype features with NA gaps. Avoid when missingness is extreme or cardinality is high; consider ActDropHighCardinalityCategorical.'
20
+ _description: str = textwrap.dedent('''\
21
+ Impute missing categorical values using the {strategy} strategy.''')
22
+ _description_long: str = textwrap.dedent('''\
23
+ Replace missing values in categorical columns with either the most frequent
24
+ observed category or a constant "missing" label. The label can be customized
25
+ via missing_label and is also used as a fallback when no mode can be computed.''')
26
+
27
+ def __init__(self) -> None:
28
+ self.columns: list[str] = []
29
+ self.fill_values: dict[str, Any] = {}
30
+ self._missing_stats: dict[str, tuple[int, int, float]] = {}
31
+
32
+ self.configuration = {
33
+ 'strategy': {
34
+ 'description': 'Imputation strategy for categorical columns.',
35
+ 'default': 'most_frequent',
36
+ 'categorical': ['most_frequent', 'missing']
37
+ },
38
+ 'missing_label': {
39
+ 'description': 'Label used when strategy="missing" or when no mode exists.',
40
+ 'default': 'missing'
41
+ }
42
+ }
43
+
44
+ def fit(self, dataset: Dataset) -> Actionable:
45
+ self.columns = self._select_columns(dataset)
46
+ self.fill_values = {}
47
+ self._missing_stats = {}
48
+ self.explanations = []
49
+
50
+ if not self.columns or dataset.X.empty:
51
+ return self
52
+
53
+ X_cat = dataset.X[self.columns]
54
+ total_rows = len(X_cat)
55
+ if total_rows == 0:
56
+ return self
57
+
58
+ missing_counts = X_cat.isna().sum()
59
+ strategy = self.get_config('strategy')
60
+ missing_label = self.get_config('missing_label')
61
+
62
+ for column in self.columns:
63
+ series = X_cat[column]
64
+ missing = int(missing_counts[column])
65
+ pct = (missing / total_rows * 100.0) if total_rows else 0.0
66
+ self._missing_stats[column] = (missing, total_rows, pct)
67
+
68
+ if strategy == 'missing':
69
+ fill_value = missing_label
70
+ else:
71
+ mode_series = series.mode(dropna=True)
72
+ if not mode_series.empty:
73
+ fill_value = mode_series.iloc[0]
74
+ else:
75
+ fill_value = missing_label
76
+
77
+ self.fill_values[column] = fill_value
78
+ if missing > 0:
79
+ self.explanations.append(
80
+ f"Filled {missing} missing values in `{column}` with {fill_value!r}."
81
+ )
82
+
83
+ return self
84
+
85
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
86
+ if not self.columns or not self.fill_values:
87
+ return X
88
+
89
+ for column, fill_value in self.fill_values.items():
90
+ if column not in X.columns:
91
+ continue
92
+ if pd.api.types.is_categorical_dtype(X[column]):
93
+ if fill_value not in X[column].cat.categories:
94
+ X[column] = X[column].cat.add_categories([fill_value])
95
+ X[column] = X[column].fillna(fill_value)
96
+
97
+ return X
98
+
99
+ def suitable(self, dataset: Dataset) -> bool:
100
+ columns = self._select_columns(dataset)
101
+ if not columns or dataset.X.empty:
102
+ return False
103
+ return bool(dataset.X[columns].isna().any().any())
104
+
105
+ def priorize(self, candidate: Candidate = None) -> float:
106
+ if candidate is None or candidate.dataset.X.empty:
107
+ return 0.0
108
+ columns = self._select_columns(candidate.dataset)
109
+ if not columns:
110
+ return 0.0
111
+ missing = candidate.dataset.X[columns].isna().sum().sum()
112
+ total = candidate.dataset.X[columns].size or 1
113
+ return min(1.0, missing / total)
114
+
115
+ def _select_columns(self, dataset: Dataset) -> list[str]:
116
+ columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
117
+ category_columns = list(dataset.X.select_dtypes(include=['category']).columns)
118
+ seen = set()
119
+ ordered = []
120
+ for column in columns + category_columns:
121
+ if column in dataset.X.columns and column not in seen:
122
+ ordered.append(column)
123
+ seen.add(column)
124
+ return ordered
@@ -0,0 +1,204 @@
1
+ """[STEP] Vectorize short text with CountVectorizer."""
2
+ import textwrap
3
+ from typing import Any
4
+
5
+ import pandas as pd
6
+ from sklearn.feature_extraction.text import CountVectorizer
7
+
8
+ from ...actionable import Actionable
9
+ from ...candidate import Candidate
10
+ from ...data_type import DataType
11
+ from ...dataset import Dataset
12
+ from ...decorators.all import is_step
13
+
14
+
15
+ @is_step('cleaning')
16
+ class ActCountVectorizer(Actionable):
17
+ """[STEP] Vectorize short text with CountVectorizer."""
18
+
19
+ name: str = 'Count Vectorizer'
20
+ _usage: str = 'Use when short text is predictive and you want bag-of-words counts instead of ActDropTextualColumn. Applicable to short text columns with a manageable vocabulary. Avoid when text is long, extremely sparse, or you would drop text entirely (ActDropTextualColumn).'
21
+ _description: str = textwrap.dedent('''\
22
+ Vectorize short text columns into bag-of-words counts with n-grams.''')
23
+ _description_long: str = textwrap.dedent('''\
24
+ Build a vocabulary on each short text column and replace it with count features
25
+ for each token or n-gram observed in the training data. Adjust the n-gram range
26
+ and vocabulary size to control sparsity and keep runtime manageable.''')
27
+
28
+ def __init__(self) -> None:
29
+ self.columns: list[tuple[str, CountVectorizer]] = []
30
+ self.configuration = {
31
+ 'ngram_min': {
32
+ 'description': 'Minimum n-gram size to include.',
33
+ 'default': 1
34
+ },
35
+ 'ngram_max': {
36
+ 'description': 'Maximum n-gram size to include.',
37
+ 'default': 2
38
+ },
39
+ 'max_features': {
40
+ 'description': 'Maximum size of the vocabulary (None keeps all).',
41
+ 'default': 2000
42
+ },
43
+ 'min_df': {
44
+ 'description': 'Minimum document frequency for a term to be kept.',
45
+ 'default': 1
46
+ },
47
+ 'max_df': {
48
+ 'description': 'Maximum document frequency for a term to be kept.',
49
+ 'default': 1.0
50
+ }
51
+ }
52
+
53
+ def fit(self, dataset: Dataset) -> Actionable:
54
+ self.columns = []
55
+ self.explanations = []
56
+
57
+ columns = dataset.get_columns_names_by_type([DataType.SHORT_TEXT])
58
+ if not columns or dataset.X.empty:
59
+ return self
60
+
61
+ params = self._build_vectorizer_params(len(dataset.X))
62
+
63
+ for column in columns:
64
+ values = dataset.X[column].fillna('').astype(str)
65
+ vectorizer = CountVectorizer(**params)
66
+ try:
67
+ vectorizer.fit(values)
68
+ except ValueError:
69
+ continue
70
+
71
+ feature_names = vectorizer.get_feature_names_out()
72
+ if len(feature_names) == 0:
73
+ continue
74
+
75
+ self.columns.append((column, vectorizer))
76
+ self.explanations.append(
77
+ f'Encoded text column **`{column}`** into **{len(feature_names)}** count features.'
78
+ )
79
+
80
+ if not self.columns and columns:
81
+ raise RuntimeError(
82
+ "Count vectorizer failed: no usable vocabulary in short text columns."
83
+ )
84
+
85
+ return self
86
+
87
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
88
+ """Apply CountVectorizer to short text columns.
89
+
90
+ :param pd.DataFrame X: DataFrame to transform.
91
+ :return: Transformed dataset.
92
+ """
93
+ if not self.columns:
94
+ return X
95
+
96
+ X = X.reset_index(drop=True)
97
+
98
+ for name, vectorizer in self.columns:
99
+ if name not in X.columns:
100
+ continue
101
+
102
+ values = X[name].fillna('').astype(str)
103
+ transformed = vectorizer.transform(values)
104
+ feature_names = vectorizer.get_feature_names_out()
105
+ if len(feature_names) == 0:
106
+ X = X.drop([name], axis=1)
107
+ continue
108
+
109
+ new_names = [f"{name}_{token}" for token in feature_names]
110
+ vector_df = pd.DataFrame(transformed.toarray(), columns=new_names)
111
+
112
+ X = pd.concat([X, vector_df], axis=1).drop([name], axis=1)
113
+
114
+ return X
115
+
116
+ def suitable(self, dataset: Dataset) -> bool:
117
+ if dataset.X.empty:
118
+ return False
119
+ return bool(dataset.get_columns_names_by_type([DataType.SHORT_TEXT]))
120
+
121
+ def priorize(self, candidate: Candidate = None) -> float:
122
+ if candidate is None or candidate.dataset.X.empty:
123
+ return 0.0
124
+
125
+ columns = candidate.dataset.get_columns_names_by_type([DataType.SHORT_TEXT])
126
+ if not columns:
127
+ return 0.0
128
+
129
+ total_columns = candidate.dataset.X.shape[1] or 1
130
+ return min(1.0, len(columns) / total_columns)
131
+
132
+ def _build_vectorizer_params(self, n_rows: int) -> dict[str, Any]:
133
+ ngram_min = self._coerce_int(self.get_config('ngram_min'), 1)
134
+ ngram_max = self._coerce_int(self.get_config('ngram_max'), max(ngram_min, 1))
135
+ ngram_min = max(1, ngram_min)
136
+ ngram_max = max(ngram_min, ngram_max)
137
+
138
+ min_df = self._coerce_df(self.get_config('min_df'), 1)
139
+ max_df = self._coerce_df(self.get_config('max_df'), 1.0)
140
+
141
+ if n_rows > 0:
142
+ min_df = self._clamp_df(min_df, n_rows)
143
+ max_df = self._clamp_df(max_df, n_rows)
144
+ if self._effective_df(min_df, n_rows) > self._effective_df(max_df, n_rows):
145
+ max_df = min_df
146
+
147
+ max_features = self._coerce_optional_int(self.get_config('max_features'))
148
+
149
+ params = {
150
+ 'ngram_range': (ngram_min, ngram_max),
151
+ 'min_df': min_df,
152
+ 'max_df': max_df
153
+ }
154
+ if max_features is not None:
155
+ params['max_features'] = max_features
156
+
157
+ return params
158
+
159
+ @staticmethod
160
+ def _coerce_int(value: Any, default: int) -> int:
161
+ try:
162
+ return int(value)
163
+ except (TypeError, ValueError):
164
+ return default
165
+
166
+ @staticmethod
167
+ def _coerce_optional_int(value: Any) -> int | None:
168
+ if value is None:
169
+ return None
170
+ try:
171
+ numeric = int(value)
172
+ except (TypeError, ValueError):
173
+ return None
174
+ if numeric <= 0:
175
+ return None
176
+ return numeric
177
+
178
+ @staticmethod
179
+ def _coerce_df(value: Any, default: float | int) -> float | int:
180
+ if value is None or isinstance(value, bool):
181
+ return default
182
+ if isinstance(value, int):
183
+ return max(0, value)
184
+ try:
185
+ numeric = float(value)
186
+ except (TypeError, ValueError):
187
+ return default
188
+ if numeric < 0:
189
+ return default
190
+ if numeric <= 1.0:
191
+ return float(numeric)
192
+ return int(round(numeric))
193
+
194
+ @staticmethod
195
+ def _clamp_df(value: float | int, n_rows: int) -> float | int:
196
+ if isinstance(value, float):
197
+ return min(max(value, 0.0), 1.0)
198
+ return min(max(value, 0), n_rows)
199
+
200
+ @staticmethod
201
+ def _effective_df(value: float | int, n_rows: int) -> float:
202
+ if isinstance(value, float):
203
+ return value * n_rows
204
+ return float(value)
@@ -0,0 +1,51 @@
1
+ """[STEP] Drop categorical columns"""
2
+ import textwrap
3
+ import pandas as pd
4
+ from ...actionable import Actionable
5
+ from ...dataset import Dataset
6
+ from ...data_type import DataType
7
+ from ...candidate import Candidate
8
+ from ...decorators.all import is_step
9
+
10
+
11
+ @is_step('cleaning', 'baseline_cleaning')
12
+ class ActDropCategoricalColumn(Actionable):
13
+ """[STEP] Drop categorical columns"""
14
+
15
+ name: str = 'Remove categorical columns'
16
+ _description: str = 'Remove all columns containing categorical data from the dataset'
17
+ _usage: str = 'Use when categorical columns must be removed for steps that cannot handle categories. Applicable to datasets with categorical or category-typed columns. Avoid when you can impute or encode categories instead (ActCategoricalImputer, ActCountVectorizer).'
18
+ _description_long: str = textwrap.dedent('''\
19
+ Remove all columns containing categorical data from the dataset.
20
+ This step is used to clean the dataset in order to perform other actions later on
21
+ that can't be applied to categorical columns.''')
22
+ can_be_disabled: bool = False
23
+
24
+ def __init__(self):
25
+ self.columns_to_drop: list[str] = None
26
+
27
+ def fit(self, dataset: Dataset) -> Actionable:
28
+ self.columns_to_drop = list(set(
29
+ dataset.get_columns_names_by_type([DataType.CATEGORICAL]) + \
30
+ list(dataset.X.select_dtypes(include=['category']).columns)
31
+ ))
32
+
33
+ self.explanations = [
34
+ f'Dropped column **`{c}`**.' for c in self.columns_to_drop
35
+ ]
36
+
37
+ return self
38
+
39
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
40
+ """Drop columns.
41
+
42
+ :param pd.DataFrame X: DataFrame to transform.
43
+ :return: Transformed DataFrame.
44
+ """
45
+ return X.drop(self.columns_to_drop, axis=1)
46
+
47
+ def priorize(self, candidate: Candidate = None) -> float:
48
+ return 0
49
+
50
+ def suitable(self, dataset: Dataset) -> bool:
51
+ return bool(dataset.get_columns_names_by_type([DataType.CATEGORICAL]))
@@ -0,0 +1,48 @@
1
+ """[STEP] Find and drop date column"""
2
+ import textwrap
3
+ import pandas as pd
4
+ from ...actionable import Actionable
5
+ from ...data_type import DataType
6
+ from ...candidate import Candidate
7
+ from ...dataset import Dataset
8
+ from ...decorators.all import is_step
9
+
10
+
11
+ @is_step('cleaning', 'baseline_cleaning')
12
+ class ActDropDateColumn(Actionable):
13
+ """Finds and drops date columns."""
14
+
15
+ name: str = 'Remove date columns'
16
+ _description: str = 'Remove all columns containing Date from the dataset'
17
+ _usage: str = 'Use when date columns are irrelevant and you want to remove them; for other types use ActDropNumericalColumn or ActDropCategoricalColumn. Applicable to columns typed as DataType.DATE. Avoid when dates carry predictive signal or need feature extraction.'
18
+ _description_long: str = textwrap.dedent('''\
19
+ Remove all columns containing Data from the dataset
20
+ This step is used to clean the dataset in order to perform other actions later on
21
+ that can't be applied to date columns.''')
22
+ can_be_disabled: bool = False
23
+
24
+ def __init__(self):
25
+ self.columns_to_drop: list[str] = None
26
+
27
+ def fit(self, dataset: Dataset) -> Actionable:
28
+ self.columns_to_drop = dataset.get_columns_names_by_type(DataType.DATE)
29
+
30
+ self.explanations = [
31
+ f'Dropped column **`{c}`**.' for c in self.columns_to_drop
32
+ ]
33
+
34
+ return self
35
+
36
+ def transform(self, X: pd.DataFrame) -> pd.DataFrame:
37
+ """Drop columns.
38
+
39
+ :param pd.DataFrame X: DataFrame to transform.
40
+ :return: Transformed DataFrame.
41
+ """
42
+ return X.drop(self.columns_to_drop, axis=1)
43
+
44
+ def priorize(self, candidate: Candidate = None) -> float:
45
+ return 0
46
+
47
+ def suitable(self, dataset: Dataset) -> bool:
48
+ return bool(dataset.get_columns_names_by_type(DataType.DATE))