PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
iaml/iaml.py ADDED
@@ -0,0 +1,1072 @@
1
+ """Integrated AutoML for Medical Labs (IAML).
2
+
3
+ Search and evaluate modular prediction pipelines for clinical research with
4
+ tabular data. Trained candidates expose their pipeline steps, evaluation metrics
5
+ and explanation methods for inspection and study reporting.
6
+ """
7
+ from copy import deepcopy
8
+ import time
9
+ import math
10
+ import textwrap
11
+ import multiprocessing
12
+ from typing import TYPE_CHECKING, Any
13
+ import numpy as np
14
+ import pandas as pd
15
+ from .timed_pool_executor import TimedPoolExecutor, TerminatedError
16
+ from .step import Step
17
+ from .cache import Cache
18
+ from .cache_keys import hash_evaluation_context
19
+ from .metastep import MetaStep
20
+ from .candidate import Candidate
21
+ from .dataset import Dataset
22
+ from .metric import Metric
23
+ from .statistic import Statistic
24
+ from .worker_manager import WorkerManager
25
+ from .splitters import kfold_splitter
26
+ from .meta_ordered_step import MetaOrderedStep
27
+ from .meta_explorer_step import MetaExplorerStep
28
+ from .meta_partial_explorer_step import MetaPartialExplorerStep
29
+ from .optimizers import Optimizer, GeneticOptimizer, RandomOptimizer, BayesianOptimizer
30
+ from .predictor import Predictor
31
+ from .logger import Logger
32
+ from .plot import StatisticPlot
33
+ from .actionables.cleaning.act_simple_imputer import ActSimpleImputer
34
+ from .actionables.normalize.act_standard_scaler import ActStandardScaler
35
+ from .sklearn_preprocessor import SklearnPreprocessor
36
+
37
+ # Default Actionables -> Must be a wildcard import to help IAML to know all available the steps
38
+ from .actionables import * # pylint: disable=unused-wildcard-import,wildcard-import
39
+
40
+ # Default Wrappers -> Must be a wildcard import to help IAML to know all available the steps
41
+ from .wrapper import * # pylint: disable=unused-wildcard-import,wildcard-import
42
+
43
+ # cuDF pandas acceleration
44
+ try:
45
+ import cudf.pandas
46
+ cudf.pandas.install()
47
+ Logger().info('cuDF is installed: using cuDF pandas accelerator mode.')
48
+ except ImportError as e:
49
+ Logger().warning('cuDF not found: falling back to standalone pandas.')
50
+
51
+ if TYPE_CHECKING:
52
+ from .iaml_pipeline import IAMLPipeline
53
+
54
+
55
+ class IAML: # pylint: disable=too-many-instance-attributes
56
+ """Configure and search prediction pipelines for a clinical research dataset.
57
+
58
+ :meth:`fit` returns trained candidates for evaluation, pipeline inspection
59
+ and explanation of predictions.
60
+
61
+ :param int, optional max_workers: Maximum parallel workers. Default to cpu count.
62
+ :param int, optional max_stage_duration: Maximum duration of a stage. Default to None.
63
+ :param callable, optional splitter: Split function to use. Default to kfold_splitter.
64
+ :param int, optional max_duration: Search time budget; -1 means no global limit.
65
+ :param int | str, optional time_before_sample_use: Time before we use sampled data.
66
+ Default to None.
67
+ :param bool, optional preprocessor: Use preprocessor. Default to False.
68
+ :param Metric, optional main_metric: Main metric instance, preserving its parameters.
69
+ Default to None.
70
+ :param Optimizer, optional optimizer: Optimizer class to use. Default to GeneticOptimizer.
71
+ :param int, optional train_on_n_samples: Limit the initial search dataset to this many
72
+ rows. None or nonpositive values use all rows.
73
+ :param bool, optional keep_training_history: If True, store detailed CV audit records for
74
+ every evaluated pipeline. Default to False.
75
+ :param bool, optional refit_on_sample: Reuse the initial train_on_n_samples sample for
76
+ final fitting. If False, refit on all input rows. Default to True; has no effect
77
+ without a positive train_on_n_samples limit.
78
+ :param initial_preprocessor: Optional clonable sklearn transformer. It must return
79
+ a numeric DataFrame with unchanged rows and index. Every generated pipeline,
80
+ including minimalist candidates, starts with this mandatory transformer.
81
+ It is fitted afresh within each CV training fold and during final fitting.
82
+ """
83
+ def __init__( # pylint: disable=too-many-arguments
84
+ self,
85
+ max_workers: int = None,
86
+ max_stage_duration: int = None,
87
+ splitter: callable = None,
88
+ max_duration: int = -1,
89
+ time_before_sample_use: int | str = None,
90
+ preprocessor: bool = False,
91
+ main_metric: Metric = None,
92
+ optimizer: Optimizer = GeneticOptimizer,
93
+ train_on_n_samples: int = None,
94
+ keep_training_history: bool = False,
95
+ refit_on_sample: bool = True,
96
+ initial_preprocessor: Any = None) -> None:
97
+ # Set pandas config to avoid SettingsWithcopyWarning
98
+ pd.options.mode.copy_on_write = True
99
+
100
+ self.preprocessor: bool = preprocessor
101
+ """Enable / Disable preprocessor"""
102
+
103
+ self.optimizer = optimizer
104
+ """Choose Optimizer"""
105
+
106
+ self.train_on_n_samples = train_on_n_samples
107
+ """If defined, pick n sample in the dataset before train"""
108
+
109
+ self.refit_on_sample: bool = refit_on_sample
110
+ """Apply the explicit search sample limit to final fitting as well."""
111
+
112
+ self.initial_preprocessor = initial_preprocessor
113
+ """Unfitted transformer template, prepended to all candidate pipelines."""
114
+
115
+ self.keep_training_history: bool = keep_training_history
116
+ """Whether to store detailed cross-validation audit records."""
117
+
118
+ self.training_history: list[dict[str, Any]] = []
119
+ """Detailed audit records for evaluated pipelines during the last fit."""
120
+
121
+ self._training_history_seen: set[tuple[Any, ...]] = set()
122
+ """Deduplicate audit records across warmup, cache hits, and repeated evaluations."""
123
+
124
+ # Set max duration of each stage
125
+ if max_stage_duration is None:
126
+ self.max_stage_duration = max(max_duration / 5, 900)
127
+ Logger().warning(
128
+ f"Max duration of each stage was set to {self.max_stage_duration} seconds")
129
+ else:
130
+ self.max_stage_duration = max_stage_duration
131
+
132
+ self.splitter: callable = splitter if splitter is not None else kfold_splitter
133
+ """Splitter callable"""
134
+
135
+ self.main_metric: Metric = main_metric
136
+ """Main metric"""
137
+
138
+ self.max_duration: int = max_duration
139
+ """Maximum training duration"""
140
+
141
+ if time_before_sample_use == 'auto' and max_duration:
142
+ self.time_before_sample_use = max(max_duration / 5, 60)
143
+ elif time_before_sample_use:
144
+ self.time_before_sample_use = time_before_sample_use
145
+ else:
146
+ self.time_before_sample_use = math.inf
147
+
148
+ self.candidates: list[Candidate] = None
149
+ """list of Candidates for this training"""
150
+
151
+ self.init_candidate: Candidate = None
152
+ """Initial candidate"""
153
+
154
+ self.first_step: Step = None # Will be the first Step of the pipeline (probably a MetaStep
155
+ """Hold the first step of the pipeline"""
156
+
157
+ self.last_stage_candidates: list[Candidate] = []
158
+ """Hold last generated candidates"""
159
+
160
+ self.executor: TimedPoolExecutor = None
161
+ """Hold TimePoolExecutor"""
162
+
163
+ self.default_pipeline() # Load default pipeline
164
+ self.max_workers = max_workers if (max_workers is not None and max_workers > 0) \
165
+ else multiprocessing.cpu_count()
166
+ """Hold maximum number of parallel workers"""
167
+
168
+ self.chosen_candidate: Candidate = None
169
+ """Hold the best candidate"""
170
+ WorkerManager(max_workers=self.max_workers)
171
+
172
+ self.descriptive_statistics: pd.DataFrame | None = None
173
+ """Cached descriptive statistics for the last fitted dataset."""
174
+
175
+ self._last_dataset: Dataset | None = None
176
+ """Dataset used for the most recent fit, for on-demand statistics."""
177
+
178
+ def __del__(self):
179
+ """Delete the TimedPoolExecutor"""
180
+ del self.executor
181
+
182
+ def load_pipeline(self, pipeline: dict) -> None:
183
+ """Load any kind of pipeline
184
+
185
+ :param dict pipeline: JSON description of the pipeline
186
+ """
187
+ self.first_step = Step.from_pipeline(pipeline)
188
+ self.minimal_predictor_step = self.__build_minimal_predictor_step()
189
+
190
+ def default_pipeline(self, fast: bool = False) -> None:
191
+ """Load the default pipeline.
192
+ Default pipeline is the recommended way to create classifier and regressor
193
+
194
+ Genetic search starts with one normalization and no resampling, then
195
+ explores alternatives through mutations. Other optimizers retain full
196
+ initial exploration because they only change hyperparameters.
197
+
198
+ :param bool, optional fast: If true, will only load fast machine learning model.
199
+ Fast mode is use to create fast pipeline and iterate
200
+ quickly when debugging code. Defaults to False.
201
+ """
202
+ self.first_step = MetaOrderedStep(tag="Main") # First step -> Contain all pipeline's stages
203
+
204
+ self.first_step.add_step(MetaStep(tag='features_precleaning',
205
+ name='Features Precleaning',
206
+ description=textwrap.dedent('''\
207
+ Converts complex columns into several columns, which helps the
208
+ model to extract information from your data.''')))
209
+ self.first_step.add_step(MetaStep(tag='cleaning',
210
+ name='Features Cleaning',
211
+ description=textwrap.dedent('''\
212
+ Improve data quality, handle missing values, extract
213
+ information from textual columns, etc.''')))
214
+ self.first_step.add_step(MetaStep(tag='features_selection',
215
+ name='Features Selection',
216
+ description=textwrap.dedent('''\
217
+ Decrease number of column to improve the models' performance.''')))
218
+ partial_exploration = (isinstance(self.optimizer, type)
219
+ and issubclass(self.optimizer, GeneticOptimizer))
220
+ explorer = MetaPartialExplorerStep if partial_exploration else MetaExplorerStep
221
+ normalization_options = {'initial_step': ActStandardScaler()} if partial_exploration else {}
222
+ imbalance_options = {} if partial_exploration else {'also_explore_without': True}
223
+ self.first_step.add_step(explorer(tag='normalize',
224
+ **normalization_options,
225
+ name='Features Normalization',
226
+ description=textwrap.dedent('''\
227
+ Normalize data to help model to give the same interest to each column''')))
228
+ self.first_step.add_step(explorer(tag='imbalance',
229
+ **imbalance_options,
230
+ name='Handle Imbalanced Data',
231
+ description=textwrap.dedent('''\
232
+ Balance the dataset to ensure the model does not favor the
233
+ majority class over the minority class''')))
234
+
235
+ if self.preprocessor:
236
+ self.first_step.add_step(
237
+ MetaExplorerStep(tag='features_preprocessing', also_explore_without=True)
238
+ )
239
+ else:
240
+ self.first_step.add_step(
241
+ MetaPartialExplorerStep(
242
+ tag='features_preprocessing',
243
+ name="Dimensionality Reduction (optional)",
244
+ description=textwrap.dedent('''\
245
+ Reduce the complexity of data and make computations
246
+ more efficient'''))
247
+ )
248
+
249
+ learning_tag = 'fast_predictor' if fast else 'predictor'
250
+
251
+ self.first_step.add_step(
252
+ MetaExplorerStep(
253
+ tag=learning_tag,
254
+ name="Machine learning models",
255
+ description="List of machine learning models IAML will try to optimize"))
256
+
257
+ self.minimal_predictor_step = self.__build_minimal_predictor_step()
258
+
259
+ def __build_minimal_predictor_step(self) -> MetaExplorerStep | None:
260
+ """Build the minimalist predictor stage if suitable models exist."""
261
+ minimal_step = MetaExplorerStep(
262
+ tag='minimal_predictor',
263
+ name='Minimalist Predictors',
264
+ description=textwrap.dedent('''\
265
+ Try high-performing boosting-style models without any preprocessing
266
+ to provide quick baseline candidates before the full pipeline is explored.'''))
267
+
268
+ if not minimal_step.steps:
269
+ return None
270
+
271
+ return minimal_step
272
+
273
+ def __callback(self, callback: callable, **kwargs: dict) -> None:
274
+ """Call callback function if defined
275
+
276
+ :param callable callback: Function to call.
277
+ :param dict, optional \\**kwargs: Additional parameters.
278
+ """
279
+ if callback and callable(callback):
280
+ callback(**kwargs)
281
+
282
+ def baseline( # pylint: disable=too-many-arguments
283
+ self,
284
+ X: pd.DataFrame,
285
+ y: pd.DataFrame,
286
+ groups: pd.DataFrame = None,
287
+ groups_columns: list[str] = None,
288
+ generation_sample_size: int = 200,
289
+ verbose: int = 1) -> Candidate:
290
+ """Run a very basic pipeline to train a model baseline
291
+
292
+ :param pd.DataFrame X: Training features
293
+ :param pd.DataFrame y: Training labels
294
+ :param pd.DataFrame, optional groups: Dataframe used to split data by groups.
295
+ Default to None.
296
+ :param list[str], optional groups_columns: List of column names used to split data by
297
+ groups. Default to None.
298
+ :param int, optional generation_sample_size: Size of the sample dataset used to generate
299
+ first generation of candidates (default 200).
300
+ :param int, optional verbose: Verbosity level. Default to 1.
301
+ :return: Baseline candidate
302
+ """
303
+ # Avoid [] dangerous default value in the signature
304
+ if groups_columns is None:
305
+ groups_columns = []
306
+
307
+ # Create a baseline pipeline
308
+ baseline_pipe = MetaOrderedStep(tag="Main") # First step -> Contain all pipeline's stages
309
+ baseline_pipe.add_step(MetaStep(tag='baseline_cleaning'))
310
+ baseline_pipe.add_step(MetaExplorerStep(tag='baseline_predictor'))
311
+
312
+ Logger().verbose = verbose # Set logger verbose
313
+
314
+ dataset:Dataset = Dataset(
315
+ deepcopy(X),
316
+ deepcopy(y),
317
+ groups=groups,
318
+ groups_columns=groups_columns)
319
+
320
+ ### INITIAL GENERATE CANDIDATE
321
+ init_candidate: Candidate = Candidate(
322
+ dataset.sample(generation_sample_size),
323
+ main_metric=self.main_metric)
324
+
325
+ # Select metrics used to evaluate performances
326
+ for metric \
327
+ in self.__metrics_selection(dataset.X, dataset.y, dataset.type_of_target):
328
+ init_candidate.add_metric(metric)
329
+
330
+ # Generate candidates
331
+ candidates = baseline_pipe.run(init_candidate)
332
+
333
+ # Remove candidate without predictor
334
+ candidates = [candidate for candidate in candidates \
335
+ if candidate.pipeline.predictor is not None]
336
+
337
+ for candidate in candidates:
338
+ candidate.pipeline.fit(dataset.X, dataset.y)
339
+
340
+ return candidates
341
+
342
+ ##################
343
+ ### PROPERTIES ###
344
+ ##################
345
+
346
+ # Dataset from candidate data
347
+ @property
348
+ def dataset(self) -> Dataset:
349
+ """Shortcut to get candidate Dataset
350
+
351
+ :return: Candidate dataset defined by .fit()
352
+ """
353
+ return self.init_candidate.dataset
354
+
355
+ ###########
356
+ ### RUN ###
357
+ ###########
358
+
359
+ def fit( # pylint: disable=too-many-arguments,too-many-locals
360
+ self,
361
+ X: pd.DataFrame,
362
+ y: pd.DataFrame,
363
+ groups: pd.DataFrame = None,
364
+ groups_columns: list[str] = None,
365
+ patience: int = -1,
366
+ generation_sample_size: int = 200,
367
+ n_candidates: int = 1,
368
+ callback: callable = None,
369
+ verbose: int = 1,
370
+ log_callback: callable = None) -> list[Candidate]:
371
+ """Run Pipeline to fit steps and models on X & y data.
372
+
373
+ :param pd.DataFrame X: Training features
374
+ :param pd.DataFrame y: Training labels
375
+ :param pd.DataFrame, optional groups: Dataframe used to split data by groups.
376
+ Default to None.
377
+ :param list[str], optional groups_columns: List of column names used to split data by
378
+ groups. Default to None.
379
+ :param int, optional patience: Max generation without improvement. Default to -1.
380
+ :param int, optional generation_sample_size: Size of the sample dataset used to generate
381
+ first generation of candidates (default 200).
382
+ :param int, optional n_candidates: Number of candidates to return. Default to 1.
383
+ :param callable, optional callback: Method call after each big step of training.
384
+ :param int, optional verbose: Verbosity level. Default to 1.
385
+ :param callable, optional log_callback: Callback for logger.
386
+ :return: List of all the generated candidates. Sorted by performances.
387
+ """
388
+ # Avoid [] dangerous default value in the signature
389
+ if groups_columns is None:
390
+ groups_columns = []
391
+
392
+ self.check_pipeline() # Raise error if the pipeline is not valid
393
+
394
+ Logger().verbose = verbose # Set logger verbose
395
+ if log_callback is not None:
396
+ Logger().set_callback(log_callback)
397
+
398
+ start_time = time.monotonic()
399
+ self.executor = TimedPoolExecutor(max_workers=self.max_workers)
400
+ self.training_history = []
401
+ self._training_history_seen = set()
402
+
403
+ def remain_time():
404
+ if self.max_duration == -1:
405
+ return math.inf
406
+ return max(0.0, self.max_duration - (time.monotonic() - start_time))
407
+
408
+ try:
409
+ refit_X, refit_y, refit_groups_columns = X, y, groups_columns
410
+ dataset:Dataset = Dataset(
411
+ deepcopy(X),
412
+ deepcopy(y),
413
+ groups=groups,
414
+ groups_columns=groups_columns)
415
+
416
+ if self.train_on_n_samples != None and self.train_on_n_samples > 0:
417
+ dataset = dataset.sample(self.train_on_n_samples)
418
+ if self.refit_on_sample:
419
+ # Preserve this sample even if the search later downsizes again.
420
+ # Dataset.X already excludes the grouping columns.
421
+ refit_X, refit_y, refit_groups_columns = dataset.X, dataset.y, []
422
+
423
+ self._last_dataset = dataset
424
+ self.descriptive_statistics = None
425
+
426
+ ### INITIAL GENERATE CANDIDATE
427
+ self.init_candidate: Candidate = Candidate(
428
+ dataset.sample(generation_sample_size),
429
+ main_metric=self.main_metric)
430
+
431
+ if self.initial_preprocessor is not None:
432
+ # Encode the generation sample so both branches can discover
433
+ # suitable predictors. Keep the raw search/refit datasets: the
434
+ # Step is cloned/refitted inside every pipeline's CV fit.
435
+ initial_step = SklearnPreprocessor(self.initial_preprocessor)
436
+ initial_step.fit(self.init_candidate.dataset)
437
+ self.init_candidate = self.init_candidate.add_to_pipeline(initial_step)
438
+
439
+ # Select metrics used to evaluate performances
440
+ for metric \
441
+ in self.__metrics_selection(dataset.X, dataset.y, dataset.type_of_target):
442
+ self.init_candidate.add_metric(metric)
443
+
444
+ minimal_candidates = self.__generate_minimal_candidates(self.init_candidate)
445
+
446
+ # Generate candidates
447
+ pipeline_candidates = self.__run(self.init_candidate)
448
+ pipeline_candidates = [candidate for candidate in pipeline_candidates \
449
+ if candidate.pipeline.predictor is not None]
450
+
451
+ candidates = minimal_candidates + pipeline_candidates
452
+ # Remove candidate without predictor
453
+ candidates = [candidate for candidate in candidates \
454
+ if candidate.pipeline.predictor is not None]
455
+ self.candidates = candidates
456
+
457
+ if minimal_candidates:
458
+ Logger().info(f"{len(minimal_candidates)} minimalist pipelines generated")
459
+ Logger().info(f"{len(candidates)} generated pipelines")
460
+
461
+
462
+ # Warmup is real CV, so it must use the same interruptible executor
463
+ # and shared search/stage budgets as every subsequent evaluation.
464
+ # Minimal candidates already lead the pool and provide a quick,
465
+ # honestly evaluated starting point without removing full pipelines.
466
+ warmup_candidate = None
467
+ if candidates and remain_time() >= 1:
468
+ # A single slow baseline must leave time to evaluate the other
469
+ # candidates. Unlimited searches retain the stage duration cap.
470
+ remaining = remain_time()
471
+ warmup_timeout = remaining / 5
472
+ Logger().info(
473
+ f"Warming up (at most {min(warmup_timeout, self.max_stage_duration):.2f}s): "
474
+ f"{candidates[0].pipeline.name}")
475
+ warmup_candidates = self.__run_evaluations(
476
+ candidates[:1], dataset, timeout=remaining, callback=callback,
477
+ stage_timeout=warmup_timeout)
478
+ if warmup_candidates:
479
+ warmup_candidate = warmup_candidates[0]
480
+ Logger().info("Warmed up !")
481
+ else:
482
+ Logger().info("No warmup result within the stage budget.")
483
+ # Try alternatives before retrying the same candidate,
484
+ # especially when only one worker is available.
485
+ candidates = candidates[1:] + candidates[:1]
486
+
487
+ ### INITIAL EVALUATION
488
+ # Evaluate candidates
489
+ gen0_candidates = []
490
+ i = 0
491
+ can_be_downsize = True
492
+ while can_be_downsize and not gen0_candidates and remain_time() >= 1:
493
+ # If process is too long and dataset big enough,
494
+ # we can downsize it to get quicker training
495
+ if i > 0:
496
+ dataset = dataset.sample(0.1)
497
+ Logger().warning(f"Training is too time consuming. \
498
+ Let's try again with dataset sample. \
499
+ New features shape {dataset.X.shape}")
500
+ i+= 1
501
+ can_be_downsize = dataset.X.shape[0] >= 500
502
+
503
+ timeout = min(remain_time(), self.time_before_sample_use) \
504
+ if can_be_downsize else remain_time()
505
+
506
+ gen0_candidates = self.__run_evaluations(candidates,
507
+ dataset, timeout=timeout, callback=callback)
508
+
509
+ if not gen0_candidates:
510
+ if warmup_candidate and warmup_candidate.computed_metrics:
511
+ Logger().warning(
512
+ "No candidates evaluated before timeout; using warmup candidate."
513
+ )
514
+ gen0_candidates = [warmup_candidate]
515
+ elif remain_time() < 1:
516
+ raise TimeoutError('IAML was unable to generate a model within the \
517
+ imposed time limit. Try increasing the processing time')
518
+ else:
519
+ raise RuntimeError('Undefined error. IAML was unable to create pipeline \
520
+ based on your data')
521
+
522
+ ### FINETUNING
523
+ candidates = self.__optimize(dataset,
524
+ gen0_candidates,
525
+ optimizer=self.__build_optimizer(remain_time()),
526
+ max_duration=remain_time(),
527
+ patience=patience,
528
+ callback=callback)
529
+
530
+ ### FINAL FIT
531
+ self.executor.shutdown()
532
+
533
+ # Fit candidates on the requested sample or all original input rows.
534
+ fit_candidates = []
535
+ last_fit_error = None
536
+ for candidate in candidates:
537
+ if len(fit_candidates) >= n_candidates:
538
+ break
539
+ Cache.reset()
540
+ current_candidate = deepcopy(candidate)
541
+ try:
542
+ Logger().info(f"Final fit: {len(refit_X)} rows, {current_candidate.pipeline.name}")
543
+ current_candidate.pipeline.fit(
544
+ refit_X,
545
+ refit_y,
546
+ groups_columns=refit_groups_columns,
547
+ metrics=current_candidate.metrics,
548
+ )
549
+ except (ValueError, np.linalg.LinAlgError) as exc:
550
+ Logger().warning(
551
+ f"Skipping candidate during final fit after failure: {exc!r}"
552
+ )
553
+ last_fit_error = exc
554
+ continue
555
+ fit_candidates.append(current_candidate)
556
+
557
+ if not fit_candidates:
558
+ details = f" Last error: {last_fit_error!r}" if last_fit_error else ""
559
+ raise ValueError(f"IAML could not fit any candidate pipeline.{details}")
560
+
561
+ self.chosen_candidate = fit_candidates[0]
562
+ self.last_stage_candidates = candidates
563
+
564
+ return fit_candidates
565
+ except TerminatedError:
566
+ Logger().info('IAML was terminated.')
567
+ except Exception as ex:
568
+ Logger().error('Error during fit')
569
+ raise ex
570
+ finally:
571
+ self.executor.shutdown()
572
+
573
+ @property
574
+ def chosen_model(self) -> 'IAMLPipeline':
575
+ """Return the best model trained with fit
576
+
577
+ :return: Best predictor pipeline
578
+ """
579
+ if not self.chosen_candidate:
580
+ return None
581
+
582
+ return self.chosen_candidate.pipeline
583
+
584
+ def visualize_descriptive_statistics(self) -> list[StatisticPlot]:
585
+ """Return a list of plots that show descriptive statistics."""
586
+ if self.descriptive_statistics is None and self._last_dataset is not None:
587
+ self.__ensure_descriptive_statistics(self._last_dataset)
588
+
589
+ if self.descriptive_statistics is None or self.descriptive_statistics.empty:
590
+ return []
591
+
592
+ plots: list[StatisticPlot] = []
593
+ stats_df = self.descriptive_statistics
594
+
595
+ def group_columns_by_feature(dataframe: pd.DataFrame) -> dict[str, list[str]]:
596
+ columns = list(dataframe.columns)
597
+ base_names: list[str] = []
598
+ for col in columns:
599
+ if isinstance(col, str) and col.endswith('_all'):
600
+ base = col[:-4]
601
+ if base not in base_names:
602
+ base_names.append(base)
603
+
604
+ groups: dict[str, list[str]] = {}
605
+ used_cols: set[str] = set()
606
+ if base_names:
607
+ for base in base_names:
608
+ group = [
609
+ col for col in columns
610
+ if col == base or (isinstance(col, str) and col.startswith(f"{base}_"))
611
+ ]
612
+ groups[base] = group
613
+ used_cols.update(group)
614
+
615
+ for col in columns:
616
+ if col not in used_cols:
617
+ groups[col] = [col]
618
+
619
+ return groups
620
+
621
+ for plot_sub_class in StatisticPlot.__subclasses__():
622
+ if not getattr(plot_sub_class, 'enabled', True):
623
+ continue
624
+ if getattr(plot_sub_class, 'group_by_feature', False):
625
+ for base, cols in group_columns_by_feature(stats_df).items():
626
+ plot = plot_sub_class().compute(stats_df[cols], base_name=base)
627
+ plots.append(plot)
628
+ else:
629
+ plots.append(plot_sub_class().compute(stats_df))
630
+
631
+ return plots
632
+
633
+ def get_descriptive_statistics(self) -> pd.DataFrame:
634
+ """Return descriptive statistics, computing them on demand if needed."""
635
+ if self.descriptive_statistics is None and self._last_dataset is not None:
636
+ self.__ensure_descriptive_statistics(self._last_dataset)
637
+
638
+ return self.descriptive_statistics if self.descriptive_statistics is not None else pd.DataFrame()
639
+
640
+ def __ensure_descriptive_statistics(self, dataset: Dataset) -> None:
641
+ """Compute descriptive statistics once, for on-demand usage."""
642
+ if self.descriptive_statistics is not None:
643
+ return
644
+
645
+ try:
646
+ self.descriptive_statistics = self.__compute_descriptive_statistics(dataset)
647
+ except Exception as exc: # noqa: BLE001
648
+ Logger().warning(f"Descriptive statistics computation failed: {exc}")
649
+ self.descriptive_statistics = pd.DataFrame()
650
+
651
+ def __compute_descriptive_statistics(self, dataset: Dataset) -> pd.DataFrame:
652
+ """Compute descriptive statistics on the dataset."""
653
+ computed_statistics = pd.DataFrame()
654
+ for statistic_sub_class in Statistic.all_subclasses():
655
+ statistic = statistic_sub_class()
656
+ if statistic.suitable(dataset):
657
+ result = statistic.compute(dataset)
658
+ if result is not None and not result.empty:
659
+ computed_statistics = pd.concat([computed_statistics, result])
660
+ return computed_statistics
661
+
662
+ def check_pipeline(self) -> None:
663
+ """Raise Exception if pipeline is not valid
664
+
665
+ :raise AttributeError: Step Pipeline must contains at least one predictor
666
+ """
667
+ steps = self.__all_steps()
668
+
669
+ # Pipeline must have at least one predictor
670
+ if not any((Predictor in s.__class__.__mro__) for s in steps if s.enable):
671
+ raise AttributeError('Step Pipeline must contains at least one predictor')
672
+
673
+ # TODO Others tests ?
674
+
675
+ def __run_evaluations(
676
+ self,
677
+ candidates: list[Candidate],
678
+ dataset: Dataset,
679
+ timeout: float = None,
680
+ stage_number: int = None,
681
+ callback: callable = None,
682
+ stage_timeout: float = None) -> list[Candidate]:
683
+ """Evaluate candidates
684
+
685
+ :param list[Candidate] candidates: Candidates to evaluate.
686
+ :param Dataset dataset: Dataset used for evaluation.
687
+ :param float, optional timeout: Budget including preparation and submission.
688
+ Defaults to the stage duration limit.
689
+ :param int, optional stage_number: Stage number running. Default to None.
690
+ :param callable, optional callback: Method called after evaluation. Default to None.
691
+ :param float, optional stage_timeout: Additional cap for this evaluation only;
692
+ the callback still reports the remaining ``timeout`` budget.
693
+ """
694
+ start_time = time.monotonic()
695
+ budget = self.max_stage_duration if timeout is None else min(timeout, self.max_stage_duration)
696
+ if stage_timeout is not None:
697
+ budget = min(budget, stage_timeout)
698
+ deadline = start_time + max(0.0, budget)
699
+ new_candidates: list[Candidate] = []
700
+ splitter_fingerprint = hash_evaluation_context(self.splitter)
701
+ dataset_key = dataset.fingerprint() if splitter_fingerprint is not None else None
702
+ evaluation_cache_keys: set[str] = set()
703
+ with Logger().progress as progress:
704
+ task = progress.add_task(
705
+ f'Stage {stage_number}' if stage_number is not None else "Initial evaluation",
706
+ total=len(candidates))
707
+
708
+ def update_progressbar(*args): # pylint: disable=unused-argument
709
+ progress.update(task, advance=1)
710
+
711
+ self.executor.set_callback(update_progressbar)
712
+ for candidate in candidates:
713
+ if time.monotonic() >= deadline:
714
+ break
715
+ cache_key = self.__evaluation_cache_key(candidate, splitter_fingerprint)
716
+ if cache_key is not None:
717
+ evaluation_cache_keys.add(cache_key)
718
+ from_cache = Cache().from_cache(cache_key, dataset_key) if cache_key else None
719
+
720
+ if from_cache:
721
+ self.__hydrate_cached_candidate(candidate, from_cache, dataset)
722
+ new_candidates.append(candidate)
723
+ update_progressbar() # Update progressbar even if data come from cache
724
+ else:
725
+ submitted = self.executor.submit(
726
+ process_executor,
727
+ candidate,
728
+ dataset,
729
+ deadline=deadline,
730
+ splitter=self.splitter,
731
+ store_audit=self.keep_training_history,
732
+ )
733
+ if not submitted:
734
+ break
735
+
736
+ # Preparation and submission have already consumed part of the budget.
737
+ new_candidates += self.executor.join(max(0.0, deadline - time.monotonic()))
738
+ self.__collect_training_history(new_candidates)
739
+
740
+ if new_candidates:
741
+ skipped = sum(1 for candidate in new_candidates if not candidate.computed_metrics)
742
+ if skipped:
743
+ Logger().warning(
744
+ f"Skipped {skipped} candidates with no computed metrics."
745
+ )
746
+ new_candidates = [candidate for candidate in new_candidates if candidate.computed_metrics]
747
+
748
+ new_candidates.sort(reverse=True)
749
+
750
+ # Add results to progressbar
751
+ if new_candidates:
752
+ progress.tasks[task].description = f'{progress.tasks[task].description} \
753
+ ({new_candidates[0].get_main_metric_value():.4f})'
754
+ else:
755
+ progress.tasks[task].description = f'{progress.tasks[task].description} \
756
+ (no result)'
757
+
758
+ # Add to cache
759
+ for candidate in new_candidates:
760
+ cache_key = self.__evaluation_cache_key(candidate, splitter_fingerprint)
761
+ if cache_key in evaluation_cache_keys and not Cache().from_cache(cache_key, dataset_key):
762
+ Cache().add_to_cache(
763
+ cache_key,
764
+ dataset_key,
765
+ self.__build_cached_candidate(candidate),
766
+ )
767
+
768
+ best_metric = new_candidates[0].get_main_metric_value() if new_candidates else None
769
+ remaining_time = (budget if timeout is None else timeout) - (time.monotonic() - start_time)
770
+
771
+ self.__callback(callback, # pylint: disable=too-many-function-args
772
+ generation = stage_number,
773
+ generation_size = len(new_candidates),
774
+ best = best_metric,
775
+ remaining_time = remaining_time,
776
+ text = f'Stage {stage_number} finished' \
777
+ if stage_number is not None else "Initial evaluation finished")
778
+
779
+ return new_candidates
780
+
781
+ def __evaluation_cache_key(
782
+ self, candidate: Candidate, splitter_fingerprint: str | None
783
+ ) -> str | None:
784
+ """Keep scores separate for each splitter and metric configuration."""
785
+ if splitter_fingerprint is None:
786
+ return None
787
+ context = hash_evaluation_context(
788
+ candidate.pipeline.fingerprint(), candidate.metrics, candidate.main_metric,
789
+ self.keep_training_history,
790
+ )
791
+ if context is None:
792
+ return None
793
+ return f"IAML_{splitter_fingerprint}_{context}"
794
+
795
+ def __build_optimizer(self, duration: float) -> Optimizer:
796
+ """Instantiate optimizer."""
797
+ return self.optimizer(duration=None if math.isinf(duration) else duration)
798
+
799
+ def __build_cached_candidate(self, candidate: Candidate) -> dict[str, Any] | dict[str, float]:
800
+ """Build the payload stored in cache for evaluated candidates."""
801
+ if self.keep_training_history and candidate.training_audit is not None:
802
+ return {
803
+ "__computed_metrics__": deepcopy(candidate.computed_metrics),
804
+ "__training_audit__": deepcopy(candidate.training_audit),
805
+ }
806
+ return deepcopy(candidate.computed_metrics)
807
+
808
+ def __hydrate_cached_candidate(
809
+ self,
810
+ candidate: Candidate,
811
+ payload: dict[str, Any] | dict[str, float],
812
+ dataset: Dataset,
813
+ ) -> None:
814
+ """Restore cached evaluation results into a candidate."""
815
+ candidate.fold_metrics = []
816
+ candidate.training_audit = None
817
+
818
+ if isinstance(payload, dict) and "__computed_metrics__" in payload:
819
+ candidate.computed_metrics = deepcopy(payload["__computed_metrics__"])
820
+ audit = payload.get("__training_audit__")
821
+ if audit is not None:
822
+ candidate.training_audit = deepcopy(audit)
823
+ candidate.fold_metrics = deepcopy(audit.get("fold_metrics", []))
824
+ return
825
+ else:
826
+ candidate.computed_metrics = deepcopy(payload)
827
+
828
+ if self.keep_training_history:
829
+ candidate.training_audit = candidate.build_training_audit(
830
+ dataset=dataset,
831
+ fold_metrics=[],
832
+ aggregated_metrics=candidate.computed_metrics,
833
+ status="success",
834
+ )
835
+
836
+ def __collect_training_history(self, candidates: list[Candidate]) -> None:
837
+ """Collect unique candidate audit records for the last fit."""
838
+ if not self.keep_training_history:
839
+ return
840
+
841
+ for candidate in candidates:
842
+ record = getattr(candidate, "training_audit", None)
843
+ if not record:
844
+ continue
845
+ key = (
846
+ record.get("pipeline_fingerprint"),
847
+ record.get("dataset_fingerprint"),
848
+ record.get("status"),
849
+ record.get("error"),
850
+ )
851
+ if key in self._training_history_seen:
852
+ continue
853
+ self._training_history_seen.add(key)
854
+ self.training_history.append(deepcopy(record))
855
+
856
+ def __optimize( # pylint: disable=too-many-arguments
857
+ self,
858
+ dataset: Dataset,
859
+ candidates: list[Candidate],
860
+ optimizer: Optimizer = Optimizer(),
861
+ patience: int = 5,
862
+ max_duration: int = -1,
863
+ callback: callable = None) -> list[Candidate]:
864
+ """Optimize candidates.
865
+
866
+ :param Dataset dataset: Dataset used for optimization.
867
+ :param list[Candidate], optional candidates: Candidates to optimize.
868
+ :param Optimizer, optional optimizer: Optimizer to use.
869
+ :param int, optional patience: Max generation without improvement. Default to 5.
870
+ :param int, optional max_duration: Maximum optimization duration. Default to -1.
871
+ :param callable, optional callback: Method to call after optimization. Default to None.
872
+ :return: list of optimized candidate.
873
+ """
874
+ if not candidates:
875
+ return []
876
+
877
+ if max_duration == -1:
878
+ max_duration = math.inf
879
+
880
+ # init
881
+ candidates.sort(reverse=True)
882
+ best_result: float = candidates[0].get_main_metric_value()
883
+ best_score: float = candidates[0].get_main_metric_score()
884
+ iterations_without_improvement: int = 0
885
+ iterations_count: int = 0
886
+ duration: int = 0
887
+ starting_time: int = time.monotonic() # seconds
888
+
889
+ # If there is not, define an arbitrary stop condition
890
+ if math.isinf(max_duration) and patience == -1:
891
+ Logger().warning('You have not defined any stop condition. \
892
+ Patient has arbitrary set to 20')
893
+ patience = 20
894
+
895
+ previous_candidates = candidates
896
+
897
+ while not(optimizer.finished) \
898
+ and (patience == -1 or iterations_without_improvement < patience) \
899
+ and max_duration > duration:
900
+ # Generate new candidates
901
+ generated_candidates = optimizer.run(previous_candidates)
902
+
903
+ Logger().info(f'Finetuning... \
904
+ stage={iterations_count} \
905
+ candidates={len(generated_candidates)} \
906
+ patience={iterations_without_improvement}/{patience}, \
907
+ duration={round(duration, 2)}/{max_duration}, \
908
+ best_result={best_result}')
909
+
910
+ # Evaluate new candidates
911
+ evaluated_candidates = self.__run_evaluations(generated_candidates,
912
+ dataset,
913
+ timeout=max_duration - (time.monotonic() - starting_time),
914
+ stage_number=iterations_count,
915
+ callback=callback)
916
+
917
+ # Remove not computed (error or timeout)
918
+ evaluated_candidates = [candidate for candidate in evaluated_candidates if candidate.computed_metrics]
919
+
920
+ if not evaluated_candidates:
921
+ Logger().warning('No candidates produced a valid evaluation; keeping previous best candidates.')
922
+ candidates = previous_candidates
923
+ break
924
+
925
+ candidates = evaluated_candidates
926
+ previous_candidates = candidates
927
+
928
+ # Improvement ?
929
+ new_score: float = candidates[0].get_main_metric_score()
930
+ if new_score > best_score:
931
+ best_score = new_score
932
+ best_result = candidates[0].get_main_metric_value()
933
+ iterations_without_improvement = 0
934
+ else:
935
+ iterations_without_improvement += 1
936
+
937
+ # Duration in seconds
938
+ duration = time.monotonic() - starting_time
939
+
940
+ # Increase Iteration count
941
+ iterations_count += 1
942
+
943
+ return candidates
944
+
945
+ def __metrics_selection(
946
+ self,
947
+ X: pd.DataFrame,
948
+ y: pd.DataFrame,
949
+ type_of_target: str) -> list[Metric]:
950
+ """Select metrics used to evaluate performances
951
+
952
+ :param pd.DataFrame X: Training features.
953
+ :param pd.DataFrame y: Training labels
954
+ :param str type_of_target: type of label. Example : continuous, binary
955
+ :return: List of selected metrics
956
+ """
957
+ metrics = []
958
+
959
+ for metric_sub_class in Metric.all_subclasses():
960
+ # Reuse the configured main metric instead of resetting its parameters.
961
+ metric = (
962
+ self.main_metric
963
+ if type(self.main_metric) is metric_sub_class
964
+ else metric_sub_class()
965
+ )
966
+ # Verify if a subclass is suitable or not
967
+ if metric is self.main_metric or metric.suitable(X, y, type_of_target):
968
+ metrics.append(metric)
969
+ return metrics
970
+
971
+ def __apply_minimal_preprocessing(self, candidate: Candidate) -> Candidate:
972
+ """Ensure minimalist candidates have a basic imputer in their pipeline."""
973
+ dataset = candidate.dataset
974
+
975
+ if dataset.X.isna().values.any():
976
+ imputer = ActSimpleImputer()
977
+ results = imputer.run(candidate)
978
+
979
+ if isinstance(results, list) and results:
980
+ return results[0]
981
+
982
+ return results
983
+
984
+ return candidate
985
+
986
+ def __generate_minimal_candidates(self, candidate: Candidate) -> list[Candidate]:
987
+ """Generate minimalist candidates using raw predictors only."""
988
+ if not self.minimal_predictor_step:
989
+ return []
990
+
991
+ Logger().info('Generate minimalist candidates...')
992
+ minimal_candidate = candidate.to_input()
993
+ minimal_candidate = self.__apply_minimal_preprocessing(minimal_candidate)
994
+
995
+ candidates = self.minimal_predictor_step.run(minimal_candidate)
996
+
997
+ return [cand for cand in candidates if cand.pipeline.predictor is not None]
998
+
999
+ def __run(self, candidate: Candidate) -> list[Candidate]:
1000
+ """Run pipeline steps
1001
+
1002
+ :param Candidate candidate: Data used to fit models and steps.
1003
+ :return: List of all the generated candidates. Sorted by performances.
1004
+ """
1005
+ Logger().info("Generate candidate...")
1006
+ self.candidates = self.first_step.run(candidate)
1007
+
1008
+ return self.candidates
1009
+
1010
+ ########################
1011
+ #### CONFIGURATIONS ####
1012
+ ########################
1013
+
1014
+ def json_pipeline(self) -> dict:
1015
+ """Create a dictionary (JSON) from the loaded Pipeline
1016
+
1017
+ :return: Loaded Pipeline in a JSON format
1018
+ """
1019
+ return self.first_step.json_pipeline()
1020
+
1021
+ def all_configurations(self) -> list[dict]:
1022
+ """Return a dict with configurations of all steps.
1023
+
1024
+ :return: Configurations of all steps.
1025
+ """
1026
+ return self.first_step.all_configurations()
1027
+
1028
+ def configure_all(self, configs: dict) -> None:
1029
+ """Configure one to many steps with a dict configuration
1030
+
1031
+ :param dict configs: key is a step_id and value is the configuration to set.
1032
+ """
1033
+ all_steps = self.__all_steps()
1034
+
1035
+ for step_id, config in configs.items():
1036
+ current_step: Step = self.__find_step_by_id(all_steps, step_id)
1037
+ if current_step:
1038
+ for key, value in config:
1039
+ current_step.configure(key, value) # pylint: disable=no-member
1040
+
1041
+ def __all_steps(self) -> list[Step]:
1042
+ """Recursive method. Return all the pipeline's steps in a list
1043
+
1044
+ :return: All flatten pipelines's steps
1045
+ """
1046
+ return self.first_step.all_steps()
1047
+
1048
+ def __find_step_by_id(self, step_list: list[Step], step_id: int) -> Step | None:
1049
+ """Find a step by id in a list of step
1050
+
1051
+ :param list[Step] step_list: The step will be searched in this list.
1052
+ :param int step_id: Identifier of the step
1053
+ :return: Found Step or None
1054
+ """
1055
+ for step in step_list:
1056
+ if id(step) == step_id:
1057
+ return step
1058
+ return None
1059
+
1060
+
1061
+ def process_executor(candidate: Candidate, *args, **kwargs) -> 'Candidate':
1062
+ """Wrap candidate training to run it in subprocess
1063
+
1064
+ :param Candidate candidate: Not trained candidate.
1065
+ :param tuple, optional \\*args: Additional parameters.
1066
+ :param dict, optional \\**kwargs: Additional parameters.
1067
+ :return: Trained candidate.
1068
+ """
1069
+ # Deepcopy -> Without it, process end is never detected. Strange...
1070
+ candidate = deepcopy(candidate)
1071
+ candidate.training_evaluate(*args, **kwargs)
1072
+ return candidate