PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
iaml/iaml.py
ADDED
|
@@ -0,0 +1,1072 @@
|
|
|
1
|
+
"""Integrated AutoML for Medical Labs (IAML).
|
|
2
|
+
|
|
3
|
+
Search and evaluate modular prediction pipelines for clinical research with
|
|
4
|
+
tabular data. Trained candidates expose their pipeline steps, evaluation metrics
|
|
5
|
+
and explanation methods for inspection and study reporting.
|
|
6
|
+
"""
|
|
7
|
+
from copy import deepcopy
|
|
8
|
+
import time
|
|
9
|
+
import math
|
|
10
|
+
import textwrap
|
|
11
|
+
import multiprocessing
|
|
12
|
+
from typing import TYPE_CHECKING, Any
|
|
13
|
+
import numpy as np
|
|
14
|
+
import pandas as pd
|
|
15
|
+
from .timed_pool_executor import TimedPoolExecutor, TerminatedError
|
|
16
|
+
from .step import Step
|
|
17
|
+
from .cache import Cache
|
|
18
|
+
from .cache_keys import hash_evaluation_context
|
|
19
|
+
from .metastep import MetaStep
|
|
20
|
+
from .candidate import Candidate
|
|
21
|
+
from .dataset import Dataset
|
|
22
|
+
from .metric import Metric
|
|
23
|
+
from .statistic import Statistic
|
|
24
|
+
from .worker_manager import WorkerManager
|
|
25
|
+
from .splitters import kfold_splitter
|
|
26
|
+
from .meta_ordered_step import MetaOrderedStep
|
|
27
|
+
from .meta_explorer_step import MetaExplorerStep
|
|
28
|
+
from .meta_partial_explorer_step import MetaPartialExplorerStep
|
|
29
|
+
from .optimizers import Optimizer, GeneticOptimizer, RandomOptimizer, BayesianOptimizer
|
|
30
|
+
from .predictor import Predictor
|
|
31
|
+
from .logger import Logger
|
|
32
|
+
from .plot import StatisticPlot
|
|
33
|
+
from .actionables.cleaning.act_simple_imputer import ActSimpleImputer
|
|
34
|
+
from .actionables.normalize.act_standard_scaler import ActStandardScaler
|
|
35
|
+
from .sklearn_preprocessor import SklearnPreprocessor
|
|
36
|
+
|
|
37
|
+
# Default Actionables -> Must be a wildcard import to help IAML to know all available the steps
|
|
38
|
+
from .actionables import * # pylint: disable=unused-wildcard-import,wildcard-import
|
|
39
|
+
|
|
40
|
+
# Default Wrappers -> Must be a wildcard import to help IAML to know all available the steps
|
|
41
|
+
from .wrapper import * # pylint: disable=unused-wildcard-import,wildcard-import
|
|
42
|
+
|
|
43
|
+
# cuDF pandas acceleration
|
|
44
|
+
try:
|
|
45
|
+
import cudf.pandas
|
|
46
|
+
cudf.pandas.install()
|
|
47
|
+
Logger().info('cuDF is installed: using cuDF pandas accelerator mode.')
|
|
48
|
+
except ImportError as e:
|
|
49
|
+
Logger().warning('cuDF not found: falling back to standalone pandas.')
|
|
50
|
+
|
|
51
|
+
if TYPE_CHECKING:
|
|
52
|
+
from .iaml_pipeline import IAMLPipeline
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class IAML: # pylint: disable=too-many-instance-attributes
|
|
56
|
+
"""Configure and search prediction pipelines for a clinical research dataset.
|
|
57
|
+
|
|
58
|
+
:meth:`fit` returns trained candidates for evaluation, pipeline inspection
|
|
59
|
+
and explanation of predictions.
|
|
60
|
+
|
|
61
|
+
:param int, optional max_workers: Maximum parallel workers. Default to cpu count.
|
|
62
|
+
:param int, optional max_stage_duration: Maximum duration of a stage. Default to None.
|
|
63
|
+
:param callable, optional splitter: Split function to use. Default to kfold_splitter.
|
|
64
|
+
:param int, optional max_duration: Search time budget; -1 means no global limit.
|
|
65
|
+
:param int | str, optional time_before_sample_use: Time before we use sampled data.
|
|
66
|
+
Default to None.
|
|
67
|
+
:param bool, optional preprocessor: Use preprocessor. Default to False.
|
|
68
|
+
:param Metric, optional main_metric: Main metric instance, preserving its parameters.
|
|
69
|
+
Default to None.
|
|
70
|
+
:param Optimizer, optional optimizer: Optimizer class to use. Default to GeneticOptimizer.
|
|
71
|
+
:param int, optional train_on_n_samples: Limit the initial search dataset to this many
|
|
72
|
+
rows. None or nonpositive values use all rows.
|
|
73
|
+
:param bool, optional keep_training_history: If True, store detailed CV audit records for
|
|
74
|
+
every evaluated pipeline. Default to False.
|
|
75
|
+
:param bool, optional refit_on_sample: Reuse the initial train_on_n_samples sample for
|
|
76
|
+
final fitting. If False, refit on all input rows. Default to True; has no effect
|
|
77
|
+
without a positive train_on_n_samples limit.
|
|
78
|
+
:param initial_preprocessor: Optional clonable sklearn transformer. It must return
|
|
79
|
+
a numeric DataFrame with unchanged rows and index. Every generated pipeline,
|
|
80
|
+
including minimalist candidates, starts with this mandatory transformer.
|
|
81
|
+
It is fitted afresh within each CV training fold and during final fitting.
|
|
82
|
+
"""
|
|
83
|
+
def __init__( # pylint: disable=too-many-arguments
|
|
84
|
+
self,
|
|
85
|
+
max_workers: int = None,
|
|
86
|
+
max_stage_duration: int = None,
|
|
87
|
+
splitter: callable = None,
|
|
88
|
+
max_duration: int = -1,
|
|
89
|
+
time_before_sample_use: int | str = None,
|
|
90
|
+
preprocessor: bool = False,
|
|
91
|
+
main_metric: Metric = None,
|
|
92
|
+
optimizer: Optimizer = GeneticOptimizer,
|
|
93
|
+
train_on_n_samples: int = None,
|
|
94
|
+
keep_training_history: bool = False,
|
|
95
|
+
refit_on_sample: bool = True,
|
|
96
|
+
initial_preprocessor: Any = None) -> None:
|
|
97
|
+
# Set pandas config to avoid SettingsWithcopyWarning
|
|
98
|
+
pd.options.mode.copy_on_write = True
|
|
99
|
+
|
|
100
|
+
self.preprocessor: bool = preprocessor
|
|
101
|
+
"""Enable / Disable preprocessor"""
|
|
102
|
+
|
|
103
|
+
self.optimizer = optimizer
|
|
104
|
+
"""Choose Optimizer"""
|
|
105
|
+
|
|
106
|
+
self.train_on_n_samples = train_on_n_samples
|
|
107
|
+
"""If defined, pick n sample in the dataset before train"""
|
|
108
|
+
|
|
109
|
+
self.refit_on_sample: bool = refit_on_sample
|
|
110
|
+
"""Apply the explicit search sample limit to final fitting as well."""
|
|
111
|
+
|
|
112
|
+
self.initial_preprocessor = initial_preprocessor
|
|
113
|
+
"""Unfitted transformer template, prepended to all candidate pipelines."""
|
|
114
|
+
|
|
115
|
+
self.keep_training_history: bool = keep_training_history
|
|
116
|
+
"""Whether to store detailed cross-validation audit records."""
|
|
117
|
+
|
|
118
|
+
self.training_history: list[dict[str, Any]] = []
|
|
119
|
+
"""Detailed audit records for evaluated pipelines during the last fit."""
|
|
120
|
+
|
|
121
|
+
self._training_history_seen: set[tuple[Any, ...]] = set()
|
|
122
|
+
"""Deduplicate audit records across warmup, cache hits, and repeated evaluations."""
|
|
123
|
+
|
|
124
|
+
# Set max duration of each stage
|
|
125
|
+
if max_stage_duration is None:
|
|
126
|
+
self.max_stage_duration = max(max_duration / 5, 900)
|
|
127
|
+
Logger().warning(
|
|
128
|
+
f"Max duration of each stage was set to {self.max_stage_duration} seconds")
|
|
129
|
+
else:
|
|
130
|
+
self.max_stage_duration = max_stage_duration
|
|
131
|
+
|
|
132
|
+
self.splitter: callable = splitter if splitter is not None else kfold_splitter
|
|
133
|
+
"""Splitter callable"""
|
|
134
|
+
|
|
135
|
+
self.main_metric: Metric = main_metric
|
|
136
|
+
"""Main metric"""
|
|
137
|
+
|
|
138
|
+
self.max_duration: int = max_duration
|
|
139
|
+
"""Maximum training duration"""
|
|
140
|
+
|
|
141
|
+
if time_before_sample_use == 'auto' and max_duration:
|
|
142
|
+
self.time_before_sample_use = max(max_duration / 5, 60)
|
|
143
|
+
elif time_before_sample_use:
|
|
144
|
+
self.time_before_sample_use = time_before_sample_use
|
|
145
|
+
else:
|
|
146
|
+
self.time_before_sample_use = math.inf
|
|
147
|
+
|
|
148
|
+
self.candidates: list[Candidate] = None
|
|
149
|
+
"""list of Candidates for this training"""
|
|
150
|
+
|
|
151
|
+
self.init_candidate: Candidate = None
|
|
152
|
+
"""Initial candidate"""
|
|
153
|
+
|
|
154
|
+
self.first_step: Step = None # Will be the first Step of the pipeline (probably a MetaStep
|
|
155
|
+
"""Hold the first step of the pipeline"""
|
|
156
|
+
|
|
157
|
+
self.last_stage_candidates: list[Candidate] = []
|
|
158
|
+
"""Hold last generated candidates"""
|
|
159
|
+
|
|
160
|
+
self.executor: TimedPoolExecutor = None
|
|
161
|
+
"""Hold TimePoolExecutor"""
|
|
162
|
+
|
|
163
|
+
self.default_pipeline() # Load default pipeline
|
|
164
|
+
self.max_workers = max_workers if (max_workers is not None and max_workers > 0) \
|
|
165
|
+
else multiprocessing.cpu_count()
|
|
166
|
+
"""Hold maximum number of parallel workers"""
|
|
167
|
+
|
|
168
|
+
self.chosen_candidate: Candidate = None
|
|
169
|
+
"""Hold the best candidate"""
|
|
170
|
+
WorkerManager(max_workers=self.max_workers)
|
|
171
|
+
|
|
172
|
+
self.descriptive_statistics: pd.DataFrame | None = None
|
|
173
|
+
"""Cached descriptive statistics for the last fitted dataset."""
|
|
174
|
+
|
|
175
|
+
self._last_dataset: Dataset | None = None
|
|
176
|
+
"""Dataset used for the most recent fit, for on-demand statistics."""
|
|
177
|
+
|
|
178
|
+
def __del__(self):
|
|
179
|
+
"""Delete the TimedPoolExecutor"""
|
|
180
|
+
del self.executor
|
|
181
|
+
|
|
182
|
+
def load_pipeline(self, pipeline: dict) -> None:
|
|
183
|
+
"""Load any kind of pipeline
|
|
184
|
+
|
|
185
|
+
:param dict pipeline: JSON description of the pipeline
|
|
186
|
+
"""
|
|
187
|
+
self.first_step = Step.from_pipeline(pipeline)
|
|
188
|
+
self.minimal_predictor_step = self.__build_minimal_predictor_step()
|
|
189
|
+
|
|
190
|
+
def default_pipeline(self, fast: bool = False) -> None:
|
|
191
|
+
"""Load the default pipeline.
|
|
192
|
+
Default pipeline is the recommended way to create classifier and regressor
|
|
193
|
+
|
|
194
|
+
Genetic search starts with one normalization and no resampling, then
|
|
195
|
+
explores alternatives through mutations. Other optimizers retain full
|
|
196
|
+
initial exploration because they only change hyperparameters.
|
|
197
|
+
|
|
198
|
+
:param bool, optional fast: If true, will only load fast machine learning model.
|
|
199
|
+
Fast mode is use to create fast pipeline and iterate
|
|
200
|
+
quickly when debugging code. Defaults to False.
|
|
201
|
+
"""
|
|
202
|
+
self.first_step = MetaOrderedStep(tag="Main") # First step -> Contain all pipeline's stages
|
|
203
|
+
|
|
204
|
+
self.first_step.add_step(MetaStep(tag='features_precleaning',
|
|
205
|
+
name='Features Precleaning',
|
|
206
|
+
description=textwrap.dedent('''\
|
|
207
|
+
Converts complex columns into several columns, which helps the
|
|
208
|
+
model to extract information from your data.''')))
|
|
209
|
+
self.first_step.add_step(MetaStep(tag='cleaning',
|
|
210
|
+
name='Features Cleaning',
|
|
211
|
+
description=textwrap.dedent('''\
|
|
212
|
+
Improve data quality, handle missing values, extract
|
|
213
|
+
information from textual columns, etc.''')))
|
|
214
|
+
self.first_step.add_step(MetaStep(tag='features_selection',
|
|
215
|
+
name='Features Selection',
|
|
216
|
+
description=textwrap.dedent('''\
|
|
217
|
+
Decrease number of column to improve the models' performance.''')))
|
|
218
|
+
partial_exploration = (isinstance(self.optimizer, type)
|
|
219
|
+
and issubclass(self.optimizer, GeneticOptimizer))
|
|
220
|
+
explorer = MetaPartialExplorerStep if partial_exploration else MetaExplorerStep
|
|
221
|
+
normalization_options = {'initial_step': ActStandardScaler()} if partial_exploration else {}
|
|
222
|
+
imbalance_options = {} if partial_exploration else {'also_explore_without': True}
|
|
223
|
+
self.first_step.add_step(explorer(tag='normalize',
|
|
224
|
+
**normalization_options,
|
|
225
|
+
name='Features Normalization',
|
|
226
|
+
description=textwrap.dedent('''\
|
|
227
|
+
Normalize data to help model to give the same interest to each column''')))
|
|
228
|
+
self.first_step.add_step(explorer(tag='imbalance',
|
|
229
|
+
**imbalance_options,
|
|
230
|
+
name='Handle Imbalanced Data',
|
|
231
|
+
description=textwrap.dedent('''\
|
|
232
|
+
Balance the dataset to ensure the model does not favor the
|
|
233
|
+
majority class over the minority class''')))
|
|
234
|
+
|
|
235
|
+
if self.preprocessor:
|
|
236
|
+
self.first_step.add_step(
|
|
237
|
+
MetaExplorerStep(tag='features_preprocessing', also_explore_without=True)
|
|
238
|
+
)
|
|
239
|
+
else:
|
|
240
|
+
self.first_step.add_step(
|
|
241
|
+
MetaPartialExplorerStep(
|
|
242
|
+
tag='features_preprocessing',
|
|
243
|
+
name="Dimensionality Reduction (optional)",
|
|
244
|
+
description=textwrap.dedent('''\
|
|
245
|
+
Reduce the complexity of data and make computations
|
|
246
|
+
more efficient'''))
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
learning_tag = 'fast_predictor' if fast else 'predictor'
|
|
250
|
+
|
|
251
|
+
self.first_step.add_step(
|
|
252
|
+
MetaExplorerStep(
|
|
253
|
+
tag=learning_tag,
|
|
254
|
+
name="Machine learning models",
|
|
255
|
+
description="List of machine learning models IAML will try to optimize"))
|
|
256
|
+
|
|
257
|
+
self.minimal_predictor_step = self.__build_minimal_predictor_step()
|
|
258
|
+
|
|
259
|
+
def __build_minimal_predictor_step(self) -> MetaExplorerStep | None:
|
|
260
|
+
"""Build the minimalist predictor stage if suitable models exist."""
|
|
261
|
+
minimal_step = MetaExplorerStep(
|
|
262
|
+
tag='minimal_predictor',
|
|
263
|
+
name='Minimalist Predictors',
|
|
264
|
+
description=textwrap.dedent('''\
|
|
265
|
+
Try high-performing boosting-style models without any preprocessing
|
|
266
|
+
to provide quick baseline candidates before the full pipeline is explored.'''))
|
|
267
|
+
|
|
268
|
+
if not minimal_step.steps:
|
|
269
|
+
return None
|
|
270
|
+
|
|
271
|
+
return minimal_step
|
|
272
|
+
|
|
273
|
+
def __callback(self, callback: callable, **kwargs: dict) -> None:
|
|
274
|
+
"""Call callback function if defined
|
|
275
|
+
|
|
276
|
+
:param callable callback: Function to call.
|
|
277
|
+
:param dict, optional \\**kwargs: Additional parameters.
|
|
278
|
+
"""
|
|
279
|
+
if callback and callable(callback):
|
|
280
|
+
callback(**kwargs)
|
|
281
|
+
|
|
282
|
+
def baseline( # pylint: disable=too-many-arguments
|
|
283
|
+
self,
|
|
284
|
+
X: pd.DataFrame,
|
|
285
|
+
y: pd.DataFrame,
|
|
286
|
+
groups: pd.DataFrame = None,
|
|
287
|
+
groups_columns: list[str] = None,
|
|
288
|
+
generation_sample_size: int = 200,
|
|
289
|
+
verbose: int = 1) -> Candidate:
|
|
290
|
+
"""Run a very basic pipeline to train a model baseline
|
|
291
|
+
|
|
292
|
+
:param pd.DataFrame X: Training features
|
|
293
|
+
:param pd.DataFrame y: Training labels
|
|
294
|
+
:param pd.DataFrame, optional groups: Dataframe used to split data by groups.
|
|
295
|
+
Default to None.
|
|
296
|
+
:param list[str], optional groups_columns: List of column names used to split data by
|
|
297
|
+
groups. Default to None.
|
|
298
|
+
:param int, optional generation_sample_size: Size of the sample dataset used to generate
|
|
299
|
+
first generation of candidates (default 200).
|
|
300
|
+
:param int, optional verbose: Verbosity level. Default to 1.
|
|
301
|
+
:return: Baseline candidate
|
|
302
|
+
"""
|
|
303
|
+
# Avoid [] dangerous default value in the signature
|
|
304
|
+
if groups_columns is None:
|
|
305
|
+
groups_columns = []
|
|
306
|
+
|
|
307
|
+
# Create a baseline pipeline
|
|
308
|
+
baseline_pipe = MetaOrderedStep(tag="Main") # First step -> Contain all pipeline's stages
|
|
309
|
+
baseline_pipe.add_step(MetaStep(tag='baseline_cleaning'))
|
|
310
|
+
baseline_pipe.add_step(MetaExplorerStep(tag='baseline_predictor'))
|
|
311
|
+
|
|
312
|
+
Logger().verbose = verbose # Set logger verbose
|
|
313
|
+
|
|
314
|
+
dataset:Dataset = Dataset(
|
|
315
|
+
deepcopy(X),
|
|
316
|
+
deepcopy(y),
|
|
317
|
+
groups=groups,
|
|
318
|
+
groups_columns=groups_columns)
|
|
319
|
+
|
|
320
|
+
### INITIAL GENERATE CANDIDATE
|
|
321
|
+
init_candidate: Candidate = Candidate(
|
|
322
|
+
dataset.sample(generation_sample_size),
|
|
323
|
+
main_metric=self.main_metric)
|
|
324
|
+
|
|
325
|
+
# Select metrics used to evaluate performances
|
|
326
|
+
for metric \
|
|
327
|
+
in self.__metrics_selection(dataset.X, dataset.y, dataset.type_of_target):
|
|
328
|
+
init_candidate.add_metric(metric)
|
|
329
|
+
|
|
330
|
+
# Generate candidates
|
|
331
|
+
candidates = baseline_pipe.run(init_candidate)
|
|
332
|
+
|
|
333
|
+
# Remove candidate without predictor
|
|
334
|
+
candidates = [candidate for candidate in candidates \
|
|
335
|
+
if candidate.pipeline.predictor is not None]
|
|
336
|
+
|
|
337
|
+
for candidate in candidates:
|
|
338
|
+
candidate.pipeline.fit(dataset.X, dataset.y)
|
|
339
|
+
|
|
340
|
+
return candidates
|
|
341
|
+
|
|
342
|
+
##################
|
|
343
|
+
### PROPERTIES ###
|
|
344
|
+
##################
|
|
345
|
+
|
|
346
|
+
# Dataset from candidate data
|
|
347
|
+
@property
|
|
348
|
+
def dataset(self) -> Dataset:
|
|
349
|
+
"""Shortcut to get candidate Dataset
|
|
350
|
+
|
|
351
|
+
:return: Candidate dataset defined by .fit()
|
|
352
|
+
"""
|
|
353
|
+
return self.init_candidate.dataset
|
|
354
|
+
|
|
355
|
+
###########
|
|
356
|
+
### RUN ###
|
|
357
|
+
###########
|
|
358
|
+
|
|
359
|
+
def fit( # pylint: disable=too-many-arguments,too-many-locals
|
|
360
|
+
self,
|
|
361
|
+
X: pd.DataFrame,
|
|
362
|
+
y: pd.DataFrame,
|
|
363
|
+
groups: pd.DataFrame = None,
|
|
364
|
+
groups_columns: list[str] = None,
|
|
365
|
+
patience: int = -1,
|
|
366
|
+
generation_sample_size: int = 200,
|
|
367
|
+
n_candidates: int = 1,
|
|
368
|
+
callback: callable = None,
|
|
369
|
+
verbose: int = 1,
|
|
370
|
+
log_callback: callable = None) -> list[Candidate]:
|
|
371
|
+
"""Run Pipeline to fit steps and models on X & y data.
|
|
372
|
+
|
|
373
|
+
:param pd.DataFrame X: Training features
|
|
374
|
+
:param pd.DataFrame y: Training labels
|
|
375
|
+
:param pd.DataFrame, optional groups: Dataframe used to split data by groups.
|
|
376
|
+
Default to None.
|
|
377
|
+
:param list[str], optional groups_columns: List of column names used to split data by
|
|
378
|
+
groups. Default to None.
|
|
379
|
+
:param int, optional patience: Max generation without improvement. Default to -1.
|
|
380
|
+
:param int, optional generation_sample_size: Size of the sample dataset used to generate
|
|
381
|
+
first generation of candidates (default 200).
|
|
382
|
+
:param int, optional n_candidates: Number of candidates to return. Default to 1.
|
|
383
|
+
:param callable, optional callback: Method call after each big step of training.
|
|
384
|
+
:param int, optional verbose: Verbosity level. Default to 1.
|
|
385
|
+
:param callable, optional log_callback: Callback for logger.
|
|
386
|
+
:return: List of all the generated candidates. Sorted by performances.
|
|
387
|
+
"""
|
|
388
|
+
# Avoid [] dangerous default value in the signature
|
|
389
|
+
if groups_columns is None:
|
|
390
|
+
groups_columns = []
|
|
391
|
+
|
|
392
|
+
self.check_pipeline() # Raise error if the pipeline is not valid
|
|
393
|
+
|
|
394
|
+
Logger().verbose = verbose # Set logger verbose
|
|
395
|
+
if log_callback is not None:
|
|
396
|
+
Logger().set_callback(log_callback)
|
|
397
|
+
|
|
398
|
+
start_time = time.monotonic()
|
|
399
|
+
self.executor = TimedPoolExecutor(max_workers=self.max_workers)
|
|
400
|
+
self.training_history = []
|
|
401
|
+
self._training_history_seen = set()
|
|
402
|
+
|
|
403
|
+
def remain_time():
|
|
404
|
+
if self.max_duration == -1:
|
|
405
|
+
return math.inf
|
|
406
|
+
return max(0.0, self.max_duration - (time.monotonic() - start_time))
|
|
407
|
+
|
|
408
|
+
try:
|
|
409
|
+
refit_X, refit_y, refit_groups_columns = X, y, groups_columns
|
|
410
|
+
dataset:Dataset = Dataset(
|
|
411
|
+
deepcopy(X),
|
|
412
|
+
deepcopy(y),
|
|
413
|
+
groups=groups,
|
|
414
|
+
groups_columns=groups_columns)
|
|
415
|
+
|
|
416
|
+
if self.train_on_n_samples != None and self.train_on_n_samples > 0:
|
|
417
|
+
dataset = dataset.sample(self.train_on_n_samples)
|
|
418
|
+
if self.refit_on_sample:
|
|
419
|
+
# Preserve this sample even if the search later downsizes again.
|
|
420
|
+
# Dataset.X already excludes the grouping columns.
|
|
421
|
+
refit_X, refit_y, refit_groups_columns = dataset.X, dataset.y, []
|
|
422
|
+
|
|
423
|
+
self._last_dataset = dataset
|
|
424
|
+
self.descriptive_statistics = None
|
|
425
|
+
|
|
426
|
+
### INITIAL GENERATE CANDIDATE
|
|
427
|
+
self.init_candidate: Candidate = Candidate(
|
|
428
|
+
dataset.sample(generation_sample_size),
|
|
429
|
+
main_metric=self.main_metric)
|
|
430
|
+
|
|
431
|
+
if self.initial_preprocessor is not None:
|
|
432
|
+
# Encode the generation sample so both branches can discover
|
|
433
|
+
# suitable predictors. Keep the raw search/refit datasets: the
|
|
434
|
+
# Step is cloned/refitted inside every pipeline's CV fit.
|
|
435
|
+
initial_step = SklearnPreprocessor(self.initial_preprocessor)
|
|
436
|
+
initial_step.fit(self.init_candidate.dataset)
|
|
437
|
+
self.init_candidate = self.init_candidate.add_to_pipeline(initial_step)
|
|
438
|
+
|
|
439
|
+
# Select metrics used to evaluate performances
|
|
440
|
+
for metric \
|
|
441
|
+
in self.__metrics_selection(dataset.X, dataset.y, dataset.type_of_target):
|
|
442
|
+
self.init_candidate.add_metric(metric)
|
|
443
|
+
|
|
444
|
+
minimal_candidates = self.__generate_minimal_candidates(self.init_candidate)
|
|
445
|
+
|
|
446
|
+
# Generate candidates
|
|
447
|
+
pipeline_candidates = self.__run(self.init_candidate)
|
|
448
|
+
pipeline_candidates = [candidate for candidate in pipeline_candidates \
|
|
449
|
+
if candidate.pipeline.predictor is not None]
|
|
450
|
+
|
|
451
|
+
candidates = minimal_candidates + pipeline_candidates
|
|
452
|
+
# Remove candidate without predictor
|
|
453
|
+
candidates = [candidate for candidate in candidates \
|
|
454
|
+
if candidate.pipeline.predictor is not None]
|
|
455
|
+
self.candidates = candidates
|
|
456
|
+
|
|
457
|
+
if minimal_candidates:
|
|
458
|
+
Logger().info(f"{len(minimal_candidates)} minimalist pipelines generated")
|
|
459
|
+
Logger().info(f"{len(candidates)} generated pipelines")
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
# Warmup is real CV, so it must use the same interruptible executor
|
|
463
|
+
# and shared search/stage budgets as every subsequent evaluation.
|
|
464
|
+
# Minimal candidates already lead the pool and provide a quick,
|
|
465
|
+
# honestly evaluated starting point without removing full pipelines.
|
|
466
|
+
warmup_candidate = None
|
|
467
|
+
if candidates and remain_time() >= 1:
|
|
468
|
+
# A single slow baseline must leave time to evaluate the other
|
|
469
|
+
# candidates. Unlimited searches retain the stage duration cap.
|
|
470
|
+
remaining = remain_time()
|
|
471
|
+
warmup_timeout = remaining / 5
|
|
472
|
+
Logger().info(
|
|
473
|
+
f"Warming up (at most {min(warmup_timeout, self.max_stage_duration):.2f}s): "
|
|
474
|
+
f"{candidates[0].pipeline.name}")
|
|
475
|
+
warmup_candidates = self.__run_evaluations(
|
|
476
|
+
candidates[:1], dataset, timeout=remaining, callback=callback,
|
|
477
|
+
stage_timeout=warmup_timeout)
|
|
478
|
+
if warmup_candidates:
|
|
479
|
+
warmup_candidate = warmup_candidates[0]
|
|
480
|
+
Logger().info("Warmed up !")
|
|
481
|
+
else:
|
|
482
|
+
Logger().info("No warmup result within the stage budget.")
|
|
483
|
+
# Try alternatives before retrying the same candidate,
|
|
484
|
+
# especially when only one worker is available.
|
|
485
|
+
candidates = candidates[1:] + candidates[:1]
|
|
486
|
+
|
|
487
|
+
### INITIAL EVALUATION
|
|
488
|
+
# Evaluate candidates
|
|
489
|
+
gen0_candidates = []
|
|
490
|
+
i = 0
|
|
491
|
+
can_be_downsize = True
|
|
492
|
+
while can_be_downsize and not gen0_candidates and remain_time() >= 1:
|
|
493
|
+
# If process is too long and dataset big enough,
|
|
494
|
+
# we can downsize it to get quicker training
|
|
495
|
+
if i > 0:
|
|
496
|
+
dataset = dataset.sample(0.1)
|
|
497
|
+
Logger().warning(f"Training is too time consuming. \
|
|
498
|
+
Let's try again with dataset sample. \
|
|
499
|
+
New features shape {dataset.X.shape}")
|
|
500
|
+
i+= 1
|
|
501
|
+
can_be_downsize = dataset.X.shape[0] >= 500
|
|
502
|
+
|
|
503
|
+
timeout = min(remain_time(), self.time_before_sample_use) \
|
|
504
|
+
if can_be_downsize else remain_time()
|
|
505
|
+
|
|
506
|
+
gen0_candidates = self.__run_evaluations(candidates,
|
|
507
|
+
dataset, timeout=timeout, callback=callback)
|
|
508
|
+
|
|
509
|
+
if not gen0_candidates:
|
|
510
|
+
if warmup_candidate and warmup_candidate.computed_metrics:
|
|
511
|
+
Logger().warning(
|
|
512
|
+
"No candidates evaluated before timeout; using warmup candidate."
|
|
513
|
+
)
|
|
514
|
+
gen0_candidates = [warmup_candidate]
|
|
515
|
+
elif remain_time() < 1:
|
|
516
|
+
raise TimeoutError('IAML was unable to generate a model within the \
|
|
517
|
+
imposed time limit. Try increasing the processing time')
|
|
518
|
+
else:
|
|
519
|
+
raise RuntimeError('Undefined error. IAML was unable to create pipeline \
|
|
520
|
+
based on your data')
|
|
521
|
+
|
|
522
|
+
### FINETUNING
|
|
523
|
+
candidates = self.__optimize(dataset,
|
|
524
|
+
gen0_candidates,
|
|
525
|
+
optimizer=self.__build_optimizer(remain_time()),
|
|
526
|
+
max_duration=remain_time(),
|
|
527
|
+
patience=patience,
|
|
528
|
+
callback=callback)
|
|
529
|
+
|
|
530
|
+
### FINAL FIT
|
|
531
|
+
self.executor.shutdown()
|
|
532
|
+
|
|
533
|
+
# Fit candidates on the requested sample or all original input rows.
|
|
534
|
+
fit_candidates = []
|
|
535
|
+
last_fit_error = None
|
|
536
|
+
for candidate in candidates:
|
|
537
|
+
if len(fit_candidates) >= n_candidates:
|
|
538
|
+
break
|
|
539
|
+
Cache.reset()
|
|
540
|
+
current_candidate = deepcopy(candidate)
|
|
541
|
+
try:
|
|
542
|
+
Logger().info(f"Final fit: {len(refit_X)} rows, {current_candidate.pipeline.name}")
|
|
543
|
+
current_candidate.pipeline.fit(
|
|
544
|
+
refit_X,
|
|
545
|
+
refit_y,
|
|
546
|
+
groups_columns=refit_groups_columns,
|
|
547
|
+
metrics=current_candidate.metrics,
|
|
548
|
+
)
|
|
549
|
+
except (ValueError, np.linalg.LinAlgError) as exc:
|
|
550
|
+
Logger().warning(
|
|
551
|
+
f"Skipping candidate during final fit after failure: {exc!r}"
|
|
552
|
+
)
|
|
553
|
+
last_fit_error = exc
|
|
554
|
+
continue
|
|
555
|
+
fit_candidates.append(current_candidate)
|
|
556
|
+
|
|
557
|
+
if not fit_candidates:
|
|
558
|
+
details = f" Last error: {last_fit_error!r}" if last_fit_error else ""
|
|
559
|
+
raise ValueError(f"IAML could not fit any candidate pipeline.{details}")
|
|
560
|
+
|
|
561
|
+
self.chosen_candidate = fit_candidates[0]
|
|
562
|
+
self.last_stage_candidates = candidates
|
|
563
|
+
|
|
564
|
+
return fit_candidates
|
|
565
|
+
except TerminatedError:
|
|
566
|
+
Logger().info('IAML was terminated.')
|
|
567
|
+
except Exception as ex:
|
|
568
|
+
Logger().error('Error during fit')
|
|
569
|
+
raise ex
|
|
570
|
+
finally:
|
|
571
|
+
self.executor.shutdown()
|
|
572
|
+
|
|
573
|
+
@property
|
|
574
|
+
def chosen_model(self) -> 'IAMLPipeline':
|
|
575
|
+
"""Return the best model trained with fit
|
|
576
|
+
|
|
577
|
+
:return: Best predictor pipeline
|
|
578
|
+
"""
|
|
579
|
+
if not self.chosen_candidate:
|
|
580
|
+
return None
|
|
581
|
+
|
|
582
|
+
return self.chosen_candidate.pipeline
|
|
583
|
+
|
|
584
|
+
def visualize_descriptive_statistics(self) -> list[StatisticPlot]:
|
|
585
|
+
"""Return a list of plots that show descriptive statistics."""
|
|
586
|
+
if self.descriptive_statistics is None and self._last_dataset is not None:
|
|
587
|
+
self.__ensure_descriptive_statistics(self._last_dataset)
|
|
588
|
+
|
|
589
|
+
if self.descriptive_statistics is None or self.descriptive_statistics.empty:
|
|
590
|
+
return []
|
|
591
|
+
|
|
592
|
+
plots: list[StatisticPlot] = []
|
|
593
|
+
stats_df = self.descriptive_statistics
|
|
594
|
+
|
|
595
|
+
def group_columns_by_feature(dataframe: pd.DataFrame) -> dict[str, list[str]]:
|
|
596
|
+
columns = list(dataframe.columns)
|
|
597
|
+
base_names: list[str] = []
|
|
598
|
+
for col in columns:
|
|
599
|
+
if isinstance(col, str) and col.endswith('_all'):
|
|
600
|
+
base = col[:-4]
|
|
601
|
+
if base not in base_names:
|
|
602
|
+
base_names.append(base)
|
|
603
|
+
|
|
604
|
+
groups: dict[str, list[str]] = {}
|
|
605
|
+
used_cols: set[str] = set()
|
|
606
|
+
if base_names:
|
|
607
|
+
for base in base_names:
|
|
608
|
+
group = [
|
|
609
|
+
col for col in columns
|
|
610
|
+
if col == base or (isinstance(col, str) and col.startswith(f"{base}_"))
|
|
611
|
+
]
|
|
612
|
+
groups[base] = group
|
|
613
|
+
used_cols.update(group)
|
|
614
|
+
|
|
615
|
+
for col in columns:
|
|
616
|
+
if col not in used_cols:
|
|
617
|
+
groups[col] = [col]
|
|
618
|
+
|
|
619
|
+
return groups
|
|
620
|
+
|
|
621
|
+
for plot_sub_class in StatisticPlot.__subclasses__():
|
|
622
|
+
if not getattr(plot_sub_class, 'enabled', True):
|
|
623
|
+
continue
|
|
624
|
+
if getattr(plot_sub_class, 'group_by_feature', False):
|
|
625
|
+
for base, cols in group_columns_by_feature(stats_df).items():
|
|
626
|
+
plot = plot_sub_class().compute(stats_df[cols], base_name=base)
|
|
627
|
+
plots.append(plot)
|
|
628
|
+
else:
|
|
629
|
+
plots.append(plot_sub_class().compute(stats_df))
|
|
630
|
+
|
|
631
|
+
return plots
|
|
632
|
+
|
|
633
|
+
def get_descriptive_statistics(self) -> pd.DataFrame:
|
|
634
|
+
"""Return descriptive statistics, computing them on demand if needed."""
|
|
635
|
+
if self.descriptive_statistics is None and self._last_dataset is not None:
|
|
636
|
+
self.__ensure_descriptive_statistics(self._last_dataset)
|
|
637
|
+
|
|
638
|
+
return self.descriptive_statistics if self.descriptive_statistics is not None else pd.DataFrame()
|
|
639
|
+
|
|
640
|
+
def __ensure_descriptive_statistics(self, dataset: Dataset) -> None:
|
|
641
|
+
"""Compute descriptive statistics once, for on-demand usage."""
|
|
642
|
+
if self.descriptive_statistics is not None:
|
|
643
|
+
return
|
|
644
|
+
|
|
645
|
+
try:
|
|
646
|
+
self.descriptive_statistics = self.__compute_descriptive_statistics(dataset)
|
|
647
|
+
except Exception as exc: # noqa: BLE001
|
|
648
|
+
Logger().warning(f"Descriptive statistics computation failed: {exc}")
|
|
649
|
+
self.descriptive_statistics = pd.DataFrame()
|
|
650
|
+
|
|
651
|
+
def __compute_descriptive_statistics(self, dataset: Dataset) -> pd.DataFrame:
|
|
652
|
+
"""Compute descriptive statistics on the dataset."""
|
|
653
|
+
computed_statistics = pd.DataFrame()
|
|
654
|
+
for statistic_sub_class in Statistic.all_subclasses():
|
|
655
|
+
statistic = statistic_sub_class()
|
|
656
|
+
if statistic.suitable(dataset):
|
|
657
|
+
result = statistic.compute(dataset)
|
|
658
|
+
if result is not None and not result.empty:
|
|
659
|
+
computed_statistics = pd.concat([computed_statistics, result])
|
|
660
|
+
return computed_statistics
|
|
661
|
+
|
|
662
|
+
def check_pipeline(self) -> None:
|
|
663
|
+
"""Raise Exception if pipeline is not valid
|
|
664
|
+
|
|
665
|
+
:raise AttributeError: Step Pipeline must contains at least one predictor
|
|
666
|
+
"""
|
|
667
|
+
steps = self.__all_steps()
|
|
668
|
+
|
|
669
|
+
# Pipeline must have at least one predictor
|
|
670
|
+
if not any((Predictor in s.__class__.__mro__) for s in steps if s.enable):
|
|
671
|
+
raise AttributeError('Step Pipeline must contains at least one predictor')
|
|
672
|
+
|
|
673
|
+
# TODO Others tests ?
|
|
674
|
+
|
|
675
|
+
def __run_evaluations(
|
|
676
|
+
self,
|
|
677
|
+
candidates: list[Candidate],
|
|
678
|
+
dataset: Dataset,
|
|
679
|
+
timeout: float = None,
|
|
680
|
+
stage_number: int = None,
|
|
681
|
+
callback: callable = None,
|
|
682
|
+
stage_timeout: float = None) -> list[Candidate]:
|
|
683
|
+
"""Evaluate candidates
|
|
684
|
+
|
|
685
|
+
:param list[Candidate] candidates: Candidates to evaluate.
|
|
686
|
+
:param Dataset dataset: Dataset used for evaluation.
|
|
687
|
+
:param float, optional timeout: Budget including preparation and submission.
|
|
688
|
+
Defaults to the stage duration limit.
|
|
689
|
+
:param int, optional stage_number: Stage number running. Default to None.
|
|
690
|
+
:param callable, optional callback: Method called after evaluation. Default to None.
|
|
691
|
+
:param float, optional stage_timeout: Additional cap for this evaluation only;
|
|
692
|
+
the callback still reports the remaining ``timeout`` budget.
|
|
693
|
+
"""
|
|
694
|
+
start_time = time.monotonic()
|
|
695
|
+
budget = self.max_stage_duration if timeout is None else min(timeout, self.max_stage_duration)
|
|
696
|
+
if stage_timeout is not None:
|
|
697
|
+
budget = min(budget, stage_timeout)
|
|
698
|
+
deadline = start_time + max(0.0, budget)
|
|
699
|
+
new_candidates: list[Candidate] = []
|
|
700
|
+
splitter_fingerprint = hash_evaluation_context(self.splitter)
|
|
701
|
+
dataset_key = dataset.fingerprint() if splitter_fingerprint is not None else None
|
|
702
|
+
evaluation_cache_keys: set[str] = set()
|
|
703
|
+
with Logger().progress as progress:
|
|
704
|
+
task = progress.add_task(
|
|
705
|
+
f'Stage {stage_number}' if stage_number is not None else "Initial evaluation",
|
|
706
|
+
total=len(candidates))
|
|
707
|
+
|
|
708
|
+
def update_progressbar(*args): # pylint: disable=unused-argument
|
|
709
|
+
progress.update(task, advance=1)
|
|
710
|
+
|
|
711
|
+
self.executor.set_callback(update_progressbar)
|
|
712
|
+
for candidate in candidates:
|
|
713
|
+
if time.monotonic() >= deadline:
|
|
714
|
+
break
|
|
715
|
+
cache_key = self.__evaluation_cache_key(candidate, splitter_fingerprint)
|
|
716
|
+
if cache_key is not None:
|
|
717
|
+
evaluation_cache_keys.add(cache_key)
|
|
718
|
+
from_cache = Cache().from_cache(cache_key, dataset_key) if cache_key else None
|
|
719
|
+
|
|
720
|
+
if from_cache:
|
|
721
|
+
self.__hydrate_cached_candidate(candidate, from_cache, dataset)
|
|
722
|
+
new_candidates.append(candidate)
|
|
723
|
+
update_progressbar() # Update progressbar even if data come from cache
|
|
724
|
+
else:
|
|
725
|
+
submitted = self.executor.submit(
|
|
726
|
+
process_executor,
|
|
727
|
+
candidate,
|
|
728
|
+
dataset,
|
|
729
|
+
deadline=deadline,
|
|
730
|
+
splitter=self.splitter,
|
|
731
|
+
store_audit=self.keep_training_history,
|
|
732
|
+
)
|
|
733
|
+
if not submitted:
|
|
734
|
+
break
|
|
735
|
+
|
|
736
|
+
# Preparation and submission have already consumed part of the budget.
|
|
737
|
+
new_candidates += self.executor.join(max(0.0, deadline - time.monotonic()))
|
|
738
|
+
self.__collect_training_history(new_candidates)
|
|
739
|
+
|
|
740
|
+
if new_candidates:
|
|
741
|
+
skipped = sum(1 for candidate in new_candidates if not candidate.computed_metrics)
|
|
742
|
+
if skipped:
|
|
743
|
+
Logger().warning(
|
|
744
|
+
f"Skipped {skipped} candidates with no computed metrics."
|
|
745
|
+
)
|
|
746
|
+
new_candidates = [candidate for candidate in new_candidates if candidate.computed_metrics]
|
|
747
|
+
|
|
748
|
+
new_candidates.sort(reverse=True)
|
|
749
|
+
|
|
750
|
+
# Add results to progressbar
|
|
751
|
+
if new_candidates:
|
|
752
|
+
progress.tasks[task].description = f'{progress.tasks[task].description} \
|
|
753
|
+
({new_candidates[0].get_main_metric_value():.4f})'
|
|
754
|
+
else:
|
|
755
|
+
progress.tasks[task].description = f'{progress.tasks[task].description} \
|
|
756
|
+
(no result)'
|
|
757
|
+
|
|
758
|
+
# Add to cache
|
|
759
|
+
for candidate in new_candidates:
|
|
760
|
+
cache_key = self.__evaluation_cache_key(candidate, splitter_fingerprint)
|
|
761
|
+
if cache_key in evaluation_cache_keys and not Cache().from_cache(cache_key, dataset_key):
|
|
762
|
+
Cache().add_to_cache(
|
|
763
|
+
cache_key,
|
|
764
|
+
dataset_key,
|
|
765
|
+
self.__build_cached_candidate(candidate),
|
|
766
|
+
)
|
|
767
|
+
|
|
768
|
+
best_metric = new_candidates[0].get_main_metric_value() if new_candidates else None
|
|
769
|
+
remaining_time = (budget if timeout is None else timeout) - (time.monotonic() - start_time)
|
|
770
|
+
|
|
771
|
+
self.__callback(callback, # pylint: disable=too-many-function-args
|
|
772
|
+
generation = stage_number,
|
|
773
|
+
generation_size = len(new_candidates),
|
|
774
|
+
best = best_metric,
|
|
775
|
+
remaining_time = remaining_time,
|
|
776
|
+
text = f'Stage {stage_number} finished' \
|
|
777
|
+
if stage_number is not None else "Initial evaluation finished")
|
|
778
|
+
|
|
779
|
+
return new_candidates
|
|
780
|
+
|
|
781
|
+
def __evaluation_cache_key(
|
|
782
|
+
self, candidate: Candidate, splitter_fingerprint: str | None
|
|
783
|
+
) -> str | None:
|
|
784
|
+
"""Keep scores separate for each splitter and metric configuration."""
|
|
785
|
+
if splitter_fingerprint is None:
|
|
786
|
+
return None
|
|
787
|
+
context = hash_evaluation_context(
|
|
788
|
+
candidate.pipeline.fingerprint(), candidate.metrics, candidate.main_metric,
|
|
789
|
+
self.keep_training_history,
|
|
790
|
+
)
|
|
791
|
+
if context is None:
|
|
792
|
+
return None
|
|
793
|
+
return f"IAML_{splitter_fingerprint}_{context}"
|
|
794
|
+
|
|
795
|
+
def __build_optimizer(self, duration: float) -> Optimizer:
|
|
796
|
+
"""Instantiate optimizer."""
|
|
797
|
+
return self.optimizer(duration=None if math.isinf(duration) else duration)
|
|
798
|
+
|
|
799
|
+
def __build_cached_candidate(self, candidate: Candidate) -> dict[str, Any] | dict[str, float]:
|
|
800
|
+
"""Build the payload stored in cache for evaluated candidates."""
|
|
801
|
+
if self.keep_training_history and candidate.training_audit is not None:
|
|
802
|
+
return {
|
|
803
|
+
"__computed_metrics__": deepcopy(candidate.computed_metrics),
|
|
804
|
+
"__training_audit__": deepcopy(candidate.training_audit),
|
|
805
|
+
}
|
|
806
|
+
return deepcopy(candidate.computed_metrics)
|
|
807
|
+
|
|
808
|
+
def __hydrate_cached_candidate(
|
|
809
|
+
self,
|
|
810
|
+
candidate: Candidate,
|
|
811
|
+
payload: dict[str, Any] | dict[str, float],
|
|
812
|
+
dataset: Dataset,
|
|
813
|
+
) -> None:
|
|
814
|
+
"""Restore cached evaluation results into a candidate."""
|
|
815
|
+
candidate.fold_metrics = []
|
|
816
|
+
candidate.training_audit = None
|
|
817
|
+
|
|
818
|
+
if isinstance(payload, dict) and "__computed_metrics__" in payload:
|
|
819
|
+
candidate.computed_metrics = deepcopy(payload["__computed_metrics__"])
|
|
820
|
+
audit = payload.get("__training_audit__")
|
|
821
|
+
if audit is not None:
|
|
822
|
+
candidate.training_audit = deepcopy(audit)
|
|
823
|
+
candidate.fold_metrics = deepcopy(audit.get("fold_metrics", []))
|
|
824
|
+
return
|
|
825
|
+
else:
|
|
826
|
+
candidate.computed_metrics = deepcopy(payload)
|
|
827
|
+
|
|
828
|
+
if self.keep_training_history:
|
|
829
|
+
candidate.training_audit = candidate.build_training_audit(
|
|
830
|
+
dataset=dataset,
|
|
831
|
+
fold_metrics=[],
|
|
832
|
+
aggregated_metrics=candidate.computed_metrics,
|
|
833
|
+
status="success",
|
|
834
|
+
)
|
|
835
|
+
|
|
836
|
+
def __collect_training_history(self, candidates: list[Candidate]) -> None:
|
|
837
|
+
"""Collect unique candidate audit records for the last fit."""
|
|
838
|
+
if not self.keep_training_history:
|
|
839
|
+
return
|
|
840
|
+
|
|
841
|
+
for candidate in candidates:
|
|
842
|
+
record = getattr(candidate, "training_audit", None)
|
|
843
|
+
if not record:
|
|
844
|
+
continue
|
|
845
|
+
key = (
|
|
846
|
+
record.get("pipeline_fingerprint"),
|
|
847
|
+
record.get("dataset_fingerprint"),
|
|
848
|
+
record.get("status"),
|
|
849
|
+
record.get("error"),
|
|
850
|
+
)
|
|
851
|
+
if key in self._training_history_seen:
|
|
852
|
+
continue
|
|
853
|
+
self._training_history_seen.add(key)
|
|
854
|
+
self.training_history.append(deepcopy(record))
|
|
855
|
+
|
|
856
|
+
def __optimize( # pylint: disable=too-many-arguments
|
|
857
|
+
self,
|
|
858
|
+
dataset: Dataset,
|
|
859
|
+
candidates: list[Candidate],
|
|
860
|
+
optimizer: Optimizer = Optimizer(),
|
|
861
|
+
patience: int = 5,
|
|
862
|
+
max_duration: int = -1,
|
|
863
|
+
callback: callable = None) -> list[Candidate]:
|
|
864
|
+
"""Optimize candidates.
|
|
865
|
+
|
|
866
|
+
:param Dataset dataset: Dataset used for optimization.
|
|
867
|
+
:param list[Candidate], optional candidates: Candidates to optimize.
|
|
868
|
+
:param Optimizer, optional optimizer: Optimizer to use.
|
|
869
|
+
:param int, optional patience: Max generation without improvement. Default to 5.
|
|
870
|
+
:param int, optional max_duration: Maximum optimization duration. Default to -1.
|
|
871
|
+
:param callable, optional callback: Method to call after optimization. Default to None.
|
|
872
|
+
:return: list of optimized candidate.
|
|
873
|
+
"""
|
|
874
|
+
if not candidates:
|
|
875
|
+
return []
|
|
876
|
+
|
|
877
|
+
if max_duration == -1:
|
|
878
|
+
max_duration = math.inf
|
|
879
|
+
|
|
880
|
+
# init
|
|
881
|
+
candidates.sort(reverse=True)
|
|
882
|
+
best_result: float = candidates[0].get_main_metric_value()
|
|
883
|
+
best_score: float = candidates[0].get_main_metric_score()
|
|
884
|
+
iterations_without_improvement: int = 0
|
|
885
|
+
iterations_count: int = 0
|
|
886
|
+
duration: int = 0
|
|
887
|
+
starting_time: int = time.monotonic() # seconds
|
|
888
|
+
|
|
889
|
+
# If there is not, define an arbitrary stop condition
|
|
890
|
+
if math.isinf(max_duration) and patience == -1:
|
|
891
|
+
Logger().warning('You have not defined any stop condition. \
|
|
892
|
+
Patient has arbitrary set to 20')
|
|
893
|
+
patience = 20
|
|
894
|
+
|
|
895
|
+
previous_candidates = candidates
|
|
896
|
+
|
|
897
|
+
while not(optimizer.finished) \
|
|
898
|
+
and (patience == -1 or iterations_without_improvement < patience) \
|
|
899
|
+
and max_duration > duration:
|
|
900
|
+
# Generate new candidates
|
|
901
|
+
generated_candidates = optimizer.run(previous_candidates)
|
|
902
|
+
|
|
903
|
+
Logger().info(f'Finetuning... \
|
|
904
|
+
stage={iterations_count} \
|
|
905
|
+
candidates={len(generated_candidates)} \
|
|
906
|
+
patience={iterations_without_improvement}/{patience}, \
|
|
907
|
+
duration={round(duration, 2)}/{max_duration}, \
|
|
908
|
+
best_result={best_result}')
|
|
909
|
+
|
|
910
|
+
# Evaluate new candidates
|
|
911
|
+
evaluated_candidates = self.__run_evaluations(generated_candidates,
|
|
912
|
+
dataset,
|
|
913
|
+
timeout=max_duration - (time.monotonic() - starting_time),
|
|
914
|
+
stage_number=iterations_count,
|
|
915
|
+
callback=callback)
|
|
916
|
+
|
|
917
|
+
# Remove not computed (error or timeout)
|
|
918
|
+
evaluated_candidates = [candidate for candidate in evaluated_candidates if candidate.computed_metrics]
|
|
919
|
+
|
|
920
|
+
if not evaluated_candidates:
|
|
921
|
+
Logger().warning('No candidates produced a valid evaluation; keeping previous best candidates.')
|
|
922
|
+
candidates = previous_candidates
|
|
923
|
+
break
|
|
924
|
+
|
|
925
|
+
candidates = evaluated_candidates
|
|
926
|
+
previous_candidates = candidates
|
|
927
|
+
|
|
928
|
+
# Improvement ?
|
|
929
|
+
new_score: float = candidates[0].get_main_metric_score()
|
|
930
|
+
if new_score > best_score:
|
|
931
|
+
best_score = new_score
|
|
932
|
+
best_result = candidates[0].get_main_metric_value()
|
|
933
|
+
iterations_without_improvement = 0
|
|
934
|
+
else:
|
|
935
|
+
iterations_without_improvement += 1
|
|
936
|
+
|
|
937
|
+
# Duration in seconds
|
|
938
|
+
duration = time.monotonic() - starting_time
|
|
939
|
+
|
|
940
|
+
# Increase Iteration count
|
|
941
|
+
iterations_count += 1
|
|
942
|
+
|
|
943
|
+
return candidates
|
|
944
|
+
|
|
945
|
+
def __metrics_selection(
|
|
946
|
+
self,
|
|
947
|
+
X: pd.DataFrame,
|
|
948
|
+
y: pd.DataFrame,
|
|
949
|
+
type_of_target: str) -> list[Metric]:
|
|
950
|
+
"""Select metrics used to evaluate performances
|
|
951
|
+
|
|
952
|
+
:param pd.DataFrame X: Training features.
|
|
953
|
+
:param pd.DataFrame y: Training labels
|
|
954
|
+
:param str type_of_target: type of label. Example : continuous, binary
|
|
955
|
+
:return: List of selected metrics
|
|
956
|
+
"""
|
|
957
|
+
metrics = []
|
|
958
|
+
|
|
959
|
+
for metric_sub_class in Metric.all_subclasses():
|
|
960
|
+
# Reuse the configured main metric instead of resetting its parameters.
|
|
961
|
+
metric = (
|
|
962
|
+
self.main_metric
|
|
963
|
+
if type(self.main_metric) is metric_sub_class
|
|
964
|
+
else metric_sub_class()
|
|
965
|
+
)
|
|
966
|
+
# Verify if a subclass is suitable or not
|
|
967
|
+
if metric is self.main_metric or metric.suitable(X, y, type_of_target):
|
|
968
|
+
metrics.append(metric)
|
|
969
|
+
return metrics
|
|
970
|
+
|
|
971
|
+
def __apply_minimal_preprocessing(self, candidate: Candidate) -> Candidate:
|
|
972
|
+
"""Ensure minimalist candidates have a basic imputer in their pipeline."""
|
|
973
|
+
dataset = candidate.dataset
|
|
974
|
+
|
|
975
|
+
if dataset.X.isna().values.any():
|
|
976
|
+
imputer = ActSimpleImputer()
|
|
977
|
+
results = imputer.run(candidate)
|
|
978
|
+
|
|
979
|
+
if isinstance(results, list) and results:
|
|
980
|
+
return results[0]
|
|
981
|
+
|
|
982
|
+
return results
|
|
983
|
+
|
|
984
|
+
return candidate
|
|
985
|
+
|
|
986
|
+
def __generate_minimal_candidates(self, candidate: Candidate) -> list[Candidate]:
|
|
987
|
+
"""Generate minimalist candidates using raw predictors only."""
|
|
988
|
+
if not self.minimal_predictor_step:
|
|
989
|
+
return []
|
|
990
|
+
|
|
991
|
+
Logger().info('Generate minimalist candidates...')
|
|
992
|
+
minimal_candidate = candidate.to_input()
|
|
993
|
+
minimal_candidate = self.__apply_minimal_preprocessing(minimal_candidate)
|
|
994
|
+
|
|
995
|
+
candidates = self.minimal_predictor_step.run(minimal_candidate)
|
|
996
|
+
|
|
997
|
+
return [cand for cand in candidates if cand.pipeline.predictor is not None]
|
|
998
|
+
|
|
999
|
+
def __run(self, candidate: Candidate) -> list[Candidate]:
|
|
1000
|
+
"""Run pipeline steps
|
|
1001
|
+
|
|
1002
|
+
:param Candidate candidate: Data used to fit models and steps.
|
|
1003
|
+
:return: List of all the generated candidates. Sorted by performances.
|
|
1004
|
+
"""
|
|
1005
|
+
Logger().info("Generate candidate...")
|
|
1006
|
+
self.candidates = self.first_step.run(candidate)
|
|
1007
|
+
|
|
1008
|
+
return self.candidates
|
|
1009
|
+
|
|
1010
|
+
########################
|
|
1011
|
+
#### CONFIGURATIONS ####
|
|
1012
|
+
########################
|
|
1013
|
+
|
|
1014
|
+
def json_pipeline(self) -> dict:
|
|
1015
|
+
"""Create a dictionary (JSON) from the loaded Pipeline
|
|
1016
|
+
|
|
1017
|
+
:return: Loaded Pipeline in a JSON format
|
|
1018
|
+
"""
|
|
1019
|
+
return self.first_step.json_pipeline()
|
|
1020
|
+
|
|
1021
|
+
def all_configurations(self) -> list[dict]:
|
|
1022
|
+
"""Return a dict with configurations of all steps.
|
|
1023
|
+
|
|
1024
|
+
:return: Configurations of all steps.
|
|
1025
|
+
"""
|
|
1026
|
+
return self.first_step.all_configurations()
|
|
1027
|
+
|
|
1028
|
+
def configure_all(self, configs: dict) -> None:
|
|
1029
|
+
"""Configure one to many steps with a dict configuration
|
|
1030
|
+
|
|
1031
|
+
:param dict configs: key is a step_id and value is the configuration to set.
|
|
1032
|
+
"""
|
|
1033
|
+
all_steps = self.__all_steps()
|
|
1034
|
+
|
|
1035
|
+
for step_id, config in configs.items():
|
|
1036
|
+
current_step: Step = self.__find_step_by_id(all_steps, step_id)
|
|
1037
|
+
if current_step:
|
|
1038
|
+
for key, value in config:
|
|
1039
|
+
current_step.configure(key, value) # pylint: disable=no-member
|
|
1040
|
+
|
|
1041
|
+
def __all_steps(self) -> list[Step]:
|
|
1042
|
+
"""Recursive method. Return all the pipeline's steps in a list
|
|
1043
|
+
|
|
1044
|
+
:return: All flatten pipelines's steps
|
|
1045
|
+
"""
|
|
1046
|
+
return self.first_step.all_steps()
|
|
1047
|
+
|
|
1048
|
+
def __find_step_by_id(self, step_list: list[Step], step_id: int) -> Step | None:
|
|
1049
|
+
"""Find a step by id in a list of step
|
|
1050
|
+
|
|
1051
|
+
:param list[Step] step_list: The step will be searched in this list.
|
|
1052
|
+
:param int step_id: Identifier of the step
|
|
1053
|
+
:return: Found Step or None
|
|
1054
|
+
"""
|
|
1055
|
+
for step in step_list:
|
|
1056
|
+
if id(step) == step_id:
|
|
1057
|
+
return step
|
|
1058
|
+
return None
|
|
1059
|
+
|
|
1060
|
+
|
|
1061
|
+
def process_executor(candidate: Candidate, *args, **kwargs) -> 'Candidate':
|
|
1062
|
+
"""Wrap candidate training to run it in subprocess
|
|
1063
|
+
|
|
1064
|
+
:param Candidate candidate: Not trained candidate.
|
|
1065
|
+
:param tuple, optional \\*args: Additional parameters.
|
|
1066
|
+
:param dict, optional \\**kwargs: Additional parameters.
|
|
1067
|
+
:return: Trained candidate.
|
|
1068
|
+
"""
|
|
1069
|
+
# Deepcopy -> Without it, process end is never detected. Strange...
|
|
1070
|
+
candidate = deepcopy(candidate)
|
|
1071
|
+
candidate.training_evaluate(*args, **kwargs)
|
|
1072
|
+
return candidate
|