PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""[METRIC] Specificity"""
|
|
2
|
+
from typing import Any
|
|
3
|
+
import textwrap
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from sklearn.metrics import confusion_matrix
|
|
6
|
+
from ..metric import Metric
|
|
7
|
+
|
|
8
|
+
class SpecificityMetric(Metric):
|
|
9
|
+
"""[METRIC] Specificity"""
|
|
10
|
+
|
|
11
|
+
name: str = "Specificity"
|
|
12
|
+
_description: str = textwrap.dedent('''\
|
|
13
|
+
Specificity is a metric used to evaluate the performance of a classification model.
|
|
14
|
+
It measures the proportion of true negatives correctly identified by the model, indicating
|
|
15
|
+
how well it can identify the negative class.''')
|
|
16
|
+
_description_long: str = textwrap.dedent('''\
|
|
17
|
+
Specificity is a metric that helps assess how well a classification model identifies the negative class.
|
|
18
|
+
For example, if you're predicting whether a medical test result is negative for a disease, specificity tells you the
|
|
19
|
+
percentage of actual negative cases that the model correctly identifies. A high specificity means the model is good at
|
|
20
|
+
avoiding false positives, while a low specificity indicates it may incorrectly label negative cases as positive.
|
|
21
|
+
''')
|
|
22
|
+
refs: list[dict[str, Any]]=[
|
|
23
|
+
{
|
|
24
|
+
'year': 1994 ,
|
|
25
|
+
'name': 'Diagnostic tests. 1: Sensitivity and specificity.',
|
|
26
|
+
'authors': [
|
|
27
|
+
'D. G. Altman',
|
|
28
|
+
'J. M. Bland'
|
|
29
|
+
],
|
|
30
|
+
'doi': 'https://doi.org/10.1136%2Fbmj.308.6943.1552',
|
|
31
|
+
'publisher': 'BMJ. 308 (6943): 1552'
|
|
32
|
+
}
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
def __str__(self) -> str:
|
|
36
|
+
return 'specificity'
|
|
37
|
+
|
|
38
|
+
def suitable(self, X: pd.DataFrame, y: pd.DataFrame, type_of_target: str) -> bool:
|
|
39
|
+
return type_of_target == 'binary'
|
|
40
|
+
|
|
41
|
+
def compute(self, y: pd.DataFrame, y_pred: pd.DataFrame, **kwargs) -> float:
|
|
42
|
+
tn, fp, _, _ = confusion_matrix(y, y_pred).ravel()
|
|
43
|
+
|
|
44
|
+
return tn / (tn + fp) if (tn + fp) != 0 else 0
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""[METRIC] Specificity Multiclass"""
|
|
2
|
+
from typing import Any
|
|
3
|
+
import textwrap
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from sklearn.metrics import confusion_matrix
|
|
6
|
+
import numpy as np
|
|
7
|
+
from ..metric import Metric
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class SpecificityMulticlassMetric(Metric):
|
|
11
|
+
"""[METRIC] Specificity Multiclass"""
|
|
12
|
+
|
|
13
|
+
name: str = "Specificity Multiclass"
|
|
14
|
+
_description: str = textwrap.dedent('''\
|
|
15
|
+
Multiclass specificity is a metric used to evaluate the performance of a classification model
|
|
16
|
+
with multiple classes. It measures how well the model identifies the negative cases for each
|
|
17
|
+
class by considering true negatives and false positives.''')
|
|
18
|
+
_description_long: str = textwrap.dedent('''\
|
|
19
|
+
Multiclass specificity assesses how effectively a classification model identifies negative cases
|
|
20
|
+
across multiple classes. For each class, it calculates the number of true negatives (correctly
|
|
21
|
+
identified negatives) and false positives (incorrectly identified positives).
|
|
22
|
+
By summing these values for all classes, you can determine an overall specificity score.
|
|
23
|
+
A high multiclass specificity indicates that the model is good at correctly identifying non-target classes,
|
|
24
|
+
while a low score suggests it may struggle with misclassifying negative cases.''')
|
|
25
|
+
refs: list[dict[str, Any]] = [
|
|
26
|
+
{
|
|
27
|
+
'year': 2020,
|
|
28
|
+
'name': 'Metrics for Multi-Class Classification: an Overview',
|
|
29
|
+
'authors': [
|
|
30
|
+
'Margherita Grandini',
|
|
31
|
+
'Enrico Bagli',
|
|
32
|
+
'Giorgio Visani'
|
|
33
|
+
],
|
|
34
|
+
'doi': 'https://doi.org/10.48550/arXiv.2008.05756',
|
|
35
|
+
'publisher': ''
|
|
36
|
+
}
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
def __str__(self) -> str:
|
|
40
|
+
return 'specificity_multiclass'
|
|
41
|
+
|
|
42
|
+
def suitable(self, X: pd.DataFrame, y: pd.DataFrame, type_of_target: str) -> bool:
|
|
43
|
+
return type_of_target == 'multiclass'
|
|
44
|
+
|
|
45
|
+
def compute(self, y: pd.DataFrame, y_pred: pd.DataFrame, **kwargs) -> float:
|
|
46
|
+
# Specificity is calculated by summing the true negartives and false
|
|
47
|
+
# positives for each class, then using these totals to obtain an overall specificity"
|
|
48
|
+
cm = confusion_matrix(y, y_pred)
|
|
49
|
+
total_tn = 0
|
|
50
|
+
total_fp = 0
|
|
51
|
+
for i in range(len(cm)):
|
|
52
|
+
total_tn += np.sum(cm) - np.sum(cm[i, :]) - np.sum(cm[:, i]) + cm[i, i]
|
|
53
|
+
total_fp += np.sum(cm[:, i]) - cm[i, i]
|
|
54
|
+
|
|
55
|
+
return total_tn / (total_tn + total_fp)
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""[METRIC] Specificity Multilabel"""
|
|
2
|
+
from typing import Any
|
|
3
|
+
import textwrap
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from sklearn.metrics import multilabel_confusion_matrix
|
|
6
|
+
import numpy as np
|
|
7
|
+
from ..metric import Metric
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class SpecificityMultilabelMetric(Metric):
|
|
11
|
+
"""[METRIC] Specificity Multilabel"""
|
|
12
|
+
|
|
13
|
+
name: str = "Specificity Multilabel"
|
|
14
|
+
_description: str = textwrap.dedent('''\
|
|
15
|
+
Multilabel specificity is a metric used to evaluate the performance of a classification model that
|
|
16
|
+
predicts multiple labels for each instance. It measures how well the model identifies negative cases
|
|
17
|
+
for each label individually.''')
|
|
18
|
+
_description_long: str = textwrap.dedent('''\
|
|
19
|
+
Multilabel specificity assesses how effectively a classification model identifies negative cases for
|
|
20
|
+
multiple labels. It calculates specificity for each label separately by determining the true negatives
|
|
21
|
+
and false positives for that label. After calculating the specificity for all labels, these values are
|
|
22
|
+
averaged to obtain an overall measure. A high multilabel specificity indicates that the model is good at
|
|
23
|
+
correctly identifying non-target labels, while a low score suggests it may misclassify negative cases.
|
|
24
|
+
In summary, multilabel specificity helps evaluate a model's ability to accurately recognize negative outcomes
|
|
25
|
+
across various labels.''')
|
|
26
|
+
refs: list[dict[str, Any]] = [
|
|
27
|
+
{
|
|
28
|
+
'year': 2021,
|
|
29
|
+
'name': 'Comprehensive Comparative Study of Multi-Label Classification Methods',
|
|
30
|
+
'authors': [
|
|
31
|
+
'Jasmin Bogatinovski',
|
|
32
|
+
'Ljupčo Todorovski',
|
|
33
|
+
'Sašo Džeroski',
|
|
34
|
+
'Dragi Kocev',
|
|
35
|
+
],
|
|
36
|
+
'doi': 'https://doi.org/10.48550/arXiv.2102.07113',
|
|
37
|
+
'publisher': ''
|
|
38
|
+
}
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
def _str__(self) -> str:
|
|
42
|
+
return 'specificity_multilabel'
|
|
43
|
+
|
|
44
|
+
def suitable(self, X: pd.DataFrame, y: pd.DataFrame, type_of_target: str) -> bool:
|
|
45
|
+
return type_of_target in ['multilabel-indicator']
|
|
46
|
+
|
|
47
|
+
def compute(self, y: pd.DataFrame, y_pred: pd.DataFrame, **kwargs) -> float:
|
|
48
|
+
# Specificity is calculated for each label separately,
|
|
49
|
+
# then averaged to obtain an overall measure.
|
|
50
|
+
# Generates a series of confusion matrices, one for each label
|
|
51
|
+
mcm = multilabel_confusion_matrix(y, y_pred)
|
|
52
|
+
specificity_per_label = []
|
|
53
|
+
for i in range(mcm.shape[0]):
|
|
54
|
+
tn, fp, _, _ = mcm[i].ravel()
|
|
55
|
+
specificity = tn / (tn + fp) if (tn + fp) != 0 else 0
|
|
56
|
+
specificity_per_label.append(specificity)
|
|
57
|
+
|
|
58
|
+
mean_specificity = np.mean(specificity_per_label)
|
|
59
|
+
|
|
60
|
+
return mean_specificity
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Bayesian Optimization-based Optimizer for IAML.
|
|
3
|
+
Optimizer receives a pool of Candidates, optimizes parameters using a surrogate model,
|
|
4
|
+
and returns a new pool of candidates.
|
|
5
|
+
"""
|
|
6
|
+
from copy import deepcopy
|
|
7
|
+
import random
|
|
8
|
+
import numpy as np
|
|
9
|
+
from skopt import Optimizer
|
|
10
|
+
from skopt.space import Real, Integer, Categorical
|
|
11
|
+
from ..candidate import Candidate
|
|
12
|
+
from .optimizer import Optimizer as BaseOptimizer
|
|
13
|
+
from ..step import Step
|
|
14
|
+
from ..logger import Logger
|
|
15
|
+
|
|
16
|
+
class BayesianOptimizer(BaseOptimizer):
|
|
17
|
+
def __init__(self, duration: int = None, max_iterations=50):
|
|
18
|
+
"""
|
|
19
|
+
Initialize the Bayesian Optimization-based Optimizer.
|
|
20
|
+
:param max_iterations: Number of optimization iterations.
|
|
21
|
+
"""
|
|
22
|
+
self.max_iterations = max_iterations
|
|
23
|
+
self.ignored_configs: list[str] = {'random_state'}
|
|
24
|
+
self.current_iteration = 0
|
|
25
|
+
self.max_candidates = 40
|
|
26
|
+
self.skopt_optimizers = {}
|
|
27
|
+
|
|
28
|
+
def _generate_structure_id(self, candidate):
|
|
29
|
+
"""Generate a unique identifier for the candidate structure."""
|
|
30
|
+
return hash(tuple((step[0], tuple(sorted(step[1].configuration.keys()))) for step in candidate.pipeline.steps))
|
|
31
|
+
|
|
32
|
+
def _initialize_search_space(self, candidates):
|
|
33
|
+
"""Define the search space for each unique candidate structure."""
|
|
34
|
+
for candidate in candidates:
|
|
35
|
+
structure_id = self._generate_structure_id(candidate)
|
|
36
|
+
if structure_id in self.skopt_optimizers:
|
|
37
|
+
continue # Avoid reinitializing existing structures
|
|
38
|
+
|
|
39
|
+
dimensions = []
|
|
40
|
+
param_keys = []
|
|
41
|
+
|
|
42
|
+
for step_idx, step in enumerate(candidate.pipeline.steps):
|
|
43
|
+
for key, config in sorted(step[1].configuration.items()):
|
|
44
|
+
if key in self.ignored_configs:
|
|
45
|
+
continue
|
|
46
|
+
param_key = f"{step_idx}_{key}"
|
|
47
|
+
|
|
48
|
+
value = config['value']
|
|
49
|
+
if isinstance(value, (np.integer, np.floating, np.bool_)):
|
|
50
|
+
value = value.item()
|
|
51
|
+
|
|
52
|
+
if 'categorical' in config:
|
|
53
|
+
dimensions.append(Categorical(config['categorical']))
|
|
54
|
+
elif isinstance(value, bool):
|
|
55
|
+
dimensions.append(Categorical([True, False]))
|
|
56
|
+
elif isinstance(value, (int, float)):
|
|
57
|
+
if 'range' in config:
|
|
58
|
+
low, high = config['range']
|
|
59
|
+
# Keep parameters with unbounded ranges fixed.
|
|
60
|
+
if low is None or high is None:
|
|
61
|
+
continue
|
|
62
|
+
if low > high:
|
|
63
|
+
low, high = high, low
|
|
64
|
+
if isinstance(value, float):
|
|
65
|
+
dimensions.append(Real(low, high))
|
|
66
|
+
else:
|
|
67
|
+
if low <= 0:
|
|
68
|
+
low = 1
|
|
69
|
+
dimensions.append(Integer(low, high))
|
|
70
|
+
else:
|
|
71
|
+
if isinstance(value, float):
|
|
72
|
+
dimensions.append(Real(-1e6, 1e6))
|
|
73
|
+
else:
|
|
74
|
+
dimensions.append(Integer(1, 1_000_000)) # TODO Pas de int négatif ?
|
|
75
|
+
else:
|
|
76
|
+
# None and other values without a search domain stay unchanged.
|
|
77
|
+
continue
|
|
78
|
+
param_keys.append(param_key)
|
|
79
|
+
|
|
80
|
+
if not dimensions:
|
|
81
|
+
continue
|
|
82
|
+
Logger().info(f"Initializing Bayesian Optimizer for structure {structure_id} with {len(dimensions)} dimensions")
|
|
83
|
+
self.skopt_optimizers[structure_id] = {'optimizer': Optimizer(dimensions), 'param_keys': param_keys, 'dimensions': dimensions}
|
|
84
|
+
|
|
85
|
+
def _suggest_new_candidates(self, candidates):
|
|
86
|
+
"""Suggest a new set of candidates using Bayesian Optimization."""
|
|
87
|
+
new_candidates = []
|
|
88
|
+
for candidate in candidates:
|
|
89
|
+
structure_id = self._generate_structure_id(candidate)
|
|
90
|
+
optimizer_data = self.skopt_optimizers.get(structure_id)
|
|
91
|
+
if not optimizer_data:
|
|
92
|
+
Logger().warning(f"Missing optimizer for structure {structure_id}")
|
|
93
|
+
continue
|
|
94
|
+
|
|
95
|
+
try:
|
|
96
|
+
new_params = optimizer_data['optimizer'].ask(n_points=1)[0]
|
|
97
|
+
except:
|
|
98
|
+
new_params = None
|
|
99
|
+
|
|
100
|
+
if new_params and len(new_params) != len(optimizer_data['param_keys']):
|
|
101
|
+
Logger().error(f"Parameter mismatch: expected {len(optimizer_data['param_keys'])}, got {len(new_params)}")
|
|
102
|
+
continue
|
|
103
|
+
|
|
104
|
+
new_candidate = deepcopy(candidate)
|
|
105
|
+
idx = 0
|
|
106
|
+
if new_params:
|
|
107
|
+
for step_idx, step in enumerate(new_candidate.pipeline.steps):
|
|
108
|
+
for key, config in sorted(step[1].configuration.items()):
|
|
109
|
+
if key in self.ignored_configs:
|
|
110
|
+
continue
|
|
111
|
+
param_key = f"{step_idx}_{key}"
|
|
112
|
+
if param_key in optimizer_data['param_keys']:
|
|
113
|
+
if isinstance(new_params[idx], (np.integer, np.floating, np.bool_)):
|
|
114
|
+
step[1].configure(key, new_params[idx].item())
|
|
115
|
+
else:
|
|
116
|
+
step[1].configure(key, new_params[idx])
|
|
117
|
+
idx += 1
|
|
118
|
+
new_candidates.append(new_candidate)
|
|
119
|
+
|
|
120
|
+
return new_candidates
|
|
121
|
+
|
|
122
|
+
def _export_params(self, candidate, optimizer_data):
|
|
123
|
+
params = [None]*len(optimizer_data['param_keys'])
|
|
124
|
+
for step_idx, step in enumerate(candidate.pipeline.steps):
|
|
125
|
+
for key, config in sorted(step[1].configuration.items()):
|
|
126
|
+
if key in self.ignored_configs:
|
|
127
|
+
continue
|
|
128
|
+
param_key = f"{step_idx}_{key}"
|
|
129
|
+
if param_key in optimizer_data['param_keys']:
|
|
130
|
+
try:
|
|
131
|
+
param_index = optimizer_data['param_keys'].index(param_key)
|
|
132
|
+
value = config['value']
|
|
133
|
+
if isinstance(value, (np.integer, np.floating, np.bool_)):
|
|
134
|
+
value = value.item()
|
|
135
|
+
|
|
136
|
+
dimension = optimizer_data['dimensions'][param_index]
|
|
137
|
+
if isinstance(dimension, (Integer, Real)) and isinstance(value, (int, float)):
|
|
138
|
+
bounds = dimension.bounds
|
|
139
|
+
value = max(min(value, bounds[1]), bounds[0])
|
|
140
|
+
params[param_index] = value
|
|
141
|
+
except IndexError as e:
|
|
142
|
+
Logger().error(f"IndexError: {str(e)} - param_key: {param_key}, param_keys: {optimizer_data['param_keys']}")
|
|
143
|
+
continue
|
|
144
|
+
return params
|
|
145
|
+
|
|
146
|
+
def run(self, candidates):
|
|
147
|
+
"""
|
|
148
|
+
Perform Bayesian Optimization iteration.
|
|
149
|
+
:param candidates: List of Candidate objects to optimize.
|
|
150
|
+
:return: New list of candidates optimized using Bayesian Optimization.
|
|
151
|
+
"""
|
|
152
|
+
if self.current_iteration == 0:
|
|
153
|
+
self._initialize_search_space(candidates)
|
|
154
|
+
|
|
155
|
+
candidates.sort(reverse=True)
|
|
156
|
+
|
|
157
|
+
for candidate in candidates:
|
|
158
|
+
structure_id = self._generate_structure_id(candidate)
|
|
159
|
+
optimizer_data = self.skopt_optimizers.get(structure_id)
|
|
160
|
+
if not optimizer_data:
|
|
161
|
+
Logger().warning(f"Skipping candidate {structure_id}, missing optimizer")
|
|
162
|
+
continue
|
|
163
|
+
|
|
164
|
+
params = self._export_params(candidate, optimizer_data)
|
|
165
|
+
|
|
166
|
+
main_metric = candidate.get_main_metric_score()
|
|
167
|
+
if isinstance(main_metric, (np.integer, np.floating)):
|
|
168
|
+
main_metric = main_metric.item()
|
|
169
|
+
if not isinstance(main_metric, (int, float)) or not np.isfinite(main_metric):
|
|
170
|
+
Logger().error(f"Invalid main_metric: expected finite scalar, got {type(main_metric)} - {main_metric}")
|
|
171
|
+
continue
|
|
172
|
+
|
|
173
|
+
try:
|
|
174
|
+
# skopt minimizes its objective, while IAML maximizes candidate scores.
|
|
175
|
+
optimizer_data['optimizer'].tell([params], [-main_metric])
|
|
176
|
+
except ValueError as e:
|
|
177
|
+
Logger().error(f"Error in skopt.tell(): {str(e)}. Params: {params}, Bounds: {optimizer_data['optimizer'].space.bounds}")
|
|
178
|
+
continue
|
|
179
|
+
|
|
180
|
+
keep_candidates = candidates[0:self.max_candidates//2]
|
|
181
|
+
new_candidates = self._suggest_new_candidates(keep_candidates)
|
|
182
|
+
self.current_iteration += 1
|
|
183
|
+
|
|
184
|
+
# return self._suggest_new_candidates(candidates) # DEBUG
|
|
185
|
+
return keep_candidates + new_candidates
|
|
186
|
+
|
|
187
|
+
@property
|
|
188
|
+
def finished(self) -> bool:
|
|
189
|
+
"""
|
|
190
|
+
Check if optimization is finished.
|
|
191
|
+
:return: Boolean indicating if optimization is complete.
|
|
192
|
+
"""
|
|
193
|
+
return self.current_iteration >= self.max_iterations
|
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
"""Pipeline optimizer based on genetic concepts"""
|
|
2
|
+
from copy import deepcopy
|
|
3
|
+
import random
|
|
4
|
+
import time
|
|
5
|
+
from typing import Any
|
|
6
|
+
from ..candidate import Candidate
|
|
7
|
+
from .optimizer import Optimizer
|
|
8
|
+
from ..step import Step
|
|
9
|
+
from ..void_step import VoidStep
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class GeneticOptimizer(Optimizer): # pylint: disable=too-many-instance-attributes
|
|
13
|
+
"""Pipeline optimizer based on genetic concepts
|
|
14
|
+
|
|
15
|
+
:param int, optional nb_candidate: Number of candidates to optimize. Default to 35.
|
|
16
|
+
:param float, optional mutation_power: Chance of a mutation hapenning. Default to 0.1.
|
|
17
|
+
:param float, optional initial_modifier: Maximum modification value possible. Default to 5.
|
|
18
|
+
:param int, optional duration: Used to compute a mutation ratio. Default to None.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def __init__(
|
|
22
|
+
self,
|
|
23
|
+
nb_candidate: int = 35,
|
|
24
|
+
mutation_power: float = 0.1,
|
|
25
|
+
initial_modifier: float = 5,
|
|
26
|
+
duration: int = None) -> None:
|
|
27
|
+
super().__init__()
|
|
28
|
+
self.number_of_candidate: int = max(nb_candidate, 4)
|
|
29
|
+
"""Maxmimum number of candicates"""
|
|
30
|
+
|
|
31
|
+
self.generation_count: int = 0
|
|
32
|
+
"""Generation coutner"""
|
|
33
|
+
|
|
34
|
+
self.ignored_configs: set[str] = {'random_state'}
|
|
35
|
+
"""Set of config keys to ignore"""
|
|
36
|
+
|
|
37
|
+
self.mutation_power: float = mutation_power
|
|
38
|
+
"""Chance of a mutation hapenning"""
|
|
39
|
+
|
|
40
|
+
self.initial_modifier: float = initial_modifier
|
|
41
|
+
"""Maximum modification value possible"""
|
|
42
|
+
|
|
43
|
+
self.max_generations: int = 200
|
|
44
|
+
"""Maximum number of generations"""
|
|
45
|
+
|
|
46
|
+
self.first_candidate_pool: list[Candidate] = None
|
|
47
|
+
"""List of candidates for the first generation"""
|
|
48
|
+
|
|
49
|
+
self.duration: int = duration
|
|
50
|
+
"""Used to compute a mutation ratio"""
|
|
51
|
+
|
|
52
|
+
self.start_time: float = time.time()
|
|
53
|
+
"""Starting time of the first generation"""
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def __mutate_ratio(self) -> float:
|
|
57
|
+
"""Define a mutation ratio"""
|
|
58
|
+
if not self.duration:
|
|
59
|
+
return 0.5
|
|
60
|
+
return min(0.9, max(0.1, ((time.time() - self.start_time) / self.duration)))
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def finished(self) -> bool:
|
|
64
|
+
"""Is optimization finished ?
|
|
65
|
+
|
|
66
|
+
:return: finished ?
|
|
67
|
+
"""
|
|
68
|
+
return self.generation_count >= self.max_generations
|
|
69
|
+
|
|
70
|
+
def run(self, candidates: list[Candidate]) -> list[Candidate]:
|
|
71
|
+
"""Run one optimisation stage. Run of Optimzer have to be overwrite (do nothing).
|
|
72
|
+
|
|
73
|
+
:param list[Candidate] candidates: List of candidates to optimize
|
|
74
|
+
:return: Optimized candidates
|
|
75
|
+
"""
|
|
76
|
+
candidates.sort(reverse=True)
|
|
77
|
+
|
|
78
|
+
# Will be used for random generation
|
|
79
|
+
if self.first_candidate_pool is None:
|
|
80
|
+
self.first_candidate_pool = candidates[0:6]
|
|
81
|
+
|
|
82
|
+
nb_to_keep: int = round(self.number_of_candidate / 4)
|
|
83
|
+
# nb_to_keep: int = max(min(4, round(self.number_of_candidate / 4)), 1)
|
|
84
|
+
mutate_ratio = self.__mutate_ratio
|
|
85
|
+
self.generation_count += 1
|
|
86
|
+
|
|
87
|
+
# Keep best pipelines
|
|
88
|
+
new_generation = [deepcopy(candidate) for candidate in candidates[0:nb_to_keep]]
|
|
89
|
+
|
|
90
|
+
for _ in range(int(self.number_of_candidate*mutate_ratio)):
|
|
91
|
+
new_generation.append(self.__mutate(random.choice(candidates[0:nb_to_keep])))
|
|
92
|
+
|
|
93
|
+
new_generation = [item for item in new_generation if item is not None] # remove None
|
|
94
|
+
|
|
95
|
+
while len(new_generation) < self.number_of_candidate:
|
|
96
|
+
new_generation.append(
|
|
97
|
+
self.__random_configuration(random.choice(candidates))
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
new_generation = [item for item in new_generation if item is not None] # remove None
|
|
101
|
+
|
|
102
|
+
return self.__unique(new_generation) # Remove duplicated
|
|
103
|
+
|
|
104
|
+
def __random_configuration(self, candidate: Candidate) -> Candidate:
|
|
105
|
+
"""From a candidate generate a new one with a full random configuration
|
|
106
|
+
|
|
107
|
+
:param Candidate candidate: Candidate used to generate a new one
|
|
108
|
+
:return: Newly generated Candidate
|
|
109
|
+
"""
|
|
110
|
+
# Deepcopy to avoid editing other Steps of the same generation
|
|
111
|
+
new_candidate: Candidate = deepcopy(candidate)
|
|
112
|
+
|
|
113
|
+
for current_step in new_candidate.pipeline.optimizable_step:
|
|
114
|
+
|
|
115
|
+
# If interchangeable -> 1/2 to change the step
|
|
116
|
+
if current_step.is_interchangeable and bool(random.getrandbits(1)):
|
|
117
|
+
new_step = self.__interchange(current_step)
|
|
118
|
+
new_candidate.pipeline.replace_step(current_step, new_step)
|
|
119
|
+
current_step = new_step
|
|
120
|
+
if not self.__config_keys(current_step):
|
|
121
|
+
continue # Nothing to optimize
|
|
122
|
+
|
|
123
|
+
# For each configuration key, we'll choose a random value
|
|
124
|
+
for key in self.__config_keys(current_step):
|
|
125
|
+
|
|
126
|
+
# 1/2 chance to let the default value unchanged
|
|
127
|
+
if bool(random.getrandbits(1)):
|
|
128
|
+
continue
|
|
129
|
+
|
|
130
|
+
config = current_step.configuration[key]
|
|
131
|
+
|
|
132
|
+
if type(config['value']) in [int, float]: # Numeric value ? Let's apply multiplier
|
|
133
|
+
is_int = isinstance(config['value'], int)
|
|
134
|
+
|
|
135
|
+
new_value = None
|
|
136
|
+
if 'range' in config: # Random in range
|
|
137
|
+
new_value = random.uniform(*config['range'])
|
|
138
|
+
else: # Kind of strong mutate
|
|
139
|
+
# Randomly choose a positive or negative editing
|
|
140
|
+
if bool(random.getrandbits(1)):
|
|
141
|
+
# Negative -> Multiply value by something between 0.01 and 1
|
|
142
|
+
change_rate = random.uniform(0.01, 1)
|
|
143
|
+
new_value = config['value']*change_rate
|
|
144
|
+
else:
|
|
145
|
+
# Positive -> Multiply value by something between 1
|
|
146
|
+
# and the max modificator in configuration
|
|
147
|
+
change_rate = random.uniform(1, self.initial_modifier)
|
|
148
|
+
new_value = config['value']*change_rate
|
|
149
|
+
|
|
150
|
+
# Value was a int ? Round it to keep it int
|
|
151
|
+
if is_int:
|
|
152
|
+
new_value = round(new_value)
|
|
153
|
+
|
|
154
|
+
if not self.__valide_config(config, new_value):
|
|
155
|
+
# Cancel is the new value is not correct.
|
|
156
|
+
new_value = config['value']
|
|
157
|
+
|
|
158
|
+
# Categorical value, choose randomly one of them
|
|
159
|
+
elif 'categorical' in config.keys():
|
|
160
|
+
new_value = random.choice(config['categorical'])
|
|
161
|
+
elif isinstance(config['value'], bool):
|
|
162
|
+
# Bool value, choose randomly beetwen True and False
|
|
163
|
+
new_value = random.choice([True, False])
|
|
164
|
+
else: # Other value ? Just keep it
|
|
165
|
+
new_value = config['value']
|
|
166
|
+
|
|
167
|
+
current_step.configure(key, new_value) # Set new configuration in the step
|
|
168
|
+
|
|
169
|
+
return new_candidate
|
|
170
|
+
|
|
171
|
+
def __mutate(self, candidate: Candidate) -> Candidate:
|
|
172
|
+
"""Mutate a candidate into a new one
|
|
173
|
+
|
|
174
|
+
:param Candidate candidate: Candidate used for the mutation.
|
|
175
|
+
:return: Newly created Candidate.
|
|
176
|
+
"""
|
|
177
|
+
new_candidate: Candidate = deepcopy(candidate)
|
|
178
|
+
steps = new_candidate.pipeline.optimizable_step
|
|
179
|
+
if not steps:
|
|
180
|
+
return None
|
|
181
|
+
step_to_mutate: Step = random.choice(steps)
|
|
182
|
+
|
|
183
|
+
mutable_keys = self.__config_keys(step_to_mutate)
|
|
184
|
+
if step_to_mutate.is_interchangeable:
|
|
185
|
+
mutable_keys.append("interchange")
|
|
186
|
+
|
|
187
|
+
if not mutable_keys:
|
|
188
|
+
return None # Nothing to optimize
|
|
189
|
+
|
|
190
|
+
# Choose a random key to mutate
|
|
191
|
+
random_key: str = random.choice(mutable_keys)
|
|
192
|
+
|
|
193
|
+
if random_key == "interchange": # Mutate by interchanging the step with sibling
|
|
194
|
+
new_step = self.__interchange(step_to_mutate)
|
|
195
|
+
new_candidate.pipeline.replace_step(step_to_mutate, new_step)
|
|
196
|
+
return new_candidate
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
random_item: dict = step_to_mutate.configuration[random_key] # Get value of the random key
|
|
200
|
+
new_value = None
|
|
201
|
+
|
|
202
|
+
if type(random_item['value']) in [int, float]: # Numeric value ? Apply multiplier
|
|
203
|
+
is_int = isinstance(random_item['value'], int)
|
|
204
|
+
|
|
205
|
+
# Find a multiplier between - mutation_power & + mutation_power
|
|
206
|
+
change_rate = random.uniform(-self.mutation_power, self.mutation_power)
|
|
207
|
+
new_value = random_item['value']*(1+change_rate) # Apply random multiplier
|
|
208
|
+
|
|
209
|
+
# Value was a int ? Round it to keep it int
|
|
210
|
+
if is_int:
|
|
211
|
+
new_value = round(new_value)
|
|
212
|
+
|
|
213
|
+
if new_value == random_item['value']: # To be sure there is a mutation
|
|
214
|
+
new_value += random.choice([-1, 1])
|
|
215
|
+
|
|
216
|
+
if not self.__valide_config(random_item, new_value):
|
|
217
|
+
# Cancel if the new value is not correct.
|
|
218
|
+
new_value = random_item['value']
|
|
219
|
+
|
|
220
|
+
elif 'categorical' in random_item.keys(): # Categorial -> Choose one
|
|
221
|
+
new_value = random.choice(random_item['categorical'])
|
|
222
|
+
elif isinstance(random_item['value'], bool): # Bool -> Choose between True and False
|
|
223
|
+
new_value = random.choice([True, False])
|
|
224
|
+
else: # Other -> Keep it
|
|
225
|
+
new_value = random_item['value']
|
|
226
|
+
|
|
227
|
+
step_to_mutate.configure(random_key, new_value) # Apply configuration
|
|
228
|
+
|
|
229
|
+
return new_candidate
|
|
230
|
+
|
|
231
|
+
@staticmethod
|
|
232
|
+
def __interchange(step: Step) -> Step:
|
|
233
|
+
"""Choose another implementation, allowing optional preprocessing to be removed."""
|
|
234
|
+
choices = [sibling for sibling in step.step_with_same_tags()
|
|
235
|
+
if sibling is not type(step)]
|
|
236
|
+
if (not isinstance(step, VoidStep) and step.can_be_disabled
|
|
237
|
+
and not hasattr(step, 'predict')
|
|
238
|
+
and (hasattr(step, 'transform') or hasattr(step, 'resample'))):
|
|
239
|
+
choices.append(VoidStep)
|
|
240
|
+
if not choices:
|
|
241
|
+
return step
|
|
242
|
+
step_class = random.choice(choices)
|
|
243
|
+
if step_class is VoidStep:
|
|
244
|
+
replacement = VoidStep(step_to_mimic=type(step)())
|
|
245
|
+
else:
|
|
246
|
+
replacement = step_class()
|
|
247
|
+
replacement.is_interchangeable = True
|
|
248
|
+
return replacement
|
|
249
|
+
|
|
250
|
+
def __config_keys(self, step: Step) -> list[str]:
|
|
251
|
+
"""Get a list of config keys used for a Step
|
|
252
|
+
|
|
253
|
+
:param Step step: The step we want the config
|
|
254
|
+
:return: List of config keys for this step
|
|
255
|
+
"""
|
|
256
|
+
if not step.optimizable:
|
|
257
|
+
return []
|
|
258
|
+
return list(set(step.configuration.keys()) - self.ignored_configs)
|
|
259
|
+
|
|
260
|
+
def __valide_config(self, config: dict, value: Any) -> bool:
|
|
261
|
+
"""Is the configuration range valid ?
|
|
262
|
+
|
|
263
|
+
:param dict config: The config to validate.
|
|
264
|
+
:return: Valid ?
|
|
265
|
+
"""
|
|
266
|
+
if 'range' not in config.keys():
|
|
267
|
+
return True
|
|
268
|
+
return config['range'][0] <= value <= config['range'][1]
|
|
269
|
+
|
|
270
|
+
def __unique(self, candidates: list[Candidate]) -> list[Candidate]:
|
|
271
|
+
"""Get a list of unique Candidates
|
|
272
|
+
|
|
273
|
+
:param list[Candidate] candidates: List of Candidate.
|
|
274
|
+
:return: List of unique Candidate
|
|
275
|
+
"""
|
|
276
|
+
unique_candidates: list[Candidate] = []
|
|
277
|
+
unique_fingerprint: list[str] = []
|
|
278
|
+
for candidate in candidates:
|
|
279
|
+
fingerprint: str = candidate.pipeline.fingerprint()
|
|
280
|
+
if fingerprint not in unique_fingerprint:
|
|
281
|
+
unique_candidates.append(candidate)
|
|
282
|
+
unique_fingerprint.append(fingerprint)
|
|
283
|
+
|
|
284
|
+
return unique_candidates
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
|
|
2
|
+
"""Base class of IAML Optimizer. Optimizer receive a pool of Candidates,
|
|
3
|
+
optimize parameters and return a new pool of candidate
|
|
4
|
+
"""
|
|
5
|
+
from ..candidate import Candidate
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class Optimizer:
|
|
9
|
+
"""Base class of IAML Optimizer. Optimizer receive a pool of Candidates,
|
|
10
|
+
optimize parameters and return a new pool of candidate
|
|
11
|
+
"""
|
|
12
|
+
def __init__(self):
|
|
13
|
+
self.__finished: bool = False
|
|
14
|
+
"""Is the optimization done ?"""
|
|
15
|
+
|
|
16
|
+
@property
|
|
17
|
+
def finished(self) -> bool:
|
|
18
|
+
"""Is the optimization finished ?
|
|
19
|
+
|
|
20
|
+
:return: Finished ?
|
|
21
|
+
"""
|
|
22
|
+
return self.__finished
|
|
23
|
+
|
|
24
|
+
def run(self, candidates: list[Candidate]) -> list[Candidate]:
|
|
25
|
+
"""Run one optimisation stage. Run of Optimzer have to be overwrite (do nothing).
|
|
26
|
+
|
|
27
|
+
:param list[Candidate] candidates: List of candidates to optimize
|
|
28
|
+
:return: Optimized candidates
|
|
29
|
+
"""
|
|
30
|
+
self.__finished = True
|
|
31
|
+
return candidates
|