PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,44 @@
1
+ """[METRIC] Specificity"""
2
+ from typing import Any
3
+ import textwrap
4
+ import pandas as pd
5
+ from sklearn.metrics import confusion_matrix
6
+ from ..metric import Metric
7
+
8
+ class SpecificityMetric(Metric):
9
+ """[METRIC] Specificity"""
10
+
11
+ name: str = "Specificity"
12
+ _description: str = textwrap.dedent('''\
13
+ Specificity is a metric used to evaluate the performance of a classification model.
14
+ It measures the proportion of true negatives correctly identified by the model, indicating
15
+ how well it can identify the negative class.''')
16
+ _description_long: str = textwrap.dedent('''\
17
+ Specificity is a metric that helps assess how well a classification model identifies the negative class.
18
+ For example, if you're predicting whether a medical test result is negative for a disease, specificity tells you the
19
+ percentage of actual negative cases that the model correctly identifies. A high specificity means the model is good at
20
+ avoiding false positives, while a low specificity indicates it may incorrectly label negative cases as positive.
21
+ ''')
22
+ refs: list[dict[str, Any]]=[
23
+ {
24
+ 'year': 1994 ,
25
+ 'name': 'Diagnostic tests. 1: Sensitivity and specificity.',
26
+ 'authors': [
27
+ 'D. G. Altman',
28
+ 'J. M. Bland'
29
+ ],
30
+ 'doi': 'https://doi.org/10.1136%2Fbmj.308.6943.1552',
31
+ 'publisher': 'BMJ. 308 (6943): 1552'
32
+ }
33
+ ]
34
+
35
+ def __str__(self) -> str:
36
+ return 'specificity'
37
+
38
+ def suitable(self, X: pd.DataFrame, y: pd.DataFrame, type_of_target: str) -> bool:
39
+ return type_of_target == 'binary'
40
+
41
+ def compute(self, y: pd.DataFrame, y_pred: pd.DataFrame, **kwargs) -> float:
42
+ tn, fp, _, _ = confusion_matrix(y, y_pred).ravel()
43
+
44
+ return tn / (tn + fp) if (tn + fp) != 0 else 0
@@ -0,0 +1,55 @@
1
+ """[METRIC] Specificity Multiclass"""
2
+ from typing import Any
3
+ import textwrap
4
+ import pandas as pd
5
+ from sklearn.metrics import confusion_matrix
6
+ import numpy as np
7
+ from ..metric import Metric
8
+
9
+
10
+ class SpecificityMulticlassMetric(Metric):
11
+ """[METRIC] Specificity Multiclass"""
12
+
13
+ name: str = "Specificity Multiclass"
14
+ _description: str = textwrap.dedent('''\
15
+ Multiclass specificity is a metric used to evaluate the performance of a classification model
16
+ with multiple classes. It measures how well the model identifies the negative cases for each
17
+ class by considering true negatives and false positives.''')
18
+ _description_long: str = textwrap.dedent('''\
19
+ Multiclass specificity assesses how effectively a classification model identifies negative cases
20
+ across multiple classes. For each class, it calculates the number of true negatives (correctly
21
+ identified negatives) and false positives (incorrectly identified positives).
22
+ By summing these values for all classes, you can determine an overall specificity score.
23
+ A high multiclass specificity indicates that the model is good at correctly identifying non-target classes,
24
+ while a low score suggests it may struggle with misclassifying negative cases.''')
25
+ refs: list[dict[str, Any]] = [
26
+ {
27
+ 'year': 2020,
28
+ 'name': 'Metrics for Multi-Class Classification: an Overview',
29
+ 'authors': [
30
+ 'Margherita Grandini',
31
+ 'Enrico Bagli',
32
+ 'Giorgio Visani'
33
+ ],
34
+ 'doi': 'https://doi.org/10.48550/arXiv.2008.05756',
35
+ 'publisher': ''
36
+ }
37
+ ]
38
+
39
+ def __str__(self) -> str:
40
+ return 'specificity_multiclass'
41
+
42
+ def suitable(self, X: pd.DataFrame, y: pd.DataFrame, type_of_target: str) -> bool:
43
+ return type_of_target == 'multiclass'
44
+
45
+ def compute(self, y: pd.DataFrame, y_pred: pd.DataFrame, **kwargs) -> float:
46
+ # Specificity is calculated by summing the true negartives and false
47
+ # positives for each class, then using these totals to obtain an overall specificity"
48
+ cm = confusion_matrix(y, y_pred)
49
+ total_tn = 0
50
+ total_fp = 0
51
+ for i in range(len(cm)):
52
+ total_tn += np.sum(cm) - np.sum(cm[i, :]) - np.sum(cm[:, i]) + cm[i, i]
53
+ total_fp += np.sum(cm[:, i]) - cm[i, i]
54
+
55
+ return total_tn / (total_tn + total_fp)
@@ -0,0 +1,60 @@
1
+ """[METRIC] Specificity Multilabel"""
2
+ from typing import Any
3
+ import textwrap
4
+ import pandas as pd
5
+ from sklearn.metrics import multilabel_confusion_matrix
6
+ import numpy as np
7
+ from ..metric import Metric
8
+
9
+
10
+ class SpecificityMultilabelMetric(Metric):
11
+ """[METRIC] Specificity Multilabel"""
12
+
13
+ name: str = "Specificity Multilabel"
14
+ _description: str = textwrap.dedent('''\
15
+ Multilabel specificity is a metric used to evaluate the performance of a classification model that
16
+ predicts multiple labels for each instance. It measures how well the model identifies negative cases
17
+ for each label individually.''')
18
+ _description_long: str = textwrap.dedent('''\
19
+ Multilabel specificity assesses how effectively a classification model identifies negative cases for
20
+ multiple labels. It calculates specificity for each label separately by determining the true negatives
21
+ and false positives for that label. After calculating the specificity for all labels, these values are
22
+ averaged to obtain an overall measure. A high multilabel specificity indicates that the model is good at
23
+ correctly identifying non-target labels, while a low score suggests it may misclassify negative cases.
24
+ In summary, multilabel specificity helps evaluate a model's ability to accurately recognize negative outcomes
25
+ across various labels.''')
26
+ refs: list[dict[str, Any]] = [
27
+ {
28
+ 'year': 2021,
29
+ 'name': 'Comprehensive Comparative Study of Multi-Label Classification Methods',
30
+ 'authors': [
31
+ 'Jasmin Bogatinovski',
32
+ 'Ljupčo Todorovski',
33
+ 'Sašo Džeroski',
34
+ 'Dragi Kocev',
35
+ ],
36
+ 'doi': 'https://doi.org/10.48550/arXiv.2102.07113',
37
+ 'publisher': ''
38
+ }
39
+ ]
40
+
41
+ def _str__(self) -> str:
42
+ return 'specificity_multilabel'
43
+
44
+ def suitable(self, X: pd.DataFrame, y: pd.DataFrame, type_of_target: str) -> bool:
45
+ return type_of_target in ['multilabel-indicator']
46
+
47
+ def compute(self, y: pd.DataFrame, y_pred: pd.DataFrame, **kwargs) -> float:
48
+ # Specificity is calculated for each label separately,
49
+ # then averaged to obtain an overall measure.
50
+ # Generates a series of confusion matrices, one for each label
51
+ mcm = multilabel_confusion_matrix(y, y_pred)
52
+ specificity_per_label = []
53
+ for i in range(mcm.shape[0]):
54
+ tn, fp, _, _ = mcm[i].ravel()
55
+ specificity = tn / (tn + fp) if (tn + fp) != 0 else 0
56
+ specificity_per_label.append(specificity)
57
+
58
+ mean_specificity = np.mean(specificity_per_label)
59
+
60
+ return mean_specificity
@@ -0,0 +1,5 @@
1
+ """All IAML optimizers"""
2
+ from .optimizer import Optimizer
3
+ from .genetic_optimizer import GeneticOptimizer
4
+ from .bayesian_optimizer import BayesianOptimizer
5
+ from .random_optimizer import RandomOptimizer
@@ -0,0 +1,193 @@
1
+ """
2
+ Bayesian Optimization-based Optimizer for IAML.
3
+ Optimizer receives a pool of Candidates, optimizes parameters using a surrogate model,
4
+ and returns a new pool of candidates.
5
+ """
6
+ from copy import deepcopy
7
+ import random
8
+ import numpy as np
9
+ from skopt import Optimizer
10
+ from skopt.space import Real, Integer, Categorical
11
+ from ..candidate import Candidate
12
+ from .optimizer import Optimizer as BaseOptimizer
13
+ from ..step import Step
14
+ from ..logger import Logger
15
+
16
+ class BayesianOptimizer(BaseOptimizer):
17
+ def __init__(self, duration: int = None, max_iterations=50):
18
+ """
19
+ Initialize the Bayesian Optimization-based Optimizer.
20
+ :param max_iterations: Number of optimization iterations.
21
+ """
22
+ self.max_iterations = max_iterations
23
+ self.ignored_configs: list[str] = {'random_state'}
24
+ self.current_iteration = 0
25
+ self.max_candidates = 40
26
+ self.skopt_optimizers = {}
27
+
28
+ def _generate_structure_id(self, candidate):
29
+ """Generate a unique identifier for the candidate structure."""
30
+ return hash(tuple((step[0], tuple(sorted(step[1].configuration.keys()))) for step in candidate.pipeline.steps))
31
+
32
+ def _initialize_search_space(self, candidates):
33
+ """Define the search space for each unique candidate structure."""
34
+ for candidate in candidates:
35
+ structure_id = self._generate_structure_id(candidate)
36
+ if structure_id in self.skopt_optimizers:
37
+ continue # Avoid reinitializing existing structures
38
+
39
+ dimensions = []
40
+ param_keys = []
41
+
42
+ for step_idx, step in enumerate(candidate.pipeline.steps):
43
+ for key, config in sorted(step[1].configuration.items()):
44
+ if key in self.ignored_configs:
45
+ continue
46
+ param_key = f"{step_idx}_{key}"
47
+
48
+ value = config['value']
49
+ if isinstance(value, (np.integer, np.floating, np.bool_)):
50
+ value = value.item()
51
+
52
+ if 'categorical' in config:
53
+ dimensions.append(Categorical(config['categorical']))
54
+ elif isinstance(value, bool):
55
+ dimensions.append(Categorical([True, False]))
56
+ elif isinstance(value, (int, float)):
57
+ if 'range' in config:
58
+ low, high = config['range']
59
+ # Keep parameters with unbounded ranges fixed.
60
+ if low is None or high is None:
61
+ continue
62
+ if low > high:
63
+ low, high = high, low
64
+ if isinstance(value, float):
65
+ dimensions.append(Real(low, high))
66
+ else:
67
+ if low <= 0:
68
+ low = 1
69
+ dimensions.append(Integer(low, high))
70
+ else:
71
+ if isinstance(value, float):
72
+ dimensions.append(Real(-1e6, 1e6))
73
+ else:
74
+ dimensions.append(Integer(1, 1_000_000)) # TODO Pas de int négatif ?
75
+ else:
76
+ # None and other values without a search domain stay unchanged.
77
+ continue
78
+ param_keys.append(param_key)
79
+
80
+ if not dimensions:
81
+ continue
82
+ Logger().info(f"Initializing Bayesian Optimizer for structure {structure_id} with {len(dimensions)} dimensions")
83
+ self.skopt_optimizers[structure_id] = {'optimizer': Optimizer(dimensions), 'param_keys': param_keys, 'dimensions': dimensions}
84
+
85
+ def _suggest_new_candidates(self, candidates):
86
+ """Suggest a new set of candidates using Bayesian Optimization."""
87
+ new_candidates = []
88
+ for candidate in candidates:
89
+ structure_id = self._generate_structure_id(candidate)
90
+ optimizer_data = self.skopt_optimizers.get(structure_id)
91
+ if not optimizer_data:
92
+ Logger().warning(f"Missing optimizer for structure {structure_id}")
93
+ continue
94
+
95
+ try:
96
+ new_params = optimizer_data['optimizer'].ask(n_points=1)[0]
97
+ except:
98
+ new_params = None
99
+
100
+ if new_params and len(new_params) != len(optimizer_data['param_keys']):
101
+ Logger().error(f"Parameter mismatch: expected {len(optimizer_data['param_keys'])}, got {len(new_params)}")
102
+ continue
103
+
104
+ new_candidate = deepcopy(candidate)
105
+ idx = 0
106
+ if new_params:
107
+ for step_idx, step in enumerate(new_candidate.pipeline.steps):
108
+ for key, config in sorted(step[1].configuration.items()):
109
+ if key in self.ignored_configs:
110
+ continue
111
+ param_key = f"{step_idx}_{key}"
112
+ if param_key in optimizer_data['param_keys']:
113
+ if isinstance(new_params[idx], (np.integer, np.floating, np.bool_)):
114
+ step[1].configure(key, new_params[idx].item())
115
+ else:
116
+ step[1].configure(key, new_params[idx])
117
+ idx += 1
118
+ new_candidates.append(new_candidate)
119
+
120
+ return new_candidates
121
+
122
+ def _export_params(self, candidate, optimizer_data):
123
+ params = [None]*len(optimizer_data['param_keys'])
124
+ for step_idx, step in enumerate(candidate.pipeline.steps):
125
+ for key, config in sorted(step[1].configuration.items()):
126
+ if key in self.ignored_configs:
127
+ continue
128
+ param_key = f"{step_idx}_{key}"
129
+ if param_key in optimizer_data['param_keys']:
130
+ try:
131
+ param_index = optimizer_data['param_keys'].index(param_key)
132
+ value = config['value']
133
+ if isinstance(value, (np.integer, np.floating, np.bool_)):
134
+ value = value.item()
135
+
136
+ dimension = optimizer_data['dimensions'][param_index]
137
+ if isinstance(dimension, (Integer, Real)) and isinstance(value, (int, float)):
138
+ bounds = dimension.bounds
139
+ value = max(min(value, bounds[1]), bounds[0])
140
+ params[param_index] = value
141
+ except IndexError as e:
142
+ Logger().error(f"IndexError: {str(e)} - param_key: {param_key}, param_keys: {optimizer_data['param_keys']}")
143
+ continue
144
+ return params
145
+
146
+ def run(self, candidates):
147
+ """
148
+ Perform Bayesian Optimization iteration.
149
+ :param candidates: List of Candidate objects to optimize.
150
+ :return: New list of candidates optimized using Bayesian Optimization.
151
+ """
152
+ if self.current_iteration == 0:
153
+ self._initialize_search_space(candidates)
154
+
155
+ candidates.sort(reverse=True)
156
+
157
+ for candidate in candidates:
158
+ structure_id = self._generate_structure_id(candidate)
159
+ optimizer_data = self.skopt_optimizers.get(structure_id)
160
+ if not optimizer_data:
161
+ Logger().warning(f"Skipping candidate {structure_id}, missing optimizer")
162
+ continue
163
+
164
+ params = self._export_params(candidate, optimizer_data)
165
+
166
+ main_metric = candidate.get_main_metric_score()
167
+ if isinstance(main_metric, (np.integer, np.floating)):
168
+ main_metric = main_metric.item()
169
+ if not isinstance(main_metric, (int, float)) or not np.isfinite(main_metric):
170
+ Logger().error(f"Invalid main_metric: expected finite scalar, got {type(main_metric)} - {main_metric}")
171
+ continue
172
+
173
+ try:
174
+ # skopt minimizes its objective, while IAML maximizes candidate scores.
175
+ optimizer_data['optimizer'].tell([params], [-main_metric])
176
+ except ValueError as e:
177
+ Logger().error(f"Error in skopt.tell(): {str(e)}. Params: {params}, Bounds: {optimizer_data['optimizer'].space.bounds}")
178
+ continue
179
+
180
+ keep_candidates = candidates[0:self.max_candidates//2]
181
+ new_candidates = self._suggest_new_candidates(keep_candidates)
182
+ self.current_iteration += 1
183
+
184
+ # return self._suggest_new_candidates(candidates) # DEBUG
185
+ return keep_candidates + new_candidates
186
+
187
+ @property
188
+ def finished(self) -> bool:
189
+ """
190
+ Check if optimization is finished.
191
+ :return: Boolean indicating if optimization is complete.
192
+ """
193
+ return self.current_iteration >= self.max_iterations
@@ -0,0 +1,284 @@
1
+ """Pipeline optimizer based on genetic concepts"""
2
+ from copy import deepcopy
3
+ import random
4
+ import time
5
+ from typing import Any
6
+ from ..candidate import Candidate
7
+ from .optimizer import Optimizer
8
+ from ..step import Step
9
+ from ..void_step import VoidStep
10
+
11
+
12
+ class GeneticOptimizer(Optimizer): # pylint: disable=too-many-instance-attributes
13
+ """Pipeline optimizer based on genetic concepts
14
+
15
+ :param int, optional nb_candidate: Number of candidates to optimize. Default to 35.
16
+ :param float, optional mutation_power: Chance of a mutation hapenning. Default to 0.1.
17
+ :param float, optional initial_modifier: Maximum modification value possible. Default to 5.
18
+ :param int, optional duration: Used to compute a mutation ratio. Default to None.
19
+ """
20
+
21
+ def __init__(
22
+ self,
23
+ nb_candidate: int = 35,
24
+ mutation_power: float = 0.1,
25
+ initial_modifier: float = 5,
26
+ duration: int = None) -> None:
27
+ super().__init__()
28
+ self.number_of_candidate: int = max(nb_candidate, 4)
29
+ """Maxmimum number of candicates"""
30
+
31
+ self.generation_count: int = 0
32
+ """Generation coutner"""
33
+
34
+ self.ignored_configs: set[str] = {'random_state'}
35
+ """Set of config keys to ignore"""
36
+
37
+ self.mutation_power: float = mutation_power
38
+ """Chance of a mutation hapenning"""
39
+
40
+ self.initial_modifier: float = initial_modifier
41
+ """Maximum modification value possible"""
42
+
43
+ self.max_generations: int = 200
44
+ """Maximum number of generations"""
45
+
46
+ self.first_candidate_pool: list[Candidate] = None
47
+ """List of candidates for the first generation"""
48
+
49
+ self.duration: int = duration
50
+ """Used to compute a mutation ratio"""
51
+
52
+ self.start_time: float = time.time()
53
+ """Starting time of the first generation"""
54
+
55
+ @property
56
+ def __mutate_ratio(self) -> float:
57
+ """Define a mutation ratio"""
58
+ if not self.duration:
59
+ return 0.5
60
+ return min(0.9, max(0.1, ((time.time() - self.start_time) / self.duration)))
61
+
62
+ @property
63
+ def finished(self) -> bool:
64
+ """Is optimization finished ?
65
+
66
+ :return: finished ?
67
+ """
68
+ return self.generation_count >= self.max_generations
69
+
70
+ def run(self, candidates: list[Candidate]) -> list[Candidate]:
71
+ """Run one optimisation stage. Run of Optimzer have to be overwrite (do nothing).
72
+
73
+ :param list[Candidate] candidates: List of candidates to optimize
74
+ :return: Optimized candidates
75
+ """
76
+ candidates.sort(reverse=True)
77
+
78
+ # Will be used for random generation
79
+ if self.first_candidate_pool is None:
80
+ self.first_candidate_pool = candidates[0:6]
81
+
82
+ nb_to_keep: int = round(self.number_of_candidate / 4)
83
+ # nb_to_keep: int = max(min(4, round(self.number_of_candidate / 4)), 1)
84
+ mutate_ratio = self.__mutate_ratio
85
+ self.generation_count += 1
86
+
87
+ # Keep best pipelines
88
+ new_generation = [deepcopy(candidate) for candidate in candidates[0:nb_to_keep]]
89
+
90
+ for _ in range(int(self.number_of_candidate*mutate_ratio)):
91
+ new_generation.append(self.__mutate(random.choice(candidates[0:nb_to_keep])))
92
+
93
+ new_generation = [item for item in new_generation if item is not None] # remove None
94
+
95
+ while len(new_generation) < self.number_of_candidate:
96
+ new_generation.append(
97
+ self.__random_configuration(random.choice(candidates))
98
+ )
99
+
100
+ new_generation = [item for item in new_generation if item is not None] # remove None
101
+
102
+ return self.__unique(new_generation) # Remove duplicated
103
+
104
+ def __random_configuration(self, candidate: Candidate) -> Candidate:
105
+ """From a candidate generate a new one with a full random configuration
106
+
107
+ :param Candidate candidate: Candidate used to generate a new one
108
+ :return: Newly generated Candidate
109
+ """
110
+ # Deepcopy to avoid editing other Steps of the same generation
111
+ new_candidate: Candidate = deepcopy(candidate)
112
+
113
+ for current_step in new_candidate.pipeline.optimizable_step:
114
+
115
+ # If interchangeable -> 1/2 to change the step
116
+ if current_step.is_interchangeable and bool(random.getrandbits(1)):
117
+ new_step = self.__interchange(current_step)
118
+ new_candidate.pipeline.replace_step(current_step, new_step)
119
+ current_step = new_step
120
+ if not self.__config_keys(current_step):
121
+ continue # Nothing to optimize
122
+
123
+ # For each configuration key, we'll choose a random value
124
+ for key in self.__config_keys(current_step):
125
+
126
+ # 1/2 chance to let the default value unchanged
127
+ if bool(random.getrandbits(1)):
128
+ continue
129
+
130
+ config = current_step.configuration[key]
131
+
132
+ if type(config['value']) in [int, float]: # Numeric value ? Let's apply multiplier
133
+ is_int = isinstance(config['value'], int)
134
+
135
+ new_value = None
136
+ if 'range' in config: # Random in range
137
+ new_value = random.uniform(*config['range'])
138
+ else: # Kind of strong mutate
139
+ # Randomly choose a positive or negative editing
140
+ if bool(random.getrandbits(1)):
141
+ # Negative -> Multiply value by something between 0.01 and 1
142
+ change_rate = random.uniform(0.01, 1)
143
+ new_value = config['value']*change_rate
144
+ else:
145
+ # Positive -> Multiply value by something between 1
146
+ # and the max modificator in configuration
147
+ change_rate = random.uniform(1, self.initial_modifier)
148
+ new_value = config['value']*change_rate
149
+
150
+ # Value was a int ? Round it to keep it int
151
+ if is_int:
152
+ new_value = round(new_value)
153
+
154
+ if not self.__valide_config(config, new_value):
155
+ # Cancel is the new value is not correct.
156
+ new_value = config['value']
157
+
158
+ # Categorical value, choose randomly one of them
159
+ elif 'categorical' in config.keys():
160
+ new_value = random.choice(config['categorical'])
161
+ elif isinstance(config['value'], bool):
162
+ # Bool value, choose randomly beetwen True and False
163
+ new_value = random.choice([True, False])
164
+ else: # Other value ? Just keep it
165
+ new_value = config['value']
166
+
167
+ current_step.configure(key, new_value) # Set new configuration in the step
168
+
169
+ return new_candidate
170
+
171
+ def __mutate(self, candidate: Candidate) -> Candidate:
172
+ """Mutate a candidate into a new one
173
+
174
+ :param Candidate candidate: Candidate used for the mutation.
175
+ :return: Newly created Candidate.
176
+ """
177
+ new_candidate: Candidate = deepcopy(candidate)
178
+ steps = new_candidate.pipeline.optimizable_step
179
+ if not steps:
180
+ return None
181
+ step_to_mutate: Step = random.choice(steps)
182
+
183
+ mutable_keys = self.__config_keys(step_to_mutate)
184
+ if step_to_mutate.is_interchangeable:
185
+ mutable_keys.append("interchange")
186
+
187
+ if not mutable_keys:
188
+ return None # Nothing to optimize
189
+
190
+ # Choose a random key to mutate
191
+ random_key: str = random.choice(mutable_keys)
192
+
193
+ if random_key == "interchange": # Mutate by interchanging the step with sibling
194
+ new_step = self.__interchange(step_to_mutate)
195
+ new_candidate.pipeline.replace_step(step_to_mutate, new_step)
196
+ return new_candidate
197
+
198
+
199
+ random_item: dict = step_to_mutate.configuration[random_key] # Get value of the random key
200
+ new_value = None
201
+
202
+ if type(random_item['value']) in [int, float]: # Numeric value ? Apply multiplier
203
+ is_int = isinstance(random_item['value'], int)
204
+
205
+ # Find a multiplier between - mutation_power & + mutation_power
206
+ change_rate = random.uniform(-self.mutation_power, self.mutation_power)
207
+ new_value = random_item['value']*(1+change_rate) # Apply random multiplier
208
+
209
+ # Value was a int ? Round it to keep it int
210
+ if is_int:
211
+ new_value = round(new_value)
212
+
213
+ if new_value == random_item['value']: # To be sure there is a mutation
214
+ new_value += random.choice([-1, 1])
215
+
216
+ if not self.__valide_config(random_item, new_value):
217
+ # Cancel if the new value is not correct.
218
+ new_value = random_item['value']
219
+
220
+ elif 'categorical' in random_item.keys(): # Categorial -> Choose one
221
+ new_value = random.choice(random_item['categorical'])
222
+ elif isinstance(random_item['value'], bool): # Bool -> Choose between True and False
223
+ new_value = random.choice([True, False])
224
+ else: # Other -> Keep it
225
+ new_value = random_item['value']
226
+
227
+ step_to_mutate.configure(random_key, new_value) # Apply configuration
228
+
229
+ return new_candidate
230
+
231
+ @staticmethod
232
+ def __interchange(step: Step) -> Step:
233
+ """Choose another implementation, allowing optional preprocessing to be removed."""
234
+ choices = [sibling for sibling in step.step_with_same_tags()
235
+ if sibling is not type(step)]
236
+ if (not isinstance(step, VoidStep) and step.can_be_disabled
237
+ and not hasattr(step, 'predict')
238
+ and (hasattr(step, 'transform') or hasattr(step, 'resample'))):
239
+ choices.append(VoidStep)
240
+ if not choices:
241
+ return step
242
+ step_class = random.choice(choices)
243
+ if step_class is VoidStep:
244
+ replacement = VoidStep(step_to_mimic=type(step)())
245
+ else:
246
+ replacement = step_class()
247
+ replacement.is_interchangeable = True
248
+ return replacement
249
+
250
+ def __config_keys(self, step: Step) -> list[str]:
251
+ """Get a list of config keys used for a Step
252
+
253
+ :param Step step: The step we want the config
254
+ :return: List of config keys for this step
255
+ """
256
+ if not step.optimizable:
257
+ return []
258
+ return list(set(step.configuration.keys()) - self.ignored_configs)
259
+
260
+ def __valide_config(self, config: dict, value: Any) -> bool:
261
+ """Is the configuration range valid ?
262
+
263
+ :param dict config: The config to validate.
264
+ :return: Valid ?
265
+ """
266
+ if 'range' not in config.keys():
267
+ return True
268
+ return config['range'][0] <= value <= config['range'][1]
269
+
270
+ def __unique(self, candidates: list[Candidate]) -> list[Candidate]:
271
+ """Get a list of unique Candidates
272
+
273
+ :param list[Candidate] candidates: List of Candidate.
274
+ :return: List of unique Candidate
275
+ """
276
+ unique_candidates: list[Candidate] = []
277
+ unique_fingerprint: list[str] = []
278
+ for candidate in candidates:
279
+ fingerprint: str = candidate.pipeline.fingerprint()
280
+ if fingerprint not in unique_fingerprint:
281
+ unique_candidates.append(candidate)
282
+ unique_fingerprint.append(fingerprint)
283
+
284
+ return unique_candidates
@@ -0,0 +1,31 @@
1
+
2
+ """Base class of IAML Optimizer. Optimizer receive a pool of Candidates,
3
+ optimize parameters and return a new pool of candidate
4
+ """
5
+ from ..candidate import Candidate
6
+
7
+
8
+ class Optimizer:
9
+ """Base class of IAML Optimizer. Optimizer receive a pool of Candidates,
10
+ optimize parameters and return a new pool of candidate
11
+ """
12
+ def __init__(self):
13
+ self.__finished: bool = False
14
+ """Is the optimization done ?"""
15
+
16
+ @property
17
+ def finished(self) -> bool:
18
+ """Is the optimization finished ?
19
+
20
+ :return: Finished ?
21
+ """
22
+ return self.__finished
23
+
24
+ def run(self, candidates: list[Candidate]) -> list[Candidate]:
25
+ """Run one optimisation stage. Run of Optimzer have to be overwrite (do nothing).
26
+
27
+ :param list[Candidate] candidates: List of candidates to optimize
28
+ :return: Optimized candidates
29
+ """
30
+ self.__finished = True
31
+ return candidates