PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,88 @@
1
+ """[STEP] SVM Classifier"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn import svm
5
+ from ....predictor import Predictor
6
+ from ....candidate import Candidate
7
+ from ....dataset import Dataset
8
+ from ....decorators.all import is_step
9
+
10
+
11
+ @is_step('predictor', 'tabular', 'classifier')
12
+ class ActSVMSVC(Predictor):
13
+ """[STEP] SVM Classifier"""
14
+
15
+ name: str = "SVM Classification"
16
+ _description: str = textwrap.dedent('''\
17
+ SVM Classifier is a machine learning algorithm that models the relationship
18
+ between input features and a categorical output variable using a support vector machine
19
+ (SVM). It can handle non-linearly separable data by using a kernel function to map the data
20
+ into a higher-dimensional space.''')
21
+ _description_long: str = textwrap.dedent('''\
22
+ SVM Classifier is a type of classification algorithm that models the
23
+ relationship between input features and a categorical output variable using a support vecto
24
+ machine (SVM). It works by finding the optimal hyperplane or boundary that separates the
25
+ data into different classes with the maximum margin.''')
26
+ _usage: str = "Use when tabular classes need nonlinear boundaries on small-to-medium data; consider ActCatBoost or ActExtraTreesClassifier for baseline alternatives. Applicable to binary, multiclass, or multilabel targets. Avoid when data is huge, very sparse, or interpretability is required."
27
+ refs: list[dict[str, Any]] = [
28
+ {
29
+ 'year': 1999,
30
+ 'name': 'Probabilistic Outputs for Support Vector Machines and Comparisons to \
31
+ Regularized Likelihood Methods',
32
+ 'authors': [
33
+ 'John C. Platt'
34
+ ],
35
+ 'doi': "https://api.semanticscholar.org/CorpusID:56563878",
36
+ 'publisher': 'Microsoft Research'
37
+ },
38
+ {
39
+ 'year': 2001,
40
+ 'name': 'LIBSVM: A Library for Support Vector Machines',
41
+ 'authors': [
42
+ 'Chih-Chung Chang',
43
+ 'Chih-Jen Lin'
44
+ ],
45
+ 'doi': 'https://doi.org/10.1145/1961189.1961199',
46
+ 'publisher': 'ACM Transactions on Intelligen Systems and Technology Vol.2 page 1--27'
47
+ },
48
+ ]
49
+
50
+ def __init__(self):
51
+ self.configuration = {
52
+ 'kernel': {
53
+ 'description': 'Kernel to use in the SVM',
54
+ 'default': 'rbf',
55
+ 'categorical': ['linear', 'poly', 'rbf', 'sigmoid']
56
+ },
57
+ 'random_state': {
58
+ 'description': 'random_state',
59
+ 'default': 42
60
+ },
61
+ 'class_weight': {
62
+ 'description': 'Can be set on "balanced" to improve results on unbalanced data',
63
+ 'default': None,
64
+ 'categorical': [None, 'balanced']
65
+ },
66
+ 'tol': {
67
+ 'description': 'The stopping criterion.',
68
+ 'default': 0.001,
69
+ 'range': [1e-05, 0.1]
70
+ }
71
+ }
72
+ self.model: svm.SVC = None
73
+
74
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
75
+ self.model = svm.SVC(
76
+ probability = True, # Enable predict_proba.
77
+ **self.passthrough_parameters()
78
+ )
79
+ self.model.fit(dataset.X, dataset.y)
80
+
81
+ return self
82
+
83
+ def suitable(self, dataset: Dataset) -> bool:
84
+ return dataset.type_of_target in \
85
+ ['binary', 'multiclass', 'multilabel-indicator']
86
+
87
+ def priorize(self, candidate: Candidate = None) -> float:
88
+ return 0.5 # neutral
@@ -0,0 +1,111 @@
1
+ """[STEP] XGBoost classifier."""
2
+ import textwrap
3
+ from typing import Any
4
+
5
+ from sklearn.metrics import accuracy_score
6
+ from sklearn.preprocessing import LabelEncoder
7
+ from sklearn.utils.multiclass import check_classification_targets
8
+ from xgboost import XGBClassifier
9
+
10
+ from .._xgboost import xgboost_features
11
+ from ....predictor import Predictor
12
+ from ....dataset import Dataset
13
+ from ....candidate import Candidate
14
+ from ....decorators.all import is_step
15
+
16
+
17
+ @is_step('predictor', 'tabular', 'classifier', 'minimal_predictor')
18
+ class ActXGBoost(Predictor):
19
+ """[STEP] XGBoost classifier."""
20
+
21
+ name: str = "XGBoost"
22
+ _description: str = textwrap.dedent('''\
23
+ XGBoost trains an ensemble of boosted decision trees for binary
24
+ or multiclass classification.''')
25
+ _description_long: str = textwrap.dedent('''\
26
+ XGBoost adds trees sequentially to improve predictions, using regularized
27
+ gradient boosting and histogram-based split finding. Class labels are
28
+ encoded during training and restored when predicting.''')
29
+ _usage: str = "Use for boosted-tree classification on numeric or encoded tabular features. Supports binary and multiclass targets."
30
+ refs: list[dict[str, Any]] = [
31
+ {
32
+ 'name': 'XGBoost: A Scalable Tree Boosting System',
33
+ 'year': 2016,
34
+ 'authors': ['Tianqi Chen', 'Carlos Guestrin'],
35
+ 'doi': 'https://doi.org/10.1145/2939672.2939785',
36
+ 'publisher': 'Proceedings of the 22nd ACM SIGKDD International Conference, pages 785–794'
37
+ }
38
+ ]
39
+
40
+ def __init__(self):
41
+ self.configuration = {
42
+ 'max_depth': {
43
+ 'description': 'Maximum depth of each tree.',
44
+ 'default': 6,
45
+ 'range': [1, 16]
46
+ },
47
+ 'random_state': {
48
+ 'description': 'Random seed for reproducibility.',
49
+ 'default': 42
50
+ },
51
+ 'learning_rate': {
52
+ 'description': 'Shrinkage applied to each boosting round.',
53
+ 'default': 0.1,
54
+ 'range': [1e-3, 1.0]
55
+ },
56
+ 'subsample': {
57
+ 'description': 'Fraction of training rows sampled for each tree.',
58
+ 'default': 1.0,
59
+ 'range': [0.1, 1.0]
60
+ },
61
+ 'n_estimators': {
62
+ 'description': 'Number of boosting rounds.',
63
+ 'default': 100,
64
+ 'range': [1, 500]
65
+ },
66
+ 'min_child_weight': {
67
+ 'description': 'Minimum sum of instance Hessians needed in a child.',
68
+ 'default': 1.0,
69
+ 'range': [0.0, 20.0]
70
+ },
71
+ 'colsample_bytree': {
72
+ 'description': 'Fraction of features sampled for each tree.',
73
+ 'default': 1.0,
74
+ 'range': [0.1, 1.0]
75
+ }
76
+ }
77
+ self.model: XGBClassifier = None
78
+ self.label_encoder = LabelEncoder()
79
+
80
+ def fit(self, dataset: Dataset):
81
+ check_classification_targets(dataset.y)
82
+ encoded_target = self.label_encoder.fit_transform(dataset.y)
83
+ # IAML schedules candidates in parallel; keep each estimator single-threaded.
84
+ self.model = XGBClassifier(
85
+ n_jobs=1, tree_method='hist', **self.passthrough_parameters()
86
+ )
87
+ self.model.fit(xgboost_features(dataset.X), encoded_target)
88
+ return self
89
+
90
+ def predict(self, X):
91
+ """Predict original class labels using XGBoost-safe feature names."""
92
+ return super().predict(xgboost_features(X))
93
+
94
+ def predict_proba(self, X):
95
+ """Predict class probabilities using XGBoost-safe feature names."""
96
+ return super().predict_proba(xgboost_features(X))
97
+
98
+ @property
99
+ def classes_(self):
100
+ """Original labels, in the order of predict_proba columns."""
101
+ return self.label_encoder.classes_
102
+
103
+ def score(self, X, y, sample_weight=None):
104
+ """Compute accuracy with the original class labels."""
105
+ return accuracy_score(y, self.predict(X), sample_weight=sample_weight)
106
+
107
+ def suitable(self, dataset: Dataset) -> bool:
108
+ return dataset.type_of_target in ['binary', 'multiclass']
109
+
110
+ def priorize(self, candidate: Candidate = None) -> float:
111
+ return 0.5
@@ -0,0 +1,27 @@
1
+ """
2
+ Regressor Predictors Actionables
3
+ """
4
+
5
+ from .act_svm_svr import ActSVMSVR
6
+ from .act_randomforest_regressor import ActRandomForestRegressor
7
+ from .act_xgboost_regressor import ActXGBoostRegressor
8
+ from .act_gboost_regressor import ActGBoostRegressor
9
+ from .act_catboost_regressor import ActCatBoostRegressor
10
+ from .act_knn_regressor import ActKNNRegressor
11
+ from .act_linear_regression import ActLinearRegression
12
+ from .act_extra_trees_regressor import ActExtraTreesRegressor
13
+ from .act_ard_regression import ActARDRegression
14
+ from .act_ada_boost_regressor import ActAdaBoostRegressor
15
+ from .act_hist_gradient_boosting_regressor import ActHistGradientBoostingRegressor
16
+ from .act_mlp_regressor import ActMLPRegressor
17
+ from .act_sgd_regressor import ActSGDRegressor
18
+ from .act_gaussian_process_regressor import ActGaussianProcessRegressor
19
+ from .act_decision_tree_regressor import ActDecisionTreeRegressor
20
+ from .act_ridge_regressor import ActRidgeRegressor
21
+ from .act_lasso_regressor import ActLassoRegressor
22
+ from .act_elastic_net_regressor import ActElasticNetRegressor
23
+ from .act_huber_regressor import ActHuberRegressor
24
+ from .act_ransac_regressor import ActRANSACRegressor
25
+ from .act_quantile_regressor import ActQuantileRegressor
26
+ from .act_poisson_regressor import ActPoissonRegressor
27
+ from .act_light_gbm_regressor import ActLightGBMRegressor
@@ -0,0 +1,75 @@
1
+ """[STEP] AdaBoost Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.ensemble import AdaBoostRegressor
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....decorators.all import is_step
9
+
10
+ @is_step('predictor', 'tabular', 'regressor')
11
+ class ActAdaBoostRegressor(Predictor):
12
+ """[STEP] AdaBoost Regressor"""
13
+
14
+ name: str = "AdaBoost Regressor"
15
+ _description: str = textwrap.dedent('''\
16
+ AdaBoostRegressor is a powerful tool that combines many simple models
17
+ to make accurate predictions for continuous outcomes.''')
18
+ _description_long: str = textwrap.dedent('''\
19
+ AdaBoostRegressor is an ensemble learning technique
20
+ used for regression problems. It works by combining multiple weak learners
21
+ (simple models) into a strong learner.''')
22
+ _usage: str = "Use when regression needs boosting and you want an alternative to ActDecisionTreeRegressor or ActExtraTreesRegressor. Applicable to continuous targets with modest features and nonlinear signal. Avoid when data is very noisy, high-dimensional, or you need strong interpretability."
23
+ refs: list[dict[str, Any]] = [
24
+ {
25
+ 'year': 1995,
26
+ 'name': (
27
+ 'A desicion-theoretic generalization of on-line learning '
28
+ 'and an application to boosting'
29
+ ),
30
+ 'authors': [
31
+ 'Yoav Freund',
32
+ 'Robert E. Schapire'
33
+ ],
34
+ 'doi': 'https://doi.org/10.1007/3-540-59119-2_166',
35
+ 'publisher': 'Springer, Berlin, Heidelberg'
36
+ }
37
+ ]
38
+ def __init__(self):
39
+ self.configuration = {
40
+ 'learning_rate': {
41
+ 'description': 'Weight applied to each regressor at each boosting iteration',
42
+ 'default': 0.1,
43
+ 'range': [0.01, 2.0]
44
+ },
45
+ 'loss': {
46
+ 'description': textwrap.dedent('''\
47
+ The loss function to use when updating the weights after
48
+ each boosting iteration.'''),
49
+ 'default': "linear",
50
+ 'categorical': ["linear", "square", "exponential"]
51
+ },
52
+ 'n_estimators': {
53
+ 'description': 'Number of threes',
54
+ 'default': 50,
55
+ 'range': [1, 500]
56
+ },
57
+ 'random_state': {
58
+ 'description': 'random_state',
59
+ 'default': 42
60
+ }
61
+ }
62
+ self.model: AdaBoostRegressor = None
63
+
64
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
65
+ self.model = AdaBoostRegressor(**self.passthrough_parameters())
66
+
67
+ self.model.fit(dataset.X, dataset.y)
68
+
69
+ return self
70
+
71
+ def suitable(self, dataset: Dataset) -> bool:
72
+ return dataset.type_of_target == 'continuous'
73
+
74
+ def priorize(self, candidate: Candidate = None) -> float:
75
+ return 0.5 # neutral
@@ -0,0 +1,95 @@
1
+ """
2
+ [STEP] ARD Regression
3
+ """
4
+
5
+ import textwrap
6
+ from sklearn.linear_model import ARDRegression
7
+ from ....predictor import Predictor
8
+ from ....dataset import Dataset
9
+ from ....candidate import Candidate
10
+ from ....decorators.all import is_step
11
+
12
+ @is_step('predictor', 'tabular', 'regressor')
13
+ class ActARDRegression(Predictor):
14
+ """
15
+ [STEP] ARD Regression
16
+ """
17
+ name = "ARD Regression"
18
+ _description = textwrap.dedent('''\
19
+ ARDRegression is a powerful tool that helps computers make accurate
20
+ predictions by giving each feature its own importance weight.''')
21
+ _description_long = textwrap.dedent('''\
22
+ ARDRegression (Automatic Relevance Determination Regression)
23
+ is a Bayesian regression technique used for predicting continuous outcomes.
24
+ It works by assigning weights to each feature, allowing some features to be
25
+ more important than others.
26
+ These weights are determined automatically during training,
27
+ hence the "automatic relevance determination.''')
28
+ _usage = "Use when you want Bayesian linear regression with automatic relevance on tabular data, as a sparse alternative to ActElasticNetRegressor. Applicable to continuous targets with many features. Avoid when strong nonlinearity or interactions suggest ActExtraTreesRegressor."
29
+ refs = [
30
+ {
31
+ 'year': 1996,
32
+ 'name': 'Bayesian Non-Linear Modeling for the Prediction Competition',
33
+ 'authors': ['David J. C. MacKay'],
34
+ 'doi': 'https://doi.org/10.1007/978-94-015-8729-7_18',
35
+ 'publisher': 'Springer, Dordrecht'
36
+ }
37
+ ]
38
+ def __init__(self):
39
+ self.configuration = {
40
+ 'alpha_1': {
41
+ 'description': textwrap.dedent('''\
42
+ Hyper-parameter : shape parameter for the Gamma
43
+ distribution prior over the alpha parameter.'''),
44
+ 'default': 1e-06,
45
+ 'range': [1e-10, 0.001]
46
+ },
47
+ 'alpha_2': {
48
+ 'description': textwrap.dedent('''\
49
+ Hyper-parameter : inverse scale parameter (rate parameter)
50
+ for the Gamma distribution prior over the alpha parameter.'''),
51
+ 'default': 1e-06,
52
+ 'range': [1e-10, 0.001]
53
+ },
54
+ 'lambda_1': {
55
+ 'description': textwrap.dedent('''\
56
+ Hyper-parameter : shape parameter for the Gamma
57
+ distribution prior over the lambda parameter.'''),
58
+ 'default': 1e-10,
59
+ 'range': [1e-10, 0.001]
60
+ },
61
+ 'lambda_2': {
62
+ 'description': textwrap.dedent('''\
63
+ Hyper-parameter : inverse scale parameter (rate parameter)
64
+ for the Gamma distribution prior over the lambda parameter.'''),
65
+ 'default': 1e-10,
66
+ 'range': [1e-10, 0.001]
67
+ },
68
+ 'threshold_lambda': {
69
+ 'description': textwrap.dedent('''\
70
+ Threshold for removing (pruning) weights with high
71
+ precision from the computation.'''),
72
+ 'default': 10000.0,
73
+ 'range': [1000.0, 100000.0]
74
+ },
75
+ 'tol': {
76
+ 'description': 'Stop the algorithm if w has converged.',
77
+ 'default': 0.001,
78
+ 'range': [1e-05, 0.1]
79
+ }
80
+ }
81
+
82
+ self.model: ARDRegression = None
83
+
84
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
85
+ self.model = ARDRegression(**self.passthrough_parameters())
86
+
87
+ self.model.fit(dataset.X, dataset.y)
88
+
89
+ return self
90
+
91
+ def suitable(self, dataset: Dataset) -> bool:
92
+ return dataset.type_of_target == 'continuous'
93
+
94
+ def priorize(self, candidate: Candidate = None) -> float:
95
+ return 0.5 # neutral
@@ -0,0 +1,134 @@
1
+ """
2
+ [STEP] CatBoost Regressor
3
+ """
4
+ import textwrap
5
+ from typing import Any
6
+ from catboost import CatBoostRegressor, CatBoostError
7
+ from ....predictor import Predictor
8
+ from ....dataset import Dataset
9
+ from ....candidate import Candidate
10
+ from ....decorators.all import is_step
11
+ from ....logger import Logger
12
+
13
+ @is_step('predictor', 'tabular', 'regressor', 'minimal_predictor')
14
+ class ActCatBoostRegressor(Predictor):
15
+ """[STEP] CatBoost Regressor"""
16
+
17
+ name: str = "CatBoost Regressor"
18
+ _usage: str = "Use when you need strong tabular regression with categorical features versus ActDecisionTreeRegressor. Applicable to continuous targets with mixed numeric/categorical columns. Avoid when interpretability, ultra-low latency, or tiny data dominate."
19
+ _description: str = textwrap.dedent('''\
20
+ CatBoostRegressor is a powerful tool that helps computers make accurate
21
+ predictions for continuous outcomes by learning from both positive and
22
+ negative examples simultaneously.''')
23
+ _description_long: str = textwrap.dedent('''\
24
+ CatBoostRegressor is a gradient boosting algorithm specifically
25
+ designed for regression tasks.''')
26
+ refs: list[dict[str, Any]] = [
27
+ {
28
+ 'year': 2017,
29
+ 'name': 'CatBoost: unbiased boosting with categorical features',
30
+ 'authors': [
31
+ 'Liudmila Prokhorenkova',
32
+ 'Gleb Gusev',
33
+ 'Aleksandr Vorobev',
34
+ 'Anna Veronika Dorogush',
35
+ 'Andrey Gulin'
36
+ ],
37
+ 'doi': 'https://doi.org/10.48550/arXiv.1706.09516',
38
+ 'publisher': 'Advances in Neural Information Processing Systems 31 (NeurIPS 2018)'
39
+ }
40
+ ]
41
+
42
+ def __init__(self):
43
+ self.configuration = {
44
+ 'iterations': {
45
+ 'description': 'The maximum number of trees that can be built.',
46
+ 'default': 1000,
47
+ 'range': [100, 10000]
48
+ },
49
+ 'learning_rate': {
50
+ 'description': 'The learning rate.',
51
+ 'default': 0.03,
52
+ 'range': [0.001, 1.0]
53
+ },
54
+ 'depth': {
55
+ 'description': 'Depth of the tree.',
56
+ 'default': 6,
57
+ 'range': [1, 16]
58
+ },
59
+ 'l2_leaf_reg': {
60
+ 'description': 'Coefficient at the L2 regularization term of the cost function.',
61
+ 'default': 3,
62
+ 'range': [0, 10]
63
+ },
64
+ 'border_count': {
65
+ 'description': 'The number of splits for numerical features.',
66
+ 'default': 254,
67
+ 'range': [1, 255]
68
+ },
69
+ 'loss_function': {
70
+ 'description': 'The metric to use in training.',
71
+ 'default': 'RMSE',
72
+ 'categorical': ['RMSE',
73
+ 'MAE',
74
+ 'Quantile',
75
+ 'LogLinQuantile',
76
+ 'Poisson',
77
+ 'MAPE']
78
+ #, 'Lq']
79
+ },
80
+ 'eval_metric': {
81
+ 'description': 'The metric to be used for validation data.',
82
+ 'default': 'RMSE',
83
+ 'categorical': ['RMSE',
84
+ 'MAE',
85
+ 'R2',
86
+ 'Quantile',
87
+ 'LogLinQuantile',
88
+ 'Poisson',
89
+ 'MAPE']
90
+ # , 'Lq']
91
+ },
92
+ 'bootstrap_type': {
93
+ 'description': 'The method for sampling the weights of objects.',
94
+ 'default': 'Bayesian',
95
+ 'categorical': ['Bayesian', 'Bernoulli', 'MVS']
96
+ },
97
+ 'leaf_estimation_iterations': {
98
+ 'description': 'The number of iterations for leaf estimation.',
99
+ 'default': 10,
100
+ 'range': [1, 50]
101
+ }
102
+ }
103
+ self.model: CatBoostRegressor = None
104
+
105
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
106
+ self.model = CatBoostRegressor(verbose=0, **self.passthrough_parameters())
107
+ try:
108
+ self.model.fit(dataset.X, dataset.y)
109
+ except CatBoostError as exc:
110
+ self._log_failure(dataset, exc)
111
+ raise ValueError(f"CatBoostRegressor training failed: {exc}") from exc
112
+ except Exception as exc: # pragma: no cover - defensive
113
+ self._log_failure(dataset, exc)
114
+ raise
115
+ return self
116
+
117
+ def _log_failure(self, dataset: Dataset, exc: Exception) -> None:
118
+ """Log enriched debug info when CatBoost crashes."""
119
+ shape = getattr(dataset.X, "shape", None)
120
+ message = (
121
+ "[CatBoostRegressor] crash detected "
122
+ f"(shape={shape}, target_len={len(dataset.y)}, "
123
+ f"params={self.passthrough_parameters()}): {exc}"
124
+ )
125
+ logger = Logger()
126
+ if logger.verbose <= 3 and logger.verbose != -1:
127
+ logger.console.log(message)
128
+ logger.error(message)
129
+
130
+ def suitable(self, dataset: Dataset) -> bool:
131
+ return dataset.type_of_target in ['continuous']
132
+
133
+ def priorize(self, candidate: Candidate = None) -> float:
134
+ return 0.5 # neutral
@@ -0,0 +1,111 @@
1
+ """[STEP] Decision Tree Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.tree import DecisionTreeRegressor
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....data_type import DataType
9
+ from ....decorators.all import is_step
10
+
11
+
12
+ @is_step('predictor', 'tabular', 'regressor')
13
+ class ActDecisionTreeRegressor(Predictor):
14
+ """[STEP] Decision Tree Regressor"""
15
+
16
+ name: str = "Decision Tree Regressor"
17
+ _description: str = textwrap.dedent('''\
18
+ DecisionTreeRegressor learns if-then rules in a tree structure
19
+ to predict a continuous target.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ DecisionTreeRegressor builds a regression tree by recursively splitting
22
+ features to reduce variance. The resulting tree is easy to inspect,
23
+ which makes it a strong baseline when interpretability matters.''')
24
+ _usage: str = "Use when you need an interpretable tree baseline; compare to ActExtraTreesRegressor for higher accuracy. Applicable to tabular numeric features with a continuous target. Avoid when linear effects dominate (ActElasticNetRegressor) or you need smoother generalization."
25
+ refs: list[dict[str, Any]] = [
26
+ {
27
+ 'year': 1984,
28
+ 'name': 'Classification and Regression Trees',
29
+ 'authors': [
30
+ 'Leo Breiman',
31
+ 'Jerome Friedman',
32
+ 'Richard Olshen',
33
+ 'Charles Stone'
34
+ ],
35
+ 'publisher': 'Wadsworth'
36
+ }
37
+ ]
38
+
39
+ def __init__(self):
40
+ self.configuration = {
41
+ 'max_depth': {
42
+ 'description': 'Maximum depth of the tree.',
43
+ 'default': 5,
44
+ 'range': [1, 50]
45
+ },
46
+ 'min_samples_leaf': {
47
+ 'description': 'Minimum number of samples required to be at a leaf node.',
48
+ 'default': 1,
49
+ 'range': [1, 20]
50
+ },
51
+ 'min_samples_split': {
52
+ 'description': 'Minimum number of samples required to split an internal node.',
53
+ 'default': 2,
54
+ 'range': [2, 50]
55
+ },
56
+ 'max_features': {
57
+ 'description': textwrap.dedent('''\
58
+ The number of features to consider when looking for the best split.
59
+ Use a float to specify a fraction of features.'''),
60
+ 'default': 1.0,
61
+ 'range': [0.1, 1.0]
62
+ },
63
+ 'criterion': {
64
+ 'description': 'Function to measure the quality of a split.',
65
+ 'default': 'squared_error',
66
+ 'categorical': [
67
+ 'squared_error',
68
+ 'absolute_error',
69
+ 'friedman_mse',
70
+ 'poisson'
71
+ ]
72
+ },
73
+ 'splitter': {
74
+ 'description': 'Strategy used to choose the split at each node.',
75
+ 'default': 'best',
76
+ 'categorical': ['best', 'random']
77
+ },
78
+ 'random_state': {
79
+ 'description': 'Random state for reproducibility.',
80
+ 'default': 42
81
+ }
82
+ }
83
+ self.model: DecisionTreeRegressor = None
84
+ self.columns: list[str] = []
85
+
86
+ def _select_features(self, X):
87
+ if self.columns and hasattr(X, 'columns'):
88
+ return X[self.columns]
89
+ return X
90
+
91
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
92
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
93
+ if not self.columns:
94
+ self.columns = dataset.features
95
+
96
+ self.model = DecisionTreeRegressor(**self.passthrough_parameters())
97
+ self.model.fit(self._select_features(dataset.X), dataset.y)
98
+ return self
99
+
100
+ def predict(self, X):
101
+ return super().predict(self._select_features(X))
102
+
103
+ def score(self, X, y=None, *args, **kwargs):
104
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
105
+
106
+ def suitable(self, dataset: Dataset) -> bool:
107
+ return dataset.type_of_target == 'continuous' \
108
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
109
+
110
+ def priorize(self, candidate: Candidate = None) -> float:
111
+ return 0.5 # neutral