PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,109 @@
1
+ """[STEP] Elastic Net Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.linear_model import ElasticNet
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....data_type import DataType
9
+ from ....decorators.all import is_step
10
+
11
+
12
+ @is_step('predictor', 'tabular', 'regressor')
13
+ class ActElasticNetRegressor(Predictor):
14
+ """[STEP] Elastic Net Regressor"""
15
+
16
+ name: str = "Elastic Net Regressor"
17
+ _description: str = textwrap.dedent('''\
18
+ ElasticNet regression combines L1 and L2 regularization to balance
19
+ sparsity and stability in linear models.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ ElasticNet regression fits a linear model with both L1 and L2 penalties.
22
+ The L1 term encourages sparsity by driving some coefficients to zero,
23
+ while the L2 term stabilizes coefficients when predictors are correlated,
24
+ making it a robust choice for tabular regression.''')
25
+ _usage: str = "Use when you need a sparse linear baseline for tabular regression and want a simpler choice than ActExtraTreesRegressor. Applicable to numeric-feature regression with correlated predictors. Avoid when strong nonlinear patterns dominate; consider ActCatBoostRegressor."
26
+ refs: list[dict[str, Any]] = [
27
+ {
28
+ 'year': 2005,
29
+ 'name': 'Regularization and Variable Selection via the Elastic Net',
30
+ 'authors': [
31
+ 'Hui Zou',
32
+ 'Trevor Hastie'
33
+ ],
34
+ 'doi': 'https://doi.org/10.1111/j.1467-9868.2005.00503.x',
35
+ 'publisher': 'Journal of the Royal Statistical Society Series B'
36
+ }
37
+ ]
38
+
39
+ def __init__(self):
40
+ self.configuration = {
41
+ 'alpha': {
42
+ 'description': 'Regularization strength.',
43
+ 'default': 1.0,
44
+ 'range': [1e-04, 10.0]
45
+ },
46
+ 'l1_ratio': {
47
+ 'description': 'Mixing parameter between L1 and L2 penalty.',
48
+ 'default': 0.5,
49
+ 'range': [0.0, 1.0]
50
+ },
51
+ 'fit_intercept': {
52
+ 'description': 'Whether to fit the intercept term.',
53
+ 'default': True,
54
+ 'categorical': [True, False]
55
+ },
56
+ 'max_iter': {
57
+ 'description': 'Maximum number of iterations.',
58
+ 'default': 1000,
59
+ 'range': [100, 5000]
60
+ },
61
+ 'tol': {
62
+ 'description': 'Stopping criterion.',
63
+ 'default': 0.0001,
64
+ 'range': [1e-05, 0.1]
65
+ },
66
+ 'selection': {
67
+ 'description': 'Coordinate descent selection strategy.',
68
+ 'default': 'cyclic',
69
+ 'categorical': ['cyclic', 'random']
70
+ },
71
+ 'positive': {
72
+ 'description': 'Force coefficients to be positive.',
73
+ 'default': False,
74
+ 'categorical': [True, False]
75
+ },
76
+ 'random_state': {
77
+ 'description': 'Random state used when selection is "random".',
78
+ 'default': 42
79
+ }
80
+ }
81
+ self.model: ElasticNet = None
82
+ self.columns: list[str] = []
83
+
84
+ def _select_features(self, X):
85
+ if self.columns and hasattr(X, 'columns'):
86
+ return X[self.columns]
87
+ return X
88
+
89
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
90
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
91
+ if not self.columns:
92
+ self.columns = dataset.features
93
+
94
+ self.model = ElasticNet(**self.passthrough_parameters())
95
+ self.model.fit(self._select_features(dataset.X), dataset.y)
96
+ return self
97
+
98
+ def predict(self, X):
99
+ return super().predict(self._select_features(X))
100
+
101
+ def score(self, X, y=None, *args, **kwargs):
102
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
103
+
104
+ def suitable(self, dataset: Dataset) -> bool:
105
+ return dataset.type_of_target == 'continuous' \
106
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
107
+
108
+ def priorize(self, candidate: Candidate = None) -> float:
109
+ return 0.5 # neutral
@@ -0,0 +1,113 @@
1
+ """[STEP] Extra Trees Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.ensemble import ExtraTreesRegressor
5
+ from ....data_type import DataType
6
+ from ....predictor import Predictor
7
+ from ....dataset import Dataset
8
+ from ....candidate import Candidate
9
+ from ....decorators.all import is_step
10
+
11
+
12
+ @is_step('predictor', 'tabular', 'regressor')
13
+ class ActExtraTreesRegressor(Predictor):
14
+ """[STEP] Extra Trees Regressor"""
15
+
16
+ name: str = "Extra Trees Regressor"
17
+ _description: str = textwrap.dedent('''\
18
+ ExtraTreesRegressor is a machine learning algorithm that makes predictions
19
+ for regression tasks by combining the outputs of multiple decision trees.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ ExtraTreesRegressor is a type of ensemble learning algorithm
22
+ that belongs to the family of decision tree-based models.
23
+ It works by building multiple decision trees, where each tree is trained on a
24
+ random subset of the input features and a random subset of the training data.
25
+ At prediction time, the algorithm aggregates the outputs of all the
26
+ decision trees to make a final prediction.''')
27
+ _usage: str = "Use when you want a strong nonparametric tabular regressor, often a better default than ActDecisionTreeRegressor or ActAdaBoostRegressor. Applicable to numeric or mixed features with continuous targets. Avoid when data is tiny, very sparse/high-dimensional, or a linear/transparent model is required."
28
+ refs: list[dict[str, Any]] = [
29
+ {
30
+ 'year': 2006,
31
+ 'name': 'Extremely randomized trees',
32
+ 'authors': [
33
+ 'Pierre Geurts',
34
+ 'Damien Ernst',
35
+ 'Louis Wehenkel'
36
+ ],
37
+ 'doi': 'https://doi.org/10.1007/s10994-006-6226-1',
38
+ 'publisher': 'Machine Learning Vol. 63 page 3--42'
39
+ }
40
+ ]
41
+ def __init__(self):
42
+ self.configuration = {
43
+ 'max_depth': {
44
+ 'description': 'Max depth of each tree',
45
+ 'default': 15,
46
+ 'range': [1, 100]
47
+ },
48
+ 'min_samples_leaf': {
49
+ 'description': 'The minimum number of samples required to be at a leaf node.',
50
+ 'default': 1,
51
+ 'range': [1, 15]
52
+ },
53
+ 'max_features': {
54
+ 'description': 'The number of features to consider when looking for the best split',
55
+ 'default': 1.0,
56
+ 'range': [0.1, 1.0]
57
+ },
58
+ 'min_samples_split': {
59
+ 'description': 'The minimum number of samples required to split an internal node',
60
+ 'default': 2,
61
+ 'range': [2, 20]
62
+ },
63
+ 'bootstrap': {
64
+ 'description': textwrap.dedent('''\
65
+ Whether bootstrap samples are used when building trees. If
66
+ False, the whole dataset is used to build each tree.'''),
67
+ 'default': False
68
+ },
69
+ 'criterion': {
70
+ 'description': 'The function to measure the quality of a split.',
71
+ 'default': "squared_error",
72
+ 'categorical': ["poisson", "friedman_mse", "absolute_error", "squared_error"]
73
+ },
74
+ 'n_estimators': {
75
+ 'description': 'Number of threes',
76
+ 'default': 100,
77
+ 'range': [1, 500]
78
+ },
79
+ 'random_state': {
80
+ 'description': 'random_state',
81
+ 'default': 42
82
+ }
83
+ }
84
+ self.model: ExtraTreesRegressor = None
85
+ self.columns: list[str] = []
86
+
87
+ def _select_features(self, X):
88
+ if self.columns and hasattr(X, 'columns'):
89
+ return X[self.columns]
90
+ return X
91
+
92
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
93
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
94
+ if not self.columns:
95
+ self.columns = dataset.features
96
+
97
+ self.model = ExtraTreesRegressor(**self.passthrough_parameters())
98
+ self.model.fit(self._select_features(dataset.X), dataset.y)
99
+
100
+ return self
101
+
102
+ def predict(self, X):
103
+ return super().predict(self._select_features(X))
104
+
105
+ def score(self, X, y=None, *args, **kwargs):
106
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
107
+
108
+ def suitable(self, dataset: Dataset) -> bool:
109
+ return dataset.type_of_target == 'continuous' \
110
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
111
+
112
+ def priorize(self, candidate: Candidate = None) -> float:
113
+ return 0.5 # neutral
@@ -0,0 +1,55 @@
1
+ """[STEP] Gaussian Process Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.gaussian_process import GaussianProcessRegressor
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....decorators.all import is_step
9
+
10
+ @is_step('predictor', 'tabular', 'regressor')
11
+ class ActGaussianProcessRegressor(Predictor):
12
+ """[STEP] Gaussian Process Regressor"""
13
+
14
+ name: str = "Gaussian Process Regressor"
15
+ _description: str = textwrap.dedent('''\
16
+ GaussianProcessRegressor is a machine learning algorithm
17
+ that makes predictions for regression tasks using Gaussian processes.''')
18
+ _description_long: str = textwrap.dedent('''\
19
+ GaussianProcessRegressor is a powerful algorithm for regression tasks,
20
+ especially when the relationship between the input features and the output variable is
21
+ complex and non-linear, and when uncertainty estimates are important.''')
22
+ _usage: str = "Use when you need nonlinear regression with uncertainty on small tabular data; compare ActCatBoostRegressor or ActExtraTreesRegressor for larger data. Applicable to continuous-target tabular features. Avoid when datasets are large/high-dimensional or latency is strict."
23
+ refs: list[dict[str, Any]] = [
24
+ {
25
+ 'year': 2006,
26
+ 'name': 'Gaussian Processes for Machine Learning',
27
+ 'authors': [
28
+ 'Carl Edward Rasmussen',
29
+ 'Christopher K. I. Williams'
30
+ ],
31
+ 'doi': 'https://doi.org/10.7551/mitpress/3206.001.0001',
32
+ 'publisher': 'MIT Press 2006'
33
+ }
34
+ ]
35
+
36
+ def __init__(self):
37
+ self.configuration = {
38
+ 'alpha': {
39
+ 'description': 'Value added to the diagonal of the kernel matrix during fitting.',
40
+ 'default': 1e-14,
41
+ 'range': [1e-08, 1.0]
42
+ }
43
+ }
44
+ self.model: GaussianProcessRegressor = None
45
+
46
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
47
+ self.model = GaussianProcessRegressor(**self.passthrough_parameters())
48
+ self.model.fit(dataset.X, dataset.y)
49
+ return self
50
+
51
+ def suitable(self, dataset: Dataset) -> bool:
52
+ return dataset.type_of_target == 'continuous'
53
+
54
+ def priorize(self, candidate: Candidate = None) -> float:
55
+ return 0.5 # neutral
@@ -0,0 +1,95 @@
1
+ """[STEP] Gradient Boosting Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+
5
+ from sklearn.ensemble import GradientBoostingRegressor
6
+
7
+ from ....predictor import Predictor
8
+ from ....dataset import Dataset
9
+ from ....candidate import Candidate
10
+ from ....decorators.all import is_step
11
+
12
+
13
+ @is_step('predictor', 'tabular', 'regressor', 'minimal_predictor')
14
+ class ActGBoostRegressor(Predictor):
15
+ """[STEP] Gradient Boosting Regressor"""
16
+
17
+ name: str = "Gradient Boosting Regressor"
18
+ _usage: str = "Use when you want a boosting baseline for regression and prefer simpler tuning vs ActCatBoostRegressor. Applicable to tabular data with continuous targets and mixed features. Avoid when you need linear interpretability or very fast training; consider ActElasticNetRegressor."
19
+ _description: str = textwrap.dedent('''\
20
+ GradientBoostingRegressor learns an ensemble of weak learners (decision trees)
21
+ in a stage-wise manner to minimize the prediction error on continuous targets.''')
22
+ _description_long: str = textwrap.dedent('''\
23
+ Gradient Boosting is an additive modeling technique where each subsequent tree
24
+ attempts to correct the residuals of the previous ensemble. The regressor is
25
+ robust to different loss formulations and usually provides a strong baseline
26
+ for tabular regression tasks.''')
27
+ refs: list[dict[str, Any]] = [
28
+ {
29
+ 'name': 'Greedy Function Approximation: A Gradient Boosting Machine',
30
+ 'year': 2001,
31
+ 'authors': ['Jerome H. Friedman'],
32
+ 'doi': 'https://doi.org/10.1214/aos/1013203451',
33
+ 'publisher': 'The Annals of Statistics, Vol.29, No.5'
34
+ }
35
+ ]
36
+
37
+ def __init__(self):
38
+ self.configuration = {
39
+ 'max_depth': {
40
+ 'description': 'Max depth of each tree',
41
+ 'default': 15,
42
+ 'range': [1, 100]
43
+ },
44
+ 'random_state': {
45
+ 'description': 'random_state',
46
+ 'default': 42
47
+ },
48
+ 'learning_rate': {
49
+ 'description': 'Learning rate',
50
+ 'default': 0.1,
51
+ 'range': [1e-8, 5.0]
52
+ },
53
+ 'n_estimators': {
54
+ 'description': 'Number of estimators',
55
+ 'default': 100,
56
+ 'range': [1, 500]
57
+ },
58
+ 'loss': {
59
+ 'description': 'Loss function to optimize.',
60
+ 'default': "squared_error",
61
+ 'categorical': ['squared_error', 'absolute_error', 'huber', 'quantile']
62
+ },
63
+ 'criterion': {
64
+ 'description': 'Split quality metric',
65
+ 'default': "friedman_mse",
66
+ 'categorical': ['friedman_mse', 'squared_error']
67
+ },
68
+ 'min_samples_leaf': {
69
+ 'description': 'Minimum samples at leaf nodes.',
70
+ 'default': 1,
71
+ 'range': [1, 15]
72
+ },
73
+ 'max_features': {
74
+ 'description': 'Fraction of features per split.',
75
+ 'default': 1.0,
76
+ 'range': [0.1, 1.0]
77
+ },
78
+ 'min_samples_split': {
79
+ 'description': 'Minimum samples to split nodes.',
80
+ 'default': 2,
81
+ 'range': [2, 20]
82
+ }
83
+ }
84
+ self.model: GradientBoostingRegressor = None
85
+
86
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
87
+ self.model = GradientBoostingRegressor(**self.passthrough_parameters())
88
+ self.model.fit(dataset.X, dataset.y)
89
+ return self
90
+
91
+ def suitable(self, dataset: Dataset) -> bool:
92
+ return dataset.type_of_target == 'continuous'
93
+
94
+ def priorize(self, candidate: Candidate = None) -> float: # pylint: disable=unused-argument
95
+ return 0.5
@@ -0,0 +1,105 @@
1
+ """[STEP] HistGradient Boosting Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.ensemble import HistGradientBoostingRegressor
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....decorators.all import is_step
9
+
10
+ @is_step('predictor', 'tabular', 'regressor')
11
+ class ActHistGradientBoostingRegressor(Predictor):
12
+ """[STEP] HistGradient Boosting Regressor"""
13
+
14
+ name: str = "HistGradient Boosting Regressor"
15
+ _description: str = textwrap.dedent('''\
16
+ HistGradientBoostingRegressor is a machine learning algorithm
17
+ that makes predictions for regression tasks using histogram-based gradient boosting.''')
18
+ _description_long: str = textwrap.dedent('''\
19
+ HistGradientBoostingRegressor is a type of gradient boosting
20
+ algorithm that uses histogram-based decision trees to model the relationship between
21
+ the input features and the output variable. It works by iteratively adding decision
22
+ trees to the model, where each tree is trained to correct the errors made by the
23
+ previous tree. The decision trees are constructed using histograms of the input
24
+ features, which allows for faster computation and more efficient memory
25
+ usage compared to other tree-based algorithms.
26
+ HistGradientBoostingRegressor also includes options for regularization,
27
+ such as L1 and L2 regularization, to prevent overfitting.''')
28
+ _usage: str = "Use when you need fast nonlinear tabular regression; leaner than ActCatBoostRegressor or ActExtraTreesRegressor. Applicable to medium to large tabular continuous targets with mostly numeric features. Avoid when data is tiny, mostly linear, or interpretability is critical."
29
+ refs: list[dict[str, Any]] = [
30
+ {
31
+ 'year': 2006,
32
+ 'name': 'Gaussian Processes for Machine Learning',
33
+ 'authors': [
34
+ 'Carl Edward Rasmussen',
35
+ 'Christopher K. I. Williams'
36
+ ],
37
+ 'doi': 'https://doi.org/10.7551/mitpress/3206.001.0001',
38
+ 'publisher': 'MIT Press 2006'
39
+ }
40
+ ]
41
+
42
+ def __init__(self):
43
+ self.configuration = {
44
+ 'l2_regularization': {
45
+ 'description': 'The L2 regularization parameter. \
46
+ Use 0 for no regularization (default).',
47
+ 'default': 1e-10,
48
+ 'range': [1e-10, 1.0]
49
+ },
50
+ 'quantile': {
51
+ 'description': 'If loss is “quantile”, this parameter specifies which quantile to \
52
+ be estimated and must be between 0 and 1.',
53
+ 'default': 0.5,
54
+ 'range': [0.1, 0.99]
55
+ },
56
+ 'learning_rate': {
57
+ 'description': 'The learning rate, also known as shrinkage.',
58
+ 'default': 0.1,
59
+ 'range': [0.01, 1.0]
60
+ },
61
+ 'max_leaf_nodes': {
62
+ 'description': 'The maximum number of leaves for each tree.',
63
+ 'default': 31,
64
+ 'range': [3, 2048]
65
+ },
66
+ 'min_samples_leaf': {
67
+ 'description': 'The minimum number of samples per leaf.',
68
+ 'default': 20,
69
+ 'range': [1, 200]
70
+ },
71
+ 'loss': {
72
+ 'description': 'The loss function to use in the boosting process.',
73
+ 'default': "squared_error",
74
+ 'categorical': ["absolute_error", "poisson", "quantile", "squared_error"]
75
+ },
76
+ 'n_iter_no_change': {
77
+ 'description': 'Used to determine when to “early stop”.',
78
+ 'default': 4,
79
+ 'range': [2, 15]
80
+ },
81
+ 'tol': {
82
+ 'description': 'The absolute tolerance to use when comparing \
83
+ scores during early stopping',
84
+ 'default': 1e-4,
85
+ 'range': [1e-8, 1e-2]
86
+ },
87
+ 'max_depth': {
88
+ 'description': 'The maximum depth of each tree',
89
+ 'default': 10,
90
+ 'range': [8, 25]
91
+ }
92
+ }
93
+ self.model: HistGradientBoostingRegressor = None
94
+
95
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
96
+ self.model = HistGradientBoostingRegressor(early_stopping=True,
97
+ **self.passthrough_parameters())
98
+ self.model.fit(dataset.X, dataset.y)
99
+ return self
100
+
101
+ def suitable(self, dataset: Dataset) -> bool:
102
+ return dataset.type_of_target == 'continuous'
103
+
104
+ def priorize(self, candidate: Candidate = None) -> float:
105
+ return 0.5 # neutral
@@ -0,0 +1,101 @@
1
+ """[STEP] Huber Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.linear_model import HuberRegressor
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....data_type import DataType
9
+ from ....decorators.all import is_step
10
+
11
+
12
+ @is_step('predictor', 'tabular', 'regressor')
13
+ class ActHuberRegressor(Predictor):
14
+ """[STEP] Huber Regressor"""
15
+
16
+ name: str = "Huber Regressor"
17
+ _description: str = textwrap.dedent('''\
18
+ HuberRegressor fits a linear model that is less sensitive to outliers
19
+ by combining squared and absolute error losses.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ HuberRegressor optimizes a loss that behaves like squared error for
22
+ small residuals and like absolute error for large residuals. This
23
+ makes it a robust choice for tabular regression when data can contain
24
+ outliers while retaining efficiency for clean data.''')
25
+ _usage: str = "Use when you need robust linear regression with outliers; prefer over ActElasticNetRegressor when outliers skew fits. Applicable to numeric tabular regression with mostly linear relationships. Avoid when nonlinear effects or categorical-heavy data suit ActCatBoostRegressor."
26
+ refs: list[dict[str, Any]] = [
27
+ {
28
+ 'year': 1964,
29
+ 'name': 'Robust Estimation of a Location Parameter',
30
+ 'authors': [
31
+ 'Peter J. Huber'
32
+ ],
33
+ 'doi': 'https://doi.org/10.1214/aoms/1177703732',
34
+ 'publisher': 'Annals of Mathematical Statistics'
35
+ }
36
+ ]
37
+
38
+ def __init__(self):
39
+ self.configuration = {
40
+ 'epsilon': {
41
+ 'description': textwrap.dedent('''\
42
+ Threshold that controls the point where the loss switches
43
+ from quadratic to linear.'''),
44
+ 'default': 1.35,
45
+ 'range': [1.01, 3.0]
46
+ },
47
+ 'alpha': {
48
+ 'description': 'L2 regularization strength.',
49
+ 'default': 0.0001,
50
+ 'range': [1e-07, 1.0]
51
+ },
52
+ 'fit_intercept': {
53
+ 'description': 'Whether to fit the intercept term.',
54
+ 'default': True,
55
+ 'categorical': [True, False]
56
+ },
57
+ 'max_iter': {
58
+ 'description': 'Maximum number of iterations.',
59
+ 'default': 100,
60
+ 'range': [50, 2000]
61
+ },
62
+ 'tol': {
63
+ 'description': 'Stopping criterion.',
64
+ 'default': 1e-05,
65
+ 'range': [1e-06, 0.1]
66
+ },
67
+ 'warm_start': {
68
+ 'description': 'Reuse solution from previous fit as initialization.',
69
+ 'default': False,
70
+ 'categorical': [True, False]
71
+ }
72
+ }
73
+ self.model: HuberRegressor = None
74
+ self.columns: list[str] = []
75
+
76
+ def _select_features(self, X):
77
+ if self.columns and hasattr(X, 'columns'):
78
+ return X[self.columns]
79
+ return X
80
+
81
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
82
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
83
+ if not self.columns:
84
+ self.columns = dataset.features
85
+
86
+ self.model = HuberRegressor(**self.passthrough_parameters())
87
+ self.model.fit(self._select_features(dataset.X), dataset.y)
88
+ return self
89
+
90
+ def predict(self, X):
91
+ return super().predict(self._select_features(X))
92
+
93
+ def score(self, X, y=None, *args, **kwargs):
94
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
95
+
96
+ def suitable(self, dataset: Dataset) -> bool:
97
+ return dataset.type_of_target == 'continuous' \
98
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
99
+
100
+ def priorize(self, candidate: Candidate = None) -> float:
101
+ return 0.5 # neutral
@@ -0,0 +1,86 @@
1
+ """[STEP] KNN"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.neighbors import KNeighborsRegressor
5
+ from ....predictor import Predictor
6
+ from ....candidate import Candidate
7
+ from ....dataset import Dataset
8
+ from ....decorators.all import is_step
9
+
10
+
11
+ @is_step('predictor', 'tabular', 'fast_predictor', 'regressor')
12
+ class ActKNNRegressor(Predictor):
13
+ """[STEP] KNN"""
14
+
15
+ name: str = "KNN"
16
+ _usage: str = "Use when local neighbor patterns matter and you need a fast baseline; compare ActDecisionTreeRegressor or ActExtraTreesRegressor. Applicable to numeric tabular regression with moderate feature scales. Avoid when data is high-dimensional, very large, or needs extrapolation."
17
+ _description: str = textwrap.dedent('''\
18
+ KNeighborsRegressor is a machine learning algorithm that makes
19
+ predictions for regression tasks using k-nearest neighbors.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ KNeighborsRegressor is a type of instance-based learning
22
+ algorithm that makes predictions for new input features based on the values of
23
+ the k-nearest neighbors in the training data. It works by calculating the distance
24
+ between the new input features and all the training data, and then selecting
25
+ the k-nearest neighbors based on that distance. The output variable for the new
26
+ input features is then calculated as the average of the output variables
27
+ for the k-nearest neighbors.''')
28
+ refs: list[dict[str, Any]] = [
29
+ {
30
+ 'year': 1951,
31
+ 'name': 'Discriminatory Analysis, Nonparametric Discrimination: Consistency Properties',
32
+ 'authors': [
33
+ 'Evelyn Fix',
34
+ 'Joseph Lawson Hodges Jr.'
35
+ ],
36
+ 'doi': 'https://doi.org/10.2307/1403797',
37
+ 'publisher': 'Technical Report 4, USAF School of Aviation Medicine, Randolph Field'
38
+ },
39
+ {
40
+ 'year': 1967,
41
+ 'name': 'Nearest neighbor pattern classification',
42
+ 'authors': [
43
+ 'Thomas M. Cover',
44
+ 'Peter E. Hart'
45
+ ],
46
+ 'doi': 'https://doi.org/10.1109/TIT.1967.1053964',
47
+ 'publisher': 'IEEE Transactions on Information Theory. 13: page 21--27'
48
+ }
49
+ ]
50
+
51
+ def __init__(self):
52
+ self.configuration = {
53
+ 'metric': {
54
+ 'description': 'Can be minkowski or manhattan',
55
+ 'default': 'minkowski',
56
+ 'categorical': ['minkowski', 'manhattan']
57
+ },
58
+ 'n_neighbors': {
59
+ 'description': 'Number of neighbors',
60
+ 'default': 5,
61
+ 'range': [1, 200],
62
+ 'passthrough': False
63
+ },
64
+ 'weights': {
65
+ 'description': 'Weight function used in prediction.',
66
+ 'default': 'uniform',
67
+ 'categorical': ['uniform', 'distance']
68
+ }
69
+ }
70
+ self.model: KNeighborsRegressor = None
71
+
72
+ def fit(self, dataset: Dataset):
73
+ self.model = KNeighborsRegressor(
74
+ n_neighbors = min(self.get_config('n_neighbors'), dataset.X.shape[0]),
75
+ **self.passthrough_parameters()
76
+ )
77
+
78
+ self.model.fit(dataset.X, dataset.y)
79
+
80
+ return self
81
+
82
+ def suitable(self, dataset: Dataset) -> bool:
83
+ return dataset.type_of_target in ['continuous']
84
+
85
+ def priorize(self, candidate: Candidate = None) -> float:
86
+ return 0.5 # neutral