PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,103 @@
1
+ """[STEP] Lasso Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.linear_model import Lasso
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....data_type import DataType
9
+ from ....decorators.all import is_step
10
+
11
+
12
+ @is_step('predictor', 'tabular', 'regressor')
13
+ class ActLassoRegressor(Predictor):
14
+ """[STEP] Lasso Regressor"""
15
+
16
+ name: str = "Lasso Regressor"
17
+ _description: str = textwrap.dedent('''\
18
+ Lasso regression uses L1 regularization to shrink coefficients and
19
+ perform automatic feature selection in linear regression.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ Lasso (Least Absolute Shrinkage and Selection Operator) fits a linear
22
+ model while adding an L1 penalty to the loss. The penalty drives some
23
+ coefficients to zero, selecting a sparse set of features and improving
24
+ interpretability for tabular regression tasks.''')
25
+ _usage: str = "Use when you want sparse linear coefficients and feature selection; compare ActElasticNetRegressor or ActARDRegression for similar linear shrinkage. Applicable to numeric tabular regression with continuous targets. Avoid when effects are strongly nonlinear or dominated by categorical features."
26
+ refs: list[dict[str, Any]] = [
27
+ {
28
+ 'year': 1996,
29
+ 'name': 'Regression Shrinkage and Selection via the Lasso',
30
+ 'authors': [
31
+ 'Robert Tibshirani'
32
+ ],
33
+ 'doi': 'https://doi.org/10.1111/j.2517-6161.1996.tb02080.x',
34
+ 'publisher': 'Journal of the Royal Statistical Society Series B'
35
+ }
36
+ ]
37
+
38
+ def __init__(self):
39
+ self.configuration = {
40
+ 'alpha': {
41
+ 'description': 'Regularization strength.',
42
+ 'default': 1.0,
43
+ 'range': [1e-04, 10.0]
44
+ },
45
+ 'fit_intercept': {
46
+ 'description': 'Whether to fit the intercept term.',
47
+ 'default': True,
48
+ 'categorical': [True, False]
49
+ },
50
+ 'max_iter': {
51
+ 'description': 'Maximum number of iterations.',
52
+ 'default': 1000,
53
+ 'range': [100, 5000]
54
+ },
55
+ 'tol': {
56
+ 'description': 'Stopping criterion.',
57
+ 'default': 0.0001,
58
+ 'range': [1e-05, 0.1]
59
+ },
60
+ 'selection': {
61
+ 'description': 'Coordinate descent selection strategy.',
62
+ 'default': 'cyclic',
63
+ 'categorical': ['cyclic', 'random']
64
+ },
65
+ 'positive': {
66
+ 'description': 'Force coefficients to be positive.',
67
+ 'default': False,
68
+ 'categorical': [True, False]
69
+ },
70
+ 'random_state': {
71
+ 'description': 'Random state used when selection is "random".',
72
+ 'default': 42
73
+ }
74
+ }
75
+ self.model: Lasso = None
76
+ self.columns: list[str] = []
77
+
78
+ def _select_features(self, X):
79
+ if self.columns and hasattr(X, 'columns'):
80
+ return X[self.columns]
81
+ return X
82
+
83
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
84
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
85
+ if not self.columns:
86
+ self.columns = dataset.features
87
+
88
+ self.model = Lasso(**self.passthrough_parameters())
89
+ self.model.fit(self._select_features(dataset.X), dataset.y)
90
+ return self
91
+
92
+ def predict(self, X):
93
+ return super().predict(self._select_features(X))
94
+
95
+ def score(self, X, y=None, *args, **kwargs):
96
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
97
+
98
+ def suitable(self, dataset: Dataset) -> bool:
99
+ return dataset.type_of_target == 'continuous' \
100
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
101
+
102
+ def priorize(self, candidate: Candidate = None) -> float:
103
+ return 0.5 # neutral
@@ -0,0 +1,201 @@
1
+ """[STEP] LightGBM Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+
5
+ try:
6
+ from lightgbm import LGBMRegressor
7
+ from lightgbm.basic import LightGBMError
8
+ _LGBM_ERRORS: tuple[type[Exception], ...] = (LightGBMError,)
9
+ except ImportError: # pragma: no cover - optional dependency
10
+ LGBMRegressor = None # type: ignore
11
+ _LGBM_ERRORS = tuple()
12
+
13
+ from ....predictor import Predictor
14
+ from ....dataset import Dataset
15
+ from ....candidate import Candidate
16
+ from ....data_type import DataType
17
+ from ....decorators.all import is_step
18
+
19
+
20
+ @is_step('predictor', 'tabular', 'regressor')
21
+ class ActLightGBMRegressor(Predictor):
22
+ """[STEP] LightGBM Regressor"""
23
+
24
+ name: str = "LightGBM Regressor"
25
+ _description: str = textwrap.dedent('''\
26
+ LightGBMRegressor is a gradient boosting algorithm that builds
27
+ decision trees efficiently for regression tasks.''')
28
+ _description_long: str = textwrap.dedent('''\
29
+ LightGBMRegressor trains an ensemble of decision trees using histogram-based
30
+ splits and leaf-wise growth. It is designed to be fast while preserving
31
+ accuracy on tabular regression problems.''')
32
+ _usage: str = "Use when you want tabular regression with mixed numeric/categorical and a booster alternative to ActCatBoostRegressor. Applicable to medium/large datasets with nonlinear interactions. Avoid when data is tiny or you need a simple linear model like ActARDRegression."
33
+ refs: list[dict[str, Any]] = [
34
+ {
35
+ 'year': 2017,
36
+ 'name': 'LightGBM: A Highly Efficient Gradient Boosting Decision Tree',
37
+ 'authors': [
38
+ 'Guolin Ke',
39
+ 'Qi Meng',
40
+ 'Thomas Finley',
41
+ 'Taifeng Wang',
42
+ 'Wei Chen',
43
+ 'Weidong Ma',
44
+ 'Qiwei Ye',
45
+ 'Tie-Yan Liu'
46
+ ],
47
+ 'doi': 'https://doi.org/10.48550/arXiv.1712.01005',
48
+ 'publisher': 'Advances in Neural Information Processing Systems 30 (NeurIPS 2017)'
49
+ }
50
+ ]
51
+
52
+ def __init__(self):
53
+ self.configuration = {
54
+ 'boosting_type': {
55
+ 'description': 'Type of boosting algorithm.',
56
+ 'default': 'gbdt',
57
+ 'categorical': ['gbdt', 'dart']
58
+ },
59
+ 'n_estimators': {
60
+ 'description': 'Number of boosting iterations.',
61
+ 'default': 200,
62
+ 'range': [50, 1000]
63
+ },
64
+ 'learning_rate': {
65
+ 'description': 'Shrinkage rate applied to each tree.',
66
+ 'default': 0.1,
67
+ 'range': [0.01, 1.0]
68
+ },
69
+ 'num_leaves': {
70
+ 'description': 'Maximum number of leaves in one tree.',
71
+ 'default': 31,
72
+ 'range': [7, 255]
73
+ },
74
+ 'max_depth': {
75
+ 'description': 'Maximum depth of a tree, -1 means no limit.',
76
+ 'default': -1,
77
+ 'categorical': [-1, 3, 5, 10, 15]
78
+ },
79
+ 'min_child_samples': {
80
+ 'description': 'Minimum number of data in one leaf.',
81
+ 'default': 20,
82
+ 'range': [5, 200]
83
+ },
84
+ 'subsample': {
85
+ 'description': 'Fraction of data to use for each boosting iteration.',
86
+ 'default': 1.0,
87
+ 'range': [0.5, 1.0]
88
+ },
89
+ 'subsample_freq': {
90
+ 'description': 'Frequency for subsampling, 0 means disabled.',
91
+ 'default': 0,
92
+ 'range': [0, 10]
93
+ },
94
+ 'colsample_bytree': {
95
+ 'description': 'Fraction of features used for each tree.',
96
+ 'default': 1.0,
97
+ 'range': [0.5, 1.0]
98
+ },
99
+ 'reg_alpha': {
100
+ 'description': 'L1 regularization.',
101
+ 'default': 0.0,
102
+ 'range': [0.0, 1.0]
103
+ },
104
+ 'reg_lambda': {
105
+ 'description': 'L2 regularization.',
106
+ 'default': 0.0,
107
+ 'range': [0.0, 1.0]
108
+ },
109
+ 'random_state': {
110
+ 'description': 'Random seed for reproducibility.',
111
+ 'default': 42
112
+ },
113
+ 'verbosity': {
114
+ 'description': 'Controls the level of LightGBM verbosity.',
115
+ 'default': -1,
116
+ 'categorical': [-1, 0, 1]
117
+ }
118
+ }
119
+ self.model: LGBMRegressor = None
120
+ self.columns: list[str] = []
121
+ self.categorical_columns: list[str] = []
122
+ self._category_levels: dict[str, list] = {}
123
+
124
+ def _select_features(self, X):
125
+ if self.columns and hasattr(X, 'columns'):
126
+ return X[self.columns]
127
+ return X
128
+
129
+ def _prepare_features(self, X, fit: bool = False):
130
+ X_selected = self._select_features(X)
131
+ if not hasattr(X_selected, 'copy'):
132
+ return X_selected
133
+ X_prepared = X_selected.copy()
134
+ if self.categorical_columns:
135
+ for column in self.categorical_columns:
136
+ if column not in X_prepared.columns:
137
+ continue
138
+ X_prepared[column] = X_prepared[column].astype('category')
139
+ if not fit and column in self._category_levels:
140
+ X_prepared[column] = X_prepared[column].cat.set_categories(
141
+ self._category_levels[column]
142
+ )
143
+ if fit:
144
+ self._category_levels = {
145
+ column: list(X_prepared[column].cat.categories)
146
+ for column in self.categorical_columns
147
+ if column in X_prepared.columns and hasattr(X_prepared[column], 'cat')
148
+ }
149
+ return X_prepared
150
+
151
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
152
+ if LGBMRegressor is None:
153
+ raise ImportError(
154
+ "lightgbm is required for ActLightGBMRegressor. "
155
+ "Install with: pip install lightgbm"
156
+ )
157
+
158
+ self.columns = dataset.get_columns_names_by_type(
159
+ [DataType.NUMERIC, DataType.CATEGORICAL]
160
+ )
161
+ if not self.columns:
162
+ self.columns = dataset.features
163
+ self.categorical_columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
164
+
165
+ X_prepared = self._prepare_features(dataset.X, fit=True)
166
+ self.model = LGBMRegressor(**self.passthrough_parameters())
167
+
168
+ categorical_features = []
169
+ if self.categorical_columns and hasattr(X_prepared, 'columns'):
170
+ categorical_features = [
171
+ col for col in self.categorical_columns if col in X_prepared.columns
172
+ ]
173
+
174
+ try:
175
+ if categorical_features:
176
+ self.model.fit(
177
+ X_prepared,
178
+ dataset.y,
179
+ categorical_feature=categorical_features
180
+ )
181
+ else:
182
+ self.model.fit(X_prepared, dataset.y)
183
+ except _LGBM_ERRORS as exc:
184
+ raise ValueError(f"LightGBMRegressor training failed: {exc}") from exc
185
+ return self
186
+
187
+ def predict(self, X):
188
+ return super().predict(self._prepare_features(X))
189
+
190
+ def score(self, X, y=None, *args, **kwargs):
191
+ return self.model.score(self._prepare_features(X), y, *args, **kwargs)
192
+
193
+ def suitable(self, dataset: Dataset) -> bool:
194
+ supported = dataset.get_columns_names_by_type(
195
+ [DataType.NUMERIC, DataType.CATEGORICAL]
196
+ )
197
+ return LGBMRegressor is not None and dataset.type_of_target in \
198
+ ['continuous'] and bool(supported)
199
+
200
+ def priorize(self, candidate: Candidate = None) -> float:
201
+ return 0.5 # neutral
@@ -0,0 +1,43 @@
1
+ """
2
+ [STEP] Linear Regression
3
+ """
4
+ import textwrap
5
+ from typing import Any
6
+ from sklearn.linear_model import LinearRegression
7
+ from ....predictor import Predictor
8
+ from ....dataset import Dataset
9
+ from ....candidate import Candidate
10
+ from ....decorators.all import is_step
11
+
12
+ @is_step('predictor', 'tabular', 'fast_predictor', 'regressor', 'baseline_predictor')
13
+ class ActLinearRegression(Predictor):
14
+ """[STEP] Linear Regression"""
15
+
16
+ name: str = "Linear Regression"
17
+ _description: str = textwrap.dedent('''\
18
+ LinearRegression is a machine learning algorithm that models the
19
+ relationship between input features and a continuous output variable using
20
+ a linear function.''')
21
+ _description_long: str = textwrap.dedent('''\
22
+ LinearRegression is a type of regression algorithm that models
23
+ the relationship between input features and a continuous output variable using
24
+ a linear function. It works by finding the best-fitting line or hyperplane
25
+ that minimizes the sum of the squared differences between the predicted
26
+ and actual output variables.''')
27
+ _usage: str = "Use when you need a fast linear baseline before ActElasticNetRegressor or ActDecisionTreeRegressor. Applicable to tabular regression with roughly linear relationships. Avoid when strong nonlinearity, interactions, or heavy regularization is needed."
28
+ refs: list[dict[str, Any]] = []
29
+
30
+ def __init__(self):
31
+ self.model: LinearRegression = None
32
+
33
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
34
+ self.model = LinearRegression()
35
+ self.model.fit(dataset.X, dataset.y)
36
+
37
+ return self
38
+
39
+ def suitable(self, dataset: Dataset) -> bool:
40
+ return dataset.type_of_target == 'continuous'
41
+
42
+ def priorize(self, candidate: Candidate = None) -> float:
43
+ return 0.5 # neutral
@@ -0,0 +1,104 @@
1
+ """[STEP] MLP Regressor"""
2
+ from typing import Any
3
+ import textwrap
4
+ from sklearn.neural_network import MLPRegressor
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....decorators.all import is_step
9
+
10
+ @is_step('predictor', 'tabular', 'regressor')
11
+ class ActMLPRegressor(Predictor):
12
+ """[STEP] MLP Regressor"""
13
+
14
+ name: str = "MLP Regressor"
15
+ _usage: str = "Use when nonlinear tabular regression needs a flexible MLP, beyond ActElasticNetRegressor or ActDecisionTreeRegressor. Applicable to continuous targets with mostly numeric, scaled features. Avoid when data are tiny, mostly categorical, or you need fast/transparent models."
16
+ _description: str = textwrap.dedent('''\
17
+ MLPRegressor is a machine learning algorithm that models the
18
+ relationship between input features and a continuous output variable using
19
+ a multi-layer perceptron neural network.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ MLPRegressor is a type of neural network algorithm that models
22
+ the relationship between input features and a continuous output variable using a
23
+ multi-layer perceptron (MLP) neural network.
24
+ It works by transforming the input features through one or more hidden
25
+ layers with non-linear activation functions, and then using a final layer with
26
+ a linear activation function to output a continuous value.''')
27
+ refs: list[dict[str, Any]] = [
28
+ {
29
+ 'year': 1989,
30
+ 'name': 'Connectionist Learning Procedures',
31
+ 'authors': [
32
+ 'Geoffrey E. Hinton'
33
+ ],
34
+ 'doi': 'https://doi.org/10.1016/0004-3702(89)90049-0',
35
+ 'publisher': 'Artificial intelligence Vol. 40.1 page 185--234'
36
+ },
37
+ {
38
+ 'year': 2010,
39
+ 'name': 'Understanding the difficulty of training deep feedforward neural networks',
40
+ 'authors': [
41
+ 'Xavier Glorot',
42
+ 'Yoshua Bengio'
43
+ ],
44
+ 'doi': '',
45
+ 'publisher': (
46
+ 'Proceedings of the Thirteenth International Conference on '
47
+ 'Artificial Intelligence and Statistics page 249--256'
48
+ )
49
+ }
50
+ ]
51
+
52
+ def __init__(self):
53
+ self.configuration = {
54
+ 'activation': {
55
+ 'description': 'Activation function for the hidden layer.',
56
+ 'default': 'relu',
57
+ 'categorical': ["tanh", "relu"]
58
+ },
59
+ 'alpha': {
60
+ 'description': textwrap.dedent('''\
61
+ Strength of the L2 regularization term. The L2
62
+ regularization term is divided by the sample size when
63
+ added to the loss.'''),
64
+ 'default': 0.0001,
65
+ 'range': [1e-07, 0.1]
66
+ },
67
+ 'hidden_layer_count': {
68
+ 'description': 'Number of hidden layer',
69
+ 'default': 1,
70
+ 'range': [1, 4],
71
+ 'passthrough': False
72
+ },
73
+ 'node_per_layer': {
74
+ 'description': 'Number of node per layer',
75
+ 'default': 32,
76
+ 'range': [16, 256],
77
+ 'passthrough': False
78
+ },
79
+ 'learning_rate_init': {
80
+ 'description': 'Learning rate schedule for weight updates',
81
+ 'default': 0.001,
82
+ 'range': [0.0001, 0.5]
83
+ }
84
+ }
85
+
86
+ self.model: MLPRegressor = None
87
+
88
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
89
+ self.model = MLPRegressor(
90
+ hidden_layer_sizes=[self.get_config('node_per_layer') \
91
+ for i in range(self.get_config('hidden_layer_count'))],
92
+ early_stopping=True,
93
+ max_iter=400,
94
+ **self.passthrough_parameters())
95
+
96
+ self.model.fit(dataset.X, dataset.y)
97
+
98
+ return self
99
+
100
+ def suitable(self, dataset: Dataset) -> bool:
101
+ return dataset.type_of_target == "continuous"
102
+
103
+ def priorize(self, candidate: Candidate = None) -> float:
104
+ return 0.5 # neutral
@@ -0,0 +1,111 @@
1
+ """[STEP] Poisson Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+
5
+ import numpy as np
6
+ from sklearn.linear_model import PoissonRegressor
7
+
8
+ from ....predictor import Predictor
9
+ from ....dataset import Dataset
10
+ from ....candidate import Candidate
11
+ from ....data_type import DataType
12
+ from ....decorators.all import is_step
13
+
14
+
15
+ @is_step('predictor', 'tabular', 'regressor')
16
+ class ActPoissonRegressor(Predictor):
17
+ """[STEP] Poisson Regressor"""
18
+
19
+ name: str = "Poisson Regressor"
20
+ _description: str = textwrap.dedent('''\
21
+ PoissonRegressor models count targets using a log link and a Poisson
22
+ likelihood to produce positive predictions.''')
23
+ _description_long: str = textwrap.dedent('''\
24
+ Poisson regression is a generalized linear model designed for
25
+ non-negative count data. It connects predictors to the expected
26
+ count through a log link, which keeps predictions positive and
27
+ is appropriate when variance grows with the mean.''')
28
+ _usage: str = "Use when modeling non-negative count targets with variance rising with mean, as a simpler option than ActElasticNetRegressor or ActDecisionTreeRegressor. Applicable to tabular numeric features with count outcomes. Avoid when targets are continuous, negative, or highly zero-inflated."
29
+ refs: list[dict[str, Any]] = [
30
+ {
31
+ 'year': 1972,
32
+ 'name': 'Generalized Linear Models',
33
+ 'authors': [
34
+ 'John A. Nelder',
35
+ 'Robert W. M. Wedderburn'
36
+ ],
37
+ 'doi': 'https://doi.org/10.2307/2344614',
38
+ 'publisher': 'Journal of the Royal Statistical Society, Series A'
39
+ }
40
+ ]
41
+
42
+ def __init__(self):
43
+ self.configuration = {
44
+ 'alpha': {
45
+ 'description': 'L2 regularization strength.',
46
+ 'default': 1.0,
47
+ 'range': [1e-06, 10.0]
48
+ },
49
+ 'fit_intercept': {
50
+ 'description': 'Whether to fit the intercept term.',
51
+ 'default': True,
52
+ 'categorical': [True, False]
53
+ },
54
+ 'max_iter': {
55
+ 'description': 'Maximum number of iterations.',
56
+ 'default': 100,
57
+ 'range': [50, 2000]
58
+ },
59
+ 'tol': {
60
+ 'description': 'Stopping criterion.',
61
+ 'default': 0.0001,
62
+ 'range': [1e-06, 0.1]
63
+ },
64
+ 'warm_start': {
65
+ 'description': 'Reuse solution from the previous fit.',
66
+ 'default': False,
67
+ 'categorical': [True, False]
68
+ }
69
+ }
70
+ self.model: PoissonRegressor = None
71
+ self.columns: list[str] = []
72
+
73
+ def _select_features(self, X):
74
+ if self.columns and hasattr(X, 'columns'):
75
+ return X[self.columns]
76
+ return X
77
+
78
+ def _is_count_target(self, y) -> bool:
79
+ y_array = np.asarray(y)
80
+ if y_array.size == 0:
81
+ return False
82
+ if not np.issubdtype(y_array.dtype, np.number):
83
+ return False
84
+ if not np.isfinite(y_array).all():
85
+ return False
86
+ if (y_array < 0).any():
87
+ return False
88
+ return np.allclose(y_array, np.round(y_array), rtol=0, atol=1e-06)
89
+
90
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
91
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
92
+ if not self.columns:
93
+ self.columns = dataset.features
94
+
95
+ self.model = PoissonRegressor(**self.passthrough_parameters())
96
+ self.model.fit(self._select_features(dataset.X), dataset.y)
97
+ return self
98
+
99
+ def predict(self, X):
100
+ return super().predict(self._select_features(X))
101
+
102
+ def score(self, X, y=None, *args, **kwargs):
103
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
104
+
105
+ def suitable(self, dataset: Dataset) -> bool:
106
+ return dataset.type_of_target == 'continuous' \
107
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC)) \
108
+ and self._is_count_target(dataset.y)
109
+
110
+ def priorize(self, candidate: Candidate = None) -> float:
111
+ return 0.5 # neutral
@@ -0,0 +1,87 @@
1
+ """[STEP] Quantile Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+
5
+ from sklearn.linear_model import QuantileRegressor
6
+
7
+ from ....predictor import Predictor
8
+ from ....dataset import Dataset
9
+ from ....candidate import Candidate
10
+ from ....data_type import DataType
11
+ from ....decorators.all import is_step
12
+
13
+
14
+ @is_step('predictor', 'tabular', 'regressor')
15
+ class ActQuantileRegressor(Predictor):
16
+ """[STEP] Quantile Regressor"""
17
+
18
+ name: str = "Quantile Regressor"
19
+ _usage: str = "Use when you need conditional quantiles or asymmetric error control; choose over ActElasticNetRegressor for interval focus. Applicable to tabular regression with continuous targets and numeric inputs. Avoid when mean prediction is enough or nonlinear structure dominates."
20
+ _description: str = textwrap.dedent('''\
21
+ QuantileRegressor estimates a conditional quantile of a continuous target,
22
+ enabling interval-style predictions and asymmetric error handling.''')
23
+ _description_long: str = textwrap.dedent('''\
24
+ Quantile regression models a chosen quantile of the response rather than the
25
+ mean, which makes it useful for prediction intervals and robust modeling.
26
+ By selecting different quantiles, the model can describe the uncertainty
27
+ around the target distribution.''')
28
+ refs: list[dict[str, Any]] = [
29
+ {
30
+ 'year': 1978,
31
+ 'name': 'Regression Quantiles',
32
+ 'authors': [
33
+ 'Roger Koenker',
34
+ 'Gilbert Bassett Jr.'
35
+ ],
36
+ 'doi': 'https://doi.org/10.2307/1913643',
37
+ 'publisher': 'Econometrica'
38
+ }
39
+ ]
40
+
41
+ def __init__(self):
42
+ self.configuration = {
43
+ 'quantile': {
44
+ 'description': 'Quantile to estimate between 0 and 1.',
45
+ 'default': 0.5,
46
+ 'range': [0.05, 0.95]
47
+ },
48
+ 'alpha': {
49
+ 'description': 'L1 regularization strength.',
50
+ 'default': 1.0,
51
+ 'range': [1e-06, 10.0]
52
+ },
53
+ 'fit_intercept': {
54
+ 'description': 'Whether to fit the intercept term.',
55
+ 'default': True,
56
+ 'categorical': [True, False]
57
+ }
58
+ }
59
+ self.model: QuantileRegressor = None
60
+ self.columns: list[str] = []
61
+
62
+ def _select_features(self, X):
63
+ if self.columns and hasattr(X, 'columns'):
64
+ return X[self.columns]
65
+ return X
66
+
67
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
68
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
69
+ if not self.columns:
70
+ self.columns = dataset.features
71
+
72
+ self.model = QuantileRegressor(**self.passthrough_parameters())
73
+ self.model.fit(self._select_features(dataset.X), dataset.y)
74
+ return self
75
+
76
+ def predict(self, X):
77
+ return super().predict(self._select_features(X))
78
+
79
+ def score(self, X, y=None, *args, **kwargs):
80
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
81
+
82
+ def suitable(self, dataset: Dataset) -> bool:
83
+ return dataset.type_of_target == 'continuous' \
84
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
85
+
86
+ def priorize(self, candidate: Candidate = None) -> float:
87
+ return 0.5 # neutral