PyIAML 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. iaml/__init__.py +56 -0
  2. iaml/actionable.py +11 -0
  3. iaml/actionables/__init__.py +21 -0
  4. iaml/actionables/boosting/__init__.py +4 -0
  5. iaml/actionables/boosting/act_adaboost.py +59 -0
  6. iaml/actionables/cleaning/__init__.py +26 -0
  7. iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
  8. iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
  9. iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
  10. iaml/actionables/cleaning/act_drop_date_column.py +48 -0
  11. iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
  12. iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
  13. iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
  14. iaml/actionables/cleaning/act_encode_target_column.py +56 -0
  15. iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
  16. iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
  17. iaml/actionables/cleaning/act_knn_imputer.py +152 -0
  18. iaml/actionables/cleaning/act_mean_column.py +79 -0
  19. iaml/actionables/cleaning/act_mice.py +464 -0
  20. iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
  21. iaml/actionables/cleaning/act_missing_indicator.py +124 -0
  22. iaml/actionables/cleaning/act_onehot.py +65 -0
  23. iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
  24. iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
  25. iaml/actionables/cleaning/act_simple_imputer.py +109 -0
  26. iaml/actionables/cleaning/act_split_date.py +68 -0
  27. iaml/actionables/cleaning/act_target_encoder.py +274 -0
  28. iaml/actionables/cleaning/act_text_normalizer.py +241 -0
  29. iaml/actionables/cleaning/act_tf_idf.py +80 -0
  30. iaml/actionables/cleaning/act_word2vec.py +150 -0
  31. iaml/actionables/features_precleaning/__init__.py +12 -0
  32. iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
  33. iaml/actionables/features_precleaning/act_date_converter.py +99 -0
  34. iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
  35. iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
  36. iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
  37. iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
  38. iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
  39. iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
  40. iaml/actionables/features_precleaning/act_trim_space.py +79 -0
  41. iaml/actionables/features_preprocessing/__init__.py +18 -0
  42. iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
  43. iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
  44. iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
  45. iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
  46. iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
  47. iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
  48. iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
  49. iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
  50. iaml/actionables/features_preprocessing/act_pca.py +77 -0
  51. iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
  52. iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
  53. iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
  54. iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
  55. iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
  56. iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
  57. iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
  58. iaml/actionables/features_selection/__init__.py +8 -0
  59. iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
  60. iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
  61. iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
  62. iaml/actionables/features_selection/act_rfe.py +214 -0
  63. iaml/actionables/features_selection/act_select_from_model.py +325 -0
  64. iaml/actionables/features_selection/act_select_k_best.py +181 -0
  65. iaml/actionables/features_selection/act_vif_selector.py +130 -0
  66. iaml/actionables/imbalance/__init__.py +10 -0
  67. iaml/actionables/imbalance/act_adasyn.py +150 -0
  68. iaml/actionables/imbalance/act_borderline_smote.py +171 -0
  69. iaml/actionables/imbalance/act_near_miss.py +158 -0
  70. iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
  71. iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
  72. iaml/actionables/imbalance/act_smote.py +162 -0
  73. iaml/actionables/imbalance/act_smote_tomek.py +182 -0
  74. iaml/actionables/imbalance/act_smoteenn.py +193 -0
  75. iaml/actionables/imbalance/act_tomek_links.py +138 -0
  76. iaml/actionables/normalize/__init__.py +6 -0
  77. iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
  78. iaml/actionables/normalize/act_minmax_scaler.py +56 -0
  79. iaml/actionables/normalize/act_normalizer.py +95 -0
  80. iaml/actionables/normalize/act_robust_scaler.py +111 -0
  81. iaml/actionables/normalize/act_standard_scaler.py +55 -0
  82. iaml/actionables/predictors/__init__.py +6 -0
  83. iaml/actionables/predictors/_xgboost.py +16 -0
  84. iaml/actionables/predictors/classifier/__init__.py +26 -0
  85. iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
  86. iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
  87. iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
  88. iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
  89. iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
  90. iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
  91. iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
  92. iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
  93. iaml/actionables/predictors/classifier/act_knn.py +86 -0
  94. iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
  95. iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
  96. iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
  97. iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
  98. iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
  99. iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
  100. iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
  101. iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
  102. iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
  103. iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
  104. iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
  105. iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
  106. iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
  107. iaml/actionables/predictors/regressor/__init__.py +27 -0
  108. iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
  109. iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
  110. iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
  111. iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
  112. iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
  113. iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
  114. iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
  115. iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
  116. iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
  117. iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
  118. iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
  119. iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
  120. iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
  121. iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
  122. iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
  123. iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
  124. iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
  125. iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
  126. iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
  127. iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
  128. iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
  129. iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
  130. iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
  131. iaml/actionables/predictors/survival/__init__.py +12 -0
  132. iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
  133. iaml/actionables/predictors/survival/act_cox.py +110 -0
  134. iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
  135. iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
  136. iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
  137. iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
  138. iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
  139. iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
  140. iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
  141. iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
  142. iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
  143. iaml/cache.py +61 -0
  144. iaml/cache_keys.py +57 -0
  145. iaml/candidate.py +736 -0
  146. iaml/core_dispatcher.py +125 -0
  147. iaml/data_type.py +11 -0
  148. iaml/dataset.py +506 -0
  149. iaml/decorators/__init__.py +3 -0
  150. iaml/decorators/all.py +4 -0
  151. iaml/decorators/is_step.py +45 -0
  152. iaml/decorators/runner.py +100 -0
  153. iaml/explanation.py +112 -0
  154. iaml/iaml.py +1072 -0
  155. iaml/iaml_pipeline.py +600 -0
  156. iaml/logger.py +138 -0
  157. iaml/meta_explorer_step.py +62 -0
  158. iaml/meta_ordered_step.py +28 -0
  159. iaml/meta_partial_explorer_step.py +34 -0
  160. iaml/meta_singleton.py +24 -0
  161. iaml/metastep.py +211 -0
  162. iaml/metric.py +111 -0
  163. iaml/metric_plot.py +82 -0
  164. iaml/metrics/__init__.py +21 -0
  165. iaml/metrics/_classification.py +28 -0
  166. iaml/metrics/_survival_times.py +22 -0
  167. iaml/metrics/accuracy_metric.py +59 -0
  168. iaml/metrics/balanced_accuracy_metric.py +67 -0
  169. iaml/metrics/brier_score.py +90 -0
  170. iaml/metrics/classification_error_metric.py +66 -0
  171. iaml/metrics/concordance_index_ipcw.py +84 -0
  172. iaml/metrics/concordance_index_metric.py +67 -0
  173. iaml/metrics/cumulative_dynamic_auc.py +119 -0
  174. iaml/metrics/f1_score_metric.py +71 -0
  175. iaml/metrics/integrated_brier_score.py +98 -0
  176. iaml/metrics/integrated_brier_score_loss.py +41 -0
  177. iaml/metrics/mean_absolute_error_metric.py +46 -0
  178. iaml/metrics/mean_squared_error_metric.py +46 -0
  179. iaml/metrics/mean_squared_log_error_metric.py +49 -0
  180. iaml/metrics/median_absolute_error_metric.py +48 -0
  181. iaml/metrics/precision_metric.py +63 -0
  182. iaml/metrics/r2_score_metric.py +45 -0
  183. iaml/metrics/recall_metric.py +65 -0
  184. iaml/metrics/roc_auc_metric.py +50 -0
  185. iaml/metrics/specificity_metric.py +44 -0
  186. iaml/metrics/specificity_multiclass_metric.py +55 -0
  187. iaml/metrics/specificity_multilabel_metric.py +60 -0
  188. iaml/optimizers/__init__.py +5 -0
  189. iaml/optimizers/bayesian_optimizer.py +193 -0
  190. iaml/optimizers/genetic_optimizer.py +284 -0
  191. iaml/optimizers/optimizer.py +31 -0
  192. iaml/optimizers/random_optimizer.py +101 -0
  193. iaml/plot.py +138 -0
  194. iaml/plots/__init__.py +32 -0
  195. iaml/plots/bar_plot.py +141 -0
  196. iaml/plots/box_plot.py +166 -0
  197. iaml/plots/class_prediction_error_plot.py +37 -0
  198. iaml/plots/classification_report_plot.py +35 -0
  199. iaml/plots/confusion_matrix_plot.py +34 -0
  200. iaml/plots/correlation_heatmap_plot.py +201 -0
  201. iaml/plots/cumulative_hazard_plot.py +72 -0
  202. iaml/plots/density_plot.py +210 -0
  203. iaml/plots/histogram_plot.py +179 -0
  204. iaml/plots/kaplan_meier_comparison_plot.py +89 -0
  205. iaml/plots/line_plot.py +70 -0
  206. iaml/plots/missingness_heatmap_plot.py +203 -0
  207. iaml/plots/outlier_plot.py +217 -0
  208. iaml/plots/pair_plot.py +228 -0
  209. iaml/plots/precision_recall_curve_plot.py +86 -0
  210. iaml/plots/prediction_error_plot.py +34 -0
  211. iaml/plots/qq_plot.py +220 -0
  212. iaml/plots/residual_plot.py +38 -0
  213. iaml/plots/roc_dynamique_curve_plot.py +79 -0
  214. iaml/plots/rocauc_plot.py +96 -0
  215. iaml/plots/shap_plot.py +187 -0
  216. iaml/plots/target_distribution_plot.py +241 -0
  217. iaml/plots/violin_plot.py +206 -0
  218. iaml/predictor.py +139 -0
  219. iaml/reference.py +65 -0
  220. iaml/shared_cache.py +90 -0
  221. iaml/sklearn_preprocessor.py +74 -0
  222. iaml/splitters/__init__.py +3 -0
  223. iaml/splitters/kfold_splitter.py +32 -0
  224. iaml/splitters/random_splitter.py +26 -0
  225. iaml/stack.py +39 -0
  226. iaml/statistic.py +66 -0
  227. iaml/statistics/__init__.py +77 -0
  228. iaml/statistics/anova_statistic.py +80 -0
  229. iaml/statistics/cardinality_ratio_statistic.py +63 -0
  230. iaml/statistics/category_cooccurrence_statistic.py +79 -0
  231. iaml/statistics/chi_square_statistic.py +81 -0
  232. iaml/statistics/coef_variation_statistic.py +72 -0
  233. iaml/statistics/correlation_with_target.py +105 -0
  234. iaml/statistics/count.py +72 -0
  235. iaml/statistics/data_type_summary_statistic.py +74 -0
  236. iaml/statistics/duplicate_row_statistic.py +56 -0
  237. iaml/statistics/effect_size_statistic.py +129 -0
  238. iaml/statistics/entropy_statistic.py +69 -0
  239. iaml/statistics/event_rate_statistic.py +52 -0
  240. iaml/statistics/grouped_mean_statistic.py +60 -0
  241. iaml/statistics/iqr_statistic.py +66 -0
  242. iaml/statistics/kurtosis.py +50 -0
  243. iaml/statistics/mad_statistic.py +66 -0
  244. iaml/statistics/mean.py +61 -0
  245. iaml/statistics/median_statistic.py +61 -0
  246. iaml/statistics/minmax.py +60 -0
  247. iaml/statistics/missing_rate_statistic.py +62 -0
  248. iaml/statistics/mode.py +47 -0
  249. iaml/statistics/most_frequent_ratio.py +81 -0
  250. iaml/statistics/outlier_count_iqr_statistic.py +76 -0
  251. iaml/statistics/quantile.py +59 -0
  252. iaml/statistics/range.py +53 -0
  253. iaml/statistics/rare_category_rate.py +92 -0
  254. iaml/statistics/skewness.py +53 -0
  255. iaml/statistics/stdev.py +50 -0
  256. iaml/statistics/summary_table_statistic.py +60 -0
  257. iaml/statistics/time_by_group_statistic.py +83 -0
  258. iaml/statistics/time_summary_statistic.py +56 -0
  259. iaml/statistics/top_k_value_counts.py +68 -0
  260. iaml/statistics/unique_count_statistic.py +57 -0
  261. iaml/statistics/value_counts.py +63 -0
  262. iaml/statistics/variance.py +51 -0
  263. iaml/statistics/violin.py +63 -0
  264. iaml/step.py +600 -0
  265. iaml/step_cache.py +87 -0
  266. iaml/step_wrapper.py +79 -0
  267. iaml/timed_pool_executor.py +492 -0
  268. iaml/type_of_target.py +68 -0
  269. iaml/void_step.py +101 -0
  270. iaml/worker_manager.py +169 -0
  271. iaml/wrapper/__init__.py +4 -0
  272. iaml/wrapper/wrap_basic_gridsearch.py +68 -0
  273. iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
  274. iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
  275. pyiaml-1.0.0.dist-info/METADATA +802 -0
  276. pyiaml-1.0.0.dist-info/RECORD +279 -0
  277. pyiaml-1.0.0.dist-info/WHEEL +5 -0
  278. pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
  279. pyiaml-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,116 @@
1
+ """[STEP] Random Forest Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.ensemble import RandomForestRegressor
5
+ from ....data_type import DataType
6
+ from ....predictor import Predictor
7
+ from ....dataset import Dataset
8
+ from ....candidate import Candidate
9
+ from ....decorators.all import is_step
10
+
11
+
12
+ @is_step('predictor', 'tabular', 'regressor')
13
+ class ActRandomForestRegressor(Predictor):
14
+ """[STEP] Random Forest Regressor"""
15
+
16
+ name: str = "Random Forest Regressor"
17
+ _description: str = textwrap.dedent('''\
18
+ RandomForestRegressor is a machine learning algorithm that
19
+ models the relationship between input features and a continuous output
20
+ variable using a collection of decision trees.''')
21
+ _description_long: str = textwrap.dedent('''\
22
+ RandomForestRegressor is a type of ensemble learning algorithm
23
+ that models the relationship between input features and a continuous output variable
24
+ using a collection of decision trees. It works by building multiple decision trees on
25
+ random subsets of the input features and data, and then averaging the predictions of each
26
+ tree to make the final prediction.''')
27
+ _usage: str = "Use when you want a robust nonlinear tabular baseline; more stable than ActDecisionTreeRegressor. Applicable to continuous targets with numeric or encoded categorical features. Avoid when data is huge or latency is tight; consider ActExtraTreesRegressor."
28
+ refs: list[dict[str, Any]] = [
29
+ {
30
+ 'year': 2001,
31
+ 'name': 'Random Forests',
32
+ 'authors': ['Leo Breiman'],
33
+ 'doi': 'https://doi.org/10.1023/A:1010933404324',
34
+ 'publisher': 'Machine Learning Vol.45 page 5--32'
35
+ },
36
+ {
37
+ 'year': 2006,
38
+ 'name': 'Extremely Randomized Trees',
39
+ 'authors': ['Pierre Geurts', 'Damien Ernst', 'Lous Wehenkel'],
40
+ 'doi': 'https://doi.org/10.1007/s10994-006-6226-1',
41
+ 'publisher': 'Machine Learning Vol.63 page 5--42'
42
+ }
43
+ ]
44
+
45
+ def __init__(self):
46
+ self.configuration = {
47
+ # 'max_depth': { # Disable before probably better with no limit in regression
48
+ # 'description': 'Max depth of each tree',
49
+ # 'default': 15,
50
+ # 'range': [1, 100]
51
+ # },
52
+ 'n_estimators': {
53
+ 'description': 'Number of threes',
54
+ 'default': 100,
55
+ 'range': [1, 500]
56
+ },
57
+ 'random_state': {
58
+ 'description': 'random_state',
59
+ 'default': 42
60
+ },
61
+ 'min_samples_leaf': {
62
+ 'description': 'The minimum number of samples required to be at a leaf node.',
63
+ 'default': 1,
64
+ 'range': [1, 15]
65
+ },
66
+ 'max_features': {
67
+ 'description': 'The number of features to consider when looking for the best split',
68
+ 'default': 1.0,
69
+ 'range': [0.1, 1.0]
70
+ },
71
+ 'min_samples_split': {
72
+ 'description': 'The minimum number of samples required to split an internal node',
73
+ 'default': 2,
74
+ 'range': [2, 20]
75
+ },
76
+ 'bootstrap': {
77
+ 'description': textwrap.dedent('''\
78
+ Whether bootstrap samples are used when building trees. If
79
+ False, the whole dataset is used to build each tree.'''),
80
+ 'default': False
81
+ },
82
+ 'criterion': {
83
+ 'description': 'The function to measure the quality of a split.',
84
+ 'default': "squared_error",
85
+ 'categorical': ["poisson", "friedman_mse", "absolute_error", "squared_error"]
86
+ }
87
+ }
88
+ self.model: RandomForestRegressor = None
89
+ self.columns: list[str] = []
90
+
91
+ def _select_features(self, X):
92
+ if self.columns and hasattr(X, 'columns'):
93
+ return X[self.columns]
94
+ return X
95
+
96
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
97
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
98
+ if not self.columns:
99
+ self.columns = dataset.features
100
+
101
+ self.model = RandomForestRegressor(**self.passthrough_parameters())
102
+ self.model.fit(self._select_features(dataset.X), dataset.y)
103
+ return self
104
+
105
+ def predict(self, X):
106
+ return super().predict(self._select_features(X))
107
+
108
+ def score(self, X, y=None, *args, **kwargs):
109
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
110
+
111
+ def suitable(self, dataset: Dataset) -> bool:
112
+ return dataset.type_of_target in ['continuous'] \
113
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
114
+
115
+ def priorize(self, candidate: Candidate = None) -> float:
116
+ return 0.5 # neutral
@@ -0,0 +1,106 @@
1
+ """[STEP] RANSAC Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.linear_model import RANSACRegressor
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....data_type import DataType
9
+ from ....decorators.all import is_step
10
+
11
+
12
+ @is_step('predictor', 'tabular', 'regressor')
13
+ class ActRANSACRegressor(Predictor):
14
+ """[STEP] RANSAC Regressor"""
15
+
16
+ name: str = "RANSAC Regressor"
17
+ _usage: str = "Use when outliers can skew a linear model and you need robustness vs ActElasticNetRegressor. Applicable to numeric tabular regression with moderate sample size. Avoid when data is clean or effects are highly nonlinear; consider ActDecisionTreeRegressor."
18
+ _description: str = textwrap.dedent('''\
19
+ RANSACRegressor fits a robust regression model by iteratively
20
+ sampling subsets of the data and keeping inliers.''')
21
+ _description_long: str = textwrap.dedent('''\
22
+ RANSACRegressor is a robust regression technique that repeatedly fits
23
+ a base estimator on random subsets, identifies inliers using a
24
+ residual threshold, and refits on the consensus set. This approach
25
+ reduces the influence of outliers in tabular regression problems.''')
26
+ refs: list[dict[str, Any]] = [
27
+ {
28
+ 'year': 1981,
29
+ 'name': (
30
+ 'Random Sample Consensus: A Paradigm for Model Fitting with '
31
+ 'Applications to Image Analysis and Automated Cartography'
32
+ ),
33
+ 'authors': [
34
+ 'Martin A. Fischler',
35
+ 'Robert C. Bolles'
36
+ ],
37
+ 'doi': 'https://doi.org/10.1145/358669.358692',
38
+ 'publisher': 'Communications of the ACM'
39
+ }
40
+ ]
41
+
42
+ def __init__(self):
43
+ self.configuration = {
44
+ 'min_samples': {
45
+ 'description': textwrap.dedent('''\
46
+ Minimum number of samples chosen for estimating the model.
47
+ When None, uses the number of features plus one.'''),
48
+ 'default': None
49
+ },
50
+ 'residual_threshold': {
51
+ 'description': textwrap.dedent('''\
52
+ Maximum residual for a data sample to be classified as an inlier.
53
+ When None, uses the median absolute deviation of the target.'''),
54
+ 'default': None
55
+ },
56
+ 'max_trials': {
57
+ 'description': 'Maximum number of iterations for random sampling.',
58
+ 'default': 100,
59
+ 'range': [10, 500]
60
+ },
61
+ 'stop_probability': {
62
+ 'description': textwrap.dedent('''\
63
+ Probability that at least one outlier-free sample has been chosen
64
+ after max_trials iterations.'''),
65
+ 'default': 0.99,
66
+ 'range': [0.8, 0.999]
67
+ },
68
+ 'loss': {
69
+ 'description': 'Loss function used to classify inliers.',
70
+ 'default': 'absolute_error',
71
+ 'categorical': ['absolute_error', 'squared_error']
72
+ },
73
+ 'random_state': {
74
+ 'description': 'Random state for reproducibility.',
75
+ 'default': 42
76
+ }
77
+ }
78
+ self.model: RANSACRegressor = None
79
+ self.columns: list[str] = []
80
+
81
+ def _select_features(self, X):
82
+ if self.columns and hasattr(X, 'columns'):
83
+ return X[self.columns]
84
+ return X
85
+
86
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
87
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
88
+ if not self.columns:
89
+ self.columns = dataset.features
90
+
91
+ self.model = RANSACRegressor(**self.passthrough_parameters())
92
+ self.model.fit(self._select_features(dataset.X), dataset.y)
93
+ return self
94
+
95
+ def predict(self, X):
96
+ return super().predict(self._select_features(X))
97
+
98
+ def score(self, X, y=None, *args, **kwargs):
99
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
100
+
101
+ def suitable(self, dataset: Dataset) -> bool:
102
+ return dataset.type_of_target == 'continuous' \
103
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
104
+
105
+ def priorize(self, candidate: Candidate = None) -> float:
106
+ return 0.5 # neutral
@@ -0,0 +1,107 @@
1
+ """[STEP] Ridge Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn.linear_model import Ridge
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....data_type import DataType
9
+ from ....decorators.all import is_step
10
+
11
+
12
+ @is_step('predictor', 'tabular', 'regressor')
13
+ class ActRidgeRegressor(Predictor):
14
+ """[STEP] Ridge Regressor"""
15
+
16
+ name: str = "Ridge Regressor"
17
+ _description: str = textwrap.dedent('''\
18
+ Ridge regression applies L2 regularization to stabilize coefficients
19
+ when predictors are correlated.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ Ridge regression fits a linear model while penalizing large coefficients
22
+ with an L2 term. This reduces variance, improves numerical stability,
23
+ and provides robust predictions for tabular regression problems.''')
24
+ _usage: str = "Use when you need a stable linear regressor for correlated numeric features; compare ActElasticNetRegressor or ActARDRegression. Applicable to tabular regression with mostly numeric inputs. Avoid when strong nonlinear effects dominate or labels are not continuous."
25
+ refs: list[dict[str, Any]] = [
26
+ {
27
+ 'year': 1970,
28
+ 'name': 'Ridge Regression: Biased Estimation for Nonorthogonal Problems',
29
+ 'authors': [
30
+ 'Arthur E. Hoerl',
31
+ 'Robert W. Kennard'
32
+ ],
33
+ 'doi': 'https://doi.org/10.2307/1267351',
34
+ 'publisher': 'Technometrics Vol. 12, No. 1, page 55--67'
35
+ }
36
+ ]
37
+
38
+ def __init__(self):
39
+ self.configuration = {
40
+ 'alpha': {
41
+ 'description': 'Regularization strength.',
42
+ 'default': 1.0,
43
+ 'range': [1e-04, 100.0]
44
+ },
45
+ 'fit_intercept': {
46
+ 'description': 'Whether to fit the intercept term.',
47
+ 'default': True,
48
+ 'categorical': [True, False]
49
+ },
50
+ 'solver': {
51
+ 'description': 'Solver to use in the ridge optimization.',
52
+ 'default': 'auto',
53
+ 'categorical': [
54
+ 'auto',
55
+ 'svd',
56
+ 'cholesky',
57
+ 'lsqr',
58
+ 'sparse_cg',
59
+ 'sag',
60
+ 'saga',
61
+ 'lbfgs'
62
+ ]
63
+ },
64
+ 'tol': {
65
+ 'description': 'Stopping criterion for iterative solvers.',
66
+ 'default': 0.0001,
67
+ 'range': [1e-05, 0.1]
68
+ },
69
+ 'max_iter': {
70
+ 'description': 'Maximum number of iterations for iterative solvers.',
71
+ 'default': 1000,
72
+ 'range': [50, 5000]
73
+ },
74
+ 'random_state': {
75
+ 'description': 'Random state for solvers that use randomness.',
76
+ 'default': 42
77
+ }
78
+ }
79
+ self.model: Ridge = None
80
+ self.columns: list[str] = []
81
+
82
+ def _select_features(self, X):
83
+ if self.columns and hasattr(X, 'columns'):
84
+ return X[self.columns]
85
+ return X
86
+
87
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
88
+ self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
89
+ if not self.columns:
90
+ self.columns = dataset.features
91
+
92
+ self.model = Ridge(**self.passthrough_parameters())
93
+ self.model.fit(self._select_features(dataset.X), dataset.y)
94
+ return self
95
+
96
+ def predict(self, X):
97
+ return super().predict(self._select_features(X))
98
+
99
+ def score(self, X, y=None, *args, **kwargs):
100
+ return self.model.score(self._select_features(X), y, *args, **kwargs)
101
+
102
+ def suitable(self, dataset: Dataset) -> bool:
103
+ return dataset.type_of_target == 'continuous' \
104
+ and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
105
+
106
+ def priorize(self, candidate: Candidate = None) -> float:
107
+ return 0.5 # neutral
@@ -0,0 +1,106 @@
1
+ """[STEP] SGD Regressor"""
2
+ from typing import Any
3
+ import textwrap
4
+ from sklearn.linear_model import SGDRegressor
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....decorators.all import is_step
9
+
10
+ @is_step('predictor', 'tabular', 'regressor')
11
+ class ActSGDRegressor(Predictor):
12
+ """[STEP] SGD Regressor"""
13
+
14
+ name: str = "SGD Regressor"
15
+ _usage: str = "Use when you need a fast linear baseline on large tabular data, e.g., vs ActElasticNetRegressor. Applicable to scaled numeric features with a continuous target. Avoid when strong nonlinearities or best accuracy is needed; prefer ActCatBoostRegressor or ActExtraTreesRegressor."
16
+ _description: str = textwrap.dedent('''\
17
+ SGDRegressor is a machine learning algorithm that models
18
+ the relationship between input features and a continuous output variable
19
+ using stochastic gradient descent.''')
20
+ _description_long: str = textwrap.dedent('''\
21
+ SGDRegressor is a type of linear model that models the relationship
22
+ between input features and a continuous output variable using stochastic gradient descent.
23
+ It works by iteratively updating the model parameters in the direction of the negative
24
+ gradient of the loss function with respect to the parameters, using a single example
25
+ at a time.''')
26
+ refs: list[dict[str, Any]] = [
27
+ {
28
+ 'year': 1951,
29
+ 'name': 'A Stochastic Approximation Method',
30
+ 'authors': [
31
+ 'Herbert Robbins',
32
+ 'Sutton Monro'
33
+ ],
34
+ 'doi': 'https://doi.org/10.1214/aoms/1177729586',
35
+ 'publisher': 'The annals of Mathematical Statistics Vol.22 No.3 page 400--407'
36
+ }
37
+ ]
38
+ def __init__(self):
39
+ self.configuration = {
40
+ 'alpha': {
41
+ 'description': 'Constant that multiplies the regularization term.',
42
+ 'default': 0.0001,
43
+ 'range': [1e-07, 0.1]
44
+ },
45
+ 'tol': {
46
+ 'description': 'The stopping criterion.',
47
+ 'default': 0.0001,
48
+ 'range': [1e-05, 0.1]
49
+ },
50
+ 'epsilon': {
51
+ 'description': 'Epsilon in the epsilon-insensitive loss functions',
52
+ 'default': 0.1,
53
+ 'range': [1e-05, 0.1]
54
+ },
55
+ 'eta0': {
56
+ 'description': textwrap.dedent('''\
57
+ The initial learning rate for the ‘constant’, ‘invscaling’
58
+ or ‘adaptive’ schedules.'''),
59
+ 'default': 0.01,
60
+ 'range': [1e-07, 0.1]
61
+ },
62
+ 'l1_ratio': {
63
+ 'description': 'The Elastic Net mixing parameter',
64
+ 'default': 0.15,
65
+ 'range': [1e-09, 1.0]
66
+ },
67
+ 'power_t': {
68
+ 'description': 'The exponent for inverse scaling learning rate.',
69
+ 'default': 0.25,
70
+ 'range': [1e-05, 1.0]
71
+ },
72
+ 'average': {
73
+ 'description': textwrap.dedent('''\
74
+ When set to True, computes the averaged SGD weights across
75
+ all updates and stores the result in the coef_ attribute.'''),
76
+ 'default': False
77
+ },
78
+ 'loss': {
79
+ 'description': 'The loss function to be used.',
80
+ 'default': "squared_error",
81
+ 'categorical': ["squared_error",
82
+ "huber",
83
+ "epsilon_insensitive",
84
+ "squared_epsilon_insensitive"]
85
+ },
86
+ 'penalty': {
87
+ 'description': 'The penalty (aka regularization term) to be used.',
88
+ 'default': "l2",
89
+ 'categorical': ["l1", "l2", "elasticnet"]
90
+ }
91
+ }
92
+
93
+ self.model: SGDRegressor = None
94
+
95
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
96
+ self.model = SGDRegressor(**self.passthrough_parameters())
97
+
98
+ self.model.fit(dataset.X, dataset.y)
99
+
100
+ return self
101
+
102
+ def suitable(self, dataset: Dataset) -> bool:
103
+ return dataset.type_of_target == 'continuous'
104
+
105
+ def priorize(self, candidate: Candidate = None) -> float:
106
+ return 0.5 # neutral
@@ -0,0 +1,81 @@
1
+ """[STEP] SVM Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from sklearn import svm
5
+ from ....predictor import Predictor
6
+ from ....dataset import Dataset
7
+ from ....candidate import Candidate
8
+ from ....decorators.all import is_step
9
+
10
+
11
+ @is_step('predictor', 'tabular', 'regressor')
12
+ class ActSVMSVR(Predictor):
13
+ """[STEP] SVM Regressor"""
14
+
15
+ name: str = "SVM Regression"
16
+ _description: str = textwrap.dedent('''\
17
+ SVM Regressor is a machine learning algorithm that models the relationship
18
+ between input features and a continuous output variable using a support vector machine
19
+ (SVM). It can handle non-linearly separable data by using a kernel function to map the data
20
+ into a higher-dimensional space.''')
21
+ _description_long: str = textwrap.dedent('''\
22
+ SVM Regressor is a type of regression algorithm that models the
23
+ relationship between input features and a continuous output variable using a support
24
+ vector machine (SVM). It works by finding the optimal hyperplane or boundary that predicts
25
+ the output variable with the minimum error.''')
26
+ _usage: str = "Use when you need kernel-based regression for moderate-sized non-linear data and want an alternative to ActElasticNetRegressor or ActDecisionTreeRegressor. Applicable to tabular continuous targets. Avoid when data are very large or interpretability is key."
27
+ refs: list[dict[str, Any]] = [
28
+ {
29
+ 'year': 1999,
30
+ 'name': 'Probabilistic Outputs for Support Vector Machines and Comparisons to \
31
+ Regularized Likelihood Methods',
32
+ 'authors': [
33
+ 'John C. Platt'
34
+ ],
35
+ 'doi': 'https://api.semanticscholar.org/CorpusID:5656387',
36
+ 'publisher': 'Microsoft Research'
37
+ },
38
+ {
39
+ 'year': 2001,
40
+ 'name': 'LIBSVM: A Library for Support Vector Machines',
41
+ 'authors': [
42
+ 'Chih-Chung Chang',
43
+ 'Chih-Jen Lin'
44
+ ],
45
+ 'doi': 'https://doi.org/10.1145/1961189.1961199',
46
+ 'publisher': 'ACM Transactions on Intelligen Systems and Technology Vol.2 page 1--27'
47
+ },
48
+ ]
49
+
50
+ def __init__(self):
51
+ self.configuration = {
52
+ 'kernel': {
53
+ 'description': 'Kernel to use in the SVM',
54
+ 'default': 'rbf',
55
+ 'categorical': ['linear', 'poly', 'rbf', 'sigmoid']
56
+ },
57
+ 'epsilon': {
58
+ 'description': 'Epsilon in the epsilon-SVR model.',
59
+ 'default': 0.1,
60
+ 'range': [1e-05, 0.1]
61
+ },
62
+ 'tol': {
63
+ 'description': 'The stopping criterion.',
64
+ 'default': 0.001,
65
+ 'range': [1e-05, 0.1]
66
+ }
67
+ }
68
+ self.model: svm.SVR = None
69
+
70
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
71
+ self.model = svm.SVR(
72
+ **self.passthrough_parameters()
73
+ )
74
+ self.model.fit(dataset.X, dataset.y)
75
+ return self
76
+
77
+ def suitable(self, dataset: Dataset) -> bool:
78
+ return dataset.type_of_target in ['continuous']
79
+
80
+ def priorize(self, candidate: Candidate = None) -> float:
81
+ return 0.5 # neutral
@@ -0,0 +1,97 @@
1
+ """[STEP] XGBoost Regressor"""
2
+ import textwrap
3
+ from typing import Any
4
+ from xgboost import XGBRegressor
5
+ from .._xgboost import xgboost_features
6
+ from ....predictor import Predictor
7
+ from ....dataset import Dataset
8
+ from ....candidate import Candidate
9
+ from ....decorators.all import is_step
10
+
11
+ @is_step('predictor', 'tabular', 'regressor', 'minimal_predictor')
12
+ class ActXGBoostRegressor(Predictor):
13
+ """[STEP] XGBoost Regressor"""
14
+
15
+ name: str = "XGBoost Regressor"
16
+ _description: str = textwrap.dedent('''\
17
+ XGBoost predicts continuous targets using regularized gradient-boosted
18
+ decision trees.''')
19
+ _description_long: str = textwrap.dedent('''\
20
+ Uses XGBoost's histogram tree builder with a squared-error objective.
21
+ Trees are trained sequentially to improve the ensemble's predictions,
22
+ with row and column sampling available to control overfitting.''')
23
+ _usage: str = "Use when you want boosted-tree regression on tabular data, balancing against ActCatBoostRegressor or ActExtraTreesRegressor. Applicable to continuous targets with numeric or encoded categorical features. Avoid when you need native categorical handling or a very fast baseline."
24
+ refs: list[dict[str, Any]] = [
25
+ {
26
+ 'name': 'XGBoost: A Scalable Tree Boosting System',
27
+ 'year': 2016,
28
+ 'authors': [
29
+ 'Tianqi Chen',
30
+ 'Carlos Guestrin'
31
+ ],
32
+ 'doi': 'https://doi.org/10.1145/2939672.2939785',
33
+ 'publisher': 'ACM SIGKDD 2016, pages 785--794'
34
+ }
35
+ ]
36
+ def __init__(self):
37
+ self.configuration = {
38
+ 'max_depth': {
39
+ 'description': 'Max depth of each tree',
40
+ 'default': 6,
41
+ 'range': [1, 16]
42
+ },
43
+ 'random_state': {
44
+ 'description': 'random_state',
45
+ 'default': 42
46
+ },
47
+ 'learning_rate': {
48
+ 'description': 'Learning rate',
49
+ 'default': 0.1,
50
+ 'range': [0.001, 1.0]
51
+ },
52
+ 'subsample': {
53
+ 'description': 'Fraction of training rows sampled for each tree',
54
+ 'default': 1.0,
55
+ 'range': [0.1, 1.0]
56
+ },
57
+ 'n_estimators': {
58
+ 'description': 'Number of estimators',
59
+ 'default': 100,
60
+ 'range': [1, 500]
61
+ },
62
+ 'min_child_weight': {
63
+ 'description': 'Minimum sum of instance Hessians required in a child',
64
+ 'default': 1.0,
65
+ 'range': [0.0, 20.0]
66
+ },
67
+ 'colsample_bytree': {
68
+ 'description': 'Fraction of feature columns sampled for each tree',
69
+ 'default': 1.0,
70
+ 'range': [0.1, 1.0]
71
+ }
72
+ }
73
+ self.model: XGBRegressor = None
74
+
75
+ def fit(self, dataset: Dataset): # pylint: disable=unused-argument
76
+ self.model = XGBRegressor(
77
+ **self.passthrough_parameters(),
78
+ objective='reg:squarederror',
79
+ tree_method='hist',
80
+ n_jobs=1,
81
+ )
82
+ self.model.fit(xgboost_features(dataset.X), dataset.y)
83
+ return self
84
+
85
+ def predict(self, X):
86
+ """Predict values using XGBoost-safe feature names."""
87
+ return super().predict(xgboost_features(X))
88
+
89
+ def score(self, X, y, sample_weight=None):
90
+ """Compute R² using XGBoost-safe feature names."""
91
+ return super().score(xgboost_features(X), y, sample_weight=sample_weight)
92
+
93
+ def suitable(self, dataset: Dataset) -> bool:
94
+ return dataset.type_of_target in ['continuous']
95
+
96
+ def priorize(self, candidate: Candidate = None) -> float:
97
+ return 0.5 # neutral
@@ -0,0 +1,12 @@
1
+ """
2
+ Survival Predictors Actionables
3
+ """
4
+ from .act_cox import ActCox
5
+ from .act_random_survival_forest import ActRandomSurvivalForest
6
+ from .act_survival_component_wise_gboost import ActComponentwiseGradientBoostingSurvivalAnalysis
7
+ from .act_extra_survival_trees import ActExtraSurvivalTrees
8
+ from .act_survival_tree import ActSurvivalTree
9
+ from .act_coxnet_survival_analysis import ActCoxnetSurvivalAnalysis
10
+ from .act_fast_survival_svm import ActFastSurvivalSVM
11
+ from .act_gradient_boosting_survival_analysis import ActGradientBoostingSurvivalAnalysis
12
+ from .act_weibull_aft import ActWeibullAFT