PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""[STEP] Random Forest Regressor"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
from sklearn.ensemble import RandomForestRegressor
|
|
5
|
+
from ....data_type import DataType
|
|
6
|
+
from ....predictor import Predictor
|
|
7
|
+
from ....dataset import Dataset
|
|
8
|
+
from ....candidate import Candidate
|
|
9
|
+
from ....decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('predictor', 'tabular', 'regressor')
|
|
13
|
+
class ActRandomForestRegressor(Predictor):
|
|
14
|
+
"""[STEP] Random Forest Regressor"""
|
|
15
|
+
|
|
16
|
+
name: str = "Random Forest Regressor"
|
|
17
|
+
_description: str = textwrap.dedent('''\
|
|
18
|
+
RandomForestRegressor is a machine learning algorithm that
|
|
19
|
+
models the relationship between input features and a continuous output
|
|
20
|
+
variable using a collection of decision trees.''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
RandomForestRegressor is a type of ensemble learning algorithm
|
|
23
|
+
that models the relationship between input features and a continuous output variable
|
|
24
|
+
using a collection of decision trees. It works by building multiple decision trees on
|
|
25
|
+
random subsets of the input features and data, and then averaging the predictions of each
|
|
26
|
+
tree to make the final prediction.''')
|
|
27
|
+
_usage: str = "Use when you want a robust nonlinear tabular baseline; more stable than ActDecisionTreeRegressor. Applicable to continuous targets with numeric or encoded categorical features. Avoid when data is huge or latency is tight; consider ActExtraTreesRegressor."
|
|
28
|
+
refs: list[dict[str, Any]] = [
|
|
29
|
+
{
|
|
30
|
+
'year': 2001,
|
|
31
|
+
'name': 'Random Forests',
|
|
32
|
+
'authors': ['Leo Breiman'],
|
|
33
|
+
'doi': 'https://doi.org/10.1023/A:1010933404324',
|
|
34
|
+
'publisher': 'Machine Learning Vol.45 page 5--32'
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
'year': 2006,
|
|
38
|
+
'name': 'Extremely Randomized Trees',
|
|
39
|
+
'authors': ['Pierre Geurts', 'Damien Ernst', 'Lous Wehenkel'],
|
|
40
|
+
'doi': 'https://doi.org/10.1007/s10994-006-6226-1',
|
|
41
|
+
'publisher': 'Machine Learning Vol.63 page 5--42'
|
|
42
|
+
}
|
|
43
|
+
]
|
|
44
|
+
|
|
45
|
+
def __init__(self):
|
|
46
|
+
self.configuration = {
|
|
47
|
+
# 'max_depth': { # Disable before probably better with no limit in regression
|
|
48
|
+
# 'description': 'Max depth of each tree',
|
|
49
|
+
# 'default': 15,
|
|
50
|
+
# 'range': [1, 100]
|
|
51
|
+
# },
|
|
52
|
+
'n_estimators': {
|
|
53
|
+
'description': 'Number of threes',
|
|
54
|
+
'default': 100,
|
|
55
|
+
'range': [1, 500]
|
|
56
|
+
},
|
|
57
|
+
'random_state': {
|
|
58
|
+
'description': 'random_state',
|
|
59
|
+
'default': 42
|
|
60
|
+
},
|
|
61
|
+
'min_samples_leaf': {
|
|
62
|
+
'description': 'The minimum number of samples required to be at a leaf node.',
|
|
63
|
+
'default': 1,
|
|
64
|
+
'range': [1, 15]
|
|
65
|
+
},
|
|
66
|
+
'max_features': {
|
|
67
|
+
'description': 'The number of features to consider when looking for the best split',
|
|
68
|
+
'default': 1.0,
|
|
69
|
+
'range': [0.1, 1.0]
|
|
70
|
+
},
|
|
71
|
+
'min_samples_split': {
|
|
72
|
+
'description': 'The minimum number of samples required to split an internal node',
|
|
73
|
+
'default': 2,
|
|
74
|
+
'range': [2, 20]
|
|
75
|
+
},
|
|
76
|
+
'bootstrap': {
|
|
77
|
+
'description': textwrap.dedent('''\
|
|
78
|
+
Whether bootstrap samples are used when building trees. If
|
|
79
|
+
False, the whole dataset is used to build each tree.'''),
|
|
80
|
+
'default': False
|
|
81
|
+
},
|
|
82
|
+
'criterion': {
|
|
83
|
+
'description': 'The function to measure the quality of a split.',
|
|
84
|
+
'default': "squared_error",
|
|
85
|
+
'categorical': ["poisson", "friedman_mse", "absolute_error", "squared_error"]
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
self.model: RandomForestRegressor = None
|
|
89
|
+
self.columns: list[str] = []
|
|
90
|
+
|
|
91
|
+
def _select_features(self, X):
|
|
92
|
+
if self.columns and hasattr(X, 'columns'):
|
|
93
|
+
return X[self.columns]
|
|
94
|
+
return X
|
|
95
|
+
|
|
96
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
97
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
98
|
+
if not self.columns:
|
|
99
|
+
self.columns = dataset.features
|
|
100
|
+
|
|
101
|
+
self.model = RandomForestRegressor(**self.passthrough_parameters())
|
|
102
|
+
self.model.fit(self._select_features(dataset.X), dataset.y)
|
|
103
|
+
return self
|
|
104
|
+
|
|
105
|
+
def predict(self, X):
|
|
106
|
+
return super().predict(self._select_features(X))
|
|
107
|
+
|
|
108
|
+
def score(self, X, y=None, *args, **kwargs):
|
|
109
|
+
return self.model.score(self._select_features(X), y, *args, **kwargs)
|
|
110
|
+
|
|
111
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
112
|
+
return dataset.type_of_target in ['continuous'] \
|
|
113
|
+
and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
|
|
114
|
+
|
|
115
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
116
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""[STEP] RANSAC Regressor"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
from sklearn.linear_model import RANSACRegressor
|
|
5
|
+
from ....predictor import Predictor
|
|
6
|
+
from ....dataset import Dataset
|
|
7
|
+
from ....candidate import Candidate
|
|
8
|
+
from ....data_type import DataType
|
|
9
|
+
from ....decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('predictor', 'tabular', 'regressor')
|
|
13
|
+
class ActRANSACRegressor(Predictor):
|
|
14
|
+
"""[STEP] RANSAC Regressor"""
|
|
15
|
+
|
|
16
|
+
name: str = "RANSAC Regressor"
|
|
17
|
+
_usage: str = "Use when outliers can skew a linear model and you need robustness vs ActElasticNetRegressor. Applicable to numeric tabular regression with moderate sample size. Avoid when data is clean or effects are highly nonlinear; consider ActDecisionTreeRegressor."
|
|
18
|
+
_description: str = textwrap.dedent('''\
|
|
19
|
+
RANSACRegressor fits a robust regression model by iteratively
|
|
20
|
+
sampling subsets of the data and keeping inliers.''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
RANSACRegressor is a robust regression technique that repeatedly fits
|
|
23
|
+
a base estimator on random subsets, identifies inliers using a
|
|
24
|
+
residual threshold, and refits on the consensus set. This approach
|
|
25
|
+
reduces the influence of outliers in tabular regression problems.''')
|
|
26
|
+
refs: list[dict[str, Any]] = [
|
|
27
|
+
{
|
|
28
|
+
'year': 1981,
|
|
29
|
+
'name': (
|
|
30
|
+
'Random Sample Consensus: A Paradigm for Model Fitting with '
|
|
31
|
+
'Applications to Image Analysis and Automated Cartography'
|
|
32
|
+
),
|
|
33
|
+
'authors': [
|
|
34
|
+
'Martin A. Fischler',
|
|
35
|
+
'Robert C. Bolles'
|
|
36
|
+
],
|
|
37
|
+
'doi': 'https://doi.org/10.1145/358669.358692',
|
|
38
|
+
'publisher': 'Communications of the ACM'
|
|
39
|
+
}
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
def __init__(self):
|
|
43
|
+
self.configuration = {
|
|
44
|
+
'min_samples': {
|
|
45
|
+
'description': textwrap.dedent('''\
|
|
46
|
+
Minimum number of samples chosen for estimating the model.
|
|
47
|
+
When None, uses the number of features plus one.'''),
|
|
48
|
+
'default': None
|
|
49
|
+
},
|
|
50
|
+
'residual_threshold': {
|
|
51
|
+
'description': textwrap.dedent('''\
|
|
52
|
+
Maximum residual for a data sample to be classified as an inlier.
|
|
53
|
+
When None, uses the median absolute deviation of the target.'''),
|
|
54
|
+
'default': None
|
|
55
|
+
},
|
|
56
|
+
'max_trials': {
|
|
57
|
+
'description': 'Maximum number of iterations for random sampling.',
|
|
58
|
+
'default': 100,
|
|
59
|
+
'range': [10, 500]
|
|
60
|
+
},
|
|
61
|
+
'stop_probability': {
|
|
62
|
+
'description': textwrap.dedent('''\
|
|
63
|
+
Probability that at least one outlier-free sample has been chosen
|
|
64
|
+
after max_trials iterations.'''),
|
|
65
|
+
'default': 0.99,
|
|
66
|
+
'range': [0.8, 0.999]
|
|
67
|
+
},
|
|
68
|
+
'loss': {
|
|
69
|
+
'description': 'Loss function used to classify inliers.',
|
|
70
|
+
'default': 'absolute_error',
|
|
71
|
+
'categorical': ['absolute_error', 'squared_error']
|
|
72
|
+
},
|
|
73
|
+
'random_state': {
|
|
74
|
+
'description': 'Random state for reproducibility.',
|
|
75
|
+
'default': 42
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
self.model: RANSACRegressor = None
|
|
79
|
+
self.columns: list[str] = []
|
|
80
|
+
|
|
81
|
+
def _select_features(self, X):
|
|
82
|
+
if self.columns and hasattr(X, 'columns'):
|
|
83
|
+
return X[self.columns]
|
|
84
|
+
return X
|
|
85
|
+
|
|
86
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
87
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
88
|
+
if not self.columns:
|
|
89
|
+
self.columns = dataset.features
|
|
90
|
+
|
|
91
|
+
self.model = RANSACRegressor(**self.passthrough_parameters())
|
|
92
|
+
self.model.fit(self._select_features(dataset.X), dataset.y)
|
|
93
|
+
return self
|
|
94
|
+
|
|
95
|
+
def predict(self, X):
|
|
96
|
+
return super().predict(self._select_features(X))
|
|
97
|
+
|
|
98
|
+
def score(self, X, y=None, *args, **kwargs):
|
|
99
|
+
return self.model.score(self._select_features(X), y, *args, **kwargs)
|
|
100
|
+
|
|
101
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
102
|
+
return dataset.type_of_target == 'continuous' \
|
|
103
|
+
and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
|
|
104
|
+
|
|
105
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
106
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""[STEP] Ridge Regressor"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
from sklearn.linear_model import Ridge
|
|
5
|
+
from ....predictor import Predictor
|
|
6
|
+
from ....dataset import Dataset
|
|
7
|
+
from ....candidate import Candidate
|
|
8
|
+
from ....data_type import DataType
|
|
9
|
+
from ....decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('predictor', 'tabular', 'regressor')
|
|
13
|
+
class ActRidgeRegressor(Predictor):
|
|
14
|
+
"""[STEP] Ridge Regressor"""
|
|
15
|
+
|
|
16
|
+
name: str = "Ridge Regressor"
|
|
17
|
+
_description: str = textwrap.dedent('''\
|
|
18
|
+
Ridge regression applies L2 regularization to stabilize coefficients
|
|
19
|
+
when predictors are correlated.''')
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
Ridge regression fits a linear model while penalizing large coefficients
|
|
22
|
+
with an L2 term. This reduces variance, improves numerical stability,
|
|
23
|
+
and provides robust predictions for tabular regression problems.''')
|
|
24
|
+
_usage: str = "Use when you need a stable linear regressor for correlated numeric features; compare ActElasticNetRegressor or ActARDRegression. Applicable to tabular regression with mostly numeric inputs. Avoid when strong nonlinear effects dominate or labels are not continuous."
|
|
25
|
+
refs: list[dict[str, Any]] = [
|
|
26
|
+
{
|
|
27
|
+
'year': 1970,
|
|
28
|
+
'name': 'Ridge Regression: Biased Estimation for Nonorthogonal Problems',
|
|
29
|
+
'authors': [
|
|
30
|
+
'Arthur E. Hoerl',
|
|
31
|
+
'Robert W. Kennard'
|
|
32
|
+
],
|
|
33
|
+
'doi': 'https://doi.org/10.2307/1267351',
|
|
34
|
+
'publisher': 'Technometrics Vol. 12, No. 1, page 55--67'
|
|
35
|
+
}
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
def __init__(self):
|
|
39
|
+
self.configuration = {
|
|
40
|
+
'alpha': {
|
|
41
|
+
'description': 'Regularization strength.',
|
|
42
|
+
'default': 1.0,
|
|
43
|
+
'range': [1e-04, 100.0]
|
|
44
|
+
},
|
|
45
|
+
'fit_intercept': {
|
|
46
|
+
'description': 'Whether to fit the intercept term.',
|
|
47
|
+
'default': True,
|
|
48
|
+
'categorical': [True, False]
|
|
49
|
+
},
|
|
50
|
+
'solver': {
|
|
51
|
+
'description': 'Solver to use in the ridge optimization.',
|
|
52
|
+
'default': 'auto',
|
|
53
|
+
'categorical': [
|
|
54
|
+
'auto',
|
|
55
|
+
'svd',
|
|
56
|
+
'cholesky',
|
|
57
|
+
'lsqr',
|
|
58
|
+
'sparse_cg',
|
|
59
|
+
'sag',
|
|
60
|
+
'saga',
|
|
61
|
+
'lbfgs'
|
|
62
|
+
]
|
|
63
|
+
},
|
|
64
|
+
'tol': {
|
|
65
|
+
'description': 'Stopping criterion for iterative solvers.',
|
|
66
|
+
'default': 0.0001,
|
|
67
|
+
'range': [1e-05, 0.1]
|
|
68
|
+
},
|
|
69
|
+
'max_iter': {
|
|
70
|
+
'description': 'Maximum number of iterations for iterative solvers.',
|
|
71
|
+
'default': 1000,
|
|
72
|
+
'range': [50, 5000]
|
|
73
|
+
},
|
|
74
|
+
'random_state': {
|
|
75
|
+
'description': 'Random state for solvers that use randomness.',
|
|
76
|
+
'default': 42
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
self.model: Ridge = None
|
|
80
|
+
self.columns: list[str] = []
|
|
81
|
+
|
|
82
|
+
def _select_features(self, X):
|
|
83
|
+
if self.columns and hasattr(X, 'columns'):
|
|
84
|
+
return X[self.columns]
|
|
85
|
+
return X
|
|
86
|
+
|
|
87
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
88
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
89
|
+
if not self.columns:
|
|
90
|
+
self.columns = dataset.features
|
|
91
|
+
|
|
92
|
+
self.model = Ridge(**self.passthrough_parameters())
|
|
93
|
+
self.model.fit(self._select_features(dataset.X), dataset.y)
|
|
94
|
+
return self
|
|
95
|
+
|
|
96
|
+
def predict(self, X):
|
|
97
|
+
return super().predict(self._select_features(X))
|
|
98
|
+
|
|
99
|
+
def score(self, X, y=None, *args, **kwargs):
|
|
100
|
+
return self.model.score(self._select_features(X), y, *args, **kwargs)
|
|
101
|
+
|
|
102
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
103
|
+
return dataset.type_of_target == 'continuous' \
|
|
104
|
+
and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
|
|
105
|
+
|
|
106
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
107
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""[STEP] SGD Regressor"""
|
|
2
|
+
from typing import Any
|
|
3
|
+
import textwrap
|
|
4
|
+
from sklearn.linear_model import SGDRegressor
|
|
5
|
+
from ....predictor import Predictor
|
|
6
|
+
from ....dataset import Dataset
|
|
7
|
+
from ....candidate import Candidate
|
|
8
|
+
from ....decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
@is_step('predictor', 'tabular', 'regressor')
|
|
11
|
+
class ActSGDRegressor(Predictor):
|
|
12
|
+
"""[STEP] SGD Regressor"""
|
|
13
|
+
|
|
14
|
+
name: str = "SGD Regressor"
|
|
15
|
+
_usage: str = "Use when you need a fast linear baseline on large tabular data, e.g., vs ActElasticNetRegressor. Applicable to scaled numeric features with a continuous target. Avoid when strong nonlinearities or best accuracy is needed; prefer ActCatBoostRegressor or ActExtraTreesRegressor."
|
|
16
|
+
_description: str = textwrap.dedent('''\
|
|
17
|
+
SGDRegressor is a machine learning algorithm that models
|
|
18
|
+
the relationship between input features and a continuous output variable
|
|
19
|
+
using stochastic gradient descent.''')
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
SGDRegressor is a type of linear model that models the relationship
|
|
22
|
+
between input features and a continuous output variable using stochastic gradient descent.
|
|
23
|
+
It works by iteratively updating the model parameters in the direction of the negative
|
|
24
|
+
gradient of the loss function with respect to the parameters, using a single example
|
|
25
|
+
at a time.''')
|
|
26
|
+
refs: list[dict[str, Any]] = [
|
|
27
|
+
{
|
|
28
|
+
'year': 1951,
|
|
29
|
+
'name': 'A Stochastic Approximation Method',
|
|
30
|
+
'authors': [
|
|
31
|
+
'Herbert Robbins',
|
|
32
|
+
'Sutton Monro'
|
|
33
|
+
],
|
|
34
|
+
'doi': 'https://doi.org/10.1214/aoms/1177729586',
|
|
35
|
+
'publisher': 'The annals of Mathematical Statistics Vol.22 No.3 page 400--407'
|
|
36
|
+
}
|
|
37
|
+
]
|
|
38
|
+
def __init__(self):
|
|
39
|
+
self.configuration = {
|
|
40
|
+
'alpha': {
|
|
41
|
+
'description': 'Constant that multiplies the regularization term.',
|
|
42
|
+
'default': 0.0001,
|
|
43
|
+
'range': [1e-07, 0.1]
|
|
44
|
+
},
|
|
45
|
+
'tol': {
|
|
46
|
+
'description': 'The stopping criterion.',
|
|
47
|
+
'default': 0.0001,
|
|
48
|
+
'range': [1e-05, 0.1]
|
|
49
|
+
},
|
|
50
|
+
'epsilon': {
|
|
51
|
+
'description': 'Epsilon in the epsilon-insensitive loss functions',
|
|
52
|
+
'default': 0.1,
|
|
53
|
+
'range': [1e-05, 0.1]
|
|
54
|
+
},
|
|
55
|
+
'eta0': {
|
|
56
|
+
'description': textwrap.dedent('''\
|
|
57
|
+
The initial learning rate for the ‘constant’, ‘invscaling’
|
|
58
|
+
or ‘adaptive’ schedules.'''),
|
|
59
|
+
'default': 0.01,
|
|
60
|
+
'range': [1e-07, 0.1]
|
|
61
|
+
},
|
|
62
|
+
'l1_ratio': {
|
|
63
|
+
'description': 'The Elastic Net mixing parameter',
|
|
64
|
+
'default': 0.15,
|
|
65
|
+
'range': [1e-09, 1.0]
|
|
66
|
+
},
|
|
67
|
+
'power_t': {
|
|
68
|
+
'description': 'The exponent for inverse scaling learning rate.',
|
|
69
|
+
'default': 0.25,
|
|
70
|
+
'range': [1e-05, 1.0]
|
|
71
|
+
},
|
|
72
|
+
'average': {
|
|
73
|
+
'description': textwrap.dedent('''\
|
|
74
|
+
When set to True, computes the averaged SGD weights across
|
|
75
|
+
all updates and stores the result in the coef_ attribute.'''),
|
|
76
|
+
'default': False
|
|
77
|
+
},
|
|
78
|
+
'loss': {
|
|
79
|
+
'description': 'The loss function to be used.',
|
|
80
|
+
'default': "squared_error",
|
|
81
|
+
'categorical': ["squared_error",
|
|
82
|
+
"huber",
|
|
83
|
+
"epsilon_insensitive",
|
|
84
|
+
"squared_epsilon_insensitive"]
|
|
85
|
+
},
|
|
86
|
+
'penalty': {
|
|
87
|
+
'description': 'The penalty (aka regularization term) to be used.',
|
|
88
|
+
'default': "l2",
|
|
89
|
+
'categorical': ["l1", "l2", "elasticnet"]
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
self.model: SGDRegressor = None
|
|
94
|
+
|
|
95
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
96
|
+
self.model = SGDRegressor(**self.passthrough_parameters())
|
|
97
|
+
|
|
98
|
+
self.model.fit(dataset.X, dataset.y)
|
|
99
|
+
|
|
100
|
+
return self
|
|
101
|
+
|
|
102
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
103
|
+
return dataset.type_of_target == 'continuous'
|
|
104
|
+
|
|
105
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
106
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""[STEP] SVM Regressor"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
from sklearn import svm
|
|
5
|
+
from ....predictor import Predictor
|
|
6
|
+
from ....dataset import Dataset
|
|
7
|
+
from ....candidate import Candidate
|
|
8
|
+
from ....decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@is_step('predictor', 'tabular', 'regressor')
|
|
12
|
+
class ActSVMSVR(Predictor):
|
|
13
|
+
"""[STEP] SVM Regressor"""
|
|
14
|
+
|
|
15
|
+
name: str = "SVM Regression"
|
|
16
|
+
_description: str = textwrap.dedent('''\
|
|
17
|
+
SVM Regressor is a machine learning algorithm that models the relationship
|
|
18
|
+
between input features and a continuous output variable using a support vector machine
|
|
19
|
+
(SVM). It can handle non-linearly separable data by using a kernel function to map the data
|
|
20
|
+
into a higher-dimensional space.''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
SVM Regressor is a type of regression algorithm that models the
|
|
23
|
+
relationship between input features and a continuous output variable using a support
|
|
24
|
+
vector machine (SVM). It works by finding the optimal hyperplane or boundary that predicts
|
|
25
|
+
the output variable with the minimum error.''')
|
|
26
|
+
_usage: str = "Use when you need kernel-based regression for moderate-sized non-linear data and want an alternative to ActElasticNetRegressor or ActDecisionTreeRegressor. Applicable to tabular continuous targets. Avoid when data are very large or interpretability is key."
|
|
27
|
+
refs: list[dict[str, Any]] = [
|
|
28
|
+
{
|
|
29
|
+
'year': 1999,
|
|
30
|
+
'name': 'Probabilistic Outputs for Support Vector Machines and Comparisons to \
|
|
31
|
+
Regularized Likelihood Methods',
|
|
32
|
+
'authors': [
|
|
33
|
+
'John C. Platt'
|
|
34
|
+
],
|
|
35
|
+
'doi': 'https://api.semanticscholar.org/CorpusID:5656387',
|
|
36
|
+
'publisher': 'Microsoft Research'
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
'year': 2001,
|
|
40
|
+
'name': 'LIBSVM: A Library for Support Vector Machines',
|
|
41
|
+
'authors': [
|
|
42
|
+
'Chih-Chung Chang',
|
|
43
|
+
'Chih-Jen Lin'
|
|
44
|
+
],
|
|
45
|
+
'doi': 'https://doi.org/10.1145/1961189.1961199',
|
|
46
|
+
'publisher': 'ACM Transactions on Intelligen Systems and Technology Vol.2 page 1--27'
|
|
47
|
+
},
|
|
48
|
+
]
|
|
49
|
+
|
|
50
|
+
def __init__(self):
|
|
51
|
+
self.configuration = {
|
|
52
|
+
'kernel': {
|
|
53
|
+
'description': 'Kernel to use in the SVM',
|
|
54
|
+
'default': 'rbf',
|
|
55
|
+
'categorical': ['linear', 'poly', 'rbf', 'sigmoid']
|
|
56
|
+
},
|
|
57
|
+
'epsilon': {
|
|
58
|
+
'description': 'Epsilon in the epsilon-SVR model.',
|
|
59
|
+
'default': 0.1,
|
|
60
|
+
'range': [1e-05, 0.1]
|
|
61
|
+
},
|
|
62
|
+
'tol': {
|
|
63
|
+
'description': 'The stopping criterion.',
|
|
64
|
+
'default': 0.001,
|
|
65
|
+
'range': [1e-05, 0.1]
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
self.model: svm.SVR = None
|
|
69
|
+
|
|
70
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
71
|
+
self.model = svm.SVR(
|
|
72
|
+
**self.passthrough_parameters()
|
|
73
|
+
)
|
|
74
|
+
self.model.fit(dataset.X, dataset.y)
|
|
75
|
+
return self
|
|
76
|
+
|
|
77
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
78
|
+
return dataset.type_of_target in ['continuous']
|
|
79
|
+
|
|
80
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
81
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""[STEP] XGBoost Regressor"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
from xgboost import XGBRegressor
|
|
5
|
+
from .._xgboost import xgboost_features
|
|
6
|
+
from ....predictor import Predictor
|
|
7
|
+
from ....dataset import Dataset
|
|
8
|
+
from ....candidate import Candidate
|
|
9
|
+
from ....decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
@is_step('predictor', 'tabular', 'regressor', 'minimal_predictor')
|
|
12
|
+
class ActXGBoostRegressor(Predictor):
|
|
13
|
+
"""[STEP] XGBoost Regressor"""
|
|
14
|
+
|
|
15
|
+
name: str = "XGBoost Regressor"
|
|
16
|
+
_description: str = textwrap.dedent('''\
|
|
17
|
+
XGBoost predicts continuous targets using regularized gradient-boosted
|
|
18
|
+
decision trees.''')
|
|
19
|
+
_description_long: str = textwrap.dedent('''\
|
|
20
|
+
Uses XGBoost's histogram tree builder with a squared-error objective.
|
|
21
|
+
Trees are trained sequentially to improve the ensemble's predictions,
|
|
22
|
+
with row and column sampling available to control overfitting.''')
|
|
23
|
+
_usage: str = "Use when you want boosted-tree regression on tabular data, balancing against ActCatBoostRegressor or ActExtraTreesRegressor. Applicable to continuous targets with numeric or encoded categorical features. Avoid when you need native categorical handling or a very fast baseline."
|
|
24
|
+
refs: list[dict[str, Any]] = [
|
|
25
|
+
{
|
|
26
|
+
'name': 'XGBoost: A Scalable Tree Boosting System',
|
|
27
|
+
'year': 2016,
|
|
28
|
+
'authors': [
|
|
29
|
+
'Tianqi Chen',
|
|
30
|
+
'Carlos Guestrin'
|
|
31
|
+
],
|
|
32
|
+
'doi': 'https://doi.org/10.1145/2939672.2939785',
|
|
33
|
+
'publisher': 'ACM SIGKDD 2016, pages 785--794'
|
|
34
|
+
}
|
|
35
|
+
]
|
|
36
|
+
def __init__(self):
|
|
37
|
+
self.configuration = {
|
|
38
|
+
'max_depth': {
|
|
39
|
+
'description': 'Max depth of each tree',
|
|
40
|
+
'default': 6,
|
|
41
|
+
'range': [1, 16]
|
|
42
|
+
},
|
|
43
|
+
'random_state': {
|
|
44
|
+
'description': 'random_state',
|
|
45
|
+
'default': 42
|
|
46
|
+
},
|
|
47
|
+
'learning_rate': {
|
|
48
|
+
'description': 'Learning rate',
|
|
49
|
+
'default': 0.1,
|
|
50
|
+
'range': [0.001, 1.0]
|
|
51
|
+
},
|
|
52
|
+
'subsample': {
|
|
53
|
+
'description': 'Fraction of training rows sampled for each tree',
|
|
54
|
+
'default': 1.0,
|
|
55
|
+
'range': [0.1, 1.0]
|
|
56
|
+
},
|
|
57
|
+
'n_estimators': {
|
|
58
|
+
'description': 'Number of estimators',
|
|
59
|
+
'default': 100,
|
|
60
|
+
'range': [1, 500]
|
|
61
|
+
},
|
|
62
|
+
'min_child_weight': {
|
|
63
|
+
'description': 'Minimum sum of instance Hessians required in a child',
|
|
64
|
+
'default': 1.0,
|
|
65
|
+
'range': [0.0, 20.0]
|
|
66
|
+
},
|
|
67
|
+
'colsample_bytree': {
|
|
68
|
+
'description': 'Fraction of feature columns sampled for each tree',
|
|
69
|
+
'default': 1.0,
|
|
70
|
+
'range': [0.1, 1.0]
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
self.model: XGBRegressor = None
|
|
74
|
+
|
|
75
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
76
|
+
self.model = XGBRegressor(
|
|
77
|
+
**self.passthrough_parameters(),
|
|
78
|
+
objective='reg:squarederror',
|
|
79
|
+
tree_method='hist',
|
|
80
|
+
n_jobs=1,
|
|
81
|
+
)
|
|
82
|
+
self.model.fit(xgboost_features(dataset.X), dataset.y)
|
|
83
|
+
return self
|
|
84
|
+
|
|
85
|
+
def predict(self, X):
|
|
86
|
+
"""Predict values using XGBoost-safe feature names."""
|
|
87
|
+
return super().predict(xgboost_features(X))
|
|
88
|
+
|
|
89
|
+
def score(self, X, y, sample_weight=None):
|
|
90
|
+
"""Compute R² using XGBoost-safe feature names."""
|
|
91
|
+
return super().score(xgboost_features(X), y, sample_weight=sample_weight)
|
|
92
|
+
|
|
93
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
94
|
+
return dataset.type_of_target in ['continuous']
|
|
95
|
+
|
|
96
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
97
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Survival Predictors Actionables
|
|
3
|
+
"""
|
|
4
|
+
from .act_cox import ActCox
|
|
5
|
+
from .act_random_survival_forest import ActRandomSurvivalForest
|
|
6
|
+
from .act_survival_component_wise_gboost import ActComponentwiseGradientBoostingSurvivalAnalysis
|
|
7
|
+
from .act_extra_survival_trees import ActExtraSurvivalTrees
|
|
8
|
+
from .act_survival_tree import ActSurvivalTree
|
|
9
|
+
from .act_coxnet_survival_analysis import ActCoxnetSurvivalAnalysis
|
|
10
|
+
from .act_fast_survival_svm import ActFastSurvivalSVM
|
|
11
|
+
from .act_gradient_boosting_survival_analysis import ActGradientBoostingSurvivalAnalysis
|
|
12
|
+
from .act_weibull_aft import ActWeibullAFT
|