PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""[STEP] Lasso Regressor"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
from sklearn.linear_model import Lasso
|
|
5
|
+
from ....predictor import Predictor
|
|
6
|
+
from ....dataset import Dataset
|
|
7
|
+
from ....candidate import Candidate
|
|
8
|
+
from ....data_type import DataType
|
|
9
|
+
from ....decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('predictor', 'tabular', 'regressor')
|
|
13
|
+
class ActLassoRegressor(Predictor):
|
|
14
|
+
"""[STEP] Lasso Regressor"""
|
|
15
|
+
|
|
16
|
+
name: str = "Lasso Regressor"
|
|
17
|
+
_description: str = textwrap.dedent('''\
|
|
18
|
+
Lasso regression uses L1 regularization to shrink coefficients and
|
|
19
|
+
perform automatic feature selection in linear regression.''')
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
Lasso (Least Absolute Shrinkage and Selection Operator) fits a linear
|
|
22
|
+
model while adding an L1 penalty to the loss. The penalty drives some
|
|
23
|
+
coefficients to zero, selecting a sparse set of features and improving
|
|
24
|
+
interpretability for tabular regression tasks.''')
|
|
25
|
+
_usage: str = "Use when you want sparse linear coefficients and feature selection; compare ActElasticNetRegressor or ActARDRegression for similar linear shrinkage. Applicable to numeric tabular regression with continuous targets. Avoid when effects are strongly nonlinear or dominated by categorical features."
|
|
26
|
+
refs: list[dict[str, Any]] = [
|
|
27
|
+
{
|
|
28
|
+
'year': 1996,
|
|
29
|
+
'name': 'Regression Shrinkage and Selection via the Lasso',
|
|
30
|
+
'authors': [
|
|
31
|
+
'Robert Tibshirani'
|
|
32
|
+
],
|
|
33
|
+
'doi': 'https://doi.org/10.1111/j.2517-6161.1996.tb02080.x',
|
|
34
|
+
'publisher': 'Journal of the Royal Statistical Society Series B'
|
|
35
|
+
}
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
def __init__(self):
|
|
39
|
+
self.configuration = {
|
|
40
|
+
'alpha': {
|
|
41
|
+
'description': 'Regularization strength.',
|
|
42
|
+
'default': 1.0,
|
|
43
|
+
'range': [1e-04, 10.0]
|
|
44
|
+
},
|
|
45
|
+
'fit_intercept': {
|
|
46
|
+
'description': 'Whether to fit the intercept term.',
|
|
47
|
+
'default': True,
|
|
48
|
+
'categorical': [True, False]
|
|
49
|
+
},
|
|
50
|
+
'max_iter': {
|
|
51
|
+
'description': 'Maximum number of iterations.',
|
|
52
|
+
'default': 1000,
|
|
53
|
+
'range': [100, 5000]
|
|
54
|
+
},
|
|
55
|
+
'tol': {
|
|
56
|
+
'description': 'Stopping criterion.',
|
|
57
|
+
'default': 0.0001,
|
|
58
|
+
'range': [1e-05, 0.1]
|
|
59
|
+
},
|
|
60
|
+
'selection': {
|
|
61
|
+
'description': 'Coordinate descent selection strategy.',
|
|
62
|
+
'default': 'cyclic',
|
|
63
|
+
'categorical': ['cyclic', 'random']
|
|
64
|
+
},
|
|
65
|
+
'positive': {
|
|
66
|
+
'description': 'Force coefficients to be positive.',
|
|
67
|
+
'default': False,
|
|
68
|
+
'categorical': [True, False]
|
|
69
|
+
},
|
|
70
|
+
'random_state': {
|
|
71
|
+
'description': 'Random state used when selection is "random".',
|
|
72
|
+
'default': 42
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
self.model: Lasso = None
|
|
76
|
+
self.columns: list[str] = []
|
|
77
|
+
|
|
78
|
+
def _select_features(self, X):
|
|
79
|
+
if self.columns and hasattr(X, 'columns'):
|
|
80
|
+
return X[self.columns]
|
|
81
|
+
return X
|
|
82
|
+
|
|
83
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
84
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
85
|
+
if not self.columns:
|
|
86
|
+
self.columns = dataset.features
|
|
87
|
+
|
|
88
|
+
self.model = Lasso(**self.passthrough_parameters())
|
|
89
|
+
self.model.fit(self._select_features(dataset.X), dataset.y)
|
|
90
|
+
return self
|
|
91
|
+
|
|
92
|
+
def predict(self, X):
|
|
93
|
+
return super().predict(self._select_features(X))
|
|
94
|
+
|
|
95
|
+
def score(self, X, y=None, *args, **kwargs):
|
|
96
|
+
return self.model.score(self._select_features(X), y, *args, **kwargs)
|
|
97
|
+
|
|
98
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
99
|
+
return dataset.type_of_target == 'continuous' \
|
|
100
|
+
and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
|
|
101
|
+
|
|
102
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
103
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""[STEP] LightGBM Regressor"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
try:
|
|
6
|
+
from lightgbm import LGBMRegressor
|
|
7
|
+
from lightgbm.basic import LightGBMError
|
|
8
|
+
_LGBM_ERRORS: tuple[type[Exception], ...] = (LightGBMError,)
|
|
9
|
+
except ImportError: # pragma: no cover - optional dependency
|
|
10
|
+
LGBMRegressor = None # type: ignore
|
|
11
|
+
_LGBM_ERRORS = tuple()
|
|
12
|
+
|
|
13
|
+
from ....predictor import Predictor
|
|
14
|
+
from ....dataset import Dataset
|
|
15
|
+
from ....candidate import Candidate
|
|
16
|
+
from ....data_type import DataType
|
|
17
|
+
from ....decorators.all import is_step
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@is_step('predictor', 'tabular', 'regressor')
|
|
21
|
+
class ActLightGBMRegressor(Predictor):
|
|
22
|
+
"""[STEP] LightGBM Regressor"""
|
|
23
|
+
|
|
24
|
+
name: str = "LightGBM Regressor"
|
|
25
|
+
_description: str = textwrap.dedent('''\
|
|
26
|
+
LightGBMRegressor is a gradient boosting algorithm that builds
|
|
27
|
+
decision trees efficiently for regression tasks.''')
|
|
28
|
+
_description_long: str = textwrap.dedent('''\
|
|
29
|
+
LightGBMRegressor trains an ensemble of decision trees using histogram-based
|
|
30
|
+
splits and leaf-wise growth. It is designed to be fast while preserving
|
|
31
|
+
accuracy on tabular regression problems.''')
|
|
32
|
+
_usage: str = "Use when you want tabular regression with mixed numeric/categorical and a booster alternative to ActCatBoostRegressor. Applicable to medium/large datasets with nonlinear interactions. Avoid when data is tiny or you need a simple linear model like ActARDRegression."
|
|
33
|
+
refs: list[dict[str, Any]] = [
|
|
34
|
+
{
|
|
35
|
+
'year': 2017,
|
|
36
|
+
'name': 'LightGBM: A Highly Efficient Gradient Boosting Decision Tree',
|
|
37
|
+
'authors': [
|
|
38
|
+
'Guolin Ke',
|
|
39
|
+
'Qi Meng',
|
|
40
|
+
'Thomas Finley',
|
|
41
|
+
'Taifeng Wang',
|
|
42
|
+
'Wei Chen',
|
|
43
|
+
'Weidong Ma',
|
|
44
|
+
'Qiwei Ye',
|
|
45
|
+
'Tie-Yan Liu'
|
|
46
|
+
],
|
|
47
|
+
'doi': 'https://doi.org/10.48550/arXiv.1712.01005',
|
|
48
|
+
'publisher': 'Advances in Neural Information Processing Systems 30 (NeurIPS 2017)'
|
|
49
|
+
}
|
|
50
|
+
]
|
|
51
|
+
|
|
52
|
+
def __init__(self):
|
|
53
|
+
self.configuration = {
|
|
54
|
+
'boosting_type': {
|
|
55
|
+
'description': 'Type of boosting algorithm.',
|
|
56
|
+
'default': 'gbdt',
|
|
57
|
+
'categorical': ['gbdt', 'dart']
|
|
58
|
+
},
|
|
59
|
+
'n_estimators': {
|
|
60
|
+
'description': 'Number of boosting iterations.',
|
|
61
|
+
'default': 200,
|
|
62
|
+
'range': [50, 1000]
|
|
63
|
+
},
|
|
64
|
+
'learning_rate': {
|
|
65
|
+
'description': 'Shrinkage rate applied to each tree.',
|
|
66
|
+
'default': 0.1,
|
|
67
|
+
'range': [0.01, 1.0]
|
|
68
|
+
},
|
|
69
|
+
'num_leaves': {
|
|
70
|
+
'description': 'Maximum number of leaves in one tree.',
|
|
71
|
+
'default': 31,
|
|
72
|
+
'range': [7, 255]
|
|
73
|
+
},
|
|
74
|
+
'max_depth': {
|
|
75
|
+
'description': 'Maximum depth of a tree, -1 means no limit.',
|
|
76
|
+
'default': -1,
|
|
77
|
+
'categorical': [-1, 3, 5, 10, 15]
|
|
78
|
+
},
|
|
79
|
+
'min_child_samples': {
|
|
80
|
+
'description': 'Minimum number of data in one leaf.',
|
|
81
|
+
'default': 20,
|
|
82
|
+
'range': [5, 200]
|
|
83
|
+
},
|
|
84
|
+
'subsample': {
|
|
85
|
+
'description': 'Fraction of data to use for each boosting iteration.',
|
|
86
|
+
'default': 1.0,
|
|
87
|
+
'range': [0.5, 1.0]
|
|
88
|
+
},
|
|
89
|
+
'subsample_freq': {
|
|
90
|
+
'description': 'Frequency for subsampling, 0 means disabled.',
|
|
91
|
+
'default': 0,
|
|
92
|
+
'range': [0, 10]
|
|
93
|
+
},
|
|
94
|
+
'colsample_bytree': {
|
|
95
|
+
'description': 'Fraction of features used for each tree.',
|
|
96
|
+
'default': 1.0,
|
|
97
|
+
'range': [0.5, 1.0]
|
|
98
|
+
},
|
|
99
|
+
'reg_alpha': {
|
|
100
|
+
'description': 'L1 regularization.',
|
|
101
|
+
'default': 0.0,
|
|
102
|
+
'range': [0.0, 1.0]
|
|
103
|
+
},
|
|
104
|
+
'reg_lambda': {
|
|
105
|
+
'description': 'L2 regularization.',
|
|
106
|
+
'default': 0.0,
|
|
107
|
+
'range': [0.0, 1.0]
|
|
108
|
+
},
|
|
109
|
+
'random_state': {
|
|
110
|
+
'description': 'Random seed for reproducibility.',
|
|
111
|
+
'default': 42
|
|
112
|
+
},
|
|
113
|
+
'verbosity': {
|
|
114
|
+
'description': 'Controls the level of LightGBM verbosity.',
|
|
115
|
+
'default': -1,
|
|
116
|
+
'categorical': [-1, 0, 1]
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
self.model: LGBMRegressor = None
|
|
120
|
+
self.columns: list[str] = []
|
|
121
|
+
self.categorical_columns: list[str] = []
|
|
122
|
+
self._category_levels: dict[str, list] = {}
|
|
123
|
+
|
|
124
|
+
def _select_features(self, X):
|
|
125
|
+
if self.columns and hasattr(X, 'columns'):
|
|
126
|
+
return X[self.columns]
|
|
127
|
+
return X
|
|
128
|
+
|
|
129
|
+
def _prepare_features(self, X, fit: bool = False):
|
|
130
|
+
X_selected = self._select_features(X)
|
|
131
|
+
if not hasattr(X_selected, 'copy'):
|
|
132
|
+
return X_selected
|
|
133
|
+
X_prepared = X_selected.copy()
|
|
134
|
+
if self.categorical_columns:
|
|
135
|
+
for column in self.categorical_columns:
|
|
136
|
+
if column not in X_prepared.columns:
|
|
137
|
+
continue
|
|
138
|
+
X_prepared[column] = X_prepared[column].astype('category')
|
|
139
|
+
if not fit and column in self._category_levels:
|
|
140
|
+
X_prepared[column] = X_prepared[column].cat.set_categories(
|
|
141
|
+
self._category_levels[column]
|
|
142
|
+
)
|
|
143
|
+
if fit:
|
|
144
|
+
self._category_levels = {
|
|
145
|
+
column: list(X_prepared[column].cat.categories)
|
|
146
|
+
for column in self.categorical_columns
|
|
147
|
+
if column in X_prepared.columns and hasattr(X_prepared[column], 'cat')
|
|
148
|
+
}
|
|
149
|
+
return X_prepared
|
|
150
|
+
|
|
151
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
152
|
+
if LGBMRegressor is None:
|
|
153
|
+
raise ImportError(
|
|
154
|
+
"lightgbm is required for ActLightGBMRegressor. "
|
|
155
|
+
"Install with: pip install lightgbm"
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
self.columns = dataset.get_columns_names_by_type(
|
|
159
|
+
[DataType.NUMERIC, DataType.CATEGORICAL]
|
|
160
|
+
)
|
|
161
|
+
if not self.columns:
|
|
162
|
+
self.columns = dataset.features
|
|
163
|
+
self.categorical_columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
164
|
+
|
|
165
|
+
X_prepared = self._prepare_features(dataset.X, fit=True)
|
|
166
|
+
self.model = LGBMRegressor(**self.passthrough_parameters())
|
|
167
|
+
|
|
168
|
+
categorical_features = []
|
|
169
|
+
if self.categorical_columns and hasattr(X_prepared, 'columns'):
|
|
170
|
+
categorical_features = [
|
|
171
|
+
col for col in self.categorical_columns if col in X_prepared.columns
|
|
172
|
+
]
|
|
173
|
+
|
|
174
|
+
try:
|
|
175
|
+
if categorical_features:
|
|
176
|
+
self.model.fit(
|
|
177
|
+
X_prepared,
|
|
178
|
+
dataset.y,
|
|
179
|
+
categorical_feature=categorical_features
|
|
180
|
+
)
|
|
181
|
+
else:
|
|
182
|
+
self.model.fit(X_prepared, dataset.y)
|
|
183
|
+
except _LGBM_ERRORS as exc:
|
|
184
|
+
raise ValueError(f"LightGBMRegressor training failed: {exc}") from exc
|
|
185
|
+
return self
|
|
186
|
+
|
|
187
|
+
def predict(self, X):
|
|
188
|
+
return super().predict(self._prepare_features(X))
|
|
189
|
+
|
|
190
|
+
def score(self, X, y=None, *args, **kwargs):
|
|
191
|
+
return self.model.score(self._prepare_features(X), y, *args, **kwargs)
|
|
192
|
+
|
|
193
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
194
|
+
supported = dataset.get_columns_names_by_type(
|
|
195
|
+
[DataType.NUMERIC, DataType.CATEGORICAL]
|
|
196
|
+
)
|
|
197
|
+
return LGBMRegressor is not None and dataset.type_of_target in \
|
|
198
|
+
['continuous'] and bool(supported)
|
|
199
|
+
|
|
200
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
201
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""
|
|
2
|
+
[STEP] Linear Regression
|
|
3
|
+
"""
|
|
4
|
+
import textwrap
|
|
5
|
+
from typing import Any
|
|
6
|
+
from sklearn.linear_model import LinearRegression
|
|
7
|
+
from ....predictor import Predictor
|
|
8
|
+
from ....dataset import Dataset
|
|
9
|
+
from ....candidate import Candidate
|
|
10
|
+
from ....decorators.all import is_step
|
|
11
|
+
|
|
12
|
+
@is_step('predictor', 'tabular', 'fast_predictor', 'regressor', 'baseline_predictor')
|
|
13
|
+
class ActLinearRegression(Predictor):
|
|
14
|
+
"""[STEP] Linear Regression"""
|
|
15
|
+
|
|
16
|
+
name: str = "Linear Regression"
|
|
17
|
+
_description: str = textwrap.dedent('''\
|
|
18
|
+
LinearRegression is a machine learning algorithm that models the
|
|
19
|
+
relationship between input features and a continuous output variable using
|
|
20
|
+
a linear function.''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
LinearRegression is a type of regression algorithm that models
|
|
23
|
+
the relationship between input features and a continuous output variable using
|
|
24
|
+
a linear function. It works by finding the best-fitting line or hyperplane
|
|
25
|
+
that minimizes the sum of the squared differences between the predicted
|
|
26
|
+
and actual output variables.''')
|
|
27
|
+
_usage: str = "Use when you need a fast linear baseline before ActElasticNetRegressor or ActDecisionTreeRegressor. Applicable to tabular regression with roughly linear relationships. Avoid when strong nonlinearity, interactions, or heavy regularization is needed."
|
|
28
|
+
refs: list[dict[str, Any]] = []
|
|
29
|
+
|
|
30
|
+
def __init__(self):
|
|
31
|
+
self.model: LinearRegression = None
|
|
32
|
+
|
|
33
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
34
|
+
self.model = LinearRegression()
|
|
35
|
+
self.model.fit(dataset.X, dataset.y)
|
|
36
|
+
|
|
37
|
+
return self
|
|
38
|
+
|
|
39
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
40
|
+
return dataset.type_of_target == 'continuous'
|
|
41
|
+
|
|
42
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
43
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""[STEP] MLP Regressor"""
|
|
2
|
+
from typing import Any
|
|
3
|
+
import textwrap
|
|
4
|
+
from sklearn.neural_network import MLPRegressor
|
|
5
|
+
from ....predictor import Predictor
|
|
6
|
+
from ....dataset import Dataset
|
|
7
|
+
from ....candidate import Candidate
|
|
8
|
+
from ....decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
@is_step('predictor', 'tabular', 'regressor')
|
|
11
|
+
class ActMLPRegressor(Predictor):
|
|
12
|
+
"""[STEP] MLP Regressor"""
|
|
13
|
+
|
|
14
|
+
name: str = "MLP Regressor"
|
|
15
|
+
_usage: str = "Use when nonlinear tabular regression needs a flexible MLP, beyond ActElasticNetRegressor or ActDecisionTreeRegressor. Applicable to continuous targets with mostly numeric, scaled features. Avoid when data are tiny, mostly categorical, or you need fast/transparent models."
|
|
16
|
+
_description: str = textwrap.dedent('''\
|
|
17
|
+
MLPRegressor is a machine learning algorithm that models the
|
|
18
|
+
relationship between input features and a continuous output variable using
|
|
19
|
+
a multi-layer perceptron neural network.''')
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
MLPRegressor is a type of neural network algorithm that models
|
|
22
|
+
the relationship between input features and a continuous output variable using a
|
|
23
|
+
multi-layer perceptron (MLP) neural network.
|
|
24
|
+
It works by transforming the input features through one or more hidden
|
|
25
|
+
layers with non-linear activation functions, and then using a final layer with
|
|
26
|
+
a linear activation function to output a continuous value.''')
|
|
27
|
+
refs: list[dict[str, Any]] = [
|
|
28
|
+
{
|
|
29
|
+
'year': 1989,
|
|
30
|
+
'name': 'Connectionist Learning Procedures',
|
|
31
|
+
'authors': [
|
|
32
|
+
'Geoffrey E. Hinton'
|
|
33
|
+
],
|
|
34
|
+
'doi': 'https://doi.org/10.1016/0004-3702(89)90049-0',
|
|
35
|
+
'publisher': 'Artificial intelligence Vol. 40.1 page 185--234'
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
'year': 2010,
|
|
39
|
+
'name': 'Understanding the difficulty of training deep feedforward neural networks',
|
|
40
|
+
'authors': [
|
|
41
|
+
'Xavier Glorot',
|
|
42
|
+
'Yoshua Bengio'
|
|
43
|
+
],
|
|
44
|
+
'doi': '',
|
|
45
|
+
'publisher': (
|
|
46
|
+
'Proceedings of the Thirteenth International Conference on '
|
|
47
|
+
'Artificial Intelligence and Statistics page 249--256'
|
|
48
|
+
)
|
|
49
|
+
}
|
|
50
|
+
]
|
|
51
|
+
|
|
52
|
+
def __init__(self):
|
|
53
|
+
self.configuration = {
|
|
54
|
+
'activation': {
|
|
55
|
+
'description': 'Activation function for the hidden layer.',
|
|
56
|
+
'default': 'relu',
|
|
57
|
+
'categorical': ["tanh", "relu"]
|
|
58
|
+
},
|
|
59
|
+
'alpha': {
|
|
60
|
+
'description': textwrap.dedent('''\
|
|
61
|
+
Strength of the L2 regularization term. The L2
|
|
62
|
+
regularization term is divided by the sample size when
|
|
63
|
+
added to the loss.'''),
|
|
64
|
+
'default': 0.0001,
|
|
65
|
+
'range': [1e-07, 0.1]
|
|
66
|
+
},
|
|
67
|
+
'hidden_layer_count': {
|
|
68
|
+
'description': 'Number of hidden layer',
|
|
69
|
+
'default': 1,
|
|
70
|
+
'range': [1, 4],
|
|
71
|
+
'passthrough': False
|
|
72
|
+
},
|
|
73
|
+
'node_per_layer': {
|
|
74
|
+
'description': 'Number of node per layer',
|
|
75
|
+
'default': 32,
|
|
76
|
+
'range': [16, 256],
|
|
77
|
+
'passthrough': False
|
|
78
|
+
},
|
|
79
|
+
'learning_rate_init': {
|
|
80
|
+
'description': 'Learning rate schedule for weight updates',
|
|
81
|
+
'default': 0.001,
|
|
82
|
+
'range': [0.0001, 0.5]
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
self.model: MLPRegressor = None
|
|
87
|
+
|
|
88
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
89
|
+
self.model = MLPRegressor(
|
|
90
|
+
hidden_layer_sizes=[self.get_config('node_per_layer') \
|
|
91
|
+
for i in range(self.get_config('hidden_layer_count'))],
|
|
92
|
+
early_stopping=True,
|
|
93
|
+
max_iter=400,
|
|
94
|
+
**self.passthrough_parameters())
|
|
95
|
+
|
|
96
|
+
self.model.fit(dataset.X, dataset.y)
|
|
97
|
+
|
|
98
|
+
return self
|
|
99
|
+
|
|
100
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
101
|
+
return dataset.type_of_target == "continuous"
|
|
102
|
+
|
|
103
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
104
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""[STEP] Poisson Regressor"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
from sklearn.linear_model import PoissonRegressor
|
|
7
|
+
|
|
8
|
+
from ....predictor import Predictor
|
|
9
|
+
from ....dataset import Dataset
|
|
10
|
+
from ....candidate import Candidate
|
|
11
|
+
from ....data_type import DataType
|
|
12
|
+
from ....decorators.all import is_step
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@is_step('predictor', 'tabular', 'regressor')
|
|
16
|
+
class ActPoissonRegressor(Predictor):
|
|
17
|
+
"""[STEP] Poisson Regressor"""
|
|
18
|
+
|
|
19
|
+
name: str = "Poisson Regressor"
|
|
20
|
+
_description: str = textwrap.dedent('''\
|
|
21
|
+
PoissonRegressor models count targets using a log link and a Poisson
|
|
22
|
+
likelihood to produce positive predictions.''')
|
|
23
|
+
_description_long: str = textwrap.dedent('''\
|
|
24
|
+
Poisson regression is a generalized linear model designed for
|
|
25
|
+
non-negative count data. It connects predictors to the expected
|
|
26
|
+
count through a log link, which keeps predictions positive and
|
|
27
|
+
is appropriate when variance grows with the mean.''')
|
|
28
|
+
_usage: str = "Use when modeling non-negative count targets with variance rising with mean, as a simpler option than ActElasticNetRegressor or ActDecisionTreeRegressor. Applicable to tabular numeric features with count outcomes. Avoid when targets are continuous, negative, or highly zero-inflated."
|
|
29
|
+
refs: list[dict[str, Any]] = [
|
|
30
|
+
{
|
|
31
|
+
'year': 1972,
|
|
32
|
+
'name': 'Generalized Linear Models',
|
|
33
|
+
'authors': [
|
|
34
|
+
'John A. Nelder',
|
|
35
|
+
'Robert W. M. Wedderburn'
|
|
36
|
+
],
|
|
37
|
+
'doi': 'https://doi.org/10.2307/2344614',
|
|
38
|
+
'publisher': 'Journal of the Royal Statistical Society, Series A'
|
|
39
|
+
}
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
def __init__(self):
|
|
43
|
+
self.configuration = {
|
|
44
|
+
'alpha': {
|
|
45
|
+
'description': 'L2 regularization strength.',
|
|
46
|
+
'default': 1.0,
|
|
47
|
+
'range': [1e-06, 10.0]
|
|
48
|
+
},
|
|
49
|
+
'fit_intercept': {
|
|
50
|
+
'description': 'Whether to fit the intercept term.',
|
|
51
|
+
'default': True,
|
|
52
|
+
'categorical': [True, False]
|
|
53
|
+
},
|
|
54
|
+
'max_iter': {
|
|
55
|
+
'description': 'Maximum number of iterations.',
|
|
56
|
+
'default': 100,
|
|
57
|
+
'range': [50, 2000]
|
|
58
|
+
},
|
|
59
|
+
'tol': {
|
|
60
|
+
'description': 'Stopping criterion.',
|
|
61
|
+
'default': 0.0001,
|
|
62
|
+
'range': [1e-06, 0.1]
|
|
63
|
+
},
|
|
64
|
+
'warm_start': {
|
|
65
|
+
'description': 'Reuse solution from the previous fit.',
|
|
66
|
+
'default': False,
|
|
67
|
+
'categorical': [True, False]
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
self.model: PoissonRegressor = None
|
|
71
|
+
self.columns: list[str] = []
|
|
72
|
+
|
|
73
|
+
def _select_features(self, X):
|
|
74
|
+
if self.columns and hasattr(X, 'columns'):
|
|
75
|
+
return X[self.columns]
|
|
76
|
+
return X
|
|
77
|
+
|
|
78
|
+
def _is_count_target(self, y) -> bool:
|
|
79
|
+
y_array = np.asarray(y)
|
|
80
|
+
if y_array.size == 0:
|
|
81
|
+
return False
|
|
82
|
+
if not np.issubdtype(y_array.dtype, np.number):
|
|
83
|
+
return False
|
|
84
|
+
if not np.isfinite(y_array).all():
|
|
85
|
+
return False
|
|
86
|
+
if (y_array < 0).any():
|
|
87
|
+
return False
|
|
88
|
+
return np.allclose(y_array, np.round(y_array), rtol=0, atol=1e-06)
|
|
89
|
+
|
|
90
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
91
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
92
|
+
if not self.columns:
|
|
93
|
+
self.columns = dataset.features
|
|
94
|
+
|
|
95
|
+
self.model = PoissonRegressor(**self.passthrough_parameters())
|
|
96
|
+
self.model.fit(self._select_features(dataset.X), dataset.y)
|
|
97
|
+
return self
|
|
98
|
+
|
|
99
|
+
def predict(self, X):
|
|
100
|
+
return super().predict(self._select_features(X))
|
|
101
|
+
|
|
102
|
+
def score(self, X, y=None, *args, **kwargs):
|
|
103
|
+
return self.model.score(self._select_features(X), y, *args, **kwargs)
|
|
104
|
+
|
|
105
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
106
|
+
return dataset.type_of_target == 'continuous' \
|
|
107
|
+
and bool(dataset.get_columns_names_by_type(DataType.NUMERIC)) \
|
|
108
|
+
and self._is_count_target(dataset.y)
|
|
109
|
+
|
|
110
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
111
|
+
return 0.5 # neutral
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""[STEP] Quantile Regressor"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from sklearn.linear_model import QuantileRegressor
|
|
6
|
+
|
|
7
|
+
from ....predictor import Predictor
|
|
8
|
+
from ....dataset import Dataset
|
|
9
|
+
from ....candidate import Candidate
|
|
10
|
+
from ....data_type import DataType
|
|
11
|
+
from ....decorators.all import is_step
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@is_step('predictor', 'tabular', 'regressor')
|
|
15
|
+
class ActQuantileRegressor(Predictor):
|
|
16
|
+
"""[STEP] Quantile Regressor"""
|
|
17
|
+
|
|
18
|
+
name: str = "Quantile Regressor"
|
|
19
|
+
_usage: str = "Use when you need conditional quantiles or asymmetric error control; choose over ActElasticNetRegressor for interval focus. Applicable to tabular regression with continuous targets and numeric inputs. Avoid when mean prediction is enough or nonlinear structure dominates."
|
|
20
|
+
_description: str = textwrap.dedent('''\
|
|
21
|
+
QuantileRegressor estimates a conditional quantile of a continuous target,
|
|
22
|
+
enabling interval-style predictions and asymmetric error handling.''')
|
|
23
|
+
_description_long: str = textwrap.dedent('''\
|
|
24
|
+
Quantile regression models a chosen quantile of the response rather than the
|
|
25
|
+
mean, which makes it useful for prediction intervals and robust modeling.
|
|
26
|
+
By selecting different quantiles, the model can describe the uncertainty
|
|
27
|
+
around the target distribution.''')
|
|
28
|
+
refs: list[dict[str, Any]] = [
|
|
29
|
+
{
|
|
30
|
+
'year': 1978,
|
|
31
|
+
'name': 'Regression Quantiles',
|
|
32
|
+
'authors': [
|
|
33
|
+
'Roger Koenker',
|
|
34
|
+
'Gilbert Bassett Jr.'
|
|
35
|
+
],
|
|
36
|
+
'doi': 'https://doi.org/10.2307/1913643',
|
|
37
|
+
'publisher': 'Econometrica'
|
|
38
|
+
}
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
def __init__(self):
|
|
42
|
+
self.configuration = {
|
|
43
|
+
'quantile': {
|
|
44
|
+
'description': 'Quantile to estimate between 0 and 1.',
|
|
45
|
+
'default': 0.5,
|
|
46
|
+
'range': [0.05, 0.95]
|
|
47
|
+
},
|
|
48
|
+
'alpha': {
|
|
49
|
+
'description': 'L1 regularization strength.',
|
|
50
|
+
'default': 1.0,
|
|
51
|
+
'range': [1e-06, 10.0]
|
|
52
|
+
},
|
|
53
|
+
'fit_intercept': {
|
|
54
|
+
'description': 'Whether to fit the intercept term.',
|
|
55
|
+
'default': True,
|
|
56
|
+
'categorical': [True, False]
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
self.model: QuantileRegressor = None
|
|
60
|
+
self.columns: list[str] = []
|
|
61
|
+
|
|
62
|
+
def _select_features(self, X):
|
|
63
|
+
if self.columns and hasattr(X, 'columns'):
|
|
64
|
+
return X[self.columns]
|
|
65
|
+
return X
|
|
66
|
+
|
|
67
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
68
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
69
|
+
if not self.columns:
|
|
70
|
+
self.columns = dataset.features
|
|
71
|
+
|
|
72
|
+
self.model = QuantileRegressor(**self.passthrough_parameters())
|
|
73
|
+
self.model.fit(self._select_features(dataset.X), dataset.y)
|
|
74
|
+
return self
|
|
75
|
+
|
|
76
|
+
def predict(self, X):
|
|
77
|
+
return super().predict(self._select_features(X))
|
|
78
|
+
|
|
79
|
+
def score(self, X, y=None, *args, **kwargs):
|
|
80
|
+
return self.model.score(self._select_features(X), y, *args, **kwargs)
|
|
81
|
+
|
|
82
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
83
|
+
return dataset.type_of_target == 'continuous' \
|
|
84
|
+
and bool(dataset.get_columns_names_by_type(DataType.NUMERIC))
|
|
85
|
+
|
|
86
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
87
|
+
return 0.5 # neutral
|