PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
"""[STEP] SMOTETomek"""
|
|
2
|
+
import inspect
|
|
3
|
+
import textwrap
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
from imblearn.combine import SMOTETomek
|
|
9
|
+
from imblearn.over_sampling import SMOTE
|
|
10
|
+
from imblearn.under_sampling import TomekLinks
|
|
11
|
+
|
|
12
|
+
from ...actionable import Actionable
|
|
13
|
+
from ...candidate import Candidate
|
|
14
|
+
from ...data_type import DataType
|
|
15
|
+
from ...dataset import Dataset
|
|
16
|
+
from ...decorators.all import is_step
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@is_step('imbalance')
|
|
20
|
+
class ActSMOTETomek(Actionable):
|
|
21
|
+
"""[STEP] SMOTETomek"""
|
|
22
|
+
|
|
23
|
+
name: str = "SMOTE Tomek"
|
|
24
|
+
_description: str = textwrap.dedent('''\
|
|
25
|
+
SMOTETomek balances data by creating synthetic minority samples
|
|
26
|
+
and removing Tomek links from overlapping classes.''')
|
|
27
|
+
_description_long: str = textwrap.dedent('''\
|
|
28
|
+
SMOTETomek combines SMOTE oversampling with Tomek links cleaning.
|
|
29
|
+
It generates synthetic minority samples, then removes nearest neighbor
|
|
30
|
+
pairs from different classes to reduce overlap and noise.''')
|
|
31
|
+
_usage: str = "Use when imbalanced numeric data has overlap and want SMOTE plus Tomek cleanup vs ActSMOTE. Applicable to binary or multiclass, all-numeric features. Avoid when categorical/text/date fields exist, minority has <2 samples, or you want pure undersampling like ActRandomUnderSampler."
|
|
32
|
+
refs: list[dict[str, Any]] = [
|
|
33
|
+
{
|
|
34
|
+
'year': 2002,
|
|
35
|
+
'name': 'SMOTE: Synthetic Minority Over-sampling Technique',
|
|
36
|
+
'authors': [
|
|
37
|
+
'Nitesh V. Chawla',
|
|
38
|
+
'Kevin W. Bowyer',
|
|
39
|
+
'Lawrence O. Hall',
|
|
40
|
+
'W. Philip Kegelmeyer'
|
|
41
|
+
],
|
|
42
|
+
'doi': 'https://doi.org/10.1613/jair.953',
|
|
43
|
+
'publisher': 'Journal of Artificial Intelligence Research Vol.16 page 321--357'
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
'year': 1976,
|
|
47
|
+
'name': 'Two Modifications of CNN',
|
|
48
|
+
'authors': [
|
|
49
|
+
'Ivan Tomek'
|
|
50
|
+
],
|
|
51
|
+
'publisher': 'IEEE Transactions on Systems, Man, and Cybernetics Vol.6 No.11 '
|
|
52
|
+
'page 769--772'
|
|
53
|
+
}
|
|
54
|
+
]
|
|
55
|
+
|
|
56
|
+
def __init__(self):
|
|
57
|
+
self.configuration = {
|
|
58
|
+
'sampling_strategy': {
|
|
59
|
+
'description': 'Sampling strategy to balance classes.',
|
|
60
|
+
'default': 'auto',
|
|
61
|
+
'categorical': ['minority', 'auto']
|
|
62
|
+
},
|
|
63
|
+
'k_neighbors': {
|
|
64
|
+
'description': 'Number of nearest neighbors used to create synthetic samples.',
|
|
65
|
+
'default': 5,
|
|
66
|
+
'range': [1, 20]
|
|
67
|
+
},
|
|
68
|
+
'random_state': {
|
|
69
|
+
'description': 'Random seed used for reproducibility.',
|
|
70
|
+
'default': 42
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
self.resampler: SMOTETomek | None = None
|
|
74
|
+
self.categorical_columns: list[str] = []
|
|
75
|
+
self.numeric_columns: list[str] = []
|
|
76
|
+
self._effective_k_neighbors: int | None = None
|
|
77
|
+
|
|
78
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
79
|
+
self.resampler = None
|
|
80
|
+
self.categorical_columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
81
|
+
self.numeric_columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
82
|
+
self._effective_k_neighbors = None
|
|
83
|
+
|
|
84
|
+
if dataset.X.empty or dataset.type_of_target not in ['binary', 'multiclass']:
|
|
85
|
+
return self
|
|
86
|
+
|
|
87
|
+
if dataset.y is None or len(dataset.y) == 0:
|
|
88
|
+
return self
|
|
89
|
+
|
|
90
|
+
unsupported = dataset.get_columns_names_by_type(
|
|
91
|
+
[DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
|
|
92
|
+
)
|
|
93
|
+
if unsupported:
|
|
94
|
+
return self
|
|
95
|
+
|
|
96
|
+
if self.categorical_columns:
|
|
97
|
+
return self
|
|
98
|
+
|
|
99
|
+
if not self.numeric_columns:
|
|
100
|
+
return self
|
|
101
|
+
|
|
102
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
103
|
+
if len(counts) < 2:
|
|
104
|
+
return self
|
|
105
|
+
|
|
106
|
+
min_count = int(counts.min())
|
|
107
|
+
if min_count <= 1:
|
|
108
|
+
return self
|
|
109
|
+
|
|
110
|
+
max_k = min_count - 1
|
|
111
|
+
k_neighbors = min(int(self.get_config('k_neighbors')), max_k)
|
|
112
|
+
k_neighbors = max(1, k_neighbors)
|
|
113
|
+
self._effective_k_neighbors = k_neighbors
|
|
114
|
+
|
|
115
|
+
smote_params = {
|
|
116
|
+
'k_neighbors': k_neighbors
|
|
117
|
+
}
|
|
118
|
+
smote_sig = inspect.signature(SMOTE).parameters
|
|
119
|
+
if 'sampling_strategy' in smote_sig:
|
|
120
|
+
smote_params['sampling_strategy'] = self.get_config('sampling_strategy')
|
|
121
|
+
if 'random_state' in smote_sig:
|
|
122
|
+
smote_params['random_state'] = self.get_config('random_state')
|
|
123
|
+
smote = SMOTE(**smote_params)
|
|
124
|
+
|
|
125
|
+
tomek = TomekLinks()
|
|
126
|
+
|
|
127
|
+
smotetomek_params: dict[str, Any] = {}
|
|
128
|
+
smotetomek_sig = inspect.signature(SMOTETomek).parameters
|
|
129
|
+
if 'smote' in smotetomek_sig:
|
|
130
|
+
smotetomek_params['smote'] = smote
|
|
131
|
+
if 'tomek' in smotetomek_sig:
|
|
132
|
+
smotetomek_params['tomek'] = tomek
|
|
133
|
+
if 'sampling_strategy' in smotetomek_sig and 'smote' not in smotetomek_params:
|
|
134
|
+
smotetomek_params['sampling_strategy'] = self.get_config('sampling_strategy')
|
|
135
|
+
if 'random_state' in smotetomek_sig and 'smote' not in smotetomek_params:
|
|
136
|
+
smotetomek_params['random_state'] = self.get_config('random_state')
|
|
137
|
+
|
|
138
|
+
self.resampler = SMOTETomek(**smotetomek_params)
|
|
139
|
+
self.resampler.fit(dataset.X, dataset.y)
|
|
140
|
+
return self
|
|
141
|
+
|
|
142
|
+
def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
143
|
+
"""Apply SMOTETomek.
|
|
144
|
+
|
|
145
|
+
:param pd.DataFrame X: Features to resample
|
|
146
|
+
:param pd.DataFrame y: Labels to resample
|
|
147
|
+
:return: Resampled X and y
|
|
148
|
+
"""
|
|
149
|
+
if self.resampler is None:
|
|
150
|
+
return X, y
|
|
151
|
+
return self.resampler.fit_resample(X, y)
|
|
152
|
+
|
|
153
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
154
|
+
if candidate is None or candidate.dataset.y is None:
|
|
155
|
+
return 0.0
|
|
156
|
+
y = candidate.dataset.y
|
|
157
|
+
if len(y) == 0:
|
|
158
|
+
return 0.0
|
|
159
|
+
_, counts = np.unique(y, return_counts=True)
|
|
160
|
+
if len(counts) < 2:
|
|
161
|
+
return 0.0
|
|
162
|
+
imbalance = 1.0 - (counts.min() / counts.max())
|
|
163
|
+
return float(min(1.0, max(0.0, imbalance)))
|
|
164
|
+
|
|
165
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
166
|
+
if dataset.type_of_target not in ['binary', 'multiclass']:
|
|
167
|
+
return False
|
|
168
|
+
if dataset.X.empty or dataset.y is None or len(dataset.y) == 0:
|
|
169
|
+
return False
|
|
170
|
+
if dataset.get_columns_names_by_type(DataType.CATEGORICAL):
|
|
171
|
+
return False
|
|
172
|
+
if not dataset.get_columns_names_by_type(DataType.NUMERIC):
|
|
173
|
+
return False
|
|
174
|
+
unsupported = dataset.get_columns_names_by_type(
|
|
175
|
+
[DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
|
|
176
|
+
)
|
|
177
|
+
if unsupported:
|
|
178
|
+
return False
|
|
179
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
180
|
+
if len(counts) < 2:
|
|
181
|
+
return False
|
|
182
|
+
return counts.min() > 1
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""[STEP] SMOTEENN"""
|
|
2
|
+
import inspect
|
|
3
|
+
import textwrap
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
from imblearn.combine import SMOTEENN
|
|
9
|
+
from imblearn.over_sampling import SMOTE
|
|
10
|
+
from imblearn.under_sampling import EditedNearestNeighbours
|
|
11
|
+
|
|
12
|
+
from ...actionable import Actionable
|
|
13
|
+
from ...candidate import Candidate
|
|
14
|
+
from ...data_type import DataType
|
|
15
|
+
from ...dataset import Dataset
|
|
16
|
+
from ...decorators.all import is_step
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@is_step('imbalance')
|
|
20
|
+
class ActSMOTEENN(Actionable):
|
|
21
|
+
"""[STEP] SMOTEENN"""
|
|
22
|
+
|
|
23
|
+
name: str = "SMOTE ENN"
|
|
24
|
+
_description: str = textwrap.dedent('''\
|
|
25
|
+
SMOTEENN balances data by creating synthetic minority samples and
|
|
26
|
+
removing ambiguous samples with Edited Nearest Neighbors.''')
|
|
27
|
+
_description_long: str = textwrap.dedent('''\
|
|
28
|
+
SMOTEENN combines SMOTE oversampling with Edited Nearest Neighbors
|
|
29
|
+
cleaning. It adds synthetic minority samples, then removes samples
|
|
30
|
+
that disagree with their neighbors to reduce noise and class overlap.''')
|
|
31
|
+
_usage: str = "Use when imbalance with noisy borders needs SMOTE plus cleaning vs ActSMOTE. Applicable to numeric-only binary or multiclass data with >=2 samples per class. Avoid when categorical/text features, very small minorities, or you want pure under-sampling like ActNearMiss."
|
|
32
|
+
refs: list[dict[str, Any]] = [
|
|
33
|
+
{
|
|
34
|
+
'year': 2004,
|
|
35
|
+
'name': 'A Study of the Behavior of Several Methods for Balancing '
|
|
36
|
+
'Machine Learning Training Data',
|
|
37
|
+
'authors': [
|
|
38
|
+
'Gustavo E. A. P. A. Batista',
|
|
39
|
+
'Ronaldo C. Prati',
|
|
40
|
+
'Maria Carolina Monard'
|
|
41
|
+
],
|
|
42
|
+
'doi': 'https://doi.org/10.1145/1007730.1007735',
|
|
43
|
+
'publisher': 'ACM SIGKDD Explorations Newsletter Vol.6 No.1 page 20--29'
|
|
44
|
+
}
|
|
45
|
+
]
|
|
46
|
+
|
|
47
|
+
def __init__(self):
|
|
48
|
+
self.configuration = {
|
|
49
|
+
'sampling_strategy': {
|
|
50
|
+
'description': 'Sampling strategy to balance classes.',
|
|
51
|
+
'default': 'auto',
|
|
52
|
+
'categorical': ['minority', 'auto']
|
|
53
|
+
},
|
|
54
|
+
'k_neighbors': {
|
|
55
|
+
'description': 'Number of nearest neighbors used to create synthetic samples.',
|
|
56
|
+
'default': 5,
|
|
57
|
+
'range': [1, 20]
|
|
58
|
+
},
|
|
59
|
+
'n_neighbors': {
|
|
60
|
+
'description': 'Number of neighbors used by Edited Nearest Neighbors.',
|
|
61
|
+
'default': 3,
|
|
62
|
+
'range': [1, 20]
|
|
63
|
+
},
|
|
64
|
+
'kind_sel': {
|
|
65
|
+
'description': 'Rule used by Edited Nearest Neighbors to select samples.',
|
|
66
|
+
'default': 'all',
|
|
67
|
+
'categorical': ['all', 'mode'],
|
|
68
|
+
'passthrough': False
|
|
69
|
+
},
|
|
70
|
+
'random_state': {
|
|
71
|
+
'description': 'Random seed used for reproducibility.',
|
|
72
|
+
'default': 42
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
self.resampler: SMOTEENN | None = None
|
|
76
|
+
self.categorical_columns: list[str] = []
|
|
77
|
+
self.numeric_columns: list[str] = []
|
|
78
|
+
self._effective_k_neighbors: int | None = None
|
|
79
|
+
self._effective_n_neighbors: int | None = None
|
|
80
|
+
|
|
81
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
82
|
+
self.resampler = None
|
|
83
|
+
self.categorical_columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
84
|
+
self.numeric_columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
85
|
+
self._effective_k_neighbors = None
|
|
86
|
+
self._effective_n_neighbors = None
|
|
87
|
+
|
|
88
|
+
if dataset.X.empty or dataset.type_of_target not in ['binary', 'multiclass']:
|
|
89
|
+
return self
|
|
90
|
+
|
|
91
|
+
if dataset.y is None or len(dataset.y) == 0:
|
|
92
|
+
return self
|
|
93
|
+
|
|
94
|
+
unsupported = dataset.get_columns_names_by_type(
|
|
95
|
+
[DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
|
|
96
|
+
)
|
|
97
|
+
if unsupported:
|
|
98
|
+
return self
|
|
99
|
+
|
|
100
|
+
if self.categorical_columns:
|
|
101
|
+
return self
|
|
102
|
+
|
|
103
|
+
if not self.numeric_columns:
|
|
104
|
+
return self
|
|
105
|
+
|
|
106
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
107
|
+
if len(counts) < 2:
|
|
108
|
+
return self
|
|
109
|
+
|
|
110
|
+
min_count = int(counts.min())
|
|
111
|
+
if min_count <= 1:
|
|
112
|
+
return self
|
|
113
|
+
|
|
114
|
+
max_k = min_count - 1
|
|
115
|
+
k_neighbors = min(int(self.get_config('k_neighbors')), max_k)
|
|
116
|
+
k_neighbors = max(1, k_neighbors)
|
|
117
|
+
self._effective_k_neighbors = k_neighbors
|
|
118
|
+
|
|
119
|
+
total_count = int(len(dataset.y))
|
|
120
|
+
max_neighbors = max(1, total_count - 1)
|
|
121
|
+
n_neighbors = min(int(self.get_config('n_neighbors')), max_neighbors)
|
|
122
|
+
n_neighbors = max(1, n_neighbors)
|
|
123
|
+
self._effective_n_neighbors = n_neighbors
|
|
124
|
+
|
|
125
|
+
smote = SMOTE(
|
|
126
|
+
sampling_strategy=self.get_config('sampling_strategy'),
|
|
127
|
+
k_neighbors=k_neighbors,
|
|
128
|
+
random_state=self.get_config('random_state')
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
enn_params = {
|
|
132
|
+
'n_neighbors': n_neighbors
|
|
133
|
+
}
|
|
134
|
+
if 'kind_sel' in inspect.signature(EditedNearestNeighbours).parameters:
|
|
135
|
+
enn_params['kind_sel'] = self.get_config('kind_sel')
|
|
136
|
+
enn = EditedNearestNeighbours(**enn_params)
|
|
137
|
+
|
|
138
|
+
smoteenn_params = {}
|
|
139
|
+
smoteenn_sig = inspect.signature(SMOTEENN).parameters
|
|
140
|
+
if 'smote' in smoteenn_sig:
|
|
141
|
+
smoteenn_params['smote'] = smote
|
|
142
|
+
if 'enn' in smoteenn_sig:
|
|
143
|
+
smoteenn_params['enn'] = enn
|
|
144
|
+
if 'sampling_strategy' in smoteenn_sig and 'smote' not in smoteenn_params:
|
|
145
|
+
smoteenn_params['sampling_strategy'] = self.get_config('sampling_strategy')
|
|
146
|
+
if 'random_state' in smoteenn_sig and 'smote' not in smoteenn_params:
|
|
147
|
+
smoteenn_params['random_state'] = self.get_config('random_state')
|
|
148
|
+
|
|
149
|
+
self.resampler = SMOTEENN(**smoteenn_params)
|
|
150
|
+
self.resampler.fit(dataset.X, dataset.y)
|
|
151
|
+
return self
|
|
152
|
+
|
|
153
|
+
def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
154
|
+
"""Apply SMOTEENN.
|
|
155
|
+
|
|
156
|
+
:param pd.DataFrame X: Features to resample
|
|
157
|
+
:param pd.DataFrame y: Labels to resample
|
|
158
|
+
:return: Resampled X and y
|
|
159
|
+
"""
|
|
160
|
+
if self.resampler is None:
|
|
161
|
+
return X, y
|
|
162
|
+
return self.resampler.fit_resample(X, y)
|
|
163
|
+
|
|
164
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
165
|
+
if candidate is None or candidate.dataset.y is None:
|
|
166
|
+
return 0.0
|
|
167
|
+
y = candidate.dataset.y
|
|
168
|
+
if len(y) == 0:
|
|
169
|
+
return 0.0
|
|
170
|
+
_, counts = np.unique(y, return_counts=True)
|
|
171
|
+
if len(counts) < 2:
|
|
172
|
+
return 0.0
|
|
173
|
+
imbalance = 1.0 - (counts.min() / counts.max())
|
|
174
|
+
return float(min(1.0, max(0.0, imbalance)))
|
|
175
|
+
|
|
176
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
177
|
+
if dataset.type_of_target not in ['binary', 'multiclass']:
|
|
178
|
+
return False
|
|
179
|
+
if dataset.X.empty or dataset.y is None or len(dataset.y) == 0:
|
|
180
|
+
return False
|
|
181
|
+
if dataset.get_columns_names_by_type(DataType.CATEGORICAL):
|
|
182
|
+
return False
|
|
183
|
+
if not dataset.get_columns_names_by_type(DataType.NUMERIC):
|
|
184
|
+
return False
|
|
185
|
+
unsupported = dataset.get_columns_names_by_type(
|
|
186
|
+
[DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
|
|
187
|
+
)
|
|
188
|
+
if unsupported:
|
|
189
|
+
return False
|
|
190
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
191
|
+
if len(counts) < 2:
|
|
192
|
+
return False
|
|
193
|
+
return counts.min() > 1
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""[STEP] Tomek Links"""
|
|
2
|
+
import inspect
|
|
3
|
+
import textwrap
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
from imblearn.under_sampling import TomekLinks
|
|
9
|
+
|
|
10
|
+
from ...actionable import Actionable
|
|
11
|
+
from ...candidate import Candidate
|
|
12
|
+
from ...data_type import DataType
|
|
13
|
+
from ...dataset import Dataset
|
|
14
|
+
from ...decorators.all import is_step
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@is_step('imbalance')
|
|
18
|
+
class ActTomekLinks(Actionable):
|
|
19
|
+
"""[STEP] Tomek Links"""
|
|
20
|
+
|
|
21
|
+
name: str = "Tomek Links"
|
|
22
|
+
_description: str = textwrap.dedent('''\
|
|
23
|
+
TomekLinks cleans class boundaries by removing samples that form
|
|
24
|
+
nearest-neighbor pairs across classes.''')
|
|
25
|
+
_description_long: str = textwrap.dedent('''\
|
|
26
|
+
TomekLinks identifies pairs of samples from different classes that are
|
|
27
|
+
each other's nearest neighbors (Tomek links). Removing the majority
|
|
28
|
+
samples in these pairs reduces overlap and cleans noisy boundaries.''')
|
|
29
|
+
_usage: str = "Use when you want light boundary cleaning instead of heavier ActNearMiss. Applicable to numeric-only binary or multiclass data. Avoid when data includes categorical/text/date or you need to add samples (ActSMOTE)."
|
|
30
|
+
refs: list[dict[str, Any]] = [
|
|
31
|
+
{
|
|
32
|
+
'year': 1976,
|
|
33
|
+
'name': 'Two Modifications of CNN',
|
|
34
|
+
'authors': [
|
|
35
|
+
'Ivan Tomek'
|
|
36
|
+
],
|
|
37
|
+
'publisher': 'IEEE Transactions on Systems, Man, and Cybernetics Vol.6 No.11 '
|
|
38
|
+
'page 769--772'
|
|
39
|
+
}
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
def __init__(self):
|
|
43
|
+
self.configuration = {
|
|
44
|
+
'sampling_strategy': {
|
|
45
|
+
'description': 'Sampling strategy to remove Tomek link pairs.',
|
|
46
|
+
'default': 'auto',
|
|
47
|
+
'categorical': ['auto', 'majority', 'all']
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
self.resampler: TomekLinks | None = None
|
|
51
|
+
self.categorical_columns: list[str] = []
|
|
52
|
+
self.numeric_columns: list[str] = []
|
|
53
|
+
self.imbalance_ratio: float = 0.0
|
|
54
|
+
|
|
55
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
56
|
+
self.resampler = None
|
|
57
|
+
self.categorical_columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
58
|
+
self.numeric_columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
59
|
+
self.imbalance_ratio = 0.0
|
|
60
|
+
|
|
61
|
+
if dataset.X.empty or dataset.type_of_target not in ['binary', 'multiclass']:
|
|
62
|
+
return self
|
|
63
|
+
|
|
64
|
+
if dataset.y is None or len(dataset.y) == 0:
|
|
65
|
+
return self
|
|
66
|
+
|
|
67
|
+
unsupported = dataset.get_columns_names_by_type(
|
|
68
|
+
[DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
|
|
69
|
+
)
|
|
70
|
+
if unsupported:
|
|
71
|
+
return self
|
|
72
|
+
|
|
73
|
+
if self.categorical_columns:
|
|
74
|
+
return self
|
|
75
|
+
|
|
76
|
+
if not self.numeric_columns:
|
|
77
|
+
return self
|
|
78
|
+
|
|
79
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
80
|
+
if len(counts) < 2:
|
|
81
|
+
return self
|
|
82
|
+
|
|
83
|
+
max_count = int(counts.max())
|
|
84
|
+
min_count = int(counts.min())
|
|
85
|
+
if max_count <= 0 or min_count <= 0:
|
|
86
|
+
return self
|
|
87
|
+
|
|
88
|
+
self.imbalance_ratio = 1.0 - (min_count / max_count)
|
|
89
|
+
|
|
90
|
+
params = self.passthrough_parameters()
|
|
91
|
+
sig_params = inspect.signature(TomekLinks).parameters
|
|
92
|
+
params = {key: value for key, value in params.items() if key in sig_params}
|
|
93
|
+
|
|
94
|
+
self.resampler = TomekLinks(**params)
|
|
95
|
+
self.resampler.fit(dataset.X, dataset.y)
|
|
96
|
+
return self
|
|
97
|
+
|
|
98
|
+
def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
99
|
+
"""Apply Tomek Links.
|
|
100
|
+
|
|
101
|
+
:param pd.DataFrame X: Features to resample
|
|
102
|
+
:param pd.DataFrame y: Labels to resample
|
|
103
|
+
:return: Resampled X and y
|
|
104
|
+
"""
|
|
105
|
+
if self.resampler is None:
|
|
106
|
+
return X, y
|
|
107
|
+
return self.resampler.fit_resample(X, y)
|
|
108
|
+
|
|
109
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
110
|
+
if candidate is None or candidate.dataset.y is None:
|
|
111
|
+
return 0.0
|
|
112
|
+
y = candidate.dataset.y
|
|
113
|
+
if len(y) == 0:
|
|
114
|
+
return 0.0
|
|
115
|
+
_, counts = np.unique(y, return_counts=True)
|
|
116
|
+
if len(counts) < 2:
|
|
117
|
+
return 0.0
|
|
118
|
+
imbalance = 1.0 - (counts.min() / counts.max())
|
|
119
|
+
return float(min(1.0, max(0.0, imbalance)))
|
|
120
|
+
|
|
121
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
122
|
+
if dataset.type_of_target not in ['binary', 'multiclass']:
|
|
123
|
+
return False
|
|
124
|
+
if dataset.X.empty or dataset.y is None or len(dataset.y) == 0:
|
|
125
|
+
return False
|
|
126
|
+
if dataset.get_columns_names_by_type(DataType.CATEGORICAL):
|
|
127
|
+
return False
|
|
128
|
+
if not dataset.get_columns_names_by_type(DataType.NUMERIC):
|
|
129
|
+
return False
|
|
130
|
+
unsupported = dataset.get_columns_names_by_type(
|
|
131
|
+
[DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
|
|
132
|
+
)
|
|
133
|
+
if unsupported:
|
|
134
|
+
return False
|
|
135
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
136
|
+
if len(counts) < 2:
|
|
137
|
+
return False
|
|
138
|
+
return True
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""Normalize and scaler Actionables"""
|
|
2
|
+
from .act_minmax_scaler import ActMinMaxScaler
|
|
3
|
+
from .act_standard_scaler import ActStandardScaler
|
|
4
|
+
from .act_robust_scaler import ActRobustScaler
|
|
5
|
+
from .act_max_abs_scaler import ActMaxAbsScaler
|
|
6
|
+
from .act_normalizer import ActNormalizer
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""[STEP] Max Abs Scaler"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from sklearn.preprocessing import MaxAbsScaler
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
from ...data_type import DataType
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('normalize')
|
|
13
|
+
class ActMaxAbsScaler(Actionable):
|
|
14
|
+
"""[STEP] Max Abs Scaler"""
|
|
15
|
+
|
|
16
|
+
name: str = "Max Abs Scaler"
|
|
17
|
+
_description: str = textwrap.dedent('''\
|
|
18
|
+
MaxAbsScaler rescales numeric data by dividing by the maximum absolute value,
|
|
19
|
+
keeping values within a [-1, 1] range.''')
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
MaxAbsScaler scales each numeric feature by its maximum absolute value
|
|
22
|
+
observed in the training data. This keeps values within [-1, 1] while
|
|
23
|
+
preserving sparsity because it does not center the data.
|
|
24
|
+
It is a good fit for sparse datasets where zeros should remain zeros.''')
|
|
25
|
+
_usage: str = "Use when numeric features are sparse and you want scale to [-1, 1] without centering; compare ActMinMaxScaler. Applicable to numeric data with many zeros or sparse matrices. Avoid when you need centering or heavy outlier handling; consider ActNormalizer or ActRobustScaler."
|
|
26
|
+
|
|
27
|
+
def __init__(self):
|
|
28
|
+
self.columns: list[str] = None
|
|
29
|
+
self.scaler: MaxAbsScaler = None
|
|
30
|
+
|
|
31
|
+
self.configuration = {
|
|
32
|
+
'copy': {
|
|
33
|
+
'description': 'Set to False to perform scaling in-place when possible.',
|
|
34
|
+
'default': True,
|
|
35
|
+
'categorical': [True, False]
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
40
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
41
|
+
if self.columns and not dataset.X.empty:
|
|
42
|
+
values = dataset.X[self.columns]
|
|
43
|
+
self.scaler = MaxAbsScaler(**self.passthrough_parameters())
|
|
44
|
+
self.scaler.fit(values)
|
|
45
|
+
else:
|
|
46
|
+
self.scaler = None
|
|
47
|
+
return self
|
|
48
|
+
|
|
49
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
50
|
+
"""Apply max abs scaler
|
|
51
|
+
|
|
52
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
53
|
+
:return: Transformed dataset
|
|
54
|
+
"""
|
|
55
|
+
if self.scaler and self.columns:
|
|
56
|
+
columns = [column for column in self.columns if column in X.columns]
|
|
57
|
+
if not columns:
|
|
58
|
+
return X
|
|
59
|
+
X[columns] = self.scaler.transform(X[columns])
|
|
60
|
+
return X
|
|
61
|
+
|
|
62
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
63
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
64
|
+
return bool(columns) and not dataset.X.empty
|
|
65
|
+
|
|
66
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
67
|
+
if candidate is None:
|
|
68
|
+
return 0.0
|
|
69
|
+
columns = candidate.dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
70
|
+
if not columns or candidate.dataset.X.empty:
|
|
71
|
+
return 0.0
|
|
72
|
+
values = candidate.dataset.X[columns]
|
|
73
|
+
total = values.size - values.isna().sum().sum()
|
|
74
|
+
if total <= 0:
|
|
75
|
+
return 0.0
|
|
76
|
+
zeros = (values == 0).sum().sum()
|
|
77
|
+
zero_ratio = zeros / total
|
|
78
|
+
return min(1.0, zero_ratio * 1.5)
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""[STEP] Min Max Scaler"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from sklearn.preprocessing import MinMaxScaler
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
from ...data_type import DataType
|
|
10
|
+
|
|
11
|
+
@is_step('normalize')
|
|
12
|
+
class ActMinMaxScaler(Actionable):
|
|
13
|
+
"""[STEP] Min Max Scaler"""
|
|
14
|
+
|
|
15
|
+
name: str = "Min Max Scaler"
|
|
16
|
+
_usage: str = "Use when you need bounded scaling of numeric features for scale-sensitive models; unlike ActNormalizer, keeps feature ranges. Applicable to continuous numeric columns with stable min/max. Avoid when outliers or range drift dominate; consider ActRobustScaler."
|
|
17
|
+
_description: str = textwrap.dedent('''\
|
|
18
|
+
MinMaxScaler is a tool that helps computers understand complex relationships
|
|
19
|
+
between things by turning them into simpler numbers within a fixed range.''')
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
MinMaxScaler is a machine learning technique used to scale numerical
|
|
22
|
+
features to a fixed range, typically between zero and one.
|
|
23
|
+
It works by finding the minimum and maximum values for each feature in the training data,
|
|
24
|
+
then scaling all values to fall between those extremes.
|
|
25
|
+
This transformation helps ensure that all features are on the same scale,
|
|
26
|
+
which can improve the performance of many machine learning algorithms.''')
|
|
27
|
+
|
|
28
|
+
def __init__(self):
|
|
29
|
+
self.columns: list[str] = None
|
|
30
|
+
self.scaler: MinMaxScaler = None
|
|
31
|
+
|
|
32
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
33
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
34
|
+
if self.columns:
|
|
35
|
+
values = dataset.X[self.columns]
|
|
36
|
+
self.scaler = MinMaxScaler()
|
|
37
|
+
self.scaler.fit(values)
|
|
38
|
+
else:
|
|
39
|
+
self.scaler = None
|
|
40
|
+
return self
|
|
41
|
+
|
|
42
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
43
|
+
"""Apply min max scaler
|
|
44
|
+
|
|
45
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
46
|
+
:return: Transformed dataset
|
|
47
|
+
"""
|
|
48
|
+
if self.scaler and self.columns:
|
|
49
|
+
columns = [column for column in self.columns if column in X.columns]
|
|
50
|
+
if not columns:
|
|
51
|
+
return X
|
|
52
|
+
X[columns] = self.scaler.transform(X[columns])
|
|
53
|
+
return X
|
|
54
|
+
|
|
55
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
56
|
+
return 0.5
|