PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
"""[STEP] Near Miss"""
|
|
2
|
+
import inspect
|
|
3
|
+
import textwrap
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
from imblearn.under_sampling import NearMiss
|
|
9
|
+
|
|
10
|
+
from ...actionable import Actionable
|
|
11
|
+
from ...candidate import Candidate
|
|
12
|
+
from ...data_type import DataType
|
|
13
|
+
from ...dataset import Dataset
|
|
14
|
+
from ...decorators.all import is_step
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@is_step('imbalance')
|
|
18
|
+
class ActNearMiss(Actionable):
|
|
19
|
+
"""[STEP] Near Miss"""
|
|
20
|
+
|
|
21
|
+
name: str = "Near Miss"
|
|
22
|
+
_description: str = textwrap.dedent('''\
|
|
23
|
+
NearMiss under-samples by keeping majority samples
|
|
24
|
+
that are closest to the minority class.''')
|
|
25
|
+
_description_long: str = textwrap.dedent('''\
|
|
26
|
+
NearMiss reduces class imbalance by selecting majority samples near
|
|
27
|
+
the minority class. Different versions use nearest-neighbor distances
|
|
28
|
+
to keep samples that are harder to separate from the minority class.''')
|
|
29
|
+
_usage: str = "Use when you want distance-based under-sampling on numeric data; compare ActRandomUnderSampler or ActSMOTE. Applicable to imbalanced binary or multiclass numeric datasets. Avoid when you have categorical/text/date features or need to retain most majority samples."
|
|
30
|
+
refs: list[dict[str, Any]] = []
|
|
31
|
+
|
|
32
|
+
def __init__(self):
|
|
33
|
+
self.configuration = {
|
|
34
|
+
'sampling_strategy': {
|
|
35
|
+
'description': 'Sampling strategy to reduce the majority class.',
|
|
36
|
+
'default': 'auto',
|
|
37
|
+
'categorical': ['auto', 'majority']
|
|
38
|
+
},
|
|
39
|
+
'version': {
|
|
40
|
+
'description': 'NearMiss version to use (1, 2, or 3).',
|
|
41
|
+
'default': 1,
|
|
42
|
+
'categorical': [1, 2, 3]
|
|
43
|
+
},
|
|
44
|
+
'n_neighbors': {
|
|
45
|
+
'description': 'Number of minority neighbors used to compute distances.',
|
|
46
|
+
'default': 3,
|
|
47
|
+
'range': [1, 20]
|
|
48
|
+
},
|
|
49
|
+
'n_neighbors_ver3': {
|
|
50
|
+
'description': 'Number of majority neighbors kept per minority sample for version 3.',
|
|
51
|
+
'default': 3,
|
|
52
|
+
'range': [1, 20]
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
self.resampler: NearMiss | None = None
|
|
56
|
+
self.categorical_columns: list[str] = []
|
|
57
|
+
self.numeric_columns: list[str] = []
|
|
58
|
+
self.imbalance_ratio: float = 0.0
|
|
59
|
+
self._effective_n_neighbors: int | None = None
|
|
60
|
+
self._effective_n_neighbors_ver3: int | None = None
|
|
61
|
+
|
|
62
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
63
|
+
self.resampler = None
|
|
64
|
+
self.categorical_columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
65
|
+
self.numeric_columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
66
|
+
self.imbalance_ratio = 0.0
|
|
67
|
+
self._effective_n_neighbors = None
|
|
68
|
+
self._effective_n_neighbors_ver3 = None
|
|
69
|
+
|
|
70
|
+
if dataset.X.empty or dataset.type_of_target not in ['binary', 'multiclass']:
|
|
71
|
+
return self
|
|
72
|
+
|
|
73
|
+
if dataset.y is None or len(dataset.y) == 0:
|
|
74
|
+
return self
|
|
75
|
+
|
|
76
|
+
unsupported = dataset.get_columns_names_by_type(
|
|
77
|
+
[DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
|
|
78
|
+
)
|
|
79
|
+
if unsupported:
|
|
80
|
+
return self
|
|
81
|
+
|
|
82
|
+
if self.categorical_columns:
|
|
83
|
+
return self
|
|
84
|
+
|
|
85
|
+
if not self.numeric_columns:
|
|
86
|
+
return self
|
|
87
|
+
|
|
88
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
89
|
+
if len(counts) < 2:
|
|
90
|
+
return self
|
|
91
|
+
|
|
92
|
+
max_count = int(counts.max())
|
|
93
|
+
min_count = int(counts.min())
|
|
94
|
+
if max_count <= min_count or min_count <= 0:
|
|
95
|
+
return self
|
|
96
|
+
|
|
97
|
+
self.imbalance_ratio = 1.0 - (min_count / max_count)
|
|
98
|
+
|
|
99
|
+
n_neighbors = min(int(self.get_config('n_neighbors')), min_count)
|
|
100
|
+
n_neighbors = max(1, n_neighbors)
|
|
101
|
+
self._effective_n_neighbors = n_neighbors
|
|
102
|
+
|
|
103
|
+
n_neighbors_ver3 = min(int(self.get_config('n_neighbors_ver3')), max_count)
|
|
104
|
+
n_neighbors_ver3 = max(1, n_neighbors_ver3)
|
|
105
|
+
self._effective_n_neighbors_ver3 = n_neighbors_ver3
|
|
106
|
+
|
|
107
|
+
params = self.passthrough_parameters()
|
|
108
|
+
params['n_neighbors'] = n_neighbors
|
|
109
|
+
params['n_neighbors_ver3'] = n_neighbors_ver3
|
|
110
|
+
|
|
111
|
+
sig_params = inspect.signature(NearMiss).parameters
|
|
112
|
+
params = {key: value for key, value in params.items() if key in sig_params}
|
|
113
|
+
|
|
114
|
+
self.resampler = NearMiss(**params)
|
|
115
|
+
self.resampler.fit(dataset.X, dataset.y)
|
|
116
|
+
return self
|
|
117
|
+
|
|
118
|
+
def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
119
|
+
"""Apply NearMiss.
|
|
120
|
+
|
|
121
|
+
:param pd.DataFrame X: Features to resample
|
|
122
|
+
:param pd.DataFrame y: Labels to resample
|
|
123
|
+
:return: Resampled X and y
|
|
124
|
+
"""
|
|
125
|
+
if self.resampler is None:
|
|
126
|
+
return X, y
|
|
127
|
+
return self.resampler.fit_resample(X, y)
|
|
128
|
+
|
|
129
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
130
|
+
if candidate is None or candidate.dataset.y is None:
|
|
131
|
+
return 0.0
|
|
132
|
+
y = candidate.dataset.y
|
|
133
|
+
if len(y) == 0:
|
|
134
|
+
return 0.0
|
|
135
|
+
_, counts = np.unique(y, return_counts=True)
|
|
136
|
+
if len(counts) < 2:
|
|
137
|
+
return 0.0
|
|
138
|
+
imbalance = 1.0 - (counts.min() / counts.max())
|
|
139
|
+
return float(min(1.0, max(0.0, imbalance)))
|
|
140
|
+
|
|
141
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
142
|
+
if dataset.type_of_target not in ['binary', 'multiclass']:
|
|
143
|
+
return False
|
|
144
|
+
if dataset.X.empty or dataset.y is None or len(dataset.y) == 0:
|
|
145
|
+
return False
|
|
146
|
+
if dataset.get_columns_names_by_type(DataType.CATEGORICAL):
|
|
147
|
+
return False
|
|
148
|
+
if not dataset.get_columns_names_by_type(DataType.NUMERIC):
|
|
149
|
+
return False
|
|
150
|
+
unsupported = dataset.get_columns_names_by_type(
|
|
151
|
+
[DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
|
|
152
|
+
)
|
|
153
|
+
if unsupported:
|
|
154
|
+
return False
|
|
155
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
156
|
+
if len(counts) < 2:
|
|
157
|
+
return False
|
|
158
|
+
return counts.max() > counts.min()
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""[STEP] Random Over Sampling"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
from imblearn.over_sampling import RandomOverSampler
|
|
5
|
+
import pandas as pd
|
|
6
|
+
from ...actionable import Actionable
|
|
7
|
+
from ...dataset import Dataset
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('imbalance')
|
|
13
|
+
class ActRandomOverSampling(Actionable):
|
|
14
|
+
"""[STEP] Random Over Sampling"""
|
|
15
|
+
|
|
16
|
+
name: str = "Random Over Sampling"
|
|
17
|
+
_description: str = textwrap.dedent('''\
|
|
18
|
+
RandomOverSampler is a tool that helps balance data by copying
|
|
19
|
+
and pasting samples from minority groups.''')
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
RandomOverSampler is a technique used to handle imbalanced datasets.
|
|
22
|
+
It works by randomly copying and pasting samples from the minority class
|
|
23
|
+
(the group with fewer samples) until it has the same number of samples as the majority class.
|
|
24
|
+
This helps ensure that all classes are represented equally in the dataset.''')
|
|
25
|
+
_usage: str = "Use when you need a simple baseline to boost rare classes without synthesis; compare with ActSMOTE or ActADASYN. Applicable to imbalanced binary or multiclass classification tables. Avoid when oversampling risks overfitting or duplicates distort signal."
|
|
26
|
+
refs: list[dict[str, Any]] = [
|
|
27
|
+
{
|
|
28
|
+
'year': 2012,
|
|
29
|
+
'name': 'Training and assessing classification rules with imbalanced data',
|
|
30
|
+
'authors': [
|
|
31
|
+
'Giovanna Menardi',
|
|
32
|
+
'Nicola Torelli'
|
|
33
|
+
],
|
|
34
|
+
'doi': 'https://doi.org/10.1007/s10618-012-0295-5',
|
|
35
|
+
'publisher': 'Data Mining and Knowledge Discovery Vol.28 page 92--122'
|
|
36
|
+
}
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
def __init__(self):
|
|
40
|
+
self.resampler: RandomOverSampler = None
|
|
41
|
+
|
|
42
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
43
|
+
self.resampler = RandomOverSampler(sampling_strategy='minority')
|
|
44
|
+
self.resampler.fit(dataset.X, dataset.y)
|
|
45
|
+
return self
|
|
46
|
+
|
|
47
|
+
def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
48
|
+
"""Apply Random Over Sampling
|
|
49
|
+
|
|
50
|
+
:param pd.DataFrame X: Features to resample
|
|
51
|
+
:param pd.DataFrame y: Labels to resample
|
|
52
|
+
:return: Resampled X and y
|
|
53
|
+
"""
|
|
54
|
+
return self.resampler.fit_resample(X, y)
|
|
55
|
+
|
|
56
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
57
|
+
return 1
|
|
58
|
+
|
|
59
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
60
|
+
return dataset.type_of_target in ['binary', 'multiclass']
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""[STEP] Random Under Sampling"""
|
|
2
|
+
import inspect
|
|
3
|
+
import textwrap
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
import pandas as pd
|
|
8
|
+
from imblearn.under_sampling import RandomUnderSampler
|
|
9
|
+
|
|
10
|
+
from ...actionable import Actionable
|
|
11
|
+
from ...candidate import Candidate
|
|
12
|
+
from ...data_type import DataType
|
|
13
|
+
from ...dataset import Dataset
|
|
14
|
+
from ...decorators.all import is_step
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@is_step('imbalance')
|
|
18
|
+
class ActRandomUnderSampler(Actionable):
|
|
19
|
+
"""[STEP] Random Under Sampling"""
|
|
20
|
+
|
|
21
|
+
name: str = "Random Under Sampling"
|
|
22
|
+
_description: str = textwrap.dedent('''\
|
|
23
|
+
RandomUnderSampler balances data by randomly removing
|
|
24
|
+
samples from the majority class.''')
|
|
25
|
+
_description_long: str = textwrap.dedent('''\
|
|
26
|
+
RandomUnderSampler reduces class imbalance by randomly dropping
|
|
27
|
+
samples from the majority class until class sizes are closer.
|
|
28
|
+
It is simple, fast, and keeps the original minority samples intact.''')
|
|
29
|
+
_usage: str = "Use when fast random majority downsampling is acceptable; compare ActNearMiss for guided removal. Applicable to binary or multiclass tabular data with a clear majority class. Avoid when minority data is scarce or information loss hurts; consider ActRandomOverSampling."
|
|
30
|
+
refs: list[dict[str, Any]] = [
|
|
31
|
+
{
|
|
32
|
+
'year': 2012,
|
|
33
|
+
'name': 'Training and assessing classification rules with imbalanced data',
|
|
34
|
+
'authors': [
|
|
35
|
+
'Giovanna Menardi',
|
|
36
|
+
'Nicola Torelli'
|
|
37
|
+
],
|
|
38
|
+
'doi': 'https://doi.org/10.1007/s10618-012-0295-5',
|
|
39
|
+
'publisher': 'Data Mining and Knowledge Discovery Vol.28 page 92--122'
|
|
40
|
+
}
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
def __init__(self):
|
|
44
|
+
self.configuration = {
|
|
45
|
+
'sampling_strategy': {
|
|
46
|
+
'description': 'Sampling strategy to reduce the majority class.',
|
|
47
|
+
'default': 'auto',
|
|
48
|
+
'categorical': ['auto', 'majority']
|
|
49
|
+
},
|
|
50
|
+
'random_state': {
|
|
51
|
+
'description': 'Random seed used for reproducibility.',
|
|
52
|
+
'default': 42
|
|
53
|
+
},
|
|
54
|
+
'replacement': {
|
|
55
|
+
'description': 'Sample with replacement when under-sampling.',
|
|
56
|
+
'default': False,
|
|
57
|
+
'categorical': [True, False]
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
self.resampler: RandomUnderSampler | None = None
|
|
61
|
+
self.feature_columns: list[str] = []
|
|
62
|
+
self.imbalance_ratio: float = 0.0
|
|
63
|
+
|
|
64
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
65
|
+
self.resampler = None
|
|
66
|
+
self.feature_columns = self.__candidate_columns(dataset)
|
|
67
|
+
self.imbalance_ratio = 0.0
|
|
68
|
+
|
|
69
|
+
if dataset.X.empty or not self.feature_columns:
|
|
70
|
+
return self
|
|
71
|
+
if dataset.type_of_target not in ['binary', 'multiclass']:
|
|
72
|
+
return self
|
|
73
|
+
if dataset.y is None or len(dataset.y) == 0:
|
|
74
|
+
return self
|
|
75
|
+
|
|
76
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
77
|
+
if len(counts) < 2:
|
|
78
|
+
return self
|
|
79
|
+
|
|
80
|
+
max_count = int(counts.max())
|
|
81
|
+
min_count = int(counts.min())
|
|
82
|
+
if max_count <= min_count:
|
|
83
|
+
return self
|
|
84
|
+
|
|
85
|
+
self.imbalance_ratio = 1.0 - (min_count / max_count)
|
|
86
|
+
|
|
87
|
+
params = self.passthrough_parameters()
|
|
88
|
+
sig_params = inspect.signature(RandomUnderSampler).parameters
|
|
89
|
+
params = {key: value for key, value in params.items() if key in sig_params}
|
|
90
|
+
|
|
91
|
+
self.resampler = RandomUnderSampler(**params)
|
|
92
|
+
self.resampler.fit(dataset.X, dataset.y)
|
|
93
|
+
return self
|
|
94
|
+
|
|
95
|
+
def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
96
|
+
"""Apply Random Under Sampling.
|
|
97
|
+
|
|
98
|
+
:param pd.DataFrame X: Features to resample
|
|
99
|
+
:param pd.DataFrame y: Labels to resample
|
|
100
|
+
:return: Resampled X and y
|
|
101
|
+
"""
|
|
102
|
+
if self.resampler is None:
|
|
103
|
+
return X, y
|
|
104
|
+
return self.resampler.fit_resample(X, y)
|
|
105
|
+
|
|
106
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
107
|
+
if candidate is None or candidate.dataset.y is None:
|
|
108
|
+
return 0.0
|
|
109
|
+
y = candidate.dataset.y
|
|
110
|
+
if len(y) == 0:
|
|
111
|
+
return 0.0
|
|
112
|
+
_, counts = np.unique(y, return_counts=True)
|
|
113
|
+
if len(counts) < 2:
|
|
114
|
+
return 0.0
|
|
115
|
+
imbalance = 1.0 - (counts.min() / counts.max())
|
|
116
|
+
return float(min(1.0, max(0.0, imbalance)))
|
|
117
|
+
|
|
118
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
119
|
+
if dataset.type_of_target not in ['binary', 'multiclass']:
|
|
120
|
+
return False
|
|
121
|
+
if dataset.X.empty or dataset.y is None or len(dataset.y) == 0:
|
|
122
|
+
return False
|
|
123
|
+
if not self.__candidate_columns(dataset):
|
|
124
|
+
return False
|
|
125
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
126
|
+
if len(counts) < 2:
|
|
127
|
+
return False
|
|
128
|
+
return counts.max() > counts.min()
|
|
129
|
+
|
|
130
|
+
def __candidate_columns(self, dataset: Dataset) -> list[str]:
|
|
131
|
+
columns = dataset.get_columns_names_by_type(list(DataType))
|
|
132
|
+
if len(columns) != dataset.X.shape[1]:
|
|
133
|
+
missing = [column for column in dataset.X.columns if column not in columns]
|
|
134
|
+
columns.extend(missing)
|
|
135
|
+
return columns
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""[STEP] SMOTE"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
from imblearn.over_sampling import SMOTE, SMOTENC
|
|
8
|
+
|
|
9
|
+
from ...actionable import Actionable
|
|
10
|
+
from ...candidate import Candidate
|
|
11
|
+
from ...data_type import DataType
|
|
12
|
+
from ...dataset import Dataset
|
|
13
|
+
from ...decorators.all import is_step
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@is_step('imbalance')
|
|
17
|
+
class ActSMOTE(Actionable):
|
|
18
|
+
"""[STEP] SMOTE"""
|
|
19
|
+
|
|
20
|
+
name: str = "SMOTE"
|
|
21
|
+
_usage: str = "Use when imbalanced classification needs synthetic minority samples; compare ActADASYN for adaptive oversampling. Applicable to binary or multiclass with numeric features (categoricals via SMOTENC). Avoid when non-classification, text/date-only, or minority count <= 1."
|
|
22
|
+
_description: str = textwrap.dedent('''\
|
|
23
|
+
SMOTE balances the minority class by creating synthetic samples
|
|
24
|
+
through interpolation.''')
|
|
25
|
+
_description_long: str = textwrap.dedent('''\
|
|
26
|
+
SMOTE (Synthetic Minority Over-sampling Technique) addresses class
|
|
27
|
+
imbalance by generating new minority samples between nearest neighbors.
|
|
28
|
+
This keeps the original data while reducing bias toward the majority class.''')
|
|
29
|
+
refs: list[dict[str, Any]] = [
|
|
30
|
+
{
|
|
31
|
+
'year': 2002,
|
|
32
|
+
'name': 'SMOTE: Synthetic Minority Over-sampling Technique',
|
|
33
|
+
'authors': [
|
|
34
|
+
'Nitesh V. Chawla',
|
|
35
|
+
'Kevin W. Bowyer',
|
|
36
|
+
'Lawrence O. Hall',
|
|
37
|
+
'W. Philip Kegelmeyer'
|
|
38
|
+
],
|
|
39
|
+
'doi': 'https://doi.org/10.1613/jair.953',
|
|
40
|
+
'publisher': 'Journal of Artificial Intelligence Research Vol.16 page 321--357'
|
|
41
|
+
}
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
def __init__(self):
|
|
45
|
+
self.configuration = {
|
|
46
|
+
'sampling_strategy': {
|
|
47
|
+
'description': 'Sampling strategy to balance classes.',
|
|
48
|
+
'default': 'minority',
|
|
49
|
+
'categorical': ['minority', 'auto']
|
|
50
|
+
},
|
|
51
|
+
'k_neighbors': {
|
|
52
|
+
'description': 'Number of nearest neighbors used to create synthetic samples.',
|
|
53
|
+
'default': 5,
|
|
54
|
+
'range': [1, 20]
|
|
55
|
+
},
|
|
56
|
+
'random_state': {
|
|
57
|
+
'description': 'Random seed used for reproducibility.',
|
|
58
|
+
'default': 42
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
self.resampler: SMOTE | SMOTENC | None = None
|
|
62
|
+
self.categorical_columns: list[str] = []
|
|
63
|
+
self.categorical_indices: list[int] = []
|
|
64
|
+
self.numeric_columns: list[str] = []
|
|
65
|
+
self._effective_k_neighbors: int | None = None
|
|
66
|
+
|
|
67
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
68
|
+
self.resampler = None
|
|
69
|
+
self.categorical_columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
70
|
+
self.numeric_columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
71
|
+
self.categorical_indices = []
|
|
72
|
+
self._effective_k_neighbors = None
|
|
73
|
+
|
|
74
|
+
if dataset.X.empty or dataset.type_of_target not in ['binary', 'multiclass']:
|
|
75
|
+
return self
|
|
76
|
+
|
|
77
|
+
if dataset.y is None or len(dataset.y) == 0:
|
|
78
|
+
return self
|
|
79
|
+
|
|
80
|
+
unsupported = dataset.get_columns_names_by_type(
|
|
81
|
+
[DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
|
|
82
|
+
)
|
|
83
|
+
if unsupported:
|
|
84
|
+
return self
|
|
85
|
+
|
|
86
|
+
if not self.numeric_columns:
|
|
87
|
+
return self
|
|
88
|
+
|
|
89
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
90
|
+
if len(counts) < 2:
|
|
91
|
+
return self
|
|
92
|
+
|
|
93
|
+
min_count = int(counts.min())
|
|
94
|
+
if min_count <= 1:
|
|
95
|
+
return self
|
|
96
|
+
|
|
97
|
+
max_k = min_count - 1
|
|
98
|
+
k_neighbors = min(int(self.get_config('k_neighbors')), max_k)
|
|
99
|
+
k_neighbors = max(1, k_neighbors)
|
|
100
|
+
self._effective_k_neighbors = k_neighbors
|
|
101
|
+
|
|
102
|
+
params = self.passthrough_parameters()
|
|
103
|
+
params['k_neighbors'] = k_neighbors
|
|
104
|
+
|
|
105
|
+
if self.categorical_columns:
|
|
106
|
+
self.categorical_indices = [
|
|
107
|
+
dataset.X.columns.get_loc(column)
|
|
108
|
+
for column in self.categorical_columns
|
|
109
|
+
if column in dataset.X.columns
|
|
110
|
+
]
|
|
111
|
+
if self.categorical_indices:
|
|
112
|
+
self.resampler = SMOTENC(
|
|
113
|
+
categorical_features=self.categorical_indices,
|
|
114
|
+
**params
|
|
115
|
+
)
|
|
116
|
+
else:
|
|
117
|
+
self.resampler = SMOTE(**params)
|
|
118
|
+
else:
|
|
119
|
+
self.resampler = SMOTE(**params)
|
|
120
|
+
|
|
121
|
+
self.resampler.fit(dataset.X, dataset.y)
|
|
122
|
+
return self
|
|
123
|
+
|
|
124
|
+
def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
125
|
+
"""Apply SMOTE.
|
|
126
|
+
|
|
127
|
+
:param pd.DataFrame X: Features to resample
|
|
128
|
+
:param pd.DataFrame y: Labels to resample
|
|
129
|
+
:return: Resampled X and y
|
|
130
|
+
"""
|
|
131
|
+
if self.resampler is None:
|
|
132
|
+
return X, y
|
|
133
|
+
return self.resampler.fit_resample(X, y)
|
|
134
|
+
|
|
135
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
136
|
+
if candidate is None or candidate.dataset.y is None:
|
|
137
|
+
return 0.0
|
|
138
|
+
y = candidate.dataset.y
|
|
139
|
+
if len(y) == 0:
|
|
140
|
+
return 0.0
|
|
141
|
+
_, counts = np.unique(y, return_counts=True)
|
|
142
|
+
if len(counts) < 2:
|
|
143
|
+
return 0.0
|
|
144
|
+
imbalance = 1.0 - (counts.min() / counts.max())
|
|
145
|
+
return float(min(1.0, max(0.0, imbalance)))
|
|
146
|
+
|
|
147
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
148
|
+
if dataset.type_of_target not in ['binary', 'multiclass']:
|
|
149
|
+
return False
|
|
150
|
+
if dataset.X.empty or dataset.y is None or len(dataset.y) == 0:
|
|
151
|
+
return False
|
|
152
|
+
if not dataset.get_columns_names_by_type(DataType.NUMERIC):
|
|
153
|
+
return False
|
|
154
|
+
unsupported = dataset.get_columns_names_by_type(
|
|
155
|
+
[DataType.TEXT, DataType.SHORT_TEXT, DataType.DATE]
|
|
156
|
+
)
|
|
157
|
+
if unsupported:
|
|
158
|
+
return False
|
|
159
|
+
_, counts = np.unique(dataset.y, return_counts=True)
|
|
160
|
+
if len(counts) < 2:
|
|
161
|
+
return False
|
|
162
|
+
return counts.min() > 1
|