PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""[STEP] Vectorize large text with HashingVectorizer."""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
from sklearn.feature_extraction.text import HashingVectorizer
|
|
7
|
+
|
|
8
|
+
from ...actionable import Actionable
|
|
9
|
+
from ...candidate import Candidate
|
|
10
|
+
from ...data_type import DataType
|
|
11
|
+
from ...dataset import Dataset
|
|
12
|
+
from ...decorators.all import is_step
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@is_step('cleaning')
|
|
16
|
+
class ActHashingVectorizer(Actionable):
|
|
17
|
+
"""[STEP] Vectorize large text with HashingVectorizer."""
|
|
18
|
+
|
|
19
|
+
name: str = 'Hashing Vectorizer'
|
|
20
|
+
_description: str = textwrap.dedent('''\
|
|
21
|
+
Vectorize text columns with the hashing trick for large vocabularies.''')
|
|
22
|
+
_description_long: str = textwrap.dedent('''\
|
|
23
|
+
Convert long text columns into fixed-size hashed feature vectors without
|
|
24
|
+
building an explicit vocabulary. This keeps memory usage bounded even for
|
|
25
|
+
very large vocabularies, at the cost of possible hash collisions.''')
|
|
26
|
+
_usage: str = "Use when text columns have huge vocabularies and you need fixed-size features, especially vs ActCountVectorizer for memory bounds. Applicable to free-text columns that you want numeric n-grams from. Avoid when token interpretability or collision-free features are required."
|
|
27
|
+
|
|
28
|
+
def __init__(self) -> None:
|
|
29
|
+
self.columns: list[str] = []
|
|
30
|
+
self.vectorizer: HashingVectorizer | None = None
|
|
31
|
+
self.configuration = {
|
|
32
|
+
'n_features': {
|
|
33
|
+
'description': 'Number of hash bins (power of two recommended).',
|
|
34
|
+
'default': 4096
|
|
35
|
+
},
|
|
36
|
+
'ngram_min': {
|
|
37
|
+
'description': 'Minimum n-gram size to include.',
|
|
38
|
+
'default': 1
|
|
39
|
+
},
|
|
40
|
+
'ngram_max': {
|
|
41
|
+
'description': 'Maximum n-gram size to include.',
|
|
42
|
+
'default': 2
|
|
43
|
+
},
|
|
44
|
+
'alternate_sign': {
|
|
45
|
+
'description': 'Use alternating signs to reduce hash collisions.',
|
|
46
|
+
'default': False
|
|
47
|
+
},
|
|
48
|
+
'binary': {
|
|
49
|
+
'description': 'If True, store binary occurrences instead of counts.',
|
|
50
|
+
'default': False
|
|
51
|
+
},
|
|
52
|
+
'norm': {
|
|
53
|
+
'description': 'Vector normalization ("l1", "l2", or None).',
|
|
54
|
+
'default': 'l2'
|
|
55
|
+
},
|
|
56
|
+
'lowercase': {
|
|
57
|
+
'description': 'Lowercase text before hashing.',
|
|
58
|
+
'default': True
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
63
|
+
self.columns = []
|
|
64
|
+
self.vectorizer = None
|
|
65
|
+
self.explanations = []
|
|
66
|
+
|
|
67
|
+
columns = dataset.get_columns_names_by_type([DataType.TEXT])
|
|
68
|
+
if not columns or dataset.X.empty:
|
|
69
|
+
return self
|
|
70
|
+
|
|
71
|
+
params = self._build_vectorizer_params()
|
|
72
|
+
self.vectorizer = HashingVectorizer(**params)
|
|
73
|
+
self.columns = [column for column in columns if column in dataset.X.columns]
|
|
74
|
+
|
|
75
|
+
n_features = params['n_features']
|
|
76
|
+
for column in self.columns:
|
|
77
|
+
self.explanations.append(
|
|
78
|
+
f'Encoded text column **`{column}`** into **{n_features}** hashed features.'
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
if not self.columns:
|
|
82
|
+
self.explanations.append(
|
|
83
|
+
'Hashing vectorizer skipped: no usable text columns.'
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
return self
|
|
87
|
+
|
|
88
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
89
|
+
"""Apply HashingVectorizer to text columns.
|
|
90
|
+
|
|
91
|
+
:param pd.DataFrame X: DataFrame to transform.
|
|
92
|
+
:return: Transformed dataset.
|
|
93
|
+
"""
|
|
94
|
+
if not self.columns or self.vectorizer is None:
|
|
95
|
+
return X
|
|
96
|
+
|
|
97
|
+
X = X.reset_index(drop=True)
|
|
98
|
+
|
|
99
|
+
for name in self.columns:
|
|
100
|
+
if name not in X.columns:
|
|
101
|
+
continue
|
|
102
|
+
|
|
103
|
+
values = X[name].fillna('').astype(str)
|
|
104
|
+
transformed = self.vectorizer.transform(values)
|
|
105
|
+
n_features = transformed.shape[1]
|
|
106
|
+
new_names = [f"{name}_hash_{i}" for i in range(n_features)]
|
|
107
|
+
vector_df = pd.DataFrame(transformed.toarray(), columns=new_names)
|
|
108
|
+
|
|
109
|
+
X = pd.concat([X, vector_df], axis=1).drop([name], axis=1)
|
|
110
|
+
|
|
111
|
+
return X
|
|
112
|
+
|
|
113
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
114
|
+
if dataset.X.empty:
|
|
115
|
+
return False
|
|
116
|
+
return bool(dataset.get_columns_names_by_type([DataType.TEXT]))
|
|
117
|
+
|
|
118
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
119
|
+
if candidate is None or candidate.dataset.X.empty:
|
|
120
|
+
return 0.0
|
|
121
|
+
|
|
122
|
+
columns = candidate.dataset.get_columns_names_by_type([DataType.TEXT])
|
|
123
|
+
if not columns:
|
|
124
|
+
return 0.0
|
|
125
|
+
|
|
126
|
+
total_columns = candidate.dataset.X.shape[1] or 1
|
|
127
|
+
return min(1.0, len(columns) / total_columns)
|
|
128
|
+
|
|
129
|
+
def _build_vectorizer_params(self) -> dict[str, Any]:
|
|
130
|
+
n_features = self._coerce_positive_int(self.get_config('n_features'), 4096)
|
|
131
|
+
ngram_min = self._coerce_int(self.get_config('ngram_min'), 1)
|
|
132
|
+
ngram_max = self._coerce_int(self.get_config('ngram_max'), max(ngram_min, 1))
|
|
133
|
+
ngram_min = max(1, ngram_min)
|
|
134
|
+
ngram_max = max(ngram_min, ngram_max)
|
|
135
|
+
|
|
136
|
+
return {
|
|
137
|
+
'n_features': n_features,
|
|
138
|
+
'ngram_range': (ngram_min, ngram_max),
|
|
139
|
+
'alternate_sign': self._coerce_bool(self.get_config('alternate_sign'), False),
|
|
140
|
+
'binary': self._coerce_bool(self.get_config('binary'), False),
|
|
141
|
+
'norm': self._coerce_norm(self.get_config('norm')),
|
|
142
|
+
'lowercase': self._coerce_bool(self.get_config('lowercase'), True)
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
@staticmethod
|
|
146
|
+
def _coerce_int(value: Any, default: int) -> int:
|
|
147
|
+
try:
|
|
148
|
+
return int(value)
|
|
149
|
+
except (TypeError, ValueError):
|
|
150
|
+
return default
|
|
151
|
+
|
|
152
|
+
@staticmethod
|
|
153
|
+
def _coerce_positive_int(value: Any, default: int) -> int:
|
|
154
|
+
try:
|
|
155
|
+
numeric = int(value)
|
|
156
|
+
except (TypeError, ValueError):
|
|
157
|
+
return default
|
|
158
|
+
if numeric <= 0:
|
|
159
|
+
return default
|
|
160
|
+
return numeric
|
|
161
|
+
|
|
162
|
+
@staticmethod
|
|
163
|
+
def _coerce_bool(value: Any, default: bool) -> bool:
|
|
164
|
+
if isinstance(value, bool):
|
|
165
|
+
return value
|
|
166
|
+
if isinstance(value, str):
|
|
167
|
+
normalized = value.strip().lower()
|
|
168
|
+
if normalized in {'true', '1', 'yes', 'y'}:
|
|
169
|
+
return True
|
|
170
|
+
if normalized in {'false', '0', 'no', 'n'}:
|
|
171
|
+
return False
|
|
172
|
+
if value is None:
|
|
173
|
+
return default
|
|
174
|
+
return bool(value)
|
|
175
|
+
|
|
176
|
+
@staticmethod
|
|
177
|
+
def _coerce_norm(value: Any) -> str | None:
|
|
178
|
+
if value is None:
|
|
179
|
+
return None
|
|
180
|
+
if isinstance(value, str):
|
|
181
|
+
normalized = value.strip().lower()
|
|
182
|
+
if normalized in {'none', ''}:
|
|
183
|
+
return None
|
|
184
|
+
if normalized in {'l1', 'l2'}:
|
|
185
|
+
return normalized
|
|
186
|
+
return 'l2'
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
"""[STEP] KNN imputer for numeric columns."""
|
|
2
|
+
import textwrap
|
|
3
|
+
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from sklearn.impute import KNNImputer
|
|
6
|
+
|
|
7
|
+
from ...actionable import Actionable
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...data_type import DataType
|
|
10
|
+
from ...dataset import Dataset
|
|
11
|
+
from ...decorators.all import is_step
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@is_step('cleaning')
|
|
15
|
+
class ActKNNImputer(Actionable):
|
|
16
|
+
"""[STEP] Impute missing numeric values using KNN."""
|
|
17
|
+
|
|
18
|
+
name: str = 'Impute missing values (KNN)'
|
|
19
|
+
_usage: str = 'Use when numeric features have gaps and you want to preserve rows vs ActDropNumericalColumn. Applicable to numeric columns with enough non-missing rows; use ActCategoricalImputer for categoricals. Avoid when missingness is extreme, dataset is tiny, or you plan to drop the column.'
|
|
20
|
+
_description: str = textwrap.dedent('''\
|
|
21
|
+
Impute missing numeric values using a k-nearest neighbors strategy.''')
|
|
22
|
+
_description_long: str = textwrap.dedent('''\
|
|
23
|
+
Uses scikit-learn KNNImputer to fill missing values in numeric columns
|
|
24
|
+
by averaging the k nearest neighbors in feature space. Non-numeric
|
|
25
|
+
columns are left untouched.''')
|
|
26
|
+
|
|
27
|
+
def __init__(self):
|
|
28
|
+
self.columns: list[str] = []
|
|
29
|
+
self.knn_columns: list[str] = []
|
|
30
|
+
self.imputer: KNNImputer | None = None
|
|
31
|
+
self._all_nan_cols: list[str] = []
|
|
32
|
+
self._fallback_values: dict[str, float] = {}
|
|
33
|
+
self._nan_stats: dict[str, tuple[int, int, float]] = {}
|
|
34
|
+
|
|
35
|
+
self.configuration = {
|
|
36
|
+
'n_neighbors': {
|
|
37
|
+
'description': 'Number of neighbors used for imputing missing values.',
|
|
38
|
+
'default': 5,
|
|
39
|
+
'range': [1, 50],
|
|
40
|
+
'passthrough': False
|
|
41
|
+
},
|
|
42
|
+
'weights': {
|
|
43
|
+
'description': 'Weight function used in prediction.',
|
|
44
|
+
'default': 'uniform',
|
|
45
|
+
'categorical': ['uniform', 'distance']
|
|
46
|
+
},
|
|
47
|
+
'metric': {
|
|
48
|
+
'description': 'Distance metric to use for missing-aware KNN.',
|
|
49
|
+
'default': 'nan_euclidean',
|
|
50
|
+
'categorical': ['nan_euclidean']
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
55
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
56
|
+
self.knn_columns = []
|
|
57
|
+
self.imputer = None
|
|
58
|
+
self._all_nan_cols = []
|
|
59
|
+
self._fallback_values = {}
|
|
60
|
+
self._nan_stats = {}
|
|
61
|
+
self.explanations = []
|
|
62
|
+
|
|
63
|
+
if not self.columns:
|
|
64
|
+
return self
|
|
65
|
+
|
|
66
|
+
X_num = dataset.X[self.columns]
|
|
67
|
+
total_rows = len(X_num)
|
|
68
|
+
if total_rows == 0:
|
|
69
|
+
self.explanations = ["Skipped KNN imputation: dataset has 0 rows."]
|
|
70
|
+
return self
|
|
71
|
+
|
|
72
|
+
missing_counts = X_num.isna().sum()
|
|
73
|
+
means = X_num.mean()
|
|
74
|
+
for column in self.columns:
|
|
75
|
+
missing = int(missing_counts[column])
|
|
76
|
+
pct = (missing / total_rows * 100.0) if total_rows else 0.0
|
|
77
|
+
self._nan_stats[column] = (missing, total_rows, pct)
|
|
78
|
+
|
|
79
|
+
mean_value = means[column]
|
|
80
|
+
if pd.isna(mean_value):
|
|
81
|
+
mean_value = 0.0
|
|
82
|
+
self._fallback_values[column] = float(mean_value)
|
|
83
|
+
|
|
84
|
+
self._all_nan_cols = [
|
|
85
|
+
column for column in self.columns if missing_counts[column] == total_rows
|
|
86
|
+
]
|
|
87
|
+
self.knn_columns = [
|
|
88
|
+
column for column in self.columns if column not in self._all_nan_cols
|
|
89
|
+
]
|
|
90
|
+
|
|
91
|
+
if self.knn_columns and total_rows >= 2:
|
|
92
|
+
n_neighbors = min(self.get_config('n_neighbors'), total_rows - 1)
|
|
93
|
+
n_neighbors = max(1, int(n_neighbors))
|
|
94
|
+
params = self.passthrough_parameters()
|
|
95
|
+
params['n_neighbors'] = n_neighbors
|
|
96
|
+
self.imputer = KNNImputer(**params)
|
|
97
|
+
self.imputer.fit(X_num[self.knn_columns])
|
|
98
|
+
|
|
99
|
+
if self.imputer is not None:
|
|
100
|
+
self.explanations = [
|
|
101
|
+
f"Imputed missing values of column **`{c}`** using **KNN** "
|
|
102
|
+
f"(**{n}** / **{t}**; **{pct:.2f}%** missing in train data)."
|
|
103
|
+
for c, (n, t, pct) in self._nan_stats.items()
|
|
104
|
+
if n > 0 and c in self.knn_columns
|
|
105
|
+
]
|
|
106
|
+
if self._all_nan_cols:
|
|
107
|
+
self.explanations.append(
|
|
108
|
+
"Filled all-NaN numeric columns with 0.0: " +
|
|
109
|
+
", ".join(f"`{c}`" for c in self._all_nan_cols) +
|
|
110
|
+
"."
|
|
111
|
+
)
|
|
112
|
+
if self.imputer is None and not self.explanations:
|
|
113
|
+
if any(n > 0 for n, _, _ in self._nan_stats.values()):
|
|
114
|
+
self.explanations.append(
|
|
115
|
+
"Skipped KNN imputation; filled numeric columns with their mean "
|
|
116
|
+
"(0.0 when undefined)."
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
return self
|
|
120
|
+
|
|
121
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
122
|
+
if not self.columns:
|
|
123
|
+
return X
|
|
124
|
+
|
|
125
|
+
if self.imputer is not None and self.knn_columns:
|
|
126
|
+
if all(column in X.columns for column in self.knn_columns):
|
|
127
|
+
X_knn = X[self.knn_columns].copy()
|
|
128
|
+
X_knn = X_knn.apply(pd.to_numeric, errors='coerce')
|
|
129
|
+
imputed = self.imputer.transform(X_knn)
|
|
130
|
+
X.loc[:, self.knn_columns] = imputed
|
|
131
|
+
|
|
132
|
+
for column, fill_value in self._fallback_values.items():
|
|
133
|
+
if column in X.columns:
|
|
134
|
+
X[column] = X[column].fillna(fill_value).infer_objects(copy=False)
|
|
135
|
+
|
|
136
|
+
return X
|
|
137
|
+
|
|
138
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
139
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
140
|
+
if not columns or dataset.X.empty:
|
|
141
|
+
return False
|
|
142
|
+
return bool(dataset.X[columns].isna().any().any())
|
|
143
|
+
|
|
144
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
145
|
+
if candidate is None or candidate.dataset.X.empty:
|
|
146
|
+
return 0.0
|
|
147
|
+
columns = candidate.dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
148
|
+
if not columns:
|
|
149
|
+
return 0.0
|
|
150
|
+
missing = candidate.dataset.X[columns].isna().sum().sum()
|
|
151
|
+
total = candidate.dataset.X[columns].size or 1
|
|
152
|
+
return min(1.0, missing / total)
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""[STEP] Fill missing values with mean"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import numpy as np
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
from ...data_type import DataType
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('cleaning', 'baseline_cleaning')
|
|
13
|
+
class ActMeanColumn(Actionable):
|
|
14
|
+
"""[STEP] Fill missing values with the mean."""
|
|
15
|
+
|
|
16
|
+
name: str = 'Fill missing values'
|
|
17
|
+
_usage: str = "Use when numeric columns have some missing values and you want a quick baseline imputation over ActDropNumericalColumn. Applicable to numerical data with moderate missingness. Avoid when missingness is high or systematic, or when ActDropNumericalColumn is safer."
|
|
18
|
+
_description: str = textwrap.dedent('''\
|
|
19
|
+
Fill missing values with the mean of non-missing values
|
|
20
|
+
when the proportion of empty rows is lower than {empty_threshold:.0%}.''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
Fill a column missings values with the mean of the columns
|
|
23
|
+
when the proportion of empty rows is lower than {empty_threshold}.
|
|
24
|
+
Work only for numerical columns.''')
|
|
25
|
+
can_be_disabled: bool = False
|
|
26
|
+
|
|
27
|
+
def __init__(self):
|
|
28
|
+
self.columns: list[str] = None
|
|
29
|
+
self.configuration:dict = {
|
|
30
|
+
'empty_threshold': {
|
|
31
|
+
'description': textwrap.dedent('''\
|
|
32
|
+
Column with less or equal proportion of empty row will be
|
|
33
|
+
fill with mean value. 1 will always fill void values'''),
|
|
34
|
+
'default': 1 # TODO Review when adding new kind of imputer
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
39
|
+
self.columns = []
|
|
40
|
+
explain = []
|
|
41
|
+
|
|
42
|
+
for column in dataset.get_columns_names_by_type(DataType.NUMERIC):
|
|
43
|
+
values = dataset.X[column]
|
|
44
|
+
nan_values_count = values.isnull().sum()
|
|
45
|
+
|
|
46
|
+
mean = values.mean()
|
|
47
|
+
if np.isnan(mean):
|
|
48
|
+
mean = 0
|
|
49
|
+
|
|
50
|
+
self.columns.append((column, mean))
|
|
51
|
+
explain.append((
|
|
52
|
+
nan_values_count,
|
|
53
|
+
len(values),
|
|
54
|
+
nan_values_count / len(values) * 100,
|
|
55
|
+
))
|
|
56
|
+
|
|
57
|
+
self.explanations = [
|
|
58
|
+
f"""Filled missing values of column **`{c}`** with **{mean:.2f}**
|
|
59
|
+
(**{v[0]}** out of **{v[1]}** values (**{v[2]:.2f}**%)
|
|
60
|
+
were missing in train data)."""
|
|
61
|
+
for (c, mean), v in zip(self.columns, explain)
|
|
62
|
+
if v[0] > 0 # hide processings that affected no values
|
|
63
|
+
]
|
|
64
|
+
|
|
65
|
+
return self
|
|
66
|
+
|
|
67
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
68
|
+
"""Fill NA values with the mean.
|
|
69
|
+
|
|
70
|
+
:param pd.DataFrame x: DataFrame to transform.
|
|
71
|
+
:return: Transformed dataset.
|
|
72
|
+
"""
|
|
73
|
+
for name, mean in self.columns:
|
|
74
|
+
X[name] = X[name].fillna(mean).infer_objects(copy=False)
|
|
75
|
+
|
|
76
|
+
return X
|
|
77
|
+
|
|
78
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
79
|
+
return 1 - (candidate.dataset.X.isnull().sum().min() / len(candidate.dataset.X))
|