PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,337 @@
|
|
|
1
|
+
"""[STEP] Drop or hash high-cardinality categorical columns."""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
from ...actionable import Actionable
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...data_type import DataType
|
|
10
|
+
from ...dataset import Dataset
|
|
11
|
+
from ...decorators.all import is_step
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@is_step('cleaning')
|
|
15
|
+
class ActDropHighCardinalityCategorical(Actionable):
|
|
16
|
+
"""[STEP] Drop or hash high-cardinality categorical columns."""
|
|
17
|
+
|
|
18
|
+
name: str = 'Handle high-cardinality categorical columns'
|
|
19
|
+
_description: str = textwrap.dedent('''\
|
|
20
|
+
Drop or hash categorical columns whose cardinality meets or exceeds {max_unique}
|
|
21
|
+
or {unique_ratio_threshold:.0%} of non-missing rows.''')
|
|
22
|
+
_description_long: str = textwrap.dedent('''\
|
|
23
|
+
High-cardinality categorical columns can create large sparse encodings.
|
|
24
|
+
Columns with too many distinct values are either removed or replaced with
|
|
25
|
+
hashed integer codes (modulo {hash_bins}) depending on strategy {strategy}.
|
|
26
|
+
Columns with fewer than {min_non_null} non-missing rows are ignored.''')
|
|
27
|
+
_usage: str = "Use when categorical columns are extremely high-cardinality and you want drop/hash instead of ActCountVectorizer. Applicable to categorical features with high unique-to-row ratios. Avoid when categories are low-cardinality or predictive, or ActCategoricalImputer is enough."
|
|
28
|
+
|
|
29
|
+
def __init__(self) -> None:
|
|
30
|
+
self.configuration = {
|
|
31
|
+
'strategy': {
|
|
32
|
+
'description': 'How to handle high-cardinality columns.',
|
|
33
|
+
'default': 'drop',
|
|
34
|
+
'categorical': ['drop', 'hash']
|
|
35
|
+
},
|
|
36
|
+
'max_unique': {
|
|
37
|
+
'description': 'Maximum number of distinct values before handling.',
|
|
38
|
+
'default': 50
|
|
39
|
+
},
|
|
40
|
+
'unique_ratio_threshold': {
|
|
41
|
+
'description': 'Minimum unique/non-missing ratio to mark as high-cardinality.',
|
|
42
|
+
'default': 0.5,
|
|
43
|
+
'range': [0.0, 1.0]
|
|
44
|
+
},
|
|
45
|
+
'min_non_null': {
|
|
46
|
+
'description': 'Minimum number of non-missing rows to evaluate.',
|
|
47
|
+
'default': 10
|
|
48
|
+
},
|
|
49
|
+
'hash_bins': {
|
|
50
|
+
'description': 'Number of hash bins when strategy="hash".',
|
|
51
|
+
'default': 64
|
|
52
|
+
},
|
|
53
|
+
'hash_key': {
|
|
54
|
+
'description': 'Stable hash key used for hashing (16 chars recommended).',
|
|
55
|
+
'default': 'iaml_hashing_key'
|
|
56
|
+
},
|
|
57
|
+
'missing_value': {
|
|
58
|
+
'description': 'Numeric value used for missing categories when hashing.',
|
|
59
|
+
'default': -1
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
self.columns_to_drop: list[str] = []
|
|
63
|
+
self.columns_to_hash: list[str] = []
|
|
64
|
+
self.cardinality_stats: dict[str, dict[str, float]] = {}
|
|
65
|
+
self.strategy: str = 'drop'
|
|
66
|
+
self.hash_bins: int = 0
|
|
67
|
+
self.hash_key: str | bytes | None = None
|
|
68
|
+
self.missing_value: int = -1
|
|
69
|
+
|
|
70
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
71
|
+
self.columns_to_drop = []
|
|
72
|
+
self.columns_to_hash = []
|
|
73
|
+
self.cardinality_stats = {}
|
|
74
|
+
self.explanations = []
|
|
75
|
+
|
|
76
|
+
if dataset.X.empty:
|
|
77
|
+
return self
|
|
78
|
+
|
|
79
|
+
columns = self._select_columns(dataset)
|
|
80
|
+
if not columns:
|
|
81
|
+
return self
|
|
82
|
+
|
|
83
|
+
self.strategy = self._coerce_strategy(self.get_config('strategy'))
|
|
84
|
+
max_unique = self._coerce_int(self.get_config('max_unique'), 0)
|
|
85
|
+
ratio_threshold = self._coerce_ratio(self.get_config('unique_ratio_threshold'), 0.0)
|
|
86
|
+
min_non_null = self._coerce_int(self.get_config('min_non_null'), 0)
|
|
87
|
+
|
|
88
|
+
if max_unique <= 0 and ratio_threshold <= 0:
|
|
89
|
+
return self
|
|
90
|
+
|
|
91
|
+
stats = self._cardinality_stats(dataset.X[columns])
|
|
92
|
+
high_columns = self._high_cardinality_columns(
|
|
93
|
+
columns,
|
|
94
|
+
stats,
|
|
95
|
+
max_unique,
|
|
96
|
+
ratio_threshold,
|
|
97
|
+
min_non_null
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
if not high_columns:
|
|
101
|
+
return self
|
|
102
|
+
|
|
103
|
+
if self.strategy == 'hash':
|
|
104
|
+
self.hash_bins = self._coerce_bins(self.get_config('hash_bins'), 64)
|
|
105
|
+
self.hash_key = self._coerce_hash_key(self.get_config('hash_key'))
|
|
106
|
+
self.missing_value = self._coerce_int(self.get_config('missing_value'), -1)
|
|
107
|
+
self.columns_to_hash = high_columns
|
|
108
|
+
else:
|
|
109
|
+
self.columns_to_drop = high_columns
|
|
110
|
+
|
|
111
|
+
self.cardinality_stats = self._stats_to_dict(stats, high_columns)
|
|
112
|
+
self._build_explanations(stats, high_columns)
|
|
113
|
+
|
|
114
|
+
return self
|
|
115
|
+
|
|
116
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
117
|
+
if self.columns_to_drop:
|
|
118
|
+
drop_columns = [col for col in self.columns_to_drop if col in X.columns]
|
|
119
|
+
if drop_columns:
|
|
120
|
+
X = X.drop(columns=drop_columns)
|
|
121
|
+
|
|
122
|
+
if not self.columns_to_hash:
|
|
123
|
+
return X
|
|
124
|
+
|
|
125
|
+
hash_bins = self.hash_bins or self._coerce_bins(self.get_config('hash_bins'), 64)
|
|
126
|
+
hash_key = self.hash_key if self.hash_key is not None else \
|
|
127
|
+
self._coerce_hash_key(self.get_config('hash_key'))
|
|
128
|
+
missing_value = self.missing_value
|
|
129
|
+
|
|
130
|
+
for column in self.columns_to_hash:
|
|
131
|
+
if column not in X.columns:
|
|
132
|
+
continue
|
|
133
|
+
X[column] = self._hash_series(X[column], hash_bins, hash_key, missing_value)
|
|
134
|
+
|
|
135
|
+
return X
|
|
136
|
+
|
|
137
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
138
|
+
if dataset.X.empty:
|
|
139
|
+
return False
|
|
140
|
+
|
|
141
|
+
columns = self._select_columns(dataset)
|
|
142
|
+
if not columns:
|
|
143
|
+
return False
|
|
144
|
+
|
|
145
|
+
max_unique = self._coerce_int(self.get_config('max_unique'), 0)
|
|
146
|
+
ratio_threshold = self._coerce_ratio(self.get_config('unique_ratio_threshold'), 0.0)
|
|
147
|
+
min_non_null = self._coerce_int(self.get_config('min_non_null'), 0)
|
|
148
|
+
|
|
149
|
+
if max_unique <= 0 and ratio_threshold <= 0:
|
|
150
|
+
return False
|
|
151
|
+
|
|
152
|
+
stats = self._cardinality_stats(dataset.X[columns])
|
|
153
|
+
high_columns = self._high_cardinality_columns(
|
|
154
|
+
columns,
|
|
155
|
+
stats,
|
|
156
|
+
max_unique,
|
|
157
|
+
ratio_threshold,
|
|
158
|
+
min_non_null
|
|
159
|
+
)
|
|
160
|
+
return bool(high_columns)
|
|
161
|
+
|
|
162
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
163
|
+
if candidate is None or candidate.dataset.X.empty:
|
|
164
|
+
return 0.0
|
|
165
|
+
|
|
166
|
+
columns = self._select_columns(candidate.dataset)
|
|
167
|
+
if not columns:
|
|
168
|
+
return 0.0
|
|
169
|
+
|
|
170
|
+
max_unique = self._coerce_int(self.get_config('max_unique'), 0)
|
|
171
|
+
ratio_threshold = self._coerce_ratio(self.get_config('unique_ratio_threshold'), 0.0)
|
|
172
|
+
min_non_null = self._coerce_int(self.get_config('min_non_null'), 0)
|
|
173
|
+
|
|
174
|
+
if max_unique <= 0 and ratio_threshold <= 0:
|
|
175
|
+
return 0.0
|
|
176
|
+
|
|
177
|
+
stats = self._cardinality_stats(candidate.dataset.X[columns])
|
|
178
|
+
high_columns = self._high_cardinality_columns(
|
|
179
|
+
columns,
|
|
180
|
+
stats,
|
|
181
|
+
max_unique,
|
|
182
|
+
ratio_threshold,
|
|
183
|
+
min_non_null
|
|
184
|
+
)
|
|
185
|
+
if not high_columns:
|
|
186
|
+
return 0.0
|
|
187
|
+
|
|
188
|
+
ratio = len(high_columns) / max(1, len(columns))
|
|
189
|
+
return min(1.5, 0.5 + ratio)
|
|
190
|
+
|
|
191
|
+
def _select_columns(self, dataset: Dataset) -> list[str]:
|
|
192
|
+
columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
193
|
+
category_columns = list(dataset.X.select_dtypes(include=['category']).columns)
|
|
194
|
+
seen: set[str] = set()
|
|
195
|
+
ordered: list[str] = []
|
|
196
|
+
for column in columns + category_columns:
|
|
197
|
+
if column in dataset.X.columns and column not in seen:
|
|
198
|
+
ordered.append(column)
|
|
199
|
+
seen.add(column)
|
|
200
|
+
return ordered
|
|
201
|
+
|
|
202
|
+
def _cardinality_stats(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
203
|
+
non_null = X.notna().sum()
|
|
204
|
+
unique = X.nunique(dropna=True)
|
|
205
|
+
ratio = unique / non_null.replace(0, pd.NA)
|
|
206
|
+
ratio = ratio.fillna(0.0)
|
|
207
|
+
return pd.DataFrame({'unique': unique, 'non_null': non_null, 'ratio': ratio})
|
|
208
|
+
|
|
209
|
+
def _high_cardinality_columns(
|
|
210
|
+
self,
|
|
211
|
+
columns: list[str],
|
|
212
|
+
stats: pd.DataFrame,
|
|
213
|
+
max_unique: int,
|
|
214
|
+
ratio_threshold: float,
|
|
215
|
+
min_non_null: int
|
|
216
|
+
) -> list[str]:
|
|
217
|
+
eligible = stats['non_null'] >= min_non_null
|
|
218
|
+
conditions = pd.Series(False, index=stats.index)
|
|
219
|
+
if max_unique > 0:
|
|
220
|
+
conditions |= stats['unique'] >= max_unique
|
|
221
|
+
if ratio_threshold > 0:
|
|
222
|
+
conditions |= stats['ratio'] >= ratio_threshold
|
|
223
|
+
high = eligible & conditions
|
|
224
|
+
return [column for column in columns if column in high.index and bool(high[column])]
|
|
225
|
+
|
|
226
|
+
def _build_explanations(self, stats: pd.DataFrame, columns: list[str]) -> None:
|
|
227
|
+
for column in columns:
|
|
228
|
+
values = stats.loc[column]
|
|
229
|
+
unique = int(values['unique'])
|
|
230
|
+
non_null = int(values['non_null'])
|
|
231
|
+
ratio = float(values['ratio']) if non_null else 0.0
|
|
232
|
+
if self.strategy == 'hash':
|
|
233
|
+
self.explanations.append(
|
|
234
|
+
f"Hashed `{column}` into {self.hash_bins} bins "
|
|
235
|
+
f"(unique {unique}/{non_null}, ratio {ratio:.2%})."
|
|
236
|
+
)
|
|
237
|
+
else:
|
|
238
|
+
self.explanations.append(
|
|
239
|
+
f"Dropped `{column}` with {unique} unique values "
|
|
240
|
+
f"({ratio:.2%} of {non_null} non-missing)."
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
@staticmethod
|
|
244
|
+
def _stats_to_dict(stats: pd.DataFrame, columns: list[str]) -> dict[str, dict[str, float]]:
|
|
245
|
+
result: dict[str, dict[str, float]] = {}
|
|
246
|
+
for column in columns:
|
|
247
|
+
values = stats.loc[column]
|
|
248
|
+
result[column] = {
|
|
249
|
+
'unique': float(values['unique']),
|
|
250
|
+
'non_null': float(values['non_null']),
|
|
251
|
+
'ratio': float(values['ratio'])
|
|
252
|
+
}
|
|
253
|
+
return result
|
|
254
|
+
|
|
255
|
+
@staticmethod
|
|
256
|
+
def _hash_series(
|
|
257
|
+
series: pd.Series,
|
|
258
|
+
bins: int,
|
|
259
|
+
hash_key: str | bytes | None,
|
|
260
|
+
missing_value: int
|
|
261
|
+
) -> pd.Series:
|
|
262
|
+
if bins <= 0:
|
|
263
|
+
return pd.Series(missing_value, index=series.index, dtype='int64')
|
|
264
|
+
|
|
265
|
+
mask = series.isna()
|
|
266
|
+
values = series.astype('object')
|
|
267
|
+
hashed = ActDropHighCardinalityCategorical._hash_values(values, hash_key)
|
|
268
|
+
hashed = (hashed % bins).astype('int64')
|
|
269
|
+
|
|
270
|
+
if missing_value is not None and mask.any():
|
|
271
|
+
hashed = hashed.where(~mask, int(missing_value))
|
|
272
|
+
|
|
273
|
+
return hashed
|
|
274
|
+
|
|
275
|
+
@staticmethod
|
|
276
|
+
def _hash_values(values: pd.Series, hash_key: str | bytes | None) -> pd.Series:
|
|
277
|
+
if hash_key is None:
|
|
278
|
+
return pd.util.hash_pandas_object(values, index=False)
|
|
279
|
+
try:
|
|
280
|
+
return pd.util.hash_pandas_object(values, index=False, hash_key=hash_key)
|
|
281
|
+
except TypeError:
|
|
282
|
+
return pd.util.hash_pandas_object(values, index=False)
|
|
283
|
+
|
|
284
|
+
@staticmethod
|
|
285
|
+
def _coerce_int(value: Any, default: int) -> int:
|
|
286
|
+
try:
|
|
287
|
+
return int(value)
|
|
288
|
+
except (TypeError, ValueError):
|
|
289
|
+
return default
|
|
290
|
+
|
|
291
|
+
@staticmethod
|
|
292
|
+
def _coerce_ratio(value: Any, default: float) -> float:
|
|
293
|
+
try:
|
|
294
|
+
ratio = float(value)
|
|
295
|
+
except (TypeError, ValueError):
|
|
296
|
+
return default
|
|
297
|
+
return min(1.0, max(0.0, ratio))
|
|
298
|
+
|
|
299
|
+
@staticmethod
|
|
300
|
+
def _coerce_bins(value: Any, default: int) -> int:
|
|
301
|
+
try:
|
|
302
|
+
bins = int(value)
|
|
303
|
+
except (TypeError, ValueError):
|
|
304
|
+
return default
|
|
305
|
+
if bins < 2:
|
|
306
|
+
return default
|
|
307
|
+
return bins
|
|
308
|
+
|
|
309
|
+
@staticmethod
|
|
310
|
+
def _coerce_strategy(value: Any) -> str:
|
|
311
|
+
if isinstance(value, str):
|
|
312
|
+
normalized = value.strip().lower()
|
|
313
|
+
if normalized in {'drop', 'hash'}:
|
|
314
|
+
return normalized
|
|
315
|
+
return 'drop'
|
|
316
|
+
|
|
317
|
+
@staticmethod
|
|
318
|
+
def _coerce_hash_key(value: Any) -> str | bytes | None:
|
|
319
|
+
if value is None:
|
|
320
|
+
return None
|
|
321
|
+
if isinstance(value, (bytes, bytearray)):
|
|
322
|
+
key = bytes(value)
|
|
323
|
+
if not key:
|
|
324
|
+
return None
|
|
325
|
+
if len(key) < 16:
|
|
326
|
+
key = key.ljust(16, b'0')
|
|
327
|
+
elif len(key) > 16:
|
|
328
|
+
key = key[:16]
|
|
329
|
+
return key
|
|
330
|
+
key = str(value)
|
|
331
|
+
if not key:
|
|
332
|
+
return None
|
|
333
|
+
if len(key) < 16:
|
|
334
|
+
key = key.ljust(16, '0')
|
|
335
|
+
elif len(key) > 16:
|
|
336
|
+
key = key[:16]
|
|
337
|
+
return key
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""[STEP] Drop Numerical Column"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from ...actionable import Actionable
|
|
5
|
+
from ...dataset import Dataset
|
|
6
|
+
from ...data_type import DataType
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@is_step('cleaning')
|
|
12
|
+
class ActDropNumericalColumn(Actionable):
|
|
13
|
+
"""[STEP] Drop Numerical Column"""
|
|
14
|
+
|
|
15
|
+
name: str = 'Remove numerical columns'
|
|
16
|
+
_description: str = textwrap.dedent('''\
|
|
17
|
+
Remove numerical columns where the proportion of empty rows
|
|
18
|
+
in the dataset is higher than {empty_threshold}.''')
|
|
19
|
+
_description_long: str = textwrap.dedent('''\
|
|
20
|
+
Remove numerical columns from the dataset where the proportion of empty
|
|
21
|
+
rows in the dataset is higher than {empty_threshold:.0%}. This ensure that every columns will
|
|
22
|
+
be relevant for the model to train on.''')
|
|
23
|
+
_usage: str = "Use when numeric columns are mostly empty and dropping is acceptable; compare ActDropCategoricalColumn for non-numeric drops. Applicable to datasets with numeric fields and high missingness. Avoid when you should impute or the numeric signal is critical."
|
|
24
|
+
|
|
25
|
+
def __init__(self):
|
|
26
|
+
self.columns_to_drop: list[str] = None
|
|
27
|
+
self.configuration = {
|
|
28
|
+
'empty_threshold': {
|
|
29
|
+
'description': textwrap.dedent('''\
|
|
30
|
+
Column with more or equal proportion of empty row will
|
|
31
|
+
dropped. 1 will drop all columns'''),
|
|
32
|
+
'default': 0.5
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
37
|
+
self.columns_to_drop = []
|
|
38
|
+
explain = []
|
|
39
|
+
|
|
40
|
+
for column in dataset.get_columns_names_by_type(DataType.NUMERIC):
|
|
41
|
+
values = dataset.X[column]
|
|
42
|
+
nan_values_count = values.isnull().sum()
|
|
43
|
+
|
|
44
|
+
if nan_values_count / len(values) >= self.get_config('empty_threshold'):
|
|
45
|
+
self.columns_to_drop.append(column)
|
|
46
|
+
explain.append((nan_values_count, len(values)))
|
|
47
|
+
|
|
48
|
+
self.explanations = [
|
|
49
|
+
f"""Dropped column **`{c}`** because **{v[0]}** values out of
|
|
50
|
+
**{v[1]}** (**{(v[0] / v[1] * 100):.2f}%**) are empty."""
|
|
51
|
+
for c, v in zip(self.columns_to_drop, explain)
|
|
52
|
+
]
|
|
53
|
+
|
|
54
|
+
return self
|
|
55
|
+
|
|
56
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
57
|
+
"""Drop columns.
|
|
58
|
+
|
|
59
|
+
:param pd.DataFrame X: DataFrame to transform.
|
|
60
|
+
:return: Transformed DataFrame.
|
|
61
|
+
"""
|
|
62
|
+
return X.drop(self.columns_to_drop, axis=1)
|
|
63
|
+
|
|
64
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
65
|
+
return 0
|
|
66
|
+
|
|
67
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
68
|
+
for column in dataset.get_columns_names_by_type(DataType.NUMERIC):
|
|
69
|
+
values = dataset.X[column]
|
|
70
|
+
nan_values_count = values.isnull().sum()
|
|
71
|
+
|
|
72
|
+
if nan_values_count / len(values) >= self.get_config('empty_threshold'):
|
|
73
|
+
return True
|
|
74
|
+
|
|
75
|
+
return False
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""[STEP] Drop Textual Column"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from ...actionable import Actionable
|
|
5
|
+
from ...dataset import Dataset
|
|
6
|
+
from ...data_type import DataType
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@is_step('cleaning', 'baseline_cleaning')
|
|
12
|
+
class ActDropTextualColumn(Actionable):
|
|
13
|
+
"""[STEP] Drop Textual Column"""
|
|
14
|
+
|
|
15
|
+
name: str = 'Remove textual columns'
|
|
16
|
+
_description: str = 'Remove all columns containing textual data from the dataset'
|
|
17
|
+
_usage: str = 'Use when text columns are irrelevant for modeling or downstream steps. Applicable to datasets with free-text or short-text fields detected as text/object. Avoid when text should be vectorized or cleaned; consider ActCountVectorizer or ActCategoricalImputer.'
|
|
18
|
+
_description_long: str = textwrap.dedent('''\
|
|
19
|
+
Remove all columns containing textual data from the dataset.
|
|
20
|
+
This step is used to clean the dataset in order to perform other actions later on
|
|
21
|
+
that can't be applied to textual columns.''')
|
|
22
|
+
can_be_disabled: bool = False
|
|
23
|
+
|
|
24
|
+
def __init__(self):
|
|
25
|
+
self.columns_to_drop: list[str] = None
|
|
26
|
+
|
|
27
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
28
|
+
self.columns_to_drop = list(set(
|
|
29
|
+
dataset.get_columns_names_by_type([DataType.TEXT, DataType.SHORT_TEXT]) + \
|
|
30
|
+
list(dataset.X.select_dtypes(include='object').columns)
|
|
31
|
+
))
|
|
32
|
+
|
|
33
|
+
self.explanations = [
|
|
34
|
+
f'Dropped column **`{c}`**.' for c in self.columns_to_drop
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
return self
|
|
38
|
+
|
|
39
|
+
def transform(self, X) -> pd.DataFrame:
|
|
40
|
+
"""Drop columns.
|
|
41
|
+
|
|
42
|
+
:param pd.DataFrame X: DataFrame to transform.
|
|
43
|
+
:return: Transformed DataFrame.
|
|
44
|
+
"""
|
|
45
|
+
return X.drop(self.columns_to_drop, axis=1)
|
|
46
|
+
|
|
47
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
48
|
+
return 0
|
|
49
|
+
|
|
50
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
51
|
+
return bool(dataset.get_columns_names_by_type([DataType.TEXT, DataType.SHORT_TEXT]))
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Experimental target encoder, available only through an explicit module import.
|
|
2
|
+
|
|
3
|
+
This prototype transforms y rather than X, recomputes the mapping on each call,
|
|
4
|
+
and cannot reverse predictions. It is incompatible with the pipeline transformer
|
|
5
|
+
contract and must remain outside automatic cleaning. See docs/component_status.rst.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import textwrap
|
|
9
|
+
import numpy as np
|
|
10
|
+
from ...actionable import Actionable
|
|
11
|
+
from ...dataset import Dataset
|
|
12
|
+
from ...candidate import Candidate
|
|
13
|
+
from ...decorators.all import is_step
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@is_step('experimental')
|
|
17
|
+
class ActCategoryStringToNumeric(Actionable):
|
|
18
|
+
"""Encode categorical target column to numeric"""
|
|
19
|
+
|
|
20
|
+
name: str = 'Textual Category To Numeric Value'
|
|
21
|
+
_description: str ='Encode categorical text data column to numeric value'
|
|
22
|
+
_usage: str = 'Use when target labels are categorical strings and models require numeric y; prefer ActDropCategoricalColumn only if target is unusable. Applicable to single target columns of low-to-moderate cardinality. Avoid when target is already numeric or when ActDropHighCardinalityCategorical is more appropriate.'
|
|
23
|
+
_description_long: str = textwrap.dedent('''\
|
|
24
|
+
Retrieve all unique values from a column, then transform those values to a numeric type.
|
|
25
|
+
Exemple: If a column contain 3 uniques values like "coffee", "tea" and "water",
|
|
26
|
+
then all the coffee values will be transformed to 0, tea to 1 and water to 2.''')
|
|
27
|
+
|
|
28
|
+
def __init__(self):
|
|
29
|
+
self.column_to_encode: list[str] = None
|
|
30
|
+
|
|
31
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
32
|
+
self.column_to_encode = [dataset.y]
|
|
33
|
+
self.explanations = [
|
|
34
|
+
f'Encoded column **`{np.unique(dataset.y)}`**. \
|
|
35
|
+
Mapping of categorical values to numerical values:\n- `{value}`: {i}'
|
|
36
|
+
for i, value in enumerate(np.unique(dataset.y)) ]
|
|
37
|
+
|
|
38
|
+
return self
|
|
39
|
+
|
|
40
|
+
def transform(self, y: np.array) -> np.array:
|
|
41
|
+
"""Convert categorical target columns of candidate's dataset to numeric.
|
|
42
|
+
|
|
43
|
+
:param np.array y: Target column to transform.
|
|
44
|
+
:return: Transformed target column.
|
|
45
|
+
"""
|
|
46
|
+
if np.issubdtype(y.dtype, object) or np.issubdtype(y.dtype, np.bool_):
|
|
47
|
+
categories = np.unique(y)
|
|
48
|
+
encoded_target = np.zeros_like(y, dtype=int)
|
|
49
|
+
for i, category in enumerate(categories):
|
|
50
|
+
encoded_target[y == category] = i
|
|
51
|
+
return encoded_target
|
|
52
|
+
|
|
53
|
+
return y
|
|
54
|
+
|
|
55
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
56
|
+
return 0.4
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""[STEP] Encode categorical features by relative frequency."""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
from ...actionable import Actionable
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...data_type import DataType
|
|
10
|
+
from ...dataset import Dataset
|
|
11
|
+
from ...decorators.all import is_step
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@is_step('cleaning')
|
|
15
|
+
class ActFrequencyEncoder(Actionable):
|
|
16
|
+
"""[STEP] Encode categorical features by relative frequency."""
|
|
17
|
+
|
|
18
|
+
name: str = 'Frequency encoding'
|
|
19
|
+
_usage: str = "Use when categorical features need numeric encoding and frequencies are stable; compare ActDropHighCardinalityCategorical. Applicable to categorical/string features with moderate cardinality. Avoid when categories are very sparse, drifting, or too few rows to estimate."
|
|
20
|
+
_description: str = textwrap.dedent('''\
|
|
21
|
+
Encode categorical columns using their relative frequency.''')
|
|
22
|
+
_description_long: str = textwrap.dedent('''\
|
|
23
|
+
Replace each category with its relative frequency observed in the training data.
|
|
24
|
+
Frequencies are computed on non-missing values and unseen or missing values are
|
|
25
|
+
mapped to {unknown_value}.''')
|
|
26
|
+
|
|
27
|
+
def __init__(self) -> None:
|
|
28
|
+
self.columns: list[str] = []
|
|
29
|
+
self.encodings: dict[str, dict[Any, float]] = {}
|
|
30
|
+
self.fallback_values: dict[str, float] = {}
|
|
31
|
+
self.unknown_value: float = 0.0
|
|
32
|
+
|
|
33
|
+
self.configuration = {
|
|
34
|
+
'unknown_value': {
|
|
35
|
+
'description': 'Value used for unseen or missing categories.',
|
|
36
|
+
'default': 0.0
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
41
|
+
self.columns = self._select_columns(dataset)
|
|
42
|
+
self.encodings = {}
|
|
43
|
+
self.fallback_values = {}
|
|
44
|
+
self.explanations = []
|
|
45
|
+
|
|
46
|
+
if not self.columns or dataset.X.empty:
|
|
47
|
+
return self
|
|
48
|
+
|
|
49
|
+
self.unknown_value = self._coerce_float(self.get_config('unknown_value'), 0.0)
|
|
50
|
+
|
|
51
|
+
for column in self.columns:
|
|
52
|
+
series = dataset.X[column]
|
|
53
|
+
counts = series.value_counts(dropna=True)
|
|
54
|
+
total = int(counts.sum())
|
|
55
|
+
if total <= 0 or counts.empty:
|
|
56
|
+
self.fallback_values[column] = self.unknown_value
|
|
57
|
+
continue
|
|
58
|
+
|
|
59
|
+
frequencies = (counts / total).to_dict()
|
|
60
|
+
self.encodings[column] = frequencies
|
|
61
|
+
self.fallback_values[column] = self.unknown_value
|
|
62
|
+
self.explanations.append(
|
|
63
|
+
f"Encoded `{column}` using {len(frequencies)} categories."
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
return self
|
|
67
|
+
|
|
68
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
69
|
+
if not self.columns:
|
|
70
|
+
return X
|
|
71
|
+
|
|
72
|
+
for column in self.columns:
|
|
73
|
+
if column not in X.columns:
|
|
74
|
+
continue
|
|
75
|
+
|
|
76
|
+
mapping = self.encodings.get(column, {})
|
|
77
|
+
fallback = self.fallback_values.get(column, self.unknown_value)
|
|
78
|
+
fallback = self._coerce_float(fallback, 0.0)
|
|
79
|
+
|
|
80
|
+
encoded = X[column].map(mapping)
|
|
81
|
+
X[column] = encoded.fillna(fallback).astype(float)
|
|
82
|
+
|
|
83
|
+
return X
|
|
84
|
+
|
|
85
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
86
|
+
columns = self._select_columns(dataset)
|
|
87
|
+
if not columns or dataset.X.empty:
|
|
88
|
+
return False
|
|
89
|
+
return True
|
|
90
|
+
|
|
91
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
92
|
+
if candidate is None or candidate.dataset.X.empty:
|
|
93
|
+
return 0.0
|
|
94
|
+
|
|
95
|
+
columns = self._select_columns(candidate.dataset)
|
|
96
|
+
if not columns:
|
|
97
|
+
return 0.0
|
|
98
|
+
|
|
99
|
+
total_rows = len(candidate.dataset.X)
|
|
100
|
+
if total_rows <= 0:
|
|
101
|
+
return 0.0
|
|
102
|
+
|
|
103
|
+
unique_counts = candidate.dataset.X[columns].nunique(dropna=True)
|
|
104
|
+
avg_cardinality = float((unique_counts / total_rows).mean())
|
|
105
|
+
|
|
106
|
+
total_columns = candidate.dataset.X.shape[1] or 1
|
|
107
|
+
cat_ratio = len(columns) / total_columns
|
|
108
|
+
|
|
109
|
+
return min(1.0, max(cat_ratio, avg_cardinality))
|
|
110
|
+
|
|
111
|
+
def _select_columns(self, dataset: Dataset) -> list[str]:
|
|
112
|
+
columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
113
|
+
category_columns = list(dataset.X.select_dtypes(include=['category']).columns)
|
|
114
|
+
seen = set()
|
|
115
|
+
ordered = []
|
|
116
|
+
for column in columns + category_columns:
|
|
117
|
+
if column in dataset.X.columns and column not in seen:
|
|
118
|
+
ordered.append(column)
|
|
119
|
+
seen.add(column)
|
|
120
|
+
return ordered
|
|
121
|
+
|
|
122
|
+
@staticmethod
|
|
123
|
+
def _coerce_float(value: Any, default: float) -> float:
|
|
124
|
+
try:
|
|
125
|
+
return float(value)
|
|
126
|
+
except (TypeError, ValueError):
|
|
127
|
+
return default
|