PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""[STEP] Vectorize textual columns with Word2Vec"""
|
|
2
|
+
from typing import Any
|
|
3
|
+
import textwrap
|
|
4
|
+
import string
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pandas as pd
|
|
7
|
+
from gensim.models import Word2Vec
|
|
8
|
+
from nltk.corpus import stopwords
|
|
9
|
+
from nltk.stem import StemmerI, PorterStemmer
|
|
10
|
+
from nltk.tokenize import word_tokenize
|
|
11
|
+
from ...actionable import Actionable
|
|
12
|
+
from ...dataset import Dataset
|
|
13
|
+
from ...candidate import Candidate
|
|
14
|
+
from ...decorators.all import is_step
|
|
15
|
+
from ...data_type import DataType
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@is_step('cleaning')
|
|
19
|
+
class ActWord2Vec(Actionable):
|
|
20
|
+
"""[STEP] Vectorize textual column with Word2Vec"""
|
|
21
|
+
|
|
22
|
+
name: str = "Word2Vec"
|
|
23
|
+
_description: str = "Process Word2Vec algorithm over a list of columns"
|
|
24
|
+
_description_long: str = textwrap.dedent('''\
|
|
25
|
+
Word2Vec is a word embedding algorithm auto-supervised algorithm.
|
|
26
|
+
This means we don't need labelled data as the algorithm discove
|
|
27
|
+
the ground truth by himself''')
|
|
28
|
+
_usage: str = "Use when you want dense semantic text embeddings rather than sparse counts (vs ActCountVectorizer). Applicable to free-form text columns with enough tokens per row to learn embeddings. Avoid when text is short or ID-like, or when token-count interpretability is required."
|
|
29
|
+
refs: list[dict[str, Any]] = [
|
|
30
|
+
{
|
|
31
|
+
'year': 2023,
|
|
32
|
+
'name': 'Efficient Estimation of Word Representations in Vector Space',
|
|
33
|
+
'authors': [
|
|
34
|
+
'Thomas Mikolov',
|
|
35
|
+
'Kai Chen',
|
|
36
|
+
'Greg Corrado',
|
|
37
|
+
'Jeffrey Dean'
|
|
38
|
+
],
|
|
39
|
+
'doi': 'https://doi.org/10.48550/arXiv.1301.3781',
|
|
40
|
+
'publisher': 'International Conference on Learning Representations'
|
|
41
|
+
}
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
def __init__(self):
|
|
45
|
+
self.columns: list[tuple[str, Word2Vec]] = None
|
|
46
|
+
self.stop_words: set[str] | None = None
|
|
47
|
+
self.stemmer: StemmerI = PorterStemmer()
|
|
48
|
+
|
|
49
|
+
@staticmethod
|
|
50
|
+
def vectorize(words: str, model: Word2Vec):
|
|
51
|
+
"""Convert the preprocessed text data to a vector representation using
|
|
52
|
+
the Word2Vec model by calculating the aveerage of the word vectors
|
|
53
|
+
present in the sentence and returns this average vector. This gives
|
|
54
|
+
a vector representation of the whole sentence.
|
|
55
|
+
|
|
56
|
+
:param str words: The preprocessed text data as a string.
|
|
57
|
+
:param Word2Vec model: The Word2Vec model used for vectorization.
|
|
58
|
+
:return: The average vector representation of the input sentence.
|
|
59
|
+
"""
|
|
60
|
+
vectors = [ model.wv[word] for word in words if word in model.wv ]
|
|
61
|
+
if len(vectors) > 0:
|
|
62
|
+
return np.mean(vectors, axis=0)
|
|
63
|
+
|
|
64
|
+
return np.zeros(model.vector_size)
|
|
65
|
+
|
|
66
|
+
def __preprocess(self, text: str) -> list[str]:
|
|
67
|
+
"""Preprocesses the input text for Word2Vec processing tasks.
|
|
68
|
+
|
|
69
|
+
Steps:
|
|
70
|
+
1. Convert the text to lowercase.
|
|
71
|
+
2. Remove punctuation and special characters from the text.
|
|
72
|
+
3. Tokenize the text into words.
|
|
73
|
+
4. Remove stopwords from the tokenized words.
|
|
74
|
+
5. Apply stemming to the remaining tokens.
|
|
75
|
+
|
|
76
|
+
:param str text: The input text to be preprocessed.
|
|
77
|
+
:return: The preprocessed text.
|
|
78
|
+
"""
|
|
79
|
+
text = text.lower()
|
|
80
|
+
|
|
81
|
+
table = str.maketrans('', '', string.punctuation)
|
|
82
|
+
text = text.translate(table)
|
|
83
|
+
|
|
84
|
+
# Punctuation has already been removed; sentence splitting is unnecessary.
|
|
85
|
+
tokens = word_tokenize(text, preserve_line=True)
|
|
86
|
+
tokens = [ self.stemmer.stem(word) for word in tokens if word not in self.stop_words ]
|
|
87
|
+
|
|
88
|
+
return tokens
|
|
89
|
+
|
|
90
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
91
|
+
"""Find columns to vectorize and fit the vectorizer.
|
|
92
|
+
|
|
93
|
+
:param Dataset dataset: Data.
|
|
94
|
+
:return: Transformed candidate.
|
|
95
|
+
:raises LookupError: The NLTK stopwords corpus is missing when processing text.
|
|
96
|
+
"""
|
|
97
|
+
self.columns = []
|
|
98
|
+
self.explanations = []
|
|
99
|
+
text_columns = dataset.get_columns_names_by_type([DataType.TEXT])
|
|
100
|
+
if not text_columns or dataset.X.empty:
|
|
101
|
+
return self
|
|
102
|
+
|
|
103
|
+
if self.stop_words is None:
|
|
104
|
+
try:
|
|
105
|
+
self.stop_words = set(stopwords.words('english'))
|
|
106
|
+
except LookupError as exc:
|
|
107
|
+
raise LookupError(
|
|
108
|
+
"Word2Vec requires the NLTK English stopwords corpus for text data. "
|
|
109
|
+
"Install it with: python -m nltk.downloader stopwords"
|
|
110
|
+
) from exc
|
|
111
|
+
|
|
112
|
+
for column in text_columns:
|
|
113
|
+
values = dataset.X[column].fillna('').apply(self.__preprocess)
|
|
114
|
+
vectorizer = Word2Vec(sentences = values,
|
|
115
|
+
vector_size = 100,
|
|
116
|
+
window = 5,
|
|
117
|
+
min_count = 1,
|
|
118
|
+
workers = 4)
|
|
119
|
+
self.columns.append((column, vectorizer))
|
|
120
|
+
|
|
121
|
+
feature_names = { c: list(v.wv.index_to_key) for c, v in self.columns }
|
|
122
|
+
self.explanations = [
|
|
123
|
+
f'Encoded text column **`{c}`** into **{len(v)}** new columns.'
|
|
124
|
+
for c, v in feature_names.items() if len(v) > 0
|
|
125
|
+
]
|
|
126
|
+
|
|
127
|
+
return self
|
|
128
|
+
|
|
129
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
130
|
+
"""Apply Word2Vec vectorization to string data.
|
|
131
|
+
|
|
132
|
+
:param pd.DataFrame X: DataFrame to transform.
|
|
133
|
+
:return: Transformed dataset.
|
|
134
|
+
"""
|
|
135
|
+
X = X.reset_index(drop=True)
|
|
136
|
+
for name, vectorizer in self.columns:
|
|
137
|
+
transformed = X[name].fillna('').apply(
|
|
138
|
+
lambda doc: self.vectorize(self.__preprocess(doc), vectorizer) # pylint: disable=cell-var-from-loop
|
|
139
|
+
)
|
|
140
|
+
features_names = [f"{name}_vec_{i}" for i in range(vectorizer.vector_size)]
|
|
141
|
+
vector_df = pd.DataFrame(transformed.tolist(), columns = features_names)
|
|
142
|
+
X = pd.concat([X, vector_df], axis = 1).drop([name], axis = 1)
|
|
143
|
+
|
|
144
|
+
return X
|
|
145
|
+
|
|
146
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
147
|
+
return 0.4
|
|
148
|
+
|
|
149
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
150
|
+
return bool(dataset.get_columns_names_by_type([DataType.TEXT]))
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Usualy first step of a pipeline, clean/transform features.
|
|
3
|
+
Example : Convert SHORT_TEXT column into datetime if possible
|
|
4
|
+
"""
|
|
5
|
+
from .act_date_converter import ActDateConverter
|
|
6
|
+
from .act_trim_space import ActTrimSpaces
|
|
7
|
+
from .act_normalize_column_names import ActNormalizeColumnNames
|
|
8
|
+
from .act_coerce_numeric_strings import ActCoerceNumericStrings
|
|
9
|
+
from .act_drop_high_missing_columns import ActDropHighMissingColumns
|
|
10
|
+
from .act_drop_duplicate_rows import ActDropDuplicateRows
|
|
11
|
+
from .act_sentinel_to_na_n import ActSentinelToNaN
|
|
12
|
+
from .act_drop_id_like_columns import ActDropIdLikeColumns
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""[STEP] Coerce numeric strings to numeric values"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...data_type import DataType
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('features_precleaning')
|
|
13
|
+
class ActCoerceNumericStrings(Actionable):
|
|
14
|
+
"""[STEP] Coerce numeric strings to numeric values"""
|
|
15
|
+
|
|
16
|
+
name: str = 'Coerce numeric strings'
|
|
17
|
+
_description: str = textwrap.dedent('''\
|
|
18
|
+
Convert text columns that look numeric (commas, spaces, %) into numeric values.''')
|
|
19
|
+
_usage: str = 'Use when text columns contain numeric-like strings with separators or %; prefer over ActDateConverter for dates. Applicable to text or categorical columns that are mostly numeric. Avoid when values are identifiers or mostly non-numeric; use ActSentinelToNaN for missing tokens.'
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
This step attempts to coerce text or categorical columns into numeric values
|
|
22
|
+
by removing spaces, thousands separators and percent signs. A column is converted
|
|
23
|
+
only if the error ratio is below {authorized_error_ratio:.0%} of the non-missing
|
|
24
|
+
values in a sample of size {sample_size}.''')
|
|
25
|
+
refs: list[dict[str, Any]] = []
|
|
26
|
+
|
|
27
|
+
def __init__(self):
|
|
28
|
+
self.configuration = {
|
|
29
|
+
'authorized_error_ratio': {
|
|
30
|
+
'description': textwrap.dedent('''\
|
|
31
|
+
Maximum ratio of non-convertible values allowed to coerce a column.'''),
|
|
32
|
+
'default': 0.1
|
|
33
|
+
},
|
|
34
|
+
'sample_size': {
|
|
35
|
+
'description': textwrap.dedent('''\
|
|
36
|
+
Size of the sample used to detect numeric strings.
|
|
37
|
+
Set to -1 to use the whole dataset.'''),
|
|
38
|
+
'default': 200
|
|
39
|
+
},
|
|
40
|
+
'percent_to_fraction': {
|
|
41
|
+
'description': 'Convert values with percent signs into fractions (divide by 100).',
|
|
42
|
+
'default': True
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
self.columns: list[tuple[str, str]] = None
|
|
46
|
+
|
|
47
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
48
|
+
sample = self.__sample(dataset.X)
|
|
49
|
+
self.columns = []
|
|
50
|
+
explanations = []
|
|
51
|
+
|
|
52
|
+
for column in self.__candidate_columns(dataset):
|
|
53
|
+
strategy, ratio = self.__choose_strategy(sample[column])
|
|
54
|
+
if strategy is None:
|
|
55
|
+
continue
|
|
56
|
+
self.columns.append((column, strategy))
|
|
57
|
+
explanations.append(
|
|
58
|
+
f'Coerced column **`{column}`** to numeric using {strategy} decimal separator '
|
|
59
|
+
f'(success {ratio:.0%} on sample).'
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
self.explanations = explanations
|
|
63
|
+
return self
|
|
64
|
+
|
|
65
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
66
|
+
"""Coerce numeric-like strings into numeric values.
|
|
67
|
+
|
|
68
|
+
:param pd.DataFrame X: DataFrame to transform.
|
|
69
|
+
:return: Transformed DataFrame.
|
|
70
|
+
"""
|
|
71
|
+
if not self.columns:
|
|
72
|
+
return X
|
|
73
|
+
|
|
74
|
+
for column, strategy in self.columns:
|
|
75
|
+
if column not in X.columns:
|
|
76
|
+
continue
|
|
77
|
+
X[column] = self.__convert_series(X[column], strategy)
|
|
78
|
+
|
|
79
|
+
return X
|
|
80
|
+
|
|
81
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
82
|
+
if candidate is None:
|
|
83
|
+
return 1.0
|
|
84
|
+
total = max(1, candidate.dataset.X.shape[1])
|
|
85
|
+
candidates = len(self.__candidate_columns(candidate.dataset))
|
|
86
|
+
return min(1.0, 0.5 + candidates / total)
|
|
87
|
+
|
|
88
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
89
|
+
sample = self.__sample(dataset.X)
|
|
90
|
+
for column in self.__candidate_columns(dataset):
|
|
91
|
+
strategy, _ = self.__choose_strategy(sample[column])
|
|
92
|
+
if strategy is not None:
|
|
93
|
+
return True
|
|
94
|
+
return False
|
|
95
|
+
|
|
96
|
+
def __candidate_columns(self, dataset: Dataset) -> list[str]:
|
|
97
|
+
return dataset.get_columns_names_by_type(
|
|
98
|
+
[DataType.SHORT_TEXT, DataType.TEXT, DataType.CATEGORICAL]
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
def __sample(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
102
|
+
if X.empty:
|
|
103
|
+
return X
|
|
104
|
+
sample_size = self.get_config('sample_size')
|
|
105
|
+
if sample_size is None or sample_size == 0:
|
|
106
|
+
return X
|
|
107
|
+
if sample_size < 0 or sample_size >= len(X):
|
|
108
|
+
return X
|
|
109
|
+
return X.sample(n=sample_size, random_state=42)
|
|
110
|
+
|
|
111
|
+
def __choose_strategy(self, series: pd.Series) -> tuple[str | None, float]:
|
|
112
|
+
non_null = series.notna().sum()
|
|
113
|
+
if non_null == 0:
|
|
114
|
+
return None, 0.0
|
|
115
|
+
|
|
116
|
+
ratios = {}
|
|
117
|
+
for strategy in ('dot', 'comma'):
|
|
118
|
+
converted = self.__convert_series(series, strategy)
|
|
119
|
+
ratios[strategy] = converted.notna().sum() / non_null
|
|
120
|
+
|
|
121
|
+
best = max(ratios, key=ratios.get)
|
|
122
|
+
best_ratio = ratios[best]
|
|
123
|
+
if best_ratio < 1 - self.get_config('authorized_error_ratio'):
|
|
124
|
+
return None, best_ratio
|
|
125
|
+
|
|
126
|
+
other = 'comma' if best == 'dot' else 'dot'
|
|
127
|
+
if abs(ratios[best] - ratios[other]) <= 0.02:
|
|
128
|
+
inferred = self.__infer_decimal_separator(series)
|
|
129
|
+
if ratios.get(inferred, 0) >= 1 - self.get_config('authorized_error_ratio'):
|
|
130
|
+
best = inferred
|
|
131
|
+
|
|
132
|
+
return best, best_ratio
|
|
133
|
+
|
|
134
|
+
def __normalize_strings(self, series: pd.Series) -> tuple[pd.Series, pd.Series]:
|
|
135
|
+
values = series.where(series.notna(), '')
|
|
136
|
+
values = values.astype(str).str.strip()
|
|
137
|
+
percent_mask = values.str.contains('%', regex=False, na=False)
|
|
138
|
+
values = values.str.replace('%', '', regex=False)
|
|
139
|
+
values = values.str.replace(r'\s+', '', regex=True)
|
|
140
|
+
return values, percent_mask
|
|
141
|
+
|
|
142
|
+
def __convert_series(self, series: pd.Series, strategy: str) -> pd.Series:
|
|
143
|
+
values, percent_mask = self.__normalize_strings(series)
|
|
144
|
+
|
|
145
|
+
if strategy == 'comma':
|
|
146
|
+
values = values.str.replace('.', '', regex=False)
|
|
147
|
+
values = values.str.replace(',', '.', regex=False)
|
|
148
|
+
else:
|
|
149
|
+
values = values.str.replace(',', '', regex=False)
|
|
150
|
+
|
|
151
|
+
numeric = pd.to_numeric(values, errors='coerce')
|
|
152
|
+
if self.get_config('percent_to_fraction'):
|
|
153
|
+
numeric = numeric.mask(percent_mask, numeric / 100)
|
|
154
|
+
|
|
155
|
+
return numeric
|
|
156
|
+
|
|
157
|
+
def __infer_decimal_separator(self, series: pd.Series) -> str:
|
|
158
|
+
values, _ = self.__normalize_strings(series)
|
|
159
|
+
|
|
160
|
+
has_comma = values.str.contains(',', regex=False, na=False)
|
|
161
|
+
has_dot = values.str.contains('.', regex=False, na=False)
|
|
162
|
+
|
|
163
|
+
both = has_comma & has_dot
|
|
164
|
+
if both.any():
|
|
165
|
+
last_comma = values[both].str.rfind(',')
|
|
166
|
+
last_dot = values[both].str.rfind('.')
|
|
167
|
+
comma_last = (last_comma > last_dot).sum()
|
|
168
|
+
dot_last = (last_dot > last_comma).sum()
|
|
169
|
+
if comma_last != dot_last:
|
|
170
|
+
return 'comma' if comma_last > dot_last else 'dot'
|
|
171
|
+
|
|
172
|
+
comma_only = has_comma & ~has_dot
|
|
173
|
+
dot_only = has_dot & ~has_comma
|
|
174
|
+
|
|
175
|
+
if comma_only.any() and not dot_only.any():
|
|
176
|
+
return self.__separator_preference(values[comma_only], ',')
|
|
177
|
+
if dot_only.any() and not comma_only.any():
|
|
178
|
+
return self.__separator_preference(values[dot_only], '.')
|
|
179
|
+
if comma_only.any() or dot_only.any():
|
|
180
|
+
return 'comma' if comma_only.sum() > dot_only.sum() else 'dot'
|
|
181
|
+
|
|
182
|
+
return 'dot'
|
|
183
|
+
|
|
184
|
+
def __separator_preference(self, values: pd.Series, separator: str) -> str:
|
|
185
|
+
after = values.str.rsplit(separator, n=1).str[-1]
|
|
186
|
+
decimal_like = after.str.len().isin([1, 2]).sum()
|
|
187
|
+
thousand_like = after.str.len().eq(3).sum()
|
|
188
|
+
|
|
189
|
+
if decimal_like > thousand_like:
|
|
190
|
+
return 'comma' if separator == ',' else 'dot'
|
|
191
|
+
if thousand_like > decimal_like:
|
|
192
|
+
return 'dot' if separator == ',' else 'comma'
|
|
193
|
+
|
|
194
|
+
return 'comma' if separator == ',' else 'dot'
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""[STEP] Convert Short text to date if possible"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from ...actionable import Actionable
|
|
5
|
+
from ...dataset import Dataset
|
|
6
|
+
from ...data_type import DataType
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@is_step('features_precleaning')
|
|
12
|
+
class ActDateConverter(Actionable):
|
|
13
|
+
"""[STEP] Convert Short text to date if possible"""
|
|
14
|
+
|
|
15
|
+
name: str = 'Text to Date Converter'
|
|
16
|
+
_description: str = 'Convert Text to Date if possible'
|
|
17
|
+
_description_long: str = textwrap.dedent('''\
|
|
18
|
+
Try to convert all text of a column to date.
|
|
19
|
+
If more than {authorized_error_ratios}% of the rows return errors, then the
|
|
20
|
+
column is not converted. As converting to date is time consuming, we will
|
|
21
|
+
perform the test on {sample_size}.
|
|
22
|
+
''')
|
|
23
|
+
_usage: str = 'Use when short text columns mostly contain parseable dates. Applicable to SHORT_TEXT columns with mixed date formats. Avoid when values are IDs or sentinels better handled by ActDropIdLikeColumns or ActSentinelToNaN.'
|
|
24
|
+
|
|
25
|
+
def __init__(self):
|
|
26
|
+
self.configuration = {
|
|
27
|
+
'authorized_error_ratios': {
|
|
28
|
+
'default': 0.05,
|
|
29
|
+
'description': 'Over this ratios, the column will not be converted into date'
|
|
30
|
+
},
|
|
31
|
+
'sample_size': {
|
|
32
|
+
'default': 200,
|
|
33
|
+
'description': textwrap.dedent('''\
|
|
34
|
+
Convert date is time consuming. To save time, date
|
|
35
|
+
detection will be done on a random sample.
|
|
36
|
+
Set to -1 to detect on the whole dataset.''')
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
self.columns: list[str] = None
|
|
41
|
+
|
|
42
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
43
|
+
sample_size = self.get_config('sample_size')
|
|
44
|
+
if sample_size > 0:
|
|
45
|
+
sample = dataset.X.sample(n=min(sample_size, len(dataset.X)))
|
|
46
|
+
else:
|
|
47
|
+
sample = dataset.X
|
|
48
|
+
|
|
49
|
+
self.columns = []
|
|
50
|
+
for column in dataset.get_columns_names_by_type(DataType.SHORT_TEXT):
|
|
51
|
+
threshold_count: float = sample[column].count() * \
|
|
52
|
+
(1 - self.get_config('authorized_error_ratios'))
|
|
53
|
+
new_columns: pd.DataFrame = sample[column].apply(self.__string_value_to_date)
|
|
54
|
+
|
|
55
|
+
if new_columns.count() >= threshold_count:
|
|
56
|
+
self.columns.append(column)
|
|
57
|
+
|
|
58
|
+
self.explanations = [
|
|
59
|
+
f'Convert text column **`{c}`** into datetime column.'
|
|
60
|
+
for c in self.columns
|
|
61
|
+
]
|
|
62
|
+
|
|
63
|
+
return self
|
|
64
|
+
|
|
65
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
66
|
+
return bool(dataset.get_columns_names_by_type(DataType.SHORT_TEXT))
|
|
67
|
+
|
|
68
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
69
|
+
"""Apply Text to date convert.
|
|
70
|
+
|
|
71
|
+
:param pd.DataFrame x: DataFrame to transform.
|
|
72
|
+
:return: Transformed dataset.
|
|
73
|
+
"""
|
|
74
|
+
for column in self.columns:
|
|
75
|
+
X[column] = pd.to_datetime(X[column], errors='coerce')
|
|
76
|
+
|
|
77
|
+
return X
|
|
78
|
+
|
|
79
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
80
|
+
return 0.5
|
|
81
|
+
|
|
82
|
+
def __string_value_to_date(self, value: str, date_formats = None) -> pd.Timestamp:
|
|
83
|
+
"""Convert a string value (with date format) to a date.
|
|
84
|
+
|
|
85
|
+
:param str value: String in a date format.
|
|
86
|
+
:return: Converted date.
|
|
87
|
+
"""
|
|
88
|
+
if date_formats is None:
|
|
89
|
+
date_formats = [None]
|
|
90
|
+
elif isinstance(date_formats, list):
|
|
91
|
+
date_formats = list(date_formats)
|
|
92
|
+
|
|
93
|
+
for date_format in date_formats:
|
|
94
|
+
try:
|
|
95
|
+
return pd.to_datetime(value, format=date_format)
|
|
96
|
+
except ValueError:
|
|
97
|
+
pass # Let's try the next date format
|
|
98
|
+
|
|
99
|
+
return pd.NaT
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Experimental row filter, available only through an explicit module import.
|
|
2
|
+
|
|
3
|
+
Its missingness threshold, minimum sample policy and alignment of resampled data
|
|
4
|
+
need integration tests before automatic use. See docs/component_status.rst.
|
|
5
|
+
"""
|
|
6
|
+
from typing import Any
|
|
7
|
+
import textwrap
|
|
8
|
+
import pandas as pd
|
|
9
|
+
from ...actionable import Actionable
|
|
10
|
+
from ...dataset import Dataset
|
|
11
|
+
from ...candidate import Candidate
|
|
12
|
+
from ...decorators.all import is_step
|
|
13
|
+
|
|
14
|
+
@is_step('experimental')
|
|
15
|
+
class ActDropBadQualityRows(Actionable):
|
|
16
|
+
"""
|
|
17
|
+
[STEP] Drop Rows with a ratio of Empty Columns
|
|
18
|
+
"""
|
|
19
|
+
name: str = "Drop Rows with Empty Columns"
|
|
20
|
+
_description: str = textwrap.dedent('''\
|
|
21
|
+
This step drops rows where the ratio of empty columns are over a threshold.
|
|
22
|
+
It helps clean the dataset by removing rows with a significant
|
|
23
|
+
amount of missing values.''')
|
|
24
|
+
_description_long: str = textwrap.dedent('''\
|
|
25
|
+
In datasets, missing data is a common issue.
|
|
26
|
+
This step drops rows where the ratio of empty columns are over a threshold.
|
|
27
|
+
This ensures that rows with too many missing values
|
|
28
|
+
are not included in the analysis, improving the quality of the dataset.''')
|
|
29
|
+
refs: list[dict[str, Any]] = []
|
|
30
|
+
|
|
31
|
+
def __init__(self):
|
|
32
|
+
self.configuration = {
|
|
33
|
+
'empty_threshold': {
|
|
34
|
+
'description': textwrap.dedent('''\
|
|
35
|
+
Row with less or equal proportion of empty column will be
|
|
36
|
+
drop. 0 will never never drop a row'''),
|
|
37
|
+
'default': 0.3
|
|
38
|
+
},
|
|
39
|
+
'min_size': {
|
|
40
|
+
'description': textwrap.dedent('''\
|
|
41
|
+
Minimal size of the resulting dataset. If new dataset is
|
|
42
|
+
smaller than this value, old one will be restored'''),
|
|
43
|
+
'default': 20
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
48
|
+
return self
|
|
49
|
+
|
|
50
|
+
def __transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
51
|
+
"""Apply the dropping of rows with at least 40% empty columns.
|
|
52
|
+
|
|
53
|
+
:param pd.DataFrame X: The dataframe to clean.
|
|
54
|
+
:return: The cleaned dataframe.
|
|
55
|
+
"""
|
|
56
|
+
threshold = self.get_config('empty_threshold') * X.shape[1] # 40% of the total columns
|
|
57
|
+
new_x = X.dropna(thresh=X.shape[1] - threshold)
|
|
58
|
+
return new_x if new_x.shape[0] >= self.get_config('min_size') else X
|
|
59
|
+
|
|
60
|
+
def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
61
|
+
"""Resample method to apply transformation to both X and y.
|
|
62
|
+
|
|
63
|
+
:param pd.DataFrame X: Features to transform.
|
|
64
|
+
:param pd.DataFrame y: Labels to keep aligned.
|
|
65
|
+
:return: The transformed X and aligned y.
|
|
66
|
+
"""
|
|
67
|
+
x_clean = self.__transform(X.reset_index(drop=True))
|
|
68
|
+
y_aligned = y[x_clean.index] # Align y with the cleaned X
|
|
69
|
+
x_clean.reset_index(drop=True, inplace=True)
|
|
70
|
+
|
|
71
|
+
return x_clean, y_aligned
|
|
72
|
+
|
|
73
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
74
|
+
return 1
|
|
75
|
+
|
|
76
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
77
|
+
return dataset.X.isnull().values.any()
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""[STEP] Drop duplicate rows"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...candidate import Candidate
|
|
7
|
+
from ...data_type import DataType
|
|
8
|
+
from ...dataset import Dataset
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('features_precleaning')
|
|
13
|
+
class ActDropDuplicateRows(Actionable):
|
|
14
|
+
"""[STEP] Drop duplicate rows"""
|
|
15
|
+
|
|
16
|
+
name: str = 'Drop duplicate rows'
|
|
17
|
+
_usage: str = "Use when duplicated feature rows indicate repeated records and you want one kept copy. Applicable to tabular data with aligned y; unlike ActDropIdLikeColumns, it drops rows not columns. Avoid when repeats are meaningful (time series, sampling) or counts must be preserved."
|
|
18
|
+
_description: str = textwrap.dedent('''\
|
|
19
|
+
Remove duplicated rows using keep={keep}.''')
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
Duplicated rows can bias model training by repeating the same signal.
|
|
22
|
+
This step removes duplicates based on feature values and keeps y/groups aligned.''')
|
|
23
|
+
refs: list[dict[str, Any]] = []
|
|
24
|
+
|
|
25
|
+
def __init__(self):
|
|
26
|
+
self.configuration = {
|
|
27
|
+
'keep': {
|
|
28
|
+
'description': "Which duplicate to keep when removing duplicates.",
|
|
29
|
+
'default': 'first',
|
|
30
|
+
'categorical': ['first', 'last', False]
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
self.columns_to_check: list[str] = []
|
|
34
|
+
self.duplicate_count: int = 0
|
|
35
|
+
self.duplicate_ratio: float = 0.0
|
|
36
|
+
|
|
37
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
38
|
+
self.columns_to_check = self.__candidate_columns(dataset)
|
|
39
|
+
if dataset.X.empty or not self.columns_to_check:
|
|
40
|
+
self.duplicate_count = 0
|
|
41
|
+
self.duplicate_ratio = 0.0
|
|
42
|
+
self.explanations = []
|
|
43
|
+
return self
|
|
44
|
+
|
|
45
|
+
duplicate_mask = dataset.X.duplicated(
|
|
46
|
+
subset=self.columns_to_check,
|
|
47
|
+
keep=self.get_config('keep')
|
|
48
|
+
)
|
|
49
|
+
self.duplicate_count = int(duplicate_mask.sum())
|
|
50
|
+
total_rows = len(dataset.X)
|
|
51
|
+
self.duplicate_ratio = self.duplicate_count / max(1, total_rows)
|
|
52
|
+
|
|
53
|
+
if self.duplicate_count:
|
|
54
|
+
self.explanations = [
|
|
55
|
+
f'Dropped {self.duplicate_count} duplicated rows '
|
|
56
|
+
f'({self.duplicate_ratio:.2%} of {total_rows}) using keep={self.get_config("keep")}.'
|
|
57
|
+
]
|
|
58
|
+
else:
|
|
59
|
+
self.explanations = []
|
|
60
|
+
|
|
61
|
+
return self
|
|
62
|
+
|
|
63
|
+
def resample(self, X: pd.DataFrame, y: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
64
|
+
"""Drop duplicate rows and align y.
|
|
65
|
+
|
|
66
|
+
:param pd.DataFrame X: Features to clean.
|
|
67
|
+
:param pd.DataFrame y: Labels to align.
|
|
68
|
+
:return: Cleaned X and aligned y.
|
|
69
|
+
"""
|
|
70
|
+
if X.empty:
|
|
71
|
+
return X, y
|
|
72
|
+
|
|
73
|
+
x_reset = X.reset_index(drop=True)
|
|
74
|
+
x_clean = self.__transform(x_reset)
|
|
75
|
+
if len(x_clean) == len(x_reset):
|
|
76
|
+
return X, y
|
|
77
|
+
|
|
78
|
+
if not self.__has_aligned_y(y, len(x_reset)):
|
|
79
|
+
x_clean.reset_index(drop=True, inplace=True)
|
|
80
|
+
return x_clean, y
|
|
81
|
+
|
|
82
|
+
y_aligned = y[x_clean.index]
|
|
83
|
+
x_clean.reset_index(drop=True, inplace=True)
|
|
84
|
+
return x_clean, y_aligned
|
|
85
|
+
|
|
86
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
87
|
+
if candidate is None:
|
|
88
|
+
return 1.0
|
|
89
|
+
columns = self.__candidate_columns(candidate.dataset)
|
|
90
|
+
if not columns or candidate.dataset.X.empty:
|
|
91
|
+
return 0.0
|
|
92
|
+
duplicate_ratio = candidate.dataset.X.duplicated(
|
|
93
|
+
subset=columns,
|
|
94
|
+
keep=self.get_config('keep')
|
|
95
|
+
).mean()
|
|
96
|
+
return min(1.5, 0.5 + duplicate_ratio)
|
|
97
|
+
|
|
98
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
99
|
+
columns = self.__candidate_columns(dataset)
|
|
100
|
+
if not columns or dataset.X.empty:
|
|
101
|
+
return False
|
|
102
|
+
return dataset.X.duplicated(
|
|
103
|
+
subset=columns,
|
|
104
|
+
keep=self.get_config('keep')
|
|
105
|
+
).any()
|
|
106
|
+
|
|
107
|
+
def __transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
108
|
+
subset = self.__subset_columns(X)
|
|
109
|
+
if not subset:
|
|
110
|
+
return X
|
|
111
|
+
return X.drop_duplicates(subset=subset, keep=self.get_config('keep'))
|
|
112
|
+
|
|
113
|
+
def __subset_columns(self, X: pd.DataFrame) -> list[str]:
|
|
114
|
+
if self.columns_to_check:
|
|
115
|
+
return [column for column in self.columns_to_check if column in X.columns]
|
|
116
|
+
return list(X.columns)
|
|
117
|
+
|
|
118
|
+
def __candidate_columns(self, dataset: Dataset) -> list[str]:
|
|
119
|
+
columns = dataset.get_columns_names_by_type(list(DataType))
|
|
120
|
+
if len(columns) != dataset.X.shape[1]:
|
|
121
|
+
missing = [column for column in dataset.X.columns if column not in columns]
|
|
122
|
+
columns.extend(missing)
|
|
123
|
+
return columns
|
|
124
|
+
|
|
125
|
+
def __has_aligned_y(self, y: Any, rows_count: int) -> bool:
|
|
126
|
+
if y is None:
|
|
127
|
+
return False
|
|
128
|
+
try:
|
|
129
|
+
return len(y) == rows_count
|
|
130
|
+
except TypeError:
|
|
131
|
+
return False
|