PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""[STEP] One hot encoding categorical features"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.preprocessing import OneHotEncoder
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...data_type import DataType
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('cleaning')
|
|
13
|
+
class ActOnehot(Actionable):
|
|
14
|
+
"""[STEP] One hot encoding categorical features"""
|
|
15
|
+
|
|
16
|
+
name: str = 'One hot encoding categorical features'
|
|
17
|
+
_description: str = 'Encode categorical data to numerical value using \
|
|
18
|
+
One Hot Encoding Algorithm'
|
|
19
|
+
_usage: str = 'Use when categorical columns have meaningful distinct values and you want indicator features, rather than ActDropCategoricalColumn. Applicable to nominal categorical data with low to moderate cardinality. Avoid when categories are very high-cardinality; consider ActDropHighCardinalityCategorical.'
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
Retrieve all unique values from a column, then transform
|
|
22
|
+
those values to multiple binary columns.
|
|
23
|
+
Exemple: If a column "A" contain 3 uniques values like "coffee", "tea" and "water",
|
|
24
|
+
this step will create binary columns "A_coffee", "A_tea" and "A_water".''')
|
|
25
|
+
|
|
26
|
+
def __init__(self):
|
|
27
|
+
self.columns: list[str] = None
|
|
28
|
+
self.encoder: OneHotEncoder = None
|
|
29
|
+
|
|
30
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
31
|
+
self.columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
32
|
+
values = dataset.X[self.columns]
|
|
33
|
+
self.encoder = OneHotEncoder(handle_unknown='ignore', sparse_output=False).fit(values)
|
|
34
|
+
|
|
35
|
+
# creates a dict with columns as keys and encoded categories as values
|
|
36
|
+
features = dict(zip(self.columns, self.encoder.categories_))
|
|
37
|
+
|
|
38
|
+
self.explanations = [
|
|
39
|
+
f'Encoded categorical column **`{c}`** into **{len(v)}** new columns.'
|
|
40
|
+
for c, v in features.items() if len(v) > 0
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
return self
|
|
44
|
+
|
|
45
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
46
|
+
"""Apply One Hot Encoding to dataframe.
|
|
47
|
+
|
|
48
|
+
:param pd.DataFrame x: DataFrame to transform.
|
|
49
|
+
:return: Transformed dataset.
|
|
50
|
+
"""
|
|
51
|
+
# Without Reset index, the join with OHE will create NAN (index mismatch)
|
|
52
|
+
X = X.reset_index(drop=True)
|
|
53
|
+
|
|
54
|
+
transformed = self.encoder.transform(X[self.columns])
|
|
55
|
+
ohe_df = pd.DataFrame(transformed, columns=self.encoder.get_feature_names_out(self.columns))
|
|
56
|
+
X = X.drop(self.columns, axis=1)
|
|
57
|
+
df = X.join(ohe_df)
|
|
58
|
+
|
|
59
|
+
return df
|
|
60
|
+
|
|
61
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
62
|
+
return bool(dataset.get_columns_names_by_type(DataType.CATEGORICAL))
|
|
63
|
+
|
|
64
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
65
|
+
return 0.5
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
"""[STEP] Ordinal encoding categorical features."""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
from ...actionable import Actionable
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...data_type import DataType
|
|
10
|
+
from ...dataset import Dataset
|
|
11
|
+
from ...decorators.all import is_step
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@is_step('cleaning')
|
|
15
|
+
class ActOrdinalEncoder(Actionable):
|
|
16
|
+
"""[STEP] Ordinal encoding categorical features."""
|
|
17
|
+
|
|
18
|
+
name: str = 'Ordinal encoding'
|
|
19
|
+
_description: str = textwrap.dedent('''\
|
|
20
|
+
Encode categorical columns into integer codes with unknown handling.''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
Replace categorical values with integer codes. Categories observed during training
|
|
23
|
+
are assigned consecutive integers starting at {start_value}. Missing or unseen
|
|
24
|
+
categories are encoded as {unknown_value}.''')
|
|
25
|
+
_usage: str = "Use when you need numeric codes for categorical features; ActCountVectorizer is for text-like categories. Applicable to low-cardinality categorical columns with stable labels. Avoid when categories are high-cardinality or should be dropped (ActDropHighCardinalityCategorical)."
|
|
26
|
+
|
|
27
|
+
def __init__(self) -> None:
|
|
28
|
+
self.columns: list[str] = []
|
|
29
|
+
self.categories: dict[str, list[Any]] = {}
|
|
30
|
+
self.unknown_value: int = -1
|
|
31
|
+
self.start_value: int = 0
|
|
32
|
+
self.sort_categories: bool = True
|
|
33
|
+
|
|
34
|
+
self.configuration = {
|
|
35
|
+
'unknown_value': {
|
|
36
|
+
'description': 'Value used for unseen or missing categories.',
|
|
37
|
+
'default': -1
|
|
38
|
+
},
|
|
39
|
+
'start_value': {
|
|
40
|
+
'description': 'Starting integer assigned to known categories.',
|
|
41
|
+
'default': 0
|
|
42
|
+
},
|
|
43
|
+
'sort_categories': {
|
|
44
|
+
'description': 'Sort categories before encoding for deterministic mapping.',
|
|
45
|
+
'default': True
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
50
|
+
self.columns = self._select_columns(dataset)
|
|
51
|
+
self.categories = {}
|
|
52
|
+
self.explanations = []
|
|
53
|
+
|
|
54
|
+
if not self.columns or dataset.X.empty:
|
|
55
|
+
return self
|
|
56
|
+
|
|
57
|
+
self.unknown_value = self._coerce_int(self.get_config('unknown_value'), -1)
|
|
58
|
+
self.start_value = self._coerce_int(self.get_config('start_value'), 0)
|
|
59
|
+
self.sort_categories = bool(self.get_config('sort_categories'))
|
|
60
|
+
|
|
61
|
+
for column in self.columns:
|
|
62
|
+
series = dataset.X[column]
|
|
63
|
+
categories = self._extract_categories(series, self.sort_categories)
|
|
64
|
+
self.categories[column] = categories
|
|
65
|
+
|
|
66
|
+
if categories:
|
|
67
|
+
self.explanations.append(
|
|
68
|
+
f"Ordinal-encoded `{column}` with {len(categories)} categories."
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
return self
|
|
72
|
+
|
|
73
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
74
|
+
if not self.columns or not self.categories:
|
|
75
|
+
return X
|
|
76
|
+
|
|
77
|
+
for column in self.columns:
|
|
78
|
+
if column not in X.columns:
|
|
79
|
+
continue
|
|
80
|
+
categories = self.categories.get(column, [])
|
|
81
|
+
X[column] = self._encode_series(
|
|
82
|
+
X[column],
|
|
83
|
+
categories,
|
|
84
|
+
self.start_value,
|
|
85
|
+
self.unknown_value
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
return X
|
|
89
|
+
|
|
90
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
91
|
+
columns = self._select_columns(dataset)
|
|
92
|
+
if not columns or dataset.X.empty:
|
|
93
|
+
return False
|
|
94
|
+
return True
|
|
95
|
+
|
|
96
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
97
|
+
if candidate is None or candidate.dataset.X.empty:
|
|
98
|
+
return 0.0
|
|
99
|
+
|
|
100
|
+
columns = self._select_columns(candidate.dataset)
|
|
101
|
+
if not columns:
|
|
102
|
+
return 0.0
|
|
103
|
+
|
|
104
|
+
total_rows = len(candidate.dataset.X)
|
|
105
|
+
if total_rows <= 0:
|
|
106
|
+
return 0.0
|
|
107
|
+
|
|
108
|
+
unique_counts = candidate.dataset.X[columns].nunique(dropna=True)
|
|
109
|
+
avg_cardinality = float((unique_counts / total_rows).mean())
|
|
110
|
+
low_cardinality = 1.0 - min(1.0, max(0.0, avg_cardinality))
|
|
111
|
+
|
|
112
|
+
total_columns = candidate.dataset.X.shape[1] or 1
|
|
113
|
+
cat_ratio = len(columns) / total_columns
|
|
114
|
+
|
|
115
|
+
return min(1.0, max(cat_ratio, low_cardinality))
|
|
116
|
+
|
|
117
|
+
def _select_columns(self, dataset: Dataset) -> list[str]:
|
|
118
|
+
columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
119
|
+
category_columns = list(dataset.X.select_dtypes(include=['category']).columns)
|
|
120
|
+
seen = set()
|
|
121
|
+
ordered = []
|
|
122
|
+
for column in columns + category_columns:
|
|
123
|
+
if column in dataset.X.columns and column not in seen:
|
|
124
|
+
ordered.append(column)
|
|
125
|
+
seen.add(column)
|
|
126
|
+
return ordered
|
|
127
|
+
|
|
128
|
+
@staticmethod
|
|
129
|
+
def _coerce_int(value: Any, default: int) -> int:
|
|
130
|
+
try:
|
|
131
|
+
return int(value)
|
|
132
|
+
except (TypeError, ValueError):
|
|
133
|
+
return default
|
|
134
|
+
|
|
135
|
+
@classmethod
|
|
136
|
+
def _extract_categories(cls, series: pd.Series, sort_categories: bool) -> list[Any]:
|
|
137
|
+
if pd.api.types.is_categorical_dtype(series):
|
|
138
|
+
categories = list(series.cat.categories)
|
|
139
|
+
else:
|
|
140
|
+
categories = series.dropna().unique().tolist()
|
|
141
|
+
|
|
142
|
+
if sort_categories and categories:
|
|
143
|
+
categories = cls._safe_sorted(categories)
|
|
144
|
+
|
|
145
|
+
return categories
|
|
146
|
+
|
|
147
|
+
@staticmethod
|
|
148
|
+
def _safe_sorted(values: list[Any]) -> list[Any]:
|
|
149
|
+
try:
|
|
150
|
+
return sorted(values)
|
|
151
|
+
except TypeError:
|
|
152
|
+
return sorted(values, key=lambda item: str(item))
|
|
153
|
+
|
|
154
|
+
@staticmethod
|
|
155
|
+
def _encode_series(
|
|
156
|
+
series: pd.Series,
|
|
157
|
+
categories: list[Any],
|
|
158
|
+
start_value: int,
|
|
159
|
+
unknown_value: int
|
|
160
|
+
) -> pd.Series:
|
|
161
|
+
if not categories:
|
|
162
|
+
return pd.Series(unknown_value, index=series.index, dtype='int64')
|
|
163
|
+
|
|
164
|
+
cat = pd.Categorical(series, categories=categories)
|
|
165
|
+
codes = pd.Series(cat.codes, index=series.index)
|
|
166
|
+
unknown_mask = codes.eq(-1)
|
|
167
|
+
|
|
168
|
+
if start_value:
|
|
169
|
+
codes = codes + start_value
|
|
170
|
+
|
|
171
|
+
if unknown_mask.any():
|
|
172
|
+
codes = codes.astype('int64', copy=False)
|
|
173
|
+
codes.loc[unknown_mask] = int(unknown_value)
|
|
174
|
+
else:
|
|
175
|
+
codes = codes.astype('int64', copy=False)
|
|
176
|
+
|
|
177
|
+
return codes
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""[STEP] Group rare categories into a shared label."""
|
|
2
|
+
import math
|
|
3
|
+
import textwrap
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
from ...actionable import Actionable
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...data_type import DataType
|
|
10
|
+
from ...dataset import Dataset
|
|
11
|
+
from ...decorators.all import is_step
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@is_step('cleaning')
|
|
15
|
+
class ActRareCategoryGrouper(Actionable):
|
|
16
|
+
"""[STEP] Group rare categories into a shared label."""
|
|
17
|
+
|
|
18
|
+
name: str = 'Group rare categories'
|
|
19
|
+
_description: str = textwrap.dedent('''\
|
|
20
|
+
Replace rare categorical values with an "other" label to limit sparsity.''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
Categories whose frequency is below {min_frequency:.0%} of the non-missing values
|
|
23
|
+
or below {min_count} occurrences are grouped into {other_label!r}. This reduces
|
|
24
|
+
the number of distinct categories before downstream encoders are applied.''')
|
|
25
|
+
_usage: str = 'Use when rare labels in categorical features add sparsity and you want grouping instead of ActDropHighCardinalityCategorical. Applicable to categorical or pandas category columns before encoding. Avoid when rare labels are important or you prefer ActDropCategoricalColumn.'
|
|
26
|
+
|
|
27
|
+
def __init__(self) -> None:
|
|
28
|
+
self.configuration = {
|
|
29
|
+
'min_frequency': {
|
|
30
|
+
'description': textwrap.dedent('''\
|
|
31
|
+
Minimum ratio of non-missing rows required to keep a category.'''),
|
|
32
|
+
'default': 0.01
|
|
33
|
+
},
|
|
34
|
+
'min_count': {
|
|
35
|
+
'description': 'Minimum absolute count required to keep a category.',
|
|
36
|
+
'default': 2
|
|
37
|
+
},
|
|
38
|
+
'other_label': {
|
|
39
|
+
'description': 'Label used to replace rare categories.',
|
|
40
|
+
'default': 'other'
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
self.columns: list[str] = []
|
|
44
|
+
self.rare_categories: dict[str, set] = {}
|
|
45
|
+
|
|
46
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
47
|
+
self.columns = self._select_columns(dataset)
|
|
48
|
+
self.rare_categories = {}
|
|
49
|
+
self.explanations = []
|
|
50
|
+
|
|
51
|
+
if not self.columns or dataset.X.empty:
|
|
52
|
+
return self
|
|
53
|
+
|
|
54
|
+
min_count = max(0, int(self.get_config('min_count')))
|
|
55
|
+
min_frequency = max(0.0, float(self.get_config('min_frequency')))
|
|
56
|
+
other_label = self.get_config('other_label')
|
|
57
|
+
|
|
58
|
+
for column in self.columns:
|
|
59
|
+
counts = dataset.X[column].value_counts(dropna=True)
|
|
60
|
+
total = int(counts.sum())
|
|
61
|
+
if total == 0 or counts.empty:
|
|
62
|
+
continue
|
|
63
|
+
|
|
64
|
+
threshold = self._threshold(min_count, min_frequency, total)
|
|
65
|
+
if threshold <= 0:
|
|
66
|
+
continue
|
|
67
|
+
|
|
68
|
+
rare_counts = counts[counts < threshold]
|
|
69
|
+
if rare_counts.empty:
|
|
70
|
+
continue
|
|
71
|
+
|
|
72
|
+
rare_labels = set(rare_counts.index.tolist())
|
|
73
|
+
self.rare_categories[column] = rare_labels
|
|
74
|
+
rare_total = int(rare_counts.sum())
|
|
75
|
+
self.explanations.append(
|
|
76
|
+
f"Grouped {len(rare_labels)} rare categories in `{column}` into "
|
|
77
|
+
f"{other_label!r} ({rare_total} rows)."
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
return self
|
|
81
|
+
|
|
82
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
83
|
+
if not self.rare_categories:
|
|
84
|
+
return X
|
|
85
|
+
|
|
86
|
+
other_label = self.get_config('other_label')
|
|
87
|
+
for column, rare_values in self.rare_categories.items():
|
|
88
|
+
if column not in X.columns or not rare_values:
|
|
89
|
+
continue
|
|
90
|
+
|
|
91
|
+
series = X[column]
|
|
92
|
+
rare_mask = series.isin(rare_values)
|
|
93
|
+
if not rare_mask.any():
|
|
94
|
+
continue
|
|
95
|
+
|
|
96
|
+
if pd.api.types.is_categorical_dtype(series):
|
|
97
|
+
if other_label not in series.cat.categories:
|
|
98
|
+
series = series.cat.add_categories([other_label])
|
|
99
|
+
series = series.where(~rare_mask, other_label)
|
|
100
|
+
X[column] = series
|
|
101
|
+
else:
|
|
102
|
+
X.loc[rare_mask, column] = other_label
|
|
103
|
+
|
|
104
|
+
return X
|
|
105
|
+
|
|
106
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
107
|
+
columns = self._select_columns(dataset)
|
|
108
|
+
if not columns or dataset.X.empty:
|
|
109
|
+
return False
|
|
110
|
+
|
|
111
|
+
min_count = max(0, int(self.get_config('min_count')))
|
|
112
|
+
min_frequency = max(0.0, float(self.get_config('min_frequency')))
|
|
113
|
+
if min_count <= 0 and min_frequency <= 0:
|
|
114
|
+
return False
|
|
115
|
+
|
|
116
|
+
for column in columns:
|
|
117
|
+
counts = dataset.X[column].value_counts(dropna=True)
|
|
118
|
+
total = int(counts.sum())
|
|
119
|
+
if total == 0 or counts.empty:
|
|
120
|
+
continue
|
|
121
|
+
|
|
122
|
+
threshold = self._threshold(min_count, min_frequency, total)
|
|
123
|
+
if threshold > 0 and (counts < threshold).any():
|
|
124
|
+
return True
|
|
125
|
+
|
|
126
|
+
return False
|
|
127
|
+
|
|
128
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
129
|
+
if candidate is None or candidate.dataset.X.empty:
|
|
130
|
+
return 0.0
|
|
131
|
+
|
|
132
|
+
columns = self._select_columns(candidate.dataset)
|
|
133
|
+
if not columns:
|
|
134
|
+
return 0.0
|
|
135
|
+
|
|
136
|
+
min_count = max(0, int(self.get_config('min_count')))
|
|
137
|
+
min_frequency = max(0.0, float(self.get_config('min_frequency')))
|
|
138
|
+
|
|
139
|
+
total_values = 0
|
|
140
|
+
rare_values = 0
|
|
141
|
+
for column in columns:
|
|
142
|
+
counts = candidate.dataset.X[column].value_counts(dropna=True)
|
|
143
|
+
total = int(counts.sum())
|
|
144
|
+
if total == 0 or counts.empty:
|
|
145
|
+
continue
|
|
146
|
+
threshold = self._threshold(min_count, min_frequency, total)
|
|
147
|
+
if threshold <= 0:
|
|
148
|
+
continue
|
|
149
|
+
rare_counts = counts[counts < threshold]
|
|
150
|
+
total_values += total
|
|
151
|
+
rare_values += int(rare_counts.sum())
|
|
152
|
+
|
|
153
|
+
if total_values == 0:
|
|
154
|
+
return 0.0
|
|
155
|
+
|
|
156
|
+
return min(1.0, max(0.0, rare_values / total_values))
|
|
157
|
+
|
|
158
|
+
def _threshold(self, min_count: int, min_frequency: float, total: int) -> int:
|
|
159
|
+
ratio_threshold = 0
|
|
160
|
+
if min_frequency > 0 and total > 0:
|
|
161
|
+
ratio_threshold = int(math.ceil(min_frequency * total))
|
|
162
|
+
return max(min_count, ratio_threshold)
|
|
163
|
+
|
|
164
|
+
def _select_columns(self, dataset: Dataset) -> list[str]:
|
|
165
|
+
columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
166
|
+
category_columns = list(dataset.X.select_dtypes(include=['category']).columns)
|
|
167
|
+
seen = set()
|
|
168
|
+
ordered = []
|
|
169
|
+
for column in columns + category_columns:
|
|
170
|
+
if column in dataset.X.columns and column not in seen:
|
|
171
|
+
ordered.append(column)
|
|
172
|
+
seen.add(column)
|
|
173
|
+
return ordered
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""[STEP] Simple imputer dedicated to minimalist pipelines."""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
from ...actionable import Actionable
|
|
8
|
+
from ...dataset import Dataset
|
|
9
|
+
from ...candidate import Candidate
|
|
10
|
+
from ...decorators.all import is_step
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@is_step('minimal_preprocessing')
|
|
14
|
+
class ActSimpleImputer(Actionable):
|
|
15
|
+
"""[STEP] Lightweight imputer that fills NaNs with mean/mode values."""
|
|
16
|
+
|
|
17
|
+
name: str = 'Simple Imputer'
|
|
18
|
+
_usage: str = 'Use when NaNs must be filled quickly for minimal models; if dropping is preferred consider ActDropNumericalColumn. Applicable to mixed numeric and categorical tables with missing values. Avoid when missingness is informative or you plan richer imputation such as ActCategoricalImputer.'
|
|
19
|
+
_description: str = textwrap.dedent('''\
|
|
20
|
+
Fill missing numeric values with the column mean and categorical values with the
|
|
21
|
+
most frequent value so minimalist predictors can run without preprocessing.''')
|
|
22
|
+
_description_long: str = textwrap.dedent('''\
|
|
23
|
+
This step mirrors a classical SimpleImputer: each numeric feature is imputed with
|
|
24
|
+
its mean (fallback to 0 when undefined) while non-numeric columns are imputed with
|
|
25
|
+
their most frequent value. It is primarily used to make minimalist predictors work
|
|
26
|
+
on raw datasets that still contain NaNs.''')
|
|
27
|
+
can_be_disabled: bool = False
|
|
28
|
+
|
|
29
|
+
def __init__(self) -> None:
|
|
30
|
+
self.numeric_fill_values: dict[str, float] = {}
|
|
31
|
+
self.categorical_fill_values: dict[str, Any] = {}
|
|
32
|
+
|
|
33
|
+
def fit(self, dataset: Dataset) -> 'ActSimpleImputer':
|
|
34
|
+
X = dataset.X
|
|
35
|
+
|
|
36
|
+
self.numeric_fill_values = {}
|
|
37
|
+
self.categorical_fill_values = {}
|
|
38
|
+
self.explanations = []
|
|
39
|
+
|
|
40
|
+
numeric_cols = X.select_dtypes(include='number').columns
|
|
41
|
+
for column in numeric_cols:
|
|
42
|
+
series = X[column]
|
|
43
|
+
missing = series.isna().sum()
|
|
44
|
+
|
|
45
|
+
fill_value = series.mean()
|
|
46
|
+
if pd.isna(fill_value):
|
|
47
|
+
fill_value = 0.0
|
|
48
|
+
|
|
49
|
+
self.numeric_fill_values[column] = float(fill_value)
|
|
50
|
+
if missing > 0:
|
|
51
|
+
self.explanations.append(
|
|
52
|
+
f"Filled {missing} numeric values in `{column}` with {fill_value:.4f}."
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
categorical_cols = X.select_dtypes(exclude='number').columns
|
|
56
|
+
for column in categorical_cols:
|
|
57
|
+
series = X[column]
|
|
58
|
+
missing = series.isna().sum()
|
|
59
|
+
|
|
60
|
+
mode_series = series.mode(dropna=True)
|
|
61
|
+
if not mode_series.empty:
|
|
62
|
+
fill_value = mode_series.iloc[0]
|
|
63
|
+
elif pd.api.types.is_bool_dtype(series):
|
|
64
|
+
fill_value = False
|
|
65
|
+
else:
|
|
66
|
+
fill_value = ''
|
|
67
|
+
|
|
68
|
+
self.categorical_fill_values[column] = fill_value
|
|
69
|
+
if missing > 0:
|
|
70
|
+
self.explanations.append(
|
|
71
|
+
f"Filled {missing} categorical values in `{column}` with {fill_value!r}."
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
return self
|
|
75
|
+
|
|
76
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
77
|
+
if self.numeric_fill_values:
|
|
78
|
+
for column, fill_value in self.numeric_fill_values.items():
|
|
79
|
+
if column in X.columns:
|
|
80
|
+
X[column] = X[column].fillna(fill_value)
|
|
81
|
+
|
|
82
|
+
if self.categorical_fill_values:
|
|
83
|
+
for column, fill_value in self.categorical_fill_values.items():
|
|
84
|
+
if column in X.columns:
|
|
85
|
+
X[column] = X[column].fillna(fill_value)
|
|
86
|
+
|
|
87
|
+
# As a last resort, replace any remaining NaNs to avoid downstream crashes.
|
|
88
|
+
if X.isna().any().any():
|
|
89
|
+
numeric_cols = X.select_dtypes(include='number').columns
|
|
90
|
+
if len(numeric_cols) > 0:
|
|
91
|
+
X[numeric_cols] = X[numeric_cols].fillna(0.0)
|
|
92
|
+
categorical_cols = X.select_dtypes(exclude='number').columns
|
|
93
|
+
if len(categorical_cols) > 0:
|
|
94
|
+
bool_cols = [c for c in categorical_cols if pd.api.types.is_bool_dtype(X[c])]
|
|
95
|
+
other_cols = [c for c in categorical_cols if c not in bool_cols]
|
|
96
|
+
if bool_cols:
|
|
97
|
+
X[bool_cols] = X[bool_cols].fillna(False)
|
|
98
|
+
if other_cols:
|
|
99
|
+
X[other_cols] = X[other_cols].fillna('')
|
|
100
|
+
|
|
101
|
+
return X
|
|
102
|
+
|
|
103
|
+
def priorize(self, candidate: Candidate = None) -> float: # pylint: disable=unused-argument
|
|
104
|
+
if candidate is None or candidate.dataset.X.empty:
|
|
105
|
+
return 0.0
|
|
106
|
+
|
|
107
|
+
missing = candidate.dataset.X.isna().sum().sum()
|
|
108
|
+
total = candidate.dataset.X.size or 1
|
|
109
|
+
return min(1.0, missing / total)
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""[STEP] Transform string column to date"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import numpy as np
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...data_type import DataType
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('cleaning')
|
|
13
|
+
class ActSplitDate(Actionable):
|
|
14
|
+
"""[STEP] Transform string column to date"""
|
|
15
|
+
|
|
16
|
+
name: str = 'Create Date Elements columns'
|
|
17
|
+
_usage: str = "Use when you want to expand a date column into components rather than ActDropDateColumn. Applicable to date-typed columns with usable timestamps. Avoid when dates are already split or you plan to drop them via ActDropDateColumn."
|
|
18
|
+
_descrption: str = textwrap.dedent('''\
|
|
19
|
+
Transform a textual date column into multiple columns
|
|
20
|
+
for day, month, year, hour, minute, second''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
Transform a textual date column into multiple columns for
|
|
23
|
+
day, month, year, hour, minute, second.
|
|
24
|
+
Exemple:
|
|
25
|
+
+-------------------+----------+------------+-----------+-----------+----------+----------+
|
|
26
|
+
|date | date_day | date_month | date_year | date_hour | date_min | date_sec |
|
|
27
|
+
+-------------------+----------+------------+-----------+-----------+----------+----------+
|
|
28
|
+
|2024-01-15 12:31:27| 15 | 01 | 2024 | 12 | 31 | 27 |
|
|
29
|
+
+-------------------+----------+------------+-----------+-----------+----------+----------+
|
|
30
|
+
''')
|
|
31
|
+
|
|
32
|
+
def __init__(self):
|
|
33
|
+
self.columns: list[str] = None
|
|
34
|
+
|
|
35
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
36
|
+
self.columns = dataset.get_columns_names_by_type(DataType.DATE)
|
|
37
|
+
|
|
38
|
+
self.explanations = [
|
|
39
|
+
f'Split date column **`{c}`** into year, month, weekday, hour, minute and second.'
|
|
40
|
+
for c in self.columns
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
return self
|
|
44
|
+
|
|
45
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
46
|
+
"""Split dates columns into columns (weekday, mount, year, hour, minute, second).
|
|
47
|
+
|
|
48
|
+
:param pd.DataFrame x: DataFrame to transform.
|
|
49
|
+
:return: Transformed dataset.
|
|
50
|
+
"""
|
|
51
|
+
for column in self.columns:
|
|
52
|
+
# Day
|
|
53
|
+
X[column + '_weekday'] = X[column].dt.dayofweek.replace(np.NaN, -1)
|
|
54
|
+
X[column + '_month'] = X[column].dt.month.replace(np.NaN, -1)
|
|
55
|
+
X[column + '_year'] = X[column].dt.year.replace(np.NaN, -1)
|
|
56
|
+
|
|
57
|
+
# Hour
|
|
58
|
+
X[column + '_hour'] = X[column].dt.hour.replace(np.NaN, -1)
|
|
59
|
+
X[column + '_minute'] = X[column].dt.minute.replace(np.NaN, -1)
|
|
60
|
+
X[column + '_second'] = X[column].dt.second.replace(np.NaN, -1)
|
|
61
|
+
|
|
62
|
+
return X
|
|
63
|
+
|
|
64
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
65
|
+
return bool(dataset.get_columns_names_by_type(DataType.DATE))
|
|
66
|
+
|
|
67
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
68
|
+
return 0.5
|