PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
iaml/__init__.py
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Integrated AutoML for Medical Labs (IAML).
|
|
2
|
+
|
|
3
|
+
IAML helps clinical research teams build, evaluate and inspect machine learning
|
|
4
|
+
pipelines for tabular classification, regression and survival analysis. Modular
|
|
5
|
+
preprocessing and modeling steps support automated search, while pipeline
|
|
6
|
+
descriptions, evaluation metrics and SHAP explanations support review of the
|
|
7
|
+
resulting models and reporting of research methods.
|
|
8
|
+
"""
|
|
9
|
+
from .iaml import IAML
|
|
10
|
+
from .core_dispatcher import CoreDispatcher
|
|
11
|
+
from .step import *
|
|
12
|
+
from .metastep import MetaStep
|
|
13
|
+
from .actionable import Actionable
|
|
14
|
+
from .candidate import Candidate
|
|
15
|
+
from .dataset import Dataset
|
|
16
|
+
from .data_type import DataType
|
|
17
|
+
from .metric_plot import MetricPlot
|
|
18
|
+
from .metric import Metric
|
|
19
|
+
from .plot import Plot, StatisticPlot
|
|
20
|
+
from .statistic import Statistic
|
|
21
|
+
from .cache import Cache
|
|
22
|
+
from .void_step import VoidStep
|
|
23
|
+
|
|
24
|
+
from .meta_ordered_step import MetaOrderedStep
|
|
25
|
+
from .meta_explorer_step import MetaExplorerStep
|
|
26
|
+
from .meta_partial_explorer_step import MetaPartialExplorerStep
|
|
27
|
+
|
|
28
|
+
# Default Actionables
|
|
29
|
+
from .actionables import *
|
|
30
|
+
|
|
31
|
+
# Wrappers
|
|
32
|
+
from .wrapper import *
|
|
33
|
+
|
|
34
|
+
# Metrics
|
|
35
|
+
from .metrics import *
|
|
36
|
+
|
|
37
|
+
# Statistics
|
|
38
|
+
from .statistics import *
|
|
39
|
+
|
|
40
|
+
# Plots
|
|
41
|
+
from .plots import *
|
|
42
|
+
|
|
43
|
+
# Stack
|
|
44
|
+
from .stack import Stack
|
|
45
|
+
|
|
46
|
+
# Optimizer
|
|
47
|
+
from .optimizers import *
|
|
48
|
+
|
|
49
|
+
# Type of target
|
|
50
|
+
from .type_of_target import type_of_target
|
|
51
|
+
|
|
52
|
+
# Ensure star imports expose all public names, even if __all__ is set elsewhere.
|
|
53
|
+
__all__ = [
|
|
54
|
+
name for name in globals()
|
|
55
|
+
if not name.startswith("_") and name != "__all__"
|
|
56
|
+
]
|
iaml/actionable.py
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""Classic kind of Step that transform, resample or predict from Candidate"""
|
|
2
|
+
|
|
3
|
+
# -> Must be a wildcard import to help IAML to know all available the steps
|
|
4
|
+
from .step import * # pylint: disable=unused-wildcard-import,wildcard-import
|
|
5
|
+
from .decorators.all import is_step
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@is_step('actionable')
|
|
9
|
+
class Actionable(Step):
|
|
10
|
+
"""Classic kind of Step that transform, resample or predict from Candidate"""
|
|
11
|
+
_usage = "Use when you need a concrete transform/resample/predict step over a Candidate. Applicable to pipeline steps that operate directly on Candidate data. Avoid when you need meta orchestration like MetaStep or wrappers like StepWrapper."
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""
|
|
2
|
+
All Step with transform, resample or predict function
|
|
3
|
+
"""
|
|
4
|
+
from . import (
|
|
5
|
+
boosting,
|
|
6
|
+
predictors,
|
|
7
|
+
features_selection,
|
|
8
|
+
cleaning,
|
|
9
|
+
normalize,
|
|
10
|
+
features_precleaning,
|
|
11
|
+
features_preprocessing,
|
|
12
|
+
imbalance,
|
|
13
|
+
)
|
|
14
|
+
from .boosting import *
|
|
15
|
+
from .predictors import *
|
|
16
|
+
from .features_selection import *
|
|
17
|
+
from .cleaning import *
|
|
18
|
+
from .normalize import *
|
|
19
|
+
from .features_precleaning import *
|
|
20
|
+
from .features_preprocessing import *
|
|
21
|
+
from .imbalance import *
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Apply AdaBoost on models"""
|
|
2
|
+
from typing import Any
|
|
3
|
+
from sklearn.ensemble import AdaBoostClassifier
|
|
4
|
+
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...decorators.all import runner
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ActAdaBoost(Actionable):
|
|
11
|
+
"""Apply Adaboost on models
|
|
12
|
+
|
|
13
|
+
Configuration:
|
|
14
|
+
* `random_state`: Random seed (defaults to 42).
|
|
15
|
+
* `n_estimator`: Number of estimators (defaults to 2000).
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
name: str = "AdaBoost Classifier"
|
|
19
|
+
refs: list[dict[str, Any]] = [
|
|
20
|
+
{
|
|
21
|
+
'year': 1995,
|
|
22
|
+
'name': (
|
|
23
|
+
'A desicion-theoretic generalization of on-line learning'
|
|
24
|
+
'and an application to boosting'
|
|
25
|
+
),
|
|
26
|
+
'authors': [
|
|
27
|
+
'Yoav Freund',
|
|
28
|
+
'Robert E. Schapire'
|
|
29
|
+
],
|
|
30
|
+
'doi': 'https://doi.org/10.1007/3-540-59119-2_166',
|
|
31
|
+
'publisher': 'Springer, Berlin, Heidelberg'
|
|
32
|
+
}
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
def __init__(self):
|
|
36
|
+
self.configuration = {
|
|
37
|
+
'random_state': {
|
|
38
|
+
'description': 'random_state',
|
|
39
|
+
'default': 42
|
|
40
|
+
},
|
|
41
|
+
'n_estimator': {
|
|
42
|
+
'description': 'Number of estimators',
|
|
43
|
+
'default': 2000
|
|
44
|
+
},
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
@runner
|
|
48
|
+
def run(self, candidate: Candidate) -> Candidate:
|
|
49
|
+
model = AdaBoostClassifier(
|
|
50
|
+
candidate.model,
|
|
51
|
+
n_estimators=self.get_config('n_estimator'),
|
|
52
|
+
random_state=self.get_config('random_state'))
|
|
53
|
+
|
|
54
|
+
model.fit(candidate.dataset.X, candidate.dataset.y)
|
|
55
|
+
|
|
56
|
+
return candidate.to_output(None, None, model)
|
|
57
|
+
|
|
58
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
59
|
+
return 0.5
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""
|
|
2
|
+
All cleaning actionables
|
|
3
|
+
"""
|
|
4
|
+
from .act_mean_column import ActMeanColumn
|
|
5
|
+
from .act_drop_numerical_column import ActDropNumericalColumn
|
|
6
|
+
from .act_drop_textual_column import ActDropTextualColumn
|
|
7
|
+
from .act_drop_categorical_column import ActDropCategoricalColumn
|
|
8
|
+
from .act_onehot import ActOnehot
|
|
9
|
+
from .act_tf_idf import ActTfIdf
|
|
10
|
+
from .act_drop_date_column import ActDropDateColumn
|
|
11
|
+
from .act_split_date import ActSplitDate
|
|
12
|
+
from .act_word2vec import ActWord2Vec
|
|
13
|
+
from .act_mice import ActMICEForestImputer
|
|
14
|
+
from .act_simple_imputer import ActSimpleImputer
|
|
15
|
+
from .act_knn_imputer import ActKNNImputer
|
|
16
|
+
from .act_categorical_imputer import ActCategoricalImputer
|
|
17
|
+
from .act_rare_category_grouper import ActRareCategoryGrouper
|
|
18
|
+
from .act_frequency_encoder import ActFrequencyEncoder
|
|
19
|
+
from .act_target_encoder import ActTargetEncoder
|
|
20
|
+
from .act_ordinal_encoder import ActOrdinalEncoder
|
|
21
|
+
from .act_count_vectorizer import ActCountVectorizer
|
|
22
|
+
from .act_hashing_vectorizer import ActHashingVectorizer
|
|
23
|
+
from .act_text_normalizer import ActTextNormalizer
|
|
24
|
+
from .act_missing_indicator import ActMissingIndicator
|
|
25
|
+
from .act_missing_count_feature import ActMissingCountFeature
|
|
26
|
+
from .act_drop_high_cardinality_categorical import ActDropHighCardinalityCategorical
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""[STEP] Impute missing categorical values."""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
from ...actionable import Actionable
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...data_type import DataType
|
|
10
|
+
from ...dataset import Dataset
|
|
11
|
+
from ...decorators.all import is_step
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@is_step('cleaning')
|
|
15
|
+
class ActCategoricalImputer(Actionable):
|
|
16
|
+
"""[STEP] Impute missing categorical values."""
|
|
17
|
+
|
|
18
|
+
name: str = 'Impute missing categorical values'
|
|
19
|
+
_usage: str = 'Use when categorical columns have missing labels you want to keep (vs ActDropCategoricalColumn). Applicable to categorical/category dtype features with NA gaps. Avoid when missingness is extreme or cardinality is high; consider ActDropHighCardinalityCategorical.'
|
|
20
|
+
_description: str = textwrap.dedent('''\
|
|
21
|
+
Impute missing categorical values using the {strategy} strategy.''')
|
|
22
|
+
_description_long: str = textwrap.dedent('''\
|
|
23
|
+
Replace missing values in categorical columns with either the most frequent
|
|
24
|
+
observed category or a constant "missing" label. The label can be customized
|
|
25
|
+
via missing_label and is also used as a fallback when no mode can be computed.''')
|
|
26
|
+
|
|
27
|
+
def __init__(self) -> None:
|
|
28
|
+
self.columns: list[str] = []
|
|
29
|
+
self.fill_values: dict[str, Any] = {}
|
|
30
|
+
self._missing_stats: dict[str, tuple[int, int, float]] = {}
|
|
31
|
+
|
|
32
|
+
self.configuration = {
|
|
33
|
+
'strategy': {
|
|
34
|
+
'description': 'Imputation strategy for categorical columns.',
|
|
35
|
+
'default': 'most_frequent',
|
|
36
|
+
'categorical': ['most_frequent', 'missing']
|
|
37
|
+
},
|
|
38
|
+
'missing_label': {
|
|
39
|
+
'description': 'Label used when strategy="missing" or when no mode exists.',
|
|
40
|
+
'default': 'missing'
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
45
|
+
self.columns = self._select_columns(dataset)
|
|
46
|
+
self.fill_values = {}
|
|
47
|
+
self._missing_stats = {}
|
|
48
|
+
self.explanations = []
|
|
49
|
+
|
|
50
|
+
if not self.columns or dataset.X.empty:
|
|
51
|
+
return self
|
|
52
|
+
|
|
53
|
+
X_cat = dataset.X[self.columns]
|
|
54
|
+
total_rows = len(X_cat)
|
|
55
|
+
if total_rows == 0:
|
|
56
|
+
return self
|
|
57
|
+
|
|
58
|
+
missing_counts = X_cat.isna().sum()
|
|
59
|
+
strategy = self.get_config('strategy')
|
|
60
|
+
missing_label = self.get_config('missing_label')
|
|
61
|
+
|
|
62
|
+
for column in self.columns:
|
|
63
|
+
series = X_cat[column]
|
|
64
|
+
missing = int(missing_counts[column])
|
|
65
|
+
pct = (missing / total_rows * 100.0) if total_rows else 0.0
|
|
66
|
+
self._missing_stats[column] = (missing, total_rows, pct)
|
|
67
|
+
|
|
68
|
+
if strategy == 'missing':
|
|
69
|
+
fill_value = missing_label
|
|
70
|
+
else:
|
|
71
|
+
mode_series = series.mode(dropna=True)
|
|
72
|
+
if not mode_series.empty:
|
|
73
|
+
fill_value = mode_series.iloc[0]
|
|
74
|
+
else:
|
|
75
|
+
fill_value = missing_label
|
|
76
|
+
|
|
77
|
+
self.fill_values[column] = fill_value
|
|
78
|
+
if missing > 0:
|
|
79
|
+
self.explanations.append(
|
|
80
|
+
f"Filled {missing} missing values in `{column}` with {fill_value!r}."
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
return self
|
|
84
|
+
|
|
85
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
86
|
+
if not self.columns or not self.fill_values:
|
|
87
|
+
return X
|
|
88
|
+
|
|
89
|
+
for column, fill_value in self.fill_values.items():
|
|
90
|
+
if column not in X.columns:
|
|
91
|
+
continue
|
|
92
|
+
if pd.api.types.is_categorical_dtype(X[column]):
|
|
93
|
+
if fill_value not in X[column].cat.categories:
|
|
94
|
+
X[column] = X[column].cat.add_categories([fill_value])
|
|
95
|
+
X[column] = X[column].fillna(fill_value)
|
|
96
|
+
|
|
97
|
+
return X
|
|
98
|
+
|
|
99
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
100
|
+
columns = self._select_columns(dataset)
|
|
101
|
+
if not columns or dataset.X.empty:
|
|
102
|
+
return False
|
|
103
|
+
return bool(dataset.X[columns].isna().any().any())
|
|
104
|
+
|
|
105
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
106
|
+
if candidate is None or candidate.dataset.X.empty:
|
|
107
|
+
return 0.0
|
|
108
|
+
columns = self._select_columns(candidate.dataset)
|
|
109
|
+
if not columns:
|
|
110
|
+
return 0.0
|
|
111
|
+
missing = candidate.dataset.X[columns].isna().sum().sum()
|
|
112
|
+
total = candidate.dataset.X[columns].size or 1
|
|
113
|
+
return min(1.0, missing / total)
|
|
114
|
+
|
|
115
|
+
def _select_columns(self, dataset: Dataset) -> list[str]:
|
|
116
|
+
columns = dataset.get_columns_names_by_type(DataType.CATEGORICAL)
|
|
117
|
+
category_columns = list(dataset.X.select_dtypes(include=['category']).columns)
|
|
118
|
+
seen = set()
|
|
119
|
+
ordered = []
|
|
120
|
+
for column in columns + category_columns:
|
|
121
|
+
if column in dataset.X.columns and column not in seen:
|
|
122
|
+
ordered.append(column)
|
|
123
|
+
seen.add(column)
|
|
124
|
+
return ordered
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""[STEP] Vectorize short text with CountVectorizer."""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
from sklearn.feature_extraction.text import CountVectorizer
|
|
7
|
+
|
|
8
|
+
from ...actionable import Actionable
|
|
9
|
+
from ...candidate import Candidate
|
|
10
|
+
from ...data_type import DataType
|
|
11
|
+
from ...dataset import Dataset
|
|
12
|
+
from ...decorators.all import is_step
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@is_step('cleaning')
|
|
16
|
+
class ActCountVectorizer(Actionable):
|
|
17
|
+
"""[STEP] Vectorize short text with CountVectorizer."""
|
|
18
|
+
|
|
19
|
+
name: str = 'Count Vectorizer'
|
|
20
|
+
_usage: str = 'Use when short text is predictive and you want bag-of-words counts instead of ActDropTextualColumn. Applicable to short text columns with a manageable vocabulary. Avoid when text is long, extremely sparse, or you would drop text entirely (ActDropTextualColumn).'
|
|
21
|
+
_description: str = textwrap.dedent('''\
|
|
22
|
+
Vectorize short text columns into bag-of-words counts with n-grams.''')
|
|
23
|
+
_description_long: str = textwrap.dedent('''\
|
|
24
|
+
Build a vocabulary on each short text column and replace it with count features
|
|
25
|
+
for each token or n-gram observed in the training data. Adjust the n-gram range
|
|
26
|
+
and vocabulary size to control sparsity and keep runtime manageable.''')
|
|
27
|
+
|
|
28
|
+
def __init__(self) -> None:
|
|
29
|
+
self.columns: list[tuple[str, CountVectorizer]] = []
|
|
30
|
+
self.configuration = {
|
|
31
|
+
'ngram_min': {
|
|
32
|
+
'description': 'Minimum n-gram size to include.',
|
|
33
|
+
'default': 1
|
|
34
|
+
},
|
|
35
|
+
'ngram_max': {
|
|
36
|
+
'description': 'Maximum n-gram size to include.',
|
|
37
|
+
'default': 2
|
|
38
|
+
},
|
|
39
|
+
'max_features': {
|
|
40
|
+
'description': 'Maximum size of the vocabulary (None keeps all).',
|
|
41
|
+
'default': 2000
|
|
42
|
+
},
|
|
43
|
+
'min_df': {
|
|
44
|
+
'description': 'Minimum document frequency for a term to be kept.',
|
|
45
|
+
'default': 1
|
|
46
|
+
},
|
|
47
|
+
'max_df': {
|
|
48
|
+
'description': 'Maximum document frequency for a term to be kept.',
|
|
49
|
+
'default': 1.0
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
54
|
+
self.columns = []
|
|
55
|
+
self.explanations = []
|
|
56
|
+
|
|
57
|
+
columns = dataset.get_columns_names_by_type([DataType.SHORT_TEXT])
|
|
58
|
+
if not columns or dataset.X.empty:
|
|
59
|
+
return self
|
|
60
|
+
|
|
61
|
+
params = self._build_vectorizer_params(len(dataset.X))
|
|
62
|
+
|
|
63
|
+
for column in columns:
|
|
64
|
+
values = dataset.X[column].fillna('').astype(str)
|
|
65
|
+
vectorizer = CountVectorizer(**params)
|
|
66
|
+
try:
|
|
67
|
+
vectorizer.fit(values)
|
|
68
|
+
except ValueError:
|
|
69
|
+
continue
|
|
70
|
+
|
|
71
|
+
feature_names = vectorizer.get_feature_names_out()
|
|
72
|
+
if len(feature_names) == 0:
|
|
73
|
+
continue
|
|
74
|
+
|
|
75
|
+
self.columns.append((column, vectorizer))
|
|
76
|
+
self.explanations.append(
|
|
77
|
+
f'Encoded text column **`{column}`** into **{len(feature_names)}** count features.'
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
if not self.columns and columns:
|
|
81
|
+
raise RuntimeError(
|
|
82
|
+
"Count vectorizer failed: no usable vocabulary in short text columns."
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
return self
|
|
86
|
+
|
|
87
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
88
|
+
"""Apply CountVectorizer to short text columns.
|
|
89
|
+
|
|
90
|
+
:param pd.DataFrame X: DataFrame to transform.
|
|
91
|
+
:return: Transformed dataset.
|
|
92
|
+
"""
|
|
93
|
+
if not self.columns:
|
|
94
|
+
return X
|
|
95
|
+
|
|
96
|
+
X = X.reset_index(drop=True)
|
|
97
|
+
|
|
98
|
+
for name, vectorizer in self.columns:
|
|
99
|
+
if name not in X.columns:
|
|
100
|
+
continue
|
|
101
|
+
|
|
102
|
+
values = X[name].fillna('').astype(str)
|
|
103
|
+
transformed = vectorizer.transform(values)
|
|
104
|
+
feature_names = vectorizer.get_feature_names_out()
|
|
105
|
+
if len(feature_names) == 0:
|
|
106
|
+
X = X.drop([name], axis=1)
|
|
107
|
+
continue
|
|
108
|
+
|
|
109
|
+
new_names = [f"{name}_{token}" for token in feature_names]
|
|
110
|
+
vector_df = pd.DataFrame(transformed.toarray(), columns=new_names)
|
|
111
|
+
|
|
112
|
+
X = pd.concat([X, vector_df], axis=1).drop([name], axis=1)
|
|
113
|
+
|
|
114
|
+
return X
|
|
115
|
+
|
|
116
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
117
|
+
if dataset.X.empty:
|
|
118
|
+
return False
|
|
119
|
+
return bool(dataset.get_columns_names_by_type([DataType.SHORT_TEXT]))
|
|
120
|
+
|
|
121
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
122
|
+
if candidate is None or candidate.dataset.X.empty:
|
|
123
|
+
return 0.0
|
|
124
|
+
|
|
125
|
+
columns = candidate.dataset.get_columns_names_by_type([DataType.SHORT_TEXT])
|
|
126
|
+
if not columns:
|
|
127
|
+
return 0.0
|
|
128
|
+
|
|
129
|
+
total_columns = candidate.dataset.X.shape[1] or 1
|
|
130
|
+
return min(1.0, len(columns) / total_columns)
|
|
131
|
+
|
|
132
|
+
def _build_vectorizer_params(self, n_rows: int) -> dict[str, Any]:
|
|
133
|
+
ngram_min = self._coerce_int(self.get_config('ngram_min'), 1)
|
|
134
|
+
ngram_max = self._coerce_int(self.get_config('ngram_max'), max(ngram_min, 1))
|
|
135
|
+
ngram_min = max(1, ngram_min)
|
|
136
|
+
ngram_max = max(ngram_min, ngram_max)
|
|
137
|
+
|
|
138
|
+
min_df = self._coerce_df(self.get_config('min_df'), 1)
|
|
139
|
+
max_df = self._coerce_df(self.get_config('max_df'), 1.0)
|
|
140
|
+
|
|
141
|
+
if n_rows > 0:
|
|
142
|
+
min_df = self._clamp_df(min_df, n_rows)
|
|
143
|
+
max_df = self._clamp_df(max_df, n_rows)
|
|
144
|
+
if self._effective_df(min_df, n_rows) > self._effective_df(max_df, n_rows):
|
|
145
|
+
max_df = min_df
|
|
146
|
+
|
|
147
|
+
max_features = self._coerce_optional_int(self.get_config('max_features'))
|
|
148
|
+
|
|
149
|
+
params = {
|
|
150
|
+
'ngram_range': (ngram_min, ngram_max),
|
|
151
|
+
'min_df': min_df,
|
|
152
|
+
'max_df': max_df
|
|
153
|
+
}
|
|
154
|
+
if max_features is not None:
|
|
155
|
+
params['max_features'] = max_features
|
|
156
|
+
|
|
157
|
+
return params
|
|
158
|
+
|
|
159
|
+
@staticmethod
|
|
160
|
+
def _coerce_int(value: Any, default: int) -> int:
|
|
161
|
+
try:
|
|
162
|
+
return int(value)
|
|
163
|
+
except (TypeError, ValueError):
|
|
164
|
+
return default
|
|
165
|
+
|
|
166
|
+
@staticmethod
|
|
167
|
+
def _coerce_optional_int(value: Any) -> int | None:
|
|
168
|
+
if value is None:
|
|
169
|
+
return None
|
|
170
|
+
try:
|
|
171
|
+
numeric = int(value)
|
|
172
|
+
except (TypeError, ValueError):
|
|
173
|
+
return None
|
|
174
|
+
if numeric <= 0:
|
|
175
|
+
return None
|
|
176
|
+
return numeric
|
|
177
|
+
|
|
178
|
+
@staticmethod
|
|
179
|
+
def _coerce_df(value: Any, default: float | int) -> float | int:
|
|
180
|
+
if value is None or isinstance(value, bool):
|
|
181
|
+
return default
|
|
182
|
+
if isinstance(value, int):
|
|
183
|
+
return max(0, value)
|
|
184
|
+
try:
|
|
185
|
+
numeric = float(value)
|
|
186
|
+
except (TypeError, ValueError):
|
|
187
|
+
return default
|
|
188
|
+
if numeric < 0:
|
|
189
|
+
return default
|
|
190
|
+
if numeric <= 1.0:
|
|
191
|
+
return float(numeric)
|
|
192
|
+
return int(round(numeric))
|
|
193
|
+
|
|
194
|
+
@staticmethod
|
|
195
|
+
def _clamp_df(value: float | int, n_rows: int) -> float | int:
|
|
196
|
+
if isinstance(value, float):
|
|
197
|
+
return min(max(value, 0.0), 1.0)
|
|
198
|
+
return min(max(value, 0), n_rows)
|
|
199
|
+
|
|
200
|
+
@staticmethod
|
|
201
|
+
def _effective_df(value: float | int, n_rows: int) -> float:
|
|
202
|
+
if isinstance(value, float):
|
|
203
|
+
return value * n_rows
|
|
204
|
+
return float(value)
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""[STEP] Drop categorical columns"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from ...actionable import Actionable
|
|
5
|
+
from ...dataset import Dataset
|
|
6
|
+
from ...data_type import DataType
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@is_step('cleaning', 'baseline_cleaning')
|
|
12
|
+
class ActDropCategoricalColumn(Actionable):
|
|
13
|
+
"""[STEP] Drop categorical columns"""
|
|
14
|
+
|
|
15
|
+
name: str = 'Remove categorical columns'
|
|
16
|
+
_description: str = 'Remove all columns containing categorical data from the dataset'
|
|
17
|
+
_usage: str = 'Use when categorical columns must be removed for steps that cannot handle categories. Applicable to datasets with categorical or category-typed columns. Avoid when you can impute or encode categories instead (ActCategoricalImputer, ActCountVectorizer).'
|
|
18
|
+
_description_long: str = textwrap.dedent('''\
|
|
19
|
+
Remove all columns containing categorical data from the dataset.
|
|
20
|
+
This step is used to clean the dataset in order to perform other actions later on
|
|
21
|
+
that can't be applied to categorical columns.''')
|
|
22
|
+
can_be_disabled: bool = False
|
|
23
|
+
|
|
24
|
+
def __init__(self):
|
|
25
|
+
self.columns_to_drop: list[str] = None
|
|
26
|
+
|
|
27
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
28
|
+
self.columns_to_drop = list(set(
|
|
29
|
+
dataset.get_columns_names_by_type([DataType.CATEGORICAL]) + \
|
|
30
|
+
list(dataset.X.select_dtypes(include=['category']).columns)
|
|
31
|
+
))
|
|
32
|
+
|
|
33
|
+
self.explanations = [
|
|
34
|
+
f'Dropped column **`{c}`**.' for c in self.columns_to_drop
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
return self
|
|
38
|
+
|
|
39
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
40
|
+
"""Drop columns.
|
|
41
|
+
|
|
42
|
+
:param pd.DataFrame X: DataFrame to transform.
|
|
43
|
+
:return: Transformed DataFrame.
|
|
44
|
+
"""
|
|
45
|
+
return X.drop(self.columns_to_drop, axis=1)
|
|
46
|
+
|
|
47
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
48
|
+
return 0
|
|
49
|
+
|
|
50
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
51
|
+
return bool(dataset.get_columns_names_by_type([DataType.CATEGORICAL]))
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""[STEP] Find and drop date column"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from ...actionable import Actionable
|
|
5
|
+
from ...data_type import DataType
|
|
6
|
+
from ...candidate import Candidate
|
|
7
|
+
from ...dataset import Dataset
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@is_step('cleaning', 'baseline_cleaning')
|
|
12
|
+
class ActDropDateColumn(Actionable):
|
|
13
|
+
"""Finds and drops date columns."""
|
|
14
|
+
|
|
15
|
+
name: str = 'Remove date columns'
|
|
16
|
+
_description: str = 'Remove all columns containing Date from the dataset'
|
|
17
|
+
_usage: str = 'Use when date columns are irrelevant and you want to remove them; for other types use ActDropNumericalColumn or ActDropCategoricalColumn. Applicable to columns typed as DataType.DATE. Avoid when dates carry predictive signal or need feature extraction.'
|
|
18
|
+
_description_long: str = textwrap.dedent('''\
|
|
19
|
+
Remove all columns containing Data from the dataset
|
|
20
|
+
This step is used to clean the dataset in order to perform other actions later on
|
|
21
|
+
that can't be applied to date columns.''')
|
|
22
|
+
can_be_disabled: bool = False
|
|
23
|
+
|
|
24
|
+
def __init__(self):
|
|
25
|
+
self.columns_to_drop: list[str] = None
|
|
26
|
+
|
|
27
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
28
|
+
self.columns_to_drop = dataset.get_columns_names_by_type(DataType.DATE)
|
|
29
|
+
|
|
30
|
+
self.explanations = [
|
|
31
|
+
f'Dropped column **`{c}`**.' for c in self.columns_to_drop
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
return self
|
|
35
|
+
|
|
36
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
37
|
+
"""Drop columns.
|
|
38
|
+
|
|
39
|
+
:param pd.DataFrame X: DataFrame to transform.
|
|
40
|
+
:return: Transformed DataFrame.
|
|
41
|
+
"""
|
|
42
|
+
return X.drop(self.columns_to_drop, axis=1)
|
|
43
|
+
|
|
44
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
45
|
+
return 0
|
|
46
|
+
|
|
47
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
48
|
+
return bool(dataset.get_columns_names_by_type(DataType.DATE))
|