PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""
|
|
2
|
+
[STEP] Decompose features with PCA
|
|
3
|
+
"""
|
|
4
|
+
import textwrap
|
|
5
|
+
import pandas as pd
|
|
6
|
+
from sklearn.decomposition import PCA
|
|
7
|
+
from ...actionable import Actionable
|
|
8
|
+
from ...dataset import Dataset
|
|
9
|
+
from ...candidate import Candidate
|
|
10
|
+
from ...decorators.all import is_step
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _is_numeric_matrix(values: pd.DataFrame) -> bool:
|
|
14
|
+
if values.empty:
|
|
15
|
+
return False
|
|
16
|
+
for column in values.columns:
|
|
17
|
+
if not pd.api.types.is_numeric_dtype(values[column]):
|
|
18
|
+
return False
|
|
19
|
+
return not values.isna().any().any()
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@is_step('features_preprocessing')
|
|
23
|
+
class ActPCA(Actionable):
|
|
24
|
+
"""[STEP] Reduce dimensions with PCA"""
|
|
25
|
+
|
|
26
|
+
name: str = "PCA"
|
|
27
|
+
_description: str = "Apply PCA for dimensionality reduction over a list of columns"
|
|
28
|
+
_usage: str = "Use when you need fast linear dimensionality reduction for numeric features; consider ActKernelPCA or ActFastICA for nonlinear or independent components. Applicable to scaled numeric matrices. Avoid when features are categorical or you must keep original feature meaning."
|
|
29
|
+
_description_long: str = textwrap.dedent('''\
|
|
30
|
+
PCA, or Principal Component Analysis, is a dimensionality reduction technique.
|
|
31
|
+
It transforms the data into a set of linearly uncorrelated components, capturing
|
|
32
|
+
the maximum variance in the data with each successive component.
|
|
33
|
+
This method is unsupervised, meaning it does not require labeled data,
|
|
34
|
+
and is particularly useful for simplifying datasets while retaining
|
|
35
|
+
as much of the underlying structure as possible.
|
|
36
|
+
''')
|
|
37
|
+
|
|
38
|
+
def __init__(self):
|
|
39
|
+
self.configuration = {
|
|
40
|
+
'n_components': {
|
|
41
|
+
'description': 'Number of components to keep.',
|
|
42
|
+
'default': 0.999,
|
|
43
|
+
'range': [0.5, 0.999]
|
|
44
|
+
},
|
|
45
|
+
'random_state': {
|
|
46
|
+
'description': 'Random State',
|
|
47
|
+
'default': 42
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
self.optimizable = True
|
|
52
|
+
self.preprocessor = None
|
|
53
|
+
|
|
54
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
55
|
+
self.preprocessor = None
|
|
56
|
+
if not _is_numeric_matrix(dataset.X):
|
|
57
|
+
return self
|
|
58
|
+
|
|
59
|
+
self.preprocessor = PCA(**self.passthrough_parameters())
|
|
60
|
+
self.preprocessor.fit(dataset.X)
|
|
61
|
+
return self
|
|
62
|
+
|
|
63
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
64
|
+
"""Apply PCA
|
|
65
|
+
|
|
66
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
67
|
+
:return: Transformed dataset
|
|
68
|
+
"""
|
|
69
|
+
if self.preprocessor is None:
|
|
70
|
+
return X
|
|
71
|
+
return pd.DataFrame(self.preprocessor.transform(X))
|
|
72
|
+
|
|
73
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
74
|
+
return _is_numeric_matrix(dataset.X)
|
|
75
|
+
|
|
76
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
77
|
+
return 0.5
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Experimental polynomial expansion, available only through an explicit import.
|
|
2
|
+
|
|
3
|
+
Unbounded output dimensionality can exhaust memory during automatic exploration.
|
|
4
|
+
Kept outside the default preprocessing stage; see docs/component_status.rst.
|
|
5
|
+
"""
|
|
6
|
+
import textwrap
|
|
7
|
+
import pandas as pd
|
|
8
|
+
from sklearn.preprocessing import PolynomialFeatures
|
|
9
|
+
from ...actionable import Actionable
|
|
10
|
+
from ...dataset import Dataset
|
|
11
|
+
from ...candidate import Candidate
|
|
12
|
+
from ...decorators.all import is_step
|
|
13
|
+
|
|
14
|
+
def _is_numeric_matrix(values: pd.DataFrame) -> bool:
|
|
15
|
+
if values.empty:
|
|
16
|
+
return False
|
|
17
|
+
for column in values.columns:
|
|
18
|
+
if not pd.api.types.is_numeric_dtype(values[column]):
|
|
19
|
+
return False
|
|
20
|
+
return not values.isna().any().any()
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@is_step('experimental')
|
|
24
|
+
class ActPolynomialFeatures(Actionable):
|
|
25
|
+
"""[STEP] Preprocess with PolynomialFeatures"""
|
|
26
|
+
|
|
27
|
+
name: str = "Preprocess with PolynomialFeatures"
|
|
28
|
+
_usage: str = "Use when you want explicit polynomial interactions for linear models; consider ActKernelPCA for projection-based nonlinearity. Applicable to numeric tabular features with moderate dimensionality. Avoid when feature count will explode or when ActKBinsDiscretizer is a better match."
|
|
29
|
+
_description: str = textwrap.dedent('''\
|
|
30
|
+
PolynomialFeatures creates new features by combining existing
|
|
31
|
+
features mathematically. It squares, cubes, and multiplies features to
|
|
32
|
+
create more complex patterns.''')
|
|
33
|
+
_description_long: str = textwrap.dedent('''\
|
|
34
|
+
PolynomialFeatures is a preprocessing technique that
|
|
35
|
+
generates new features based on polynomial relationships between existing
|
|
36
|
+
features. This helps capture non-linear relationships in the data that may
|
|
37
|
+
not be apparent from the original features alone. PolynomialFeatures is
|
|
38
|
+
particularly useful when you suspect the underlying relationship in your data
|
|
39
|
+
might not be straightforward or linear.''')
|
|
40
|
+
|
|
41
|
+
def __init__(self):
|
|
42
|
+
self.configuration = {
|
|
43
|
+
'include_bias': {
|
|
44
|
+
'description': 'If True (default), then include a bias \
|
|
45
|
+
column, the feature in which all polynomial powers are zero',
|
|
46
|
+
'default': True
|
|
47
|
+
},
|
|
48
|
+
'interaction_only': {
|
|
49
|
+
'description': 'If True, only interaction features are produced',
|
|
50
|
+
'default': False
|
|
51
|
+
},
|
|
52
|
+
'degree': {
|
|
53
|
+
'description': 'Degree of the polynomial kernel.',
|
|
54
|
+
'default': 3,
|
|
55
|
+
'range': [2, 5]
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
self.optimizable: bool = True
|
|
60
|
+
self.preprocessor: bool = None
|
|
61
|
+
|
|
62
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
63
|
+
self.preprocessor = None
|
|
64
|
+
if not _is_numeric_matrix(dataset.X):
|
|
65
|
+
return self
|
|
66
|
+
|
|
67
|
+
self.preprocessor = PolynomialFeatures(**self.passthrough_parameters())
|
|
68
|
+
self.preprocessor.fit(dataset.X)
|
|
69
|
+
|
|
70
|
+
return self
|
|
71
|
+
|
|
72
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
73
|
+
"""Apply PolynomialFeatures
|
|
74
|
+
|
|
75
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
76
|
+
:return: Transformed dataset
|
|
77
|
+
"""
|
|
78
|
+
if self.preprocessor is None:
|
|
79
|
+
return X
|
|
80
|
+
return pd.DataFrame(self.preprocessor.transform(X))
|
|
81
|
+
|
|
82
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
83
|
+
return 0.5
|
|
84
|
+
|
|
85
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
86
|
+
return _is_numeric_matrix(dataset.X)
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""[STEP] Preprocess with PowerTransformer"""
|
|
2
|
+
from typing import Any
|
|
3
|
+
import textwrap
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from sklearn.preprocessing import PowerTransformer
|
|
6
|
+
from ...actionable import Actionable
|
|
7
|
+
from ...dataset import Dataset
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _is_numeric_matrix(values: pd.DataFrame) -> bool:
|
|
13
|
+
if values.empty:
|
|
14
|
+
return False
|
|
15
|
+
for column in values.columns:
|
|
16
|
+
if not pd.api.types.is_numeric_dtype(values[column]):
|
|
17
|
+
return False
|
|
18
|
+
return not values.isna().any().any()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@is_step('features_preprocessing')
|
|
22
|
+
class ActPowerTransformer(Actionable):
|
|
23
|
+
"""[STEP] Preprocess with PowerTransformer"""
|
|
24
|
+
|
|
25
|
+
name: str = "Preprocess with PowerTransformer"
|
|
26
|
+
_description: str = textwrap.dedent('''\
|
|
27
|
+
PowerTransformer changes data to make it more "bell-curve" shaped.
|
|
28
|
+
It uses special math tricks to flatten out irregular distributions and make the data
|
|
29
|
+
behave more like a normal distribution.''')
|
|
30
|
+
_description_long: str = textwrap.dedent('''\
|
|
31
|
+
PowerTransformer is a preprocessing technique that applies a power
|
|
32
|
+
transformation to make data more Gaussian-like. PowerTransformer is useful when you want to apply
|
|
33
|
+
machine learning models that assume normal distribution,
|
|
34
|
+
even if your original data doesn't meet this assumption.
|
|
35
|
+
It helps make your data more compatible with many common ML algorithms.''')
|
|
36
|
+
_usage: str = "Use when numeric features are skewed and need Gaussian-like scaling, instead of ActKernelPCA or ActFastICA. Applicable to continuous numeric data with unimodal, non-normal distributions. Avoid when data are categorical/one-hot, already near-normal, or highly multimodal."
|
|
37
|
+
refs: list[dict[str, Any]] = [
|
|
38
|
+
{
|
|
39
|
+
'year': 1964,
|
|
40
|
+
'name': 'An Analysis of Transformations',
|
|
41
|
+
'authors': [
|
|
42
|
+
'G. E. P. Box',
|
|
43
|
+
'D. R. Cox'
|
|
44
|
+
],
|
|
45
|
+
'doi': 'https://doi.org/10.1111/j.2517-6161.1964.tb00553.x',
|
|
46
|
+
'publisher': 'Journal of the Royal Statistical Society: Series B (Methodological), \
|
|
47
|
+
Vol.26, No.2 page 211--243'
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
'year': 2000,
|
|
51
|
+
'name': 'A New Family of Power Transformations to Improve Normality or Symmetry',
|
|
52
|
+
'authors': [
|
|
53
|
+
'In-Kwon Yeo',
|
|
54
|
+
'Richard A. Johnson'
|
|
55
|
+
],
|
|
56
|
+
'doi': 'https://doi.org/10.1093/biomet/87.4.954',
|
|
57
|
+
'publisher': 'Oxford University Press, Biometrika Vol.87 No.4 page 954--959'
|
|
58
|
+
},
|
|
59
|
+
]
|
|
60
|
+
|
|
61
|
+
def __init__(self):
|
|
62
|
+
self.configuration = {
|
|
63
|
+
'method': {
|
|
64
|
+
'description': 'The power transform method.',
|
|
65
|
+
'default': 'yeo-johnson',
|
|
66
|
+
'categorical': ['yeo-johnson', 'box-cox']
|
|
67
|
+
},
|
|
68
|
+
'standardize': {
|
|
69
|
+
'description': 'Set to True to apply zero-mean, \
|
|
70
|
+
unit-variance normalization to the transformed output.',
|
|
71
|
+
'default': True
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
self.optimizable: bool = True
|
|
76
|
+
self.preprocessor: bool = None
|
|
77
|
+
|
|
78
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
79
|
+
self.preprocessor = None
|
|
80
|
+
if not _is_numeric_matrix(dataset.X):
|
|
81
|
+
return self
|
|
82
|
+
self.preprocessor = PowerTransformer(**self.passthrough_parameters())
|
|
83
|
+
self.preprocessor.fit(dataset.X)
|
|
84
|
+
|
|
85
|
+
return self
|
|
86
|
+
|
|
87
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
88
|
+
"""Apply PowerTransformer
|
|
89
|
+
|
|
90
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
91
|
+
:return: Transformed dataset
|
|
92
|
+
"""
|
|
93
|
+
|
|
94
|
+
if self.preprocessor is None:
|
|
95
|
+
return X
|
|
96
|
+
return pd.DataFrame(self.preprocessor.transform(X))
|
|
97
|
+
|
|
98
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
99
|
+
return 0.5
|
|
100
|
+
|
|
101
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
102
|
+
if not _is_numeric_matrix(dataset.X):
|
|
103
|
+
return False
|
|
104
|
+
if self.get_config('method') == 'box-cox' and not (dataset.X < 0).any().any():
|
|
105
|
+
self.configure('method', 'yeo-johnson') # pylint: disable=too-many-function-args
|
|
106
|
+
return True
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""[STEP] Preprocess with QuantileTransformer"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.preprocessing import QuantileTransformer
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...data_type import DataType
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('features_preprocessing')
|
|
13
|
+
class ActQuantileTransformer(Actionable):
|
|
14
|
+
"""[STEP] Preprocess with QuantileTransformer"""
|
|
15
|
+
|
|
16
|
+
name: str = "Preprocess with QuantileTransformer"
|
|
17
|
+
_description: str = textwrap.dedent('''\
|
|
18
|
+
QuantileTransformer remaps numeric features to a uniform or normal distribution,
|
|
19
|
+
reducing skew and making feature scales more comparable.''')
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
QuantileTransformer estimates the empirical cumulative distribution for each
|
|
22
|
+
numeric feature and maps values to a chosen target distribution.
|
|
23
|
+
This non-linear transformation can reduce the impact of outliers and
|
|
24
|
+
produce more Gaussian-like features for models that benefit from it.''')
|
|
25
|
+
_usage: str = "Use when numeric features are skewed or unevenly scaled; compare ActKBinsDiscretizer. Applicable to continuous numeric columns before models preferring near-normal inputs. Avoid when original units or interpretability must stay, or data is very sparse; compare ActKernelPCA."
|
|
26
|
+
|
|
27
|
+
def __init__(self):
|
|
28
|
+
self.columns: list[str] = None
|
|
29
|
+
self.preprocessor: QuantileTransformer = None
|
|
30
|
+
|
|
31
|
+
self.configuration = {
|
|
32
|
+
'n_quantiles': {
|
|
33
|
+
'description': 'Number of quantiles to estimate.',
|
|
34
|
+
'default': 1000,
|
|
35
|
+
'range': [10, 1000]
|
|
36
|
+
},
|
|
37
|
+
'output_distribution': {
|
|
38
|
+
'description': 'Target distribution for the transformed data.',
|
|
39
|
+
'default': 'normal',
|
|
40
|
+
'categorical': ['uniform', 'normal']
|
|
41
|
+
},
|
|
42
|
+
'subsample': {
|
|
43
|
+
'description': 'Maximum number of samples used to estimate quantiles.',
|
|
44
|
+
'default': 100000,
|
|
45
|
+
'range': [1000, 200000]
|
|
46
|
+
},
|
|
47
|
+
'random_state': {
|
|
48
|
+
'description': 'Random state used when subsampling.',
|
|
49
|
+
'default': 42
|
|
50
|
+
},
|
|
51
|
+
'copy': {
|
|
52
|
+
'description': 'Set to False to perform transformation in-place when possible.',
|
|
53
|
+
'default': True,
|
|
54
|
+
'categorical': [True, False]
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
self.optimizable: bool = True
|
|
59
|
+
|
|
60
|
+
def _build_transformer(self, n_samples: int) -> QuantileTransformer:
|
|
61
|
+
params = self.passthrough_parameters()
|
|
62
|
+
n_samples = max(1, int(n_samples))
|
|
63
|
+
n_quantiles = int(params.pop('n_quantiles'))
|
|
64
|
+
subsample = int(params.pop('subsample'))
|
|
65
|
+
n_quantiles = max(1, min(n_quantiles, n_samples))
|
|
66
|
+
subsample = max(1, min(subsample, n_samples))
|
|
67
|
+
params['n_quantiles'] = n_quantiles
|
|
68
|
+
params['subsample'] = subsample
|
|
69
|
+
return QuantileTransformer(**params)
|
|
70
|
+
|
|
71
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
72
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
73
|
+
if self.columns and not dataset.X.empty:
|
|
74
|
+
values = dataset.X[self.columns]
|
|
75
|
+
if values.isna().any().any():
|
|
76
|
+
self.preprocessor = None
|
|
77
|
+
return self
|
|
78
|
+
self.preprocessor = self._build_transformer(values.shape[0])
|
|
79
|
+
self.preprocessor.fit(values)
|
|
80
|
+
else:
|
|
81
|
+
self.preprocessor = None
|
|
82
|
+
return self
|
|
83
|
+
|
|
84
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
85
|
+
"""Apply QuantileTransformer
|
|
86
|
+
|
|
87
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
88
|
+
:return: Transformed dataset
|
|
89
|
+
"""
|
|
90
|
+
if self.preprocessor and self.columns:
|
|
91
|
+
X[self.columns] = self.preprocessor.transform(X[self.columns])
|
|
92
|
+
return X
|
|
93
|
+
|
|
94
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
95
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
96
|
+
if not columns or dataset.X.empty:
|
|
97
|
+
return False
|
|
98
|
+
values = dataset.X[columns]
|
|
99
|
+
return not values.isna().any().any()
|
|
100
|
+
|
|
101
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
102
|
+
if candidate is None:
|
|
103
|
+
return 0.0
|
|
104
|
+
columns = candidate.dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
105
|
+
if not columns or candidate.dataset.X.empty:
|
|
106
|
+
return 0.0
|
|
107
|
+
values = candidate.dataset.X[columns]
|
|
108
|
+
if values.empty:
|
|
109
|
+
return 0.0
|
|
110
|
+
skewness = values.skew().abs().fillna(0.0)
|
|
111
|
+
if skewness.empty:
|
|
112
|
+
return 0.0
|
|
113
|
+
mean_skew = float(skewness.mean())
|
|
114
|
+
return min(1.0, mean_skew / 2.0)
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""[STEP] Decompose features with RBFSampler"""
|
|
2
|
+
from typing import Any
|
|
3
|
+
import textwrap
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from sklearn.kernel_approximation import RBFSampler
|
|
6
|
+
from ...actionable import Actionable
|
|
7
|
+
from ...dataset import Dataset
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
def _is_numeric_matrix(values: pd.DataFrame) -> bool:
|
|
12
|
+
if values.empty:
|
|
13
|
+
return False
|
|
14
|
+
for column in values.columns:
|
|
15
|
+
if not pd.api.types.is_numeric_dtype(values[column]):
|
|
16
|
+
return False
|
|
17
|
+
return not values.isna().any().any()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@is_step('features_preprocessing')
|
|
21
|
+
class ActRBFSampler(Actionable):
|
|
22
|
+
"""[STEP] Approximate with RBFSampler"""
|
|
23
|
+
|
|
24
|
+
name: str = "Approximate with RBFSampler"
|
|
25
|
+
_usage: str = "Use when you want a fast nonlinear kernel approximation for numeric features, as a lighter alternative to ActKernelPCA. Applicable to dense tabular data where scaling is reasonable. Avoid when data are categorical heavy, very sparse, or when you need exact kernel features."
|
|
26
|
+
_description: str = textwrap.dedent('''\
|
|
27
|
+
RBFSampler is a tool that helps computers understand complex relationships
|
|
28
|
+
between things by turning them into simpler numbers.''')
|
|
29
|
+
_description_long: str = textwrap.dedent('''\
|
|
30
|
+
RBFSampler is a machine learning technique that transforms data into
|
|
31
|
+
a higher-dimensional space where it's easier for algorithms to find patterns.
|
|
32
|
+
It works by creating random projections of the original data onto a new set of axes.
|
|
33
|
+
This allows it to approximate the effects of a radial basis function kernel, which is a
|
|
34
|
+
mathematical way of measuring similarity between data points.''')
|
|
35
|
+
refs: list[dict[str, Any]] =[
|
|
36
|
+
{
|
|
37
|
+
'year': 2008,
|
|
38
|
+
'name': 'Weighted Sums of Random Kitchen Sinks: Replacing minimization with \
|
|
39
|
+
randomization in learning',
|
|
40
|
+
'authors': [
|
|
41
|
+
'Ali Rahimi',
|
|
42
|
+
'Benjamin Recht'
|
|
43
|
+
],
|
|
44
|
+
'doi': None,
|
|
45
|
+
'publisher': 'Advances in Neural Information Processing Systems 21 page 1313--1320'
|
|
46
|
+
}
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
def __init__(self):
|
|
50
|
+
self.configuration = {
|
|
51
|
+
'n_components': {
|
|
52
|
+
'description': 'Number of components to keep.',
|
|
53
|
+
'default': 100,
|
|
54
|
+
'range': [50, 10000]
|
|
55
|
+
},
|
|
56
|
+
'random_state': {
|
|
57
|
+
'description': 'Random State',
|
|
58
|
+
'default': 42
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
self.optimizable: bool = True
|
|
63
|
+
self.preprocessor: bool = None
|
|
64
|
+
|
|
65
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
66
|
+
self.preprocessor = None
|
|
67
|
+
if not _is_numeric_matrix(dataset.X):
|
|
68
|
+
return self
|
|
69
|
+
self.preprocessor = RBFSampler(**self.passthrough_parameters())
|
|
70
|
+
self.preprocessor.fit(dataset.X)
|
|
71
|
+
|
|
72
|
+
return self
|
|
73
|
+
|
|
74
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
75
|
+
"""Apply RBFSampler
|
|
76
|
+
|
|
77
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
78
|
+
:return: Transformed dataset
|
|
79
|
+
"""
|
|
80
|
+
if self.preprocessor is None:
|
|
81
|
+
return X
|
|
82
|
+
return pd.DataFrame(self.preprocessor.transform(X))
|
|
83
|
+
|
|
84
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
85
|
+
return 0.5
|
|
86
|
+
|
|
87
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
88
|
+
return _is_numeric_matrix(dataset.X)
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""[STEP] Decompose features with SelectPercentile"""
|
|
2
|
+
|
|
3
|
+
import textwrap
|
|
4
|
+
from collections.abc import Callable
|
|
5
|
+
|
|
6
|
+
import pandas as pd
|
|
7
|
+
from sklearn.feature_selection import SelectPercentile, chi2, f_classif
|
|
8
|
+
from ...actionable import Actionable
|
|
9
|
+
from ...dataset import Dataset
|
|
10
|
+
from ...candidate import Candidate
|
|
11
|
+
from ...decorators.all import is_step
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _is_numeric_matrix(values: pd.DataFrame) -> bool:
|
|
15
|
+
if values.empty:
|
|
16
|
+
return False
|
|
17
|
+
for column in values.columns:
|
|
18
|
+
if not pd.api.types.is_numeric_dtype(values[column]):
|
|
19
|
+
return False
|
|
20
|
+
return not values.isna().any().any()
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _resolve_score_func(value) -> Callable | None:
|
|
24
|
+
if callable(value):
|
|
25
|
+
return value
|
|
26
|
+
if isinstance(value, str):
|
|
27
|
+
lowered = value.strip().lower()
|
|
28
|
+
if lowered == 'chi2':
|
|
29
|
+
return chi2
|
|
30
|
+
if lowered == 'f_classif':
|
|
31
|
+
return f_classif
|
|
32
|
+
return None
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@is_step('features_preprocessing')
|
|
36
|
+
class ActSelectPercentile(Actionable):
|
|
37
|
+
"""[STEP] Preprocess with SelectPercentile"""
|
|
38
|
+
|
|
39
|
+
name: str = "Preprocess with SelectPercentile"
|
|
40
|
+
_usage: str = "Use when you need fast univariate feature selection by percentile, rather than ActKernelPCA. Applicable to non-negative features with classification targets (chi2 or f_classif). Avoid when you want feature engineering via clustering like ActKMeansFeatures."
|
|
41
|
+
_description: str = textwrap.dedent('''\
|
|
42
|
+
SelectPercentile is a tool that helps choose important features from a
|
|
43
|
+
group of variables by looking at how well each one predicts the outcome.''')
|
|
44
|
+
_description_long: str = textwrap.dedent('''\
|
|
45
|
+
SelectPercentile is a feature selection technique used in machine
|
|
46
|
+
learning. It works by assigning scores to each feature based on how well it predicts
|
|
47
|
+
the outcome. Then, it selects only the top-scoring percentage of features.
|
|
48
|
+
This helps reduce the number of variables while keeping the most informative ones.''')
|
|
49
|
+
|
|
50
|
+
def __init__(self):
|
|
51
|
+
self.configuration = {
|
|
52
|
+
'score_func': {
|
|
53
|
+
'description': 'function taking two arrays X and y, \
|
|
54
|
+
and returning a pair of arrays',
|
|
55
|
+
'default': chi2,
|
|
56
|
+
'categorical': [chi2, f_classif]
|
|
57
|
+
},
|
|
58
|
+
'percentile': {
|
|
59
|
+
'description': 'Percent of features to keep.',
|
|
60
|
+
'default': 50.0,
|
|
61
|
+
'range': [1.0, 99.0]
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
self.optimizable: bool = True
|
|
66
|
+
self.preprocessor: SelectPercentile | None = None
|
|
67
|
+
|
|
68
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
69
|
+
self.preprocessor = None
|
|
70
|
+
if dataset.y is None or not _is_numeric_matrix(dataset.X):
|
|
71
|
+
return self
|
|
72
|
+
|
|
73
|
+
score_func = _resolve_score_func(self.get_config('score_func'))
|
|
74
|
+
if score_func is None:
|
|
75
|
+
return self
|
|
76
|
+
|
|
77
|
+
params = self.passthrough_parameters()
|
|
78
|
+
params['score_func'] = score_func
|
|
79
|
+
self.preprocessor = SelectPercentile(**params)
|
|
80
|
+
self.preprocessor.fit(dataset.X, dataset.y)
|
|
81
|
+
|
|
82
|
+
return self
|
|
83
|
+
|
|
84
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
85
|
+
"""Apply SelectPercentile
|
|
86
|
+
|
|
87
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
88
|
+
:return: Transformed dataset
|
|
89
|
+
"""
|
|
90
|
+
if self.preprocessor is None:
|
|
91
|
+
return X
|
|
92
|
+
return pd.DataFrame(self.preprocessor.transform(X))
|
|
93
|
+
|
|
94
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
95
|
+
# Negative values are not supported
|
|
96
|
+
if dataset.y is None or dataset.type_of_target is None:
|
|
97
|
+
return False
|
|
98
|
+
if not _is_numeric_matrix(dataset.X):
|
|
99
|
+
return False
|
|
100
|
+
score_func = _resolve_score_func(self.get_config('score_func'))
|
|
101
|
+
if score_func is None:
|
|
102
|
+
return False
|
|
103
|
+
if dataset.type_of_target not in [
|
|
104
|
+
'binary', 'multiclass', 'multilabel-indicator'
|
|
105
|
+
]:
|
|
106
|
+
return False
|
|
107
|
+
if score_func == chi2:
|
|
108
|
+
return not (dataset.X < 0).any().any()
|
|
109
|
+
return True
|
|
110
|
+
|
|
111
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
112
|
+
return 0.5
|