PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,296 @@
|
|
|
1
|
+
"""[STEP] Add KMeans distance/cluster features."""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.cluster import KMeans
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...candidate import Candidate
|
|
7
|
+
from ...data_type import DataType
|
|
8
|
+
from ...dataset import Dataset
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('features_preprocessing')
|
|
13
|
+
class ActKMeansFeatures(Actionable):
|
|
14
|
+
"""[STEP] Add KMeans distance/cluster features."""
|
|
15
|
+
|
|
16
|
+
name: str = "KMeans Features"
|
|
17
|
+
_description: str = "Add KMeans distance and cluster assignment features"
|
|
18
|
+
_usage: str = "Use when numeric features may form clusters and you want distance/cluster signals (vs ActKernelPCA or ActFeatureAgglomeration). Applicable to numeric tables with enough rows for k clusters. Avoid when data are tiny, mostly categorical, or clustering adds noise."
|
|
19
|
+
_description_long: str = textwrap.dedent('''\
|
|
20
|
+
KMeans groups numeric observations into k clusters by minimizing the
|
|
21
|
+
within-cluster variance. This step fits KMeans on numeric features and
|
|
22
|
+
appends distance-to-centroid features and, optionally, the assigned
|
|
23
|
+
cluster label. These additional features can surface non-linear structure
|
|
24
|
+
in the numeric feature space for downstream models.
|
|
25
|
+
''')
|
|
26
|
+
|
|
27
|
+
def __init__(self) -> None:
|
|
28
|
+
self.configuration = {
|
|
29
|
+
'n_clusters': {
|
|
30
|
+
'description': 'Number of clusters to form.',
|
|
31
|
+
'default': 8,
|
|
32
|
+
'range': [2, 200]
|
|
33
|
+
},
|
|
34
|
+
'init': {
|
|
35
|
+
'description': 'Initialization method.',
|
|
36
|
+
'default': 'k-means++',
|
|
37
|
+
'categorical': ['k-means++', 'random']
|
|
38
|
+
},
|
|
39
|
+
'n_init': {
|
|
40
|
+
'description': 'Number of time the k-means algorithm will be run.',
|
|
41
|
+
'default': 10,
|
|
42
|
+
'range': [1, 20]
|
|
43
|
+
},
|
|
44
|
+
'max_iter': {
|
|
45
|
+
'description': 'Maximum number of iterations per run.',
|
|
46
|
+
'default': 300,
|
|
47
|
+
'range': [50, 1000]
|
|
48
|
+
},
|
|
49
|
+
'tol': {
|
|
50
|
+
'description': 'Relative tolerance with regards to inertia.',
|
|
51
|
+
'default': 0.0001,
|
|
52
|
+
'range': [1e-05, 0.01]
|
|
53
|
+
},
|
|
54
|
+
'algorithm': {
|
|
55
|
+
'description': 'KMeans algorithm variant.',
|
|
56
|
+
'default': 'lloyd',
|
|
57
|
+
'categorical': ['lloyd', 'elkan']
|
|
58
|
+
},
|
|
59
|
+
'random_state': {
|
|
60
|
+
'description': 'Random state for reproducibility.',
|
|
61
|
+
'default': 42
|
|
62
|
+
},
|
|
63
|
+
'add_distances': {
|
|
64
|
+
'description': 'Whether to add distance-to-centroid features.',
|
|
65
|
+
'default': True,
|
|
66
|
+
'passthrough': False
|
|
67
|
+
},
|
|
68
|
+
'distance_mode': {
|
|
69
|
+
'description': 'Distance features to add: all or closest.',
|
|
70
|
+
'default': 'all',
|
|
71
|
+
'categorical': ['all', 'closest'],
|
|
72
|
+
'passthrough': False
|
|
73
|
+
},
|
|
74
|
+
'add_cluster_label': {
|
|
75
|
+
'description': 'Whether to add the cluster assignment feature.',
|
|
76
|
+
'default': True,
|
|
77
|
+
'passthrough': False
|
|
78
|
+
},
|
|
79
|
+
'feature_prefix': {
|
|
80
|
+
'description': 'Prefix for generated KMeans features.',
|
|
81
|
+
'default': 'kmeans',
|
|
82
|
+
'passthrough': False
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
self.columns: list[str] = []
|
|
87
|
+
self.active_columns: list[str] = []
|
|
88
|
+
self.preprocessor: KMeans | None = None
|
|
89
|
+
self.distance_feature_names: list[str] = []
|
|
90
|
+
self.label_column: str | None = None
|
|
91
|
+
self.feature_prefix: str = 'kmeans'
|
|
92
|
+
self.add_distances: bool = True
|
|
93
|
+
self.add_cluster_label: bool = True
|
|
94
|
+
self.distance_mode: str = 'all'
|
|
95
|
+
self.impute_values: pd.Series | None = None
|
|
96
|
+
self.optimizable: bool = True
|
|
97
|
+
|
|
98
|
+
@staticmethod
|
|
99
|
+
def _coerce_int(value: object, default: int) -> int:
|
|
100
|
+
try:
|
|
101
|
+
return int(value)
|
|
102
|
+
except (TypeError, ValueError):
|
|
103
|
+
return default
|
|
104
|
+
|
|
105
|
+
@staticmethod
|
|
106
|
+
def _coerce_bool(value: object, default: bool) -> bool:
|
|
107
|
+
if isinstance(value, bool):
|
|
108
|
+
return value
|
|
109
|
+
if isinstance(value, str):
|
|
110
|
+
lowered = value.strip().lower()
|
|
111
|
+
if lowered in ('1', 'true', 'yes', 'y'):
|
|
112
|
+
return True
|
|
113
|
+
if lowered in ('0', 'false', 'no', 'n'):
|
|
114
|
+
return False
|
|
115
|
+
return default
|
|
116
|
+
|
|
117
|
+
@staticmethod
|
|
118
|
+
def _coerce_str(value: object, default: str) -> str:
|
|
119
|
+
if value is None:
|
|
120
|
+
return default
|
|
121
|
+
text = str(value).strip()
|
|
122
|
+
return text or default
|
|
123
|
+
|
|
124
|
+
@staticmethod
|
|
125
|
+
def _resolve_distance_mode(value: object) -> str:
|
|
126
|
+
if value is None:
|
|
127
|
+
return 'all'
|
|
128
|
+
text = str(value).strip().lower()
|
|
129
|
+
if text in ('closest', 'min', 'nearest'):
|
|
130
|
+
return 'closest'
|
|
131
|
+
return 'all'
|
|
132
|
+
|
|
133
|
+
@staticmethod
|
|
134
|
+
def _select_active_columns(values: pd.DataFrame) -> list[str]:
|
|
135
|
+
if values.empty:
|
|
136
|
+
return []
|
|
137
|
+
unique_counts = values.nunique(dropna=True)
|
|
138
|
+
return [col for col in values.columns if unique_counts.get(col, 0) > 1]
|
|
139
|
+
|
|
140
|
+
@staticmethod
|
|
141
|
+
def _unique_name(name: str, reserved: set[str]) -> str:
|
|
142
|
+
if name not in reserved:
|
|
143
|
+
return name
|
|
144
|
+
idx = 1
|
|
145
|
+
candidate = f"{name}_{idx}"
|
|
146
|
+
while candidate in reserved:
|
|
147
|
+
idx += 1
|
|
148
|
+
candidate = f"{name}_{idx}"
|
|
149
|
+
return candidate
|
|
150
|
+
|
|
151
|
+
def _resolve_n_clusters(self, n_samples: int) -> int | None:
|
|
152
|
+
if n_samples < 2:
|
|
153
|
+
return None
|
|
154
|
+
n_clusters = self._coerce_int(self.get_config('n_clusters'), 8)
|
|
155
|
+
n_clusters = max(2, n_clusters)
|
|
156
|
+
return min(n_clusters, n_samples)
|
|
157
|
+
|
|
158
|
+
def _build_kmeans(self, n_clusters: int) -> KMeans:
|
|
159
|
+
params = self.passthrough_parameters()
|
|
160
|
+
params['n_clusters'] = int(n_clusters)
|
|
161
|
+
params['n_init'] = self._coerce_int(params.get('n_init'), 10)
|
|
162
|
+
params['max_iter'] = self._coerce_int(params.get('max_iter'), 300)
|
|
163
|
+
params['tol'] = float(params.get('tol', 0.0001))
|
|
164
|
+
try:
|
|
165
|
+
return KMeans(**params)
|
|
166
|
+
except TypeError:
|
|
167
|
+
params.pop('algorithm', None)
|
|
168
|
+
return KMeans(**params)
|
|
169
|
+
|
|
170
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
171
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
172
|
+
self.active_columns = []
|
|
173
|
+
self.preprocessor = None
|
|
174
|
+
self.distance_feature_names = []
|
|
175
|
+
self.label_column = None
|
|
176
|
+
self.impute_values = None
|
|
177
|
+
self.feature_prefix = self._coerce_str(self.get_config('feature_prefix'), 'kmeans')
|
|
178
|
+
self.add_distances = self._coerce_bool(self.get_config('add_distances'), True)
|
|
179
|
+
self.add_cluster_label = self._coerce_bool(self.get_config('add_cluster_label'), True)
|
|
180
|
+
self.distance_mode = self._resolve_distance_mode(self.get_config('distance_mode'))
|
|
181
|
+
self.explanations = []
|
|
182
|
+
|
|
183
|
+
if not self.columns or dataset.X.empty:
|
|
184
|
+
return self
|
|
185
|
+
|
|
186
|
+
values = dataset.X[self.columns]
|
|
187
|
+
self.active_columns = self._select_active_columns(values)
|
|
188
|
+
if not self.active_columns:
|
|
189
|
+
return self
|
|
190
|
+
|
|
191
|
+
if not self.add_distances and not self.add_cluster_label:
|
|
192
|
+
return self
|
|
193
|
+
|
|
194
|
+
active_values = values[self.active_columns]
|
|
195
|
+
n_clusters = self._resolve_n_clusters(active_values.shape[0])
|
|
196
|
+
if n_clusters is None:
|
|
197
|
+
return self
|
|
198
|
+
|
|
199
|
+
self.configure('n_clusters', n_clusters) # pylint: disable=too-many-function-args
|
|
200
|
+
self.impute_values = active_values.median(numeric_only=True)
|
|
201
|
+
active_values = active_values.fillna(self.impute_values)
|
|
202
|
+
|
|
203
|
+
self.preprocessor = self._build_kmeans(n_clusters)
|
|
204
|
+
self.preprocessor.fit(active_values)
|
|
205
|
+
|
|
206
|
+
reserved = set(dataset.X.columns)
|
|
207
|
+
if self.add_distances:
|
|
208
|
+
if self.distance_mode == 'closest':
|
|
209
|
+
name = self._unique_name(f"{self.feature_prefix}_cluster_dist", reserved)
|
|
210
|
+
self.distance_feature_names = [name]
|
|
211
|
+
reserved.add(name)
|
|
212
|
+
else:
|
|
213
|
+
self.distance_feature_names = []
|
|
214
|
+
for idx in range(n_clusters):
|
|
215
|
+
name = self._unique_name(f"{self.feature_prefix}_cluster_{idx}_dist", reserved)
|
|
216
|
+
self.distance_feature_names.append(name)
|
|
217
|
+
reserved.add(name)
|
|
218
|
+
|
|
219
|
+
if self.add_cluster_label:
|
|
220
|
+
self.label_column = self._unique_name(f"{self.feature_prefix}_cluster", reserved)
|
|
221
|
+
reserved.add(self.label_column)
|
|
222
|
+
|
|
223
|
+
if self.add_distances and self.distance_feature_names:
|
|
224
|
+
self.explanations.append(
|
|
225
|
+
f"Added {len(self.distance_feature_names)} KMeans distance features."
|
|
226
|
+
)
|
|
227
|
+
if self.add_cluster_label and self.label_column:
|
|
228
|
+
self.explanations.append(
|
|
229
|
+
f"Added KMeans cluster label feature `{self.label_column}`."
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
return self
|
|
233
|
+
|
|
234
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
235
|
+
"""Apply KMeans distance/cluster features.
|
|
236
|
+
|
|
237
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
238
|
+
:return: Transformed dataset
|
|
239
|
+
"""
|
|
240
|
+
if self.preprocessor is None or not self.active_columns:
|
|
241
|
+
return X
|
|
242
|
+
|
|
243
|
+
if any(column not in X.columns for column in self.active_columns):
|
|
244
|
+
return X
|
|
245
|
+
|
|
246
|
+
values = X[self.active_columns]
|
|
247
|
+
if self.impute_values is not None:
|
|
248
|
+
values = values.fillna(self.impute_values)
|
|
249
|
+
|
|
250
|
+
if self.add_distances and self.distance_feature_names:
|
|
251
|
+
distances = self.preprocessor.transform(values)
|
|
252
|
+
if self.distance_mode == 'closest':
|
|
253
|
+
distances = distances.min(axis=1).reshape(-1, 1)
|
|
254
|
+
distance_df = pd.DataFrame(
|
|
255
|
+
distances,
|
|
256
|
+
columns=self.distance_feature_names,
|
|
257
|
+
index=X.index
|
|
258
|
+
)
|
|
259
|
+
X = pd.concat([X, distance_df], axis=1)
|
|
260
|
+
|
|
261
|
+
if self.add_cluster_label and self.label_column:
|
|
262
|
+
X[self.label_column] = self.preprocessor.predict(values)
|
|
263
|
+
|
|
264
|
+
return X
|
|
265
|
+
|
|
266
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
267
|
+
if dataset.X.empty:
|
|
268
|
+
return False
|
|
269
|
+
if not self._coerce_bool(self.get_config('add_distances'), True) \
|
|
270
|
+
and not self._coerce_bool(self.get_config('add_cluster_label'), True):
|
|
271
|
+
return False
|
|
272
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
273
|
+
if not columns:
|
|
274
|
+
return False
|
|
275
|
+
values = dataset.X[columns]
|
|
276
|
+
active = self._select_active_columns(values)
|
|
277
|
+
if not active:
|
|
278
|
+
return False
|
|
279
|
+
return self._resolve_n_clusters(values.shape[0]) is not None
|
|
280
|
+
|
|
281
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
282
|
+
if candidate is None:
|
|
283
|
+
return 0.0
|
|
284
|
+
dataset = candidate.dataset
|
|
285
|
+
if not self.suitable(dataset):
|
|
286
|
+
return 0.0
|
|
287
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
288
|
+
if not columns:
|
|
289
|
+
return 0.0
|
|
290
|
+
values = dataset.X[columns]
|
|
291
|
+
active = self._select_active_columns(values)
|
|
292
|
+
if not active:
|
|
293
|
+
return 0.0
|
|
294
|
+
n_samples = max(1, values.shape[0])
|
|
295
|
+
ratio = len(active) / n_samples
|
|
296
|
+
return min(1.0, 0.2 + ratio)
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""[STEP] Decompose features with KernelPCA"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.decomposition import KernelPCA
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _is_numeric_matrix(values: pd.DataFrame) -> bool:
|
|
12
|
+
if values.empty:
|
|
13
|
+
return False
|
|
14
|
+
for column in values.columns:
|
|
15
|
+
if not pd.api.types.is_numeric_dtype(values[column]):
|
|
16
|
+
return False
|
|
17
|
+
return not values.isna().any().any()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@is_step('features_preprocessing')
|
|
21
|
+
class ActKernelPCA(Actionable):
|
|
22
|
+
"""[STEP] Apply KernelPCA for dimensionality reduction"""
|
|
23
|
+
name = "KernelPCA"
|
|
24
|
+
_description = "Perform Kernel Principal Component Analysis (KernelPCA) on a dataset"
|
|
25
|
+
_description_long = textwrap.dedent('''\
|
|
26
|
+
KernelPCA is a dimensionality reduction technique that extends Principal Component Analysis (PCA)
|
|
27
|
+
using kernel methods. It projects data into a higher-dimensional space before performing PCA,
|
|
28
|
+
enabling it to capture complex, non-linear structures in the data. KernelPCA is useful for reducing
|
|
29
|
+
dimensionality while preserving intricate patterns and relationships within the data.
|
|
30
|
+
''')
|
|
31
|
+
_usage = "Use when non-linear structure matters and linear reductions are insufficient; consider ActFastICA if you want independent components. Applicable to dense numeric, scaled features. Avoid when data is very large, sparse, or interpretability is required."
|
|
32
|
+
|
|
33
|
+
refs = [
|
|
34
|
+
{
|
|
35
|
+
'year': 1997,
|
|
36
|
+
'name': 'Kernel principal component analysis',
|
|
37
|
+
'authors': [
|
|
38
|
+
'Bernhard Schölkopf',
|
|
39
|
+
'Alexander Smola',
|
|
40
|
+
'Klaus-Robert Müller'
|
|
41
|
+
],
|
|
42
|
+
'doi': 'https://doi.org/10.1007/BFb0020217',
|
|
43
|
+
'publisher': 'Springer, Berlin, Heidelberg'
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
'year': 2003,
|
|
47
|
+
'name': 'Learning to find pre-images',
|
|
48
|
+
'authors': [
|
|
49
|
+
'Jason Weston',
|
|
50
|
+
'Bernhard Schölkopf',
|
|
51
|
+
'Gökhan Bakir'
|
|
52
|
+
],
|
|
53
|
+
'doi': 'https://proceedings.neurips.cc/paper_files/paper/2003/file/ \
|
|
54
|
+
ac1ad983e08ad3304a97e147f522747e-Paper.pdf',
|
|
55
|
+
'publisher': 'Advances in neural information processing systems 16 (2004) page 449--456'
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
'year': 2009,
|
|
59
|
+
'name': 'Finding structure with randomness: Probabilistic algorithms for constructing \
|
|
60
|
+
approximate matrix decompositions',
|
|
61
|
+
'authors': [
|
|
62
|
+
'Nathan Halko',
|
|
63
|
+
'Per-Gunnar Martinsson',
|
|
64
|
+
'Joel A. Tropp'
|
|
65
|
+
],
|
|
66
|
+
'doi': 'https://doi.org/10.48550/arXiv.0909.4061',
|
|
67
|
+
'publisher': 'SIAM Rev., Survey and Review section, Vol.53, No.2 page 217--288'
|
|
68
|
+
},
|
|
69
|
+
{
|
|
70
|
+
'year': 2011,
|
|
71
|
+
'name': 'A randomized algorithm for the decomposition of matrices',
|
|
72
|
+
'authors': [
|
|
73
|
+
'Per-Gunnar Martinsson',
|
|
74
|
+
'Vladimir Rokhlin',
|
|
75
|
+
'Mark Tygert'
|
|
76
|
+
],
|
|
77
|
+
'doi': 'https://doi.org/10.1016/j.acha.2010.02.003',
|
|
78
|
+
'publisher': 'Applied and Computational Harmonic Analysis, Vol.30, No.1 page 47--68'
|
|
79
|
+
}
|
|
80
|
+
]
|
|
81
|
+
def __init__(self):
|
|
82
|
+
self.configuration = {
|
|
83
|
+
'kernel': {
|
|
84
|
+
'description': 'Kernel used for PCA.',
|
|
85
|
+
'default': 'rbf',
|
|
86
|
+
'categorical': ['poly', 'rbf', 'sigmoid', 'cosine']
|
|
87
|
+
},
|
|
88
|
+
'n_components': {
|
|
89
|
+
'description': 'Number of components to keep.',
|
|
90
|
+
'default': 100,
|
|
91
|
+
'range': [10, 2000]
|
|
92
|
+
},
|
|
93
|
+
'coef0': {
|
|
94
|
+
'description': textwrap.dedent('''\
|
|
95
|
+
Independent term in poly and sigmoid kernels. Ignored by
|
|
96
|
+
other kernels.'''),
|
|
97
|
+
'default': 1.0,
|
|
98
|
+
'range': [-1.0, 1.0]
|
|
99
|
+
},
|
|
100
|
+
'degree': {
|
|
101
|
+
'description': 'Degree for poly kernels. Ignored by other kernels.',
|
|
102
|
+
'default': 3,
|
|
103
|
+
'range': [2, 5]
|
|
104
|
+
},
|
|
105
|
+
'random_state': {
|
|
106
|
+
'description': 'Random State',
|
|
107
|
+
'default': 42
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
self.optimizable: bool = True
|
|
111
|
+
self.preprocessor: bool = None
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
115
|
+
self.preprocessor = None
|
|
116
|
+
if not _is_numeric_matrix(dataset.X):
|
|
117
|
+
return self
|
|
118
|
+
try:
|
|
119
|
+
self.preprocessor = KernelPCA(**self.passthrough_parameters())
|
|
120
|
+
self.preprocessor.fit(dataset.X)
|
|
121
|
+
except ValueError:
|
|
122
|
+
higher_gamma = 1 / dataset.X.shape[1] + 0.05
|
|
123
|
+
self.preprocessor = KernelPCA(gamma=higher_gamma, **self.passthrough_parameters())
|
|
124
|
+
self.preprocessor.fit(dataset.X)
|
|
125
|
+
|
|
126
|
+
return self
|
|
127
|
+
|
|
128
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
129
|
+
"""Apply KernelPCA
|
|
130
|
+
|
|
131
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
132
|
+
:return: Transformed dataset
|
|
133
|
+
"""
|
|
134
|
+
|
|
135
|
+
if self.preprocessor is None:
|
|
136
|
+
return X
|
|
137
|
+
return pd.DataFrame(self.preprocessor.transform(X))
|
|
138
|
+
|
|
139
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
140
|
+
return _is_numeric_matrix(dataset.X)
|
|
141
|
+
|
|
142
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
143
|
+
return 0.5
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""[STEP] Apply log1p to skewed numeric features."""
|
|
2
|
+
import textwrap
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...candidate import Candidate
|
|
7
|
+
from ...dataset import Dataset
|
|
8
|
+
from ...data_type import DataType
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('features_preprocessing')
|
|
13
|
+
class ActLogTransformer(Actionable):
|
|
14
|
+
"""[STEP] Apply log1p to skewed numeric features."""
|
|
15
|
+
|
|
16
|
+
name: str = "Log1p Transformer"
|
|
17
|
+
_description: str = "Apply log1p to highly skewed numeric columns"
|
|
18
|
+
_usage: str = "Use when numeric features are highly right-skewed and >= -1, often before ActKBinsDiscretizer or ActKernelPCA. Applicable to continuous numeric columns with long-tailed distributions. Avoid when values are <= -1 or already log/scale transformed."
|
|
19
|
+
_description_long: str = textwrap.dedent('''\
|
|
20
|
+
This step detects numeric features with strong positive skew
|
|
21
|
+
and applies a log1p (log(1+x)) transformation to compress
|
|
22
|
+
extreme values. The transformation is automatic and only
|
|
23
|
+
applied to columns that meet the skewness threshold and have
|
|
24
|
+
values above the configured minimum.
|
|
25
|
+
''')
|
|
26
|
+
|
|
27
|
+
def __init__(self):
|
|
28
|
+
self.columns: list[str] = []
|
|
29
|
+
|
|
30
|
+
self.configuration = {
|
|
31
|
+
'skew_threshold': {
|
|
32
|
+
'description': 'Minimum skewness required to apply log1p.',
|
|
33
|
+
'default': 1.5,
|
|
34
|
+
'range': [0.5, 5.0]
|
|
35
|
+
},
|
|
36
|
+
'min_value': {
|
|
37
|
+
'description': 'Minimum allowed value for applying log1p.',
|
|
38
|
+
'default': 0.0,
|
|
39
|
+
'range': [-0.99, 1.0]
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
self.optimizable: bool = True
|
|
44
|
+
|
|
45
|
+
@staticmethod
|
|
46
|
+
def _coerce_float(value: object, default: float) -> float:
|
|
47
|
+
try:
|
|
48
|
+
return float(value)
|
|
49
|
+
except (TypeError, ValueError):
|
|
50
|
+
return default
|
|
51
|
+
|
|
52
|
+
def _resolve_skew_threshold(self) -> float:
|
|
53
|
+
threshold = self._coerce_float(self.get_config('skew_threshold'), 1.5)
|
|
54
|
+
if threshold < 0:
|
|
55
|
+
threshold = 0.0
|
|
56
|
+
return threshold
|
|
57
|
+
|
|
58
|
+
def _resolve_min_value(self) -> float:
|
|
59
|
+
min_value = self._coerce_float(self.get_config('min_value'), 0.0)
|
|
60
|
+
if min_value <= -1.0:
|
|
61
|
+
min_value = -0.999
|
|
62
|
+
return min_value
|
|
63
|
+
|
|
64
|
+
def _select_columns(self, values: pd.DataFrame) -> list[str]:
|
|
65
|
+
if values.empty:
|
|
66
|
+
return []
|
|
67
|
+
skewness = values.skew().fillna(0.0)
|
|
68
|
+
min_values = values.min(skipna=True)
|
|
69
|
+
threshold = self._resolve_skew_threshold()
|
|
70
|
+
min_value = self._resolve_min_value()
|
|
71
|
+
eligible = (skewness >= threshold) & (min_values >= min_value)
|
|
72
|
+
return [col for col in values.columns if bool(eligible.get(col, False))]
|
|
73
|
+
|
|
74
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
75
|
+
self.columns = []
|
|
76
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
77
|
+
if not columns or dataset.X.empty:
|
|
78
|
+
return self
|
|
79
|
+
values = dataset.X[columns]
|
|
80
|
+
self.columns = self._select_columns(values)
|
|
81
|
+
return self
|
|
82
|
+
|
|
83
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
84
|
+
"""Apply log1p
|
|
85
|
+
|
|
86
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
87
|
+
:return: Transformed dataset
|
|
88
|
+
"""
|
|
89
|
+
if not self.columns:
|
|
90
|
+
return X
|
|
91
|
+
columns = [col for col in self.columns if col in X.columns]
|
|
92
|
+
if not columns:
|
|
93
|
+
return X
|
|
94
|
+
min_value = self._resolve_min_value()
|
|
95
|
+
values = X[columns].clip(lower=min_value)
|
|
96
|
+
X[columns] = np.log1p(values)
|
|
97
|
+
return X
|
|
98
|
+
|
|
99
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
100
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
101
|
+
if not columns or dataset.X.empty:
|
|
102
|
+
return False
|
|
103
|
+
values = dataset.X[columns]
|
|
104
|
+
return bool(self._select_columns(values))
|
|
105
|
+
|
|
106
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
107
|
+
if candidate is None:
|
|
108
|
+
return 0.0
|
|
109
|
+
dataset = candidate.dataset
|
|
110
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
111
|
+
if not columns or dataset.X.empty:
|
|
112
|
+
return 0.0
|
|
113
|
+
values = dataset.X[columns]
|
|
114
|
+
selected = self._select_columns(values)
|
|
115
|
+
if not selected:
|
|
116
|
+
return 0.0
|
|
117
|
+
skewness = values[selected].skew().fillna(0.0)
|
|
118
|
+
threshold = self._resolve_skew_threshold()
|
|
119
|
+
if threshold <= 0:
|
|
120
|
+
return 1.0
|
|
121
|
+
mean_skew = float(skewness.mean()) if not skewness.empty else 0.0
|
|
122
|
+
return min(1.0, mean_skew / (threshold * 2.0))
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""[STEP] Decompose features with Nystroem"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.kernel_approximation import Nystroem
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _is_numeric_matrix(values: pd.DataFrame) -> bool:
|
|
12
|
+
if values.empty:
|
|
13
|
+
return False
|
|
14
|
+
for column in values.columns:
|
|
15
|
+
if not pd.api.types.is_numeric_dtype(values[column]):
|
|
16
|
+
return False
|
|
17
|
+
return not values.isna().any().any()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@is_step('features_preprocessing')
|
|
21
|
+
class ActNystroem(Actionable):
|
|
22
|
+
"""[STEP] Apply Nystroem method for dimensionality reduction"""
|
|
23
|
+
|
|
24
|
+
name: str = "Nystroem"
|
|
25
|
+
_description: str = "Apply the Nystroem method for dimensionality reduction \
|
|
26
|
+
over a list of columns"
|
|
27
|
+
_usage: str = "Use when you need a fast nonlinear kernel map approximation for large numeric data, as a lighter option than ActKernelPCA. Applicable to scaled continuous features with many rows. Avoid when you need independent components or tiny data where ActFastICA or ActKernelPCA is fine."
|
|
28
|
+
_description_long: str = textwrap.dedent('''\
|
|
29
|
+
The Nystroem method is a technique used for approximating kernel methods,
|
|
30
|
+
which helps in reducing the computational cost of kernel-based algorithms.
|
|
31
|
+
It approximates a kernel map using a subset of the data, making it suitable
|
|
32
|
+
for large datasets. This approach enables dimensionality reduction by creating
|
|
33
|
+
a low-rank approximation of the original kernel matrix.
|
|
34
|
+
''')
|
|
35
|
+
|
|
36
|
+
def __init__(self):
|
|
37
|
+
self.configuration = {
|
|
38
|
+
'kernel': {
|
|
39
|
+
'description': 'Kernel map to be approximated.',
|
|
40
|
+
'default': 'rbf',
|
|
41
|
+
'categorical': ["poly", "rbf", "sigmoid", "cosine", "chi2"]
|
|
42
|
+
},
|
|
43
|
+
'n_components': {
|
|
44
|
+
'description': 'Number of components to keep.',
|
|
45
|
+
'default': 100,
|
|
46
|
+
'range': [50, 10000]
|
|
47
|
+
},
|
|
48
|
+
'coef0': {
|
|
49
|
+
'description': 'Zero coefficient for polynomial and sigmoid kernels.',
|
|
50
|
+
'default': 0.0,
|
|
51
|
+
'range': [-1.0, 1.0]
|
|
52
|
+
},
|
|
53
|
+
'degree': {
|
|
54
|
+
'description': 'Degree of the polynomial kernel.',
|
|
55
|
+
'default': 3,
|
|
56
|
+
'range': [2, 5]
|
|
57
|
+
},
|
|
58
|
+
'gamma': {
|
|
59
|
+
'description': textwrap.dedent('''\
|
|
60
|
+
Gamma parameter for the RBF, laplacian, polynomial,
|
|
61
|
+
exponential chi2 and sigmoid kernels.'''),
|
|
62
|
+
'default': 0.1,
|
|
63
|
+
'range': [3.06e-05, 8.0]
|
|
64
|
+
},
|
|
65
|
+
'random_state': {
|
|
66
|
+
'description': 'Random State',
|
|
67
|
+
'default': 42
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
self.optimizable: bool = True
|
|
71
|
+
self.preprocessor: bool = None
|
|
72
|
+
|
|
73
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
74
|
+
self.preprocessor = None
|
|
75
|
+
if not _is_numeric_matrix(dataset.X):
|
|
76
|
+
return self
|
|
77
|
+
self.preprocessor = Nystroem(**self.passthrough_parameters())
|
|
78
|
+
self.preprocessor.fit(dataset.X)
|
|
79
|
+
return self
|
|
80
|
+
|
|
81
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
82
|
+
"""Apply Nystroem
|
|
83
|
+
|
|
84
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
85
|
+
:return: Transformed dataset
|
|
86
|
+
"""
|
|
87
|
+
if self.preprocessor is None:
|
|
88
|
+
return X
|
|
89
|
+
return pd.DataFrame(self.preprocessor.transform(X))
|
|
90
|
+
|
|
91
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
92
|
+
return 0.5
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
96
|
+
if not _is_numeric_matrix(dataset.X):
|
|
97
|
+
return False
|
|
98
|
+
if self.get_config('kernel') == 'chi2' and not (dataset.X < 0).any().any():
|
|
99
|
+
self.configure('kernel', 'rbf') # pylint: disable=too-many-function-args
|
|
100
|
+
return True
|