PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
"""[STEP] Reduce dimensions with FastICA"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from sklearn.decomposition import FastICA
|
|
6
|
+
from sklearn.utils.validation import check_array
|
|
7
|
+
from ...actionable import Actionable
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...dataset import Dataset
|
|
10
|
+
from ...data_type import DataType
|
|
11
|
+
from ...decorators.all import is_step
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@is_step('features_preprocessing')
|
|
15
|
+
class ActFastICA(Actionable):
|
|
16
|
+
"""[STEP] Reduce dimensions with FastICA"""
|
|
17
|
+
|
|
18
|
+
name: str = "FastICA"
|
|
19
|
+
_description: str = "Apply FastICA to extract independent components from numeric features"
|
|
20
|
+
_description_long: str = textwrap.dedent('''\
|
|
21
|
+
FastICA is a linear decomposition technique that separates mixed signals
|
|
22
|
+
into statistically independent components. It can be used as a
|
|
23
|
+
dimensionality reduction step by projecting numeric features into a
|
|
24
|
+
smaller set of latent sources while keeping the transformation
|
|
25
|
+
deterministic and efficient.
|
|
26
|
+
''')
|
|
27
|
+
_usage: str = "Use when linear independent components from mixed numeric signals are needed instead of ActKernelPCA. Applicable to continuous numeric features with enough samples. Avoid when samples or features are too few, or when ActFeatureAgglomeration is a better fit."
|
|
28
|
+
|
|
29
|
+
def __init__(self):
|
|
30
|
+
self.columns: list[str] = []
|
|
31
|
+
self.component_names: list[str] = []
|
|
32
|
+
self.preprocessor: FastICA | None = None
|
|
33
|
+
|
|
34
|
+
self.configuration = {
|
|
35
|
+
'n_components': {
|
|
36
|
+
'description': 'Number of independent components to estimate.',
|
|
37
|
+
'default': 10,
|
|
38
|
+
'range': [2, 2000]
|
|
39
|
+
},
|
|
40
|
+
'algorithm': {
|
|
41
|
+
'description': 'ICA algorithm to use.',
|
|
42
|
+
'default': 'parallel',
|
|
43
|
+
'categorical': ['parallel', 'deflation']
|
|
44
|
+
},
|
|
45
|
+
'whiten': {
|
|
46
|
+
'description': 'Whether to whiten data before applying ICA.',
|
|
47
|
+
'default': 'unit-variance',
|
|
48
|
+
'categorical': ['unit-variance', 'arbitrary-variance', False]
|
|
49
|
+
},
|
|
50
|
+
'fun': {
|
|
51
|
+
'description': 'Functional form of the G function.',
|
|
52
|
+
'default': 'logcosh',
|
|
53
|
+
'categorical': ['logcosh', 'exp', 'cube']
|
|
54
|
+
},
|
|
55
|
+
'max_iter': {
|
|
56
|
+
'description': 'Maximum number of iterations during optimization.',
|
|
57
|
+
'default': 200,
|
|
58
|
+
'range': [100, 1000]
|
|
59
|
+
},
|
|
60
|
+
'tol': {
|
|
61
|
+
'description': 'Convergence tolerance.',
|
|
62
|
+
'default': 0.0001,
|
|
63
|
+
'range': [1e-05, 0.01]
|
|
64
|
+
},
|
|
65
|
+
'random_state': {
|
|
66
|
+
'description': 'Random State',
|
|
67
|
+
'default': 42
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
self.optimizable: bool = True
|
|
72
|
+
|
|
73
|
+
@staticmethod
|
|
74
|
+
def _coerce_int(value: object, default: int) -> int:
|
|
75
|
+
try:
|
|
76
|
+
return int(value)
|
|
77
|
+
except (TypeError, ValueError):
|
|
78
|
+
return default
|
|
79
|
+
|
|
80
|
+
def _resolve_n_components(self, values: pd.DataFrame) -> int | None:
|
|
81
|
+
max_components = min(values.shape)
|
|
82
|
+
if max_components >= 2 and self.get_config('whiten'):
|
|
83
|
+
# Whitening can only use nonzero directions after centering.
|
|
84
|
+
numeric = check_array(values, dtype=[np.float64, np.float32])
|
|
85
|
+
max_components = int(np.linalg.matrix_rank(numeric - numeric.mean(axis=0)))
|
|
86
|
+
if max_components < 2:
|
|
87
|
+
return None
|
|
88
|
+
n_components = self._coerce_int(self.get_config('n_components'), max_components)
|
|
89
|
+
n_components = max(2, n_components)
|
|
90
|
+
return min(n_components, max_components)
|
|
91
|
+
|
|
92
|
+
def _build_transformer(self) -> FastICA:
|
|
93
|
+
params = self.passthrough_parameters()
|
|
94
|
+
params['n_components'] = int(params['n_components'])
|
|
95
|
+
params['max_iter'] = int(params['max_iter'])
|
|
96
|
+
params['tol'] = float(params['tol'])
|
|
97
|
+
return FastICA(**params)
|
|
98
|
+
|
|
99
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
100
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
101
|
+
self.preprocessor = None
|
|
102
|
+
self.component_names = []
|
|
103
|
+
|
|
104
|
+
if not self.columns or dataset.X.empty:
|
|
105
|
+
return self
|
|
106
|
+
|
|
107
|
+
values = dataset.X[self.columns]
|
|
108
|
+
if values.isna().any().any():
|
|
109
|
+
return self
|
|
110
|
+
n_components = self._resolve_n_components(values)
|
|
111
|
+
if n_components is None:
|
|
112
|
+
return self
|
|
113
|
+
|
|
114
|
+
self.configure('n_components', n_components) # pylint: disable=too-many-function-args
|
|
115
|
+
self.preprocessor = self._build_transformer()
|
|
116
|
+
self.preprocessor.fit(values)
|
|
117
|
+
|
|
118
|
+
self.component_names = [f"ica_{i}" for i in range(n_components)]
|
|
119
|
+
return self
|
|
120
|
+
|
|
121
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
122
|
+
"""Apply FastICA
|
|
123
|
+
|
|
124
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
125
|
+
:return: Transformed dataset
|
|
126
|
+
"""
|
|
127
|
+
if self.preprocessor is None or not self.columns:
|
|
128
|
+
return X
|
|
129
|
+
|
|
130
|
+
values = X[self.columns]
|
|
131
|
+
transformed = self.preprocessor.transform(values)
|
|
132
|
+
component_names = self.component_names or [
|
|
133
|
+
f"ica_{i}" for i in range(transformed.shape[1])
|
|
134
|
+
]
|
|
135
|
+
ica_df = pd.DataFrame(transformed, columns=component_names, index=X.index)
|
|
136
|
+
|
|
137
|
+
X = X.drop(columns=self.columns)
|
|
138
|
+
return pd.concat([X, ica_df], axis=1)
|
|
139
|
+
|
|
140
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
141
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
142
|
+
if not columns or dataset.X.empty:
|
|
143
|
+
return False
|
|
144
|
+
values = dataset.X[columns]
|
|
145
|
+
if values.isna().any().any():
|
|
146
|
+
return False
|
|
147
|
+
return min(values.shape) > 1
|
|
148
|
+
|
|
149
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
150
|
+
if candidate is None:
|
|
151
|
+
return 0.0
|
|
152
|
+
dataset = candidate.dataset
|
|
153
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
154
|
+
if not columns or dataset.X.empty:
|
|
155
|
+
return 0.0
|
|
156
|
+
values = dataset.X[columns]
|
|
157
|
+
if min(values.shape) <= 1:
|
|
158
|
+
return 0.0
|
|
159
|
+
n_features = values.shape[1]
|
|
160
|
+
n_samples = values.shape[0]
|
|
161
|
+
return min(1.0, n_features / max(1, n_samples))
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""[STEP] Decompose features with FeatureAgglomeration"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import numpy as np
|
|
5
|
+
from sklearn.cluster import FeatureAgglomeration
|
|
6
|
+
from ...actionable import Actionable
|
|
7
|
+
from ...dataset import Dataset
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _is_numeric_matrix(values: pd.DataFrame) -> bool:
|
|
13
|
+
if values.empty:
|
|
14
|
+
return False
|
|
15
|
+
for column in values.columns:
|
|
16
|
+
if not pd.api.types.is_numeric_dtype(values[column]):
|
|
17
|
+
return False
|
|
18
|
+
return not values.isna().any().any()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@is_step('features_preprocessing')
|
|
22
|
+
class ActFeatureAgglomeration(Actionable):
|
|
23
|
+
"""[STEP] Apply FeatureAgglomeration for dimensionality reduction"""
|
|
24
|
+
name: str = "FeatureAgglomeration"
|
|
25
|
+
_description: str = "Process FeatureAgglomeration algorithm over a set of features"
|
|
26
|
+
_description_long: str = textwrap.dedent('''\
|
|
27
|
+
FeatureAgglomeration is a clustering-based dimensionality reduction technique.
|
|
28
|
+
It groups similar features together using a hierarchical clustering approach,
|
|
29
|
+
which can help reduce the dimensionality of the dataset while preserving
|
|
30
|
+
essential information. This technique is unsupervised, meaning it does not
|
|
31
|
+
require labeled data, as it identifies clusters of features based on similarity.
|
|
32
|
+
''')
|
|
33
|
+
_usage: str = "Use when you want to cluster highly correlated numeric features for dimensionality reduction, instead of ActKernelPCA or ActFastICA. Applicable to wide tabular data with many continuous features and no labels. Avoid when features are mostly categorical or you need interpretable original features."
|
|
34
|
+
|
|
35
|
+
def __init__(self):
|
|
36
|
+
self.configuration: dict = {
|
|
37
|
+
'n_clusters': {
|
|
38
|
+
'description': 'The number of clusters to find',
|
|
39
|
+
'default': 25,
|
|
40
|
+
'range': [2, 400]
|
|
41
|
+
},
|
|
42
|
+
'metric': {
|
|
43
|
+
'description': 'Metric used to compute the linkage.',
|
|
44
|
+
'default': 'euclidean',
|
|
45
|
+
'categorical': ['euclidean']
|
|
46
|
+
# 'categorical': ['euclidean', 'l1', 'l2', 'manhattan', 'cosine', 'precomputed']
|
|
47
|
+
},
|
|
48
|
+
'linkage': {
|
|
49
|
+
'description': 'Which linkage criterion to use.',
|
|
50
|
+
'default': 'ward',
|
|
51
|
+
'categorical': ['ward', 'complete', 'average', 'single']
|
|
52
|
+
# 'categorical': ['ward', 'complete', 'average', 'single']
|
|
53
|
+
},
|
|
54
|
+
'pooling_func': {
|
|
55
|
+
'description': 'Which linkage criterion to use.',
|
|
56
|
+
'default': np.mean,
|
|
57
|
+
'categorical': [np.mean, np.median, np.max]
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
self.optimizable: bool = True
|
|
62
|
+
self.preprocessor: FeatureAgglomeration | None = None
|
|
63
|
+
|
|
64
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
65
|
+
# pylint: disable=too-many-function-args
|
|
66
|
+
self.preprocessor = None
|
|
67
|
+
if not _is_numeric_matrix(dataset.X):
|
|
68
|
+
return self
|
|
69
|
+
self.configure('n_clusters', min(self.get_config('n_clusters'), dataset.X.shape[1]))
|
|
70
|
+
|
|
71
|
+
self.preprocessor = FeatureAgglomeration(**self.passthrough_parameters())
|
|
72
|
+
self.preprocessor.fit(dataset.X)
|
|
73
|
+
|
|
74
|
+
return self
|
|
75
|
+
|
|
76
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
77
|
+
"""Apply FeatureAgglomeration
|
|
78
|
+
|
|
79
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
80
|
+
:return: Transformed dataset
|
|
81
|
+
"""
|
|
82
|
+
if self.preprocessor is None:
|
|
83
|
+
return X
|
|
84
|
+
return pd.DataFrame(self.preprocessor.transform(X))
|
|
85
|
+
|
|
86
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
87
|
+
return _is_numeric_matrix(dataset.X)
|
|
88
|
+
|
|
89
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
90
|
+
return 0.5
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
"""[STEP] Discretize numeric features with KBinsDiscretizer"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.preprocessing import KBinsDiscretizer
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...candidate import Candidate
|
|
7
|
+
from ...dataset import Dataset
|
|
8
|
+
from ...data_type import DataType
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('features_preprocessing')
|
|
13
|
+
class ActKBinsDiscretizer(Actionable):
|
|
14
|
+
"""[STEP] Discretize numeric features with KBinsDiscretizer"""
|
|
15
|
+
|
|
16
|
+
name: str = "KBins Discretizer"
|
|
17
|
+
_description: str = "Discretize numeric columns into uniform, quantile, or k-means bins"
|
|
18
|
+
_usage: str = "Use when you want numeric features binned for simpler models or outlier robustness, rather than ActLogTransformer or ActKernelPCA. Applicable to continuous numeric columns with more than one unique value. Avoid when fine-grained ordering or exact values must be preserved."
|
|
19
|
+
_description_long: str = textwrap.dedent('''\
|
|
20
|
+
KBinsDiscretizer converts continuous numeric features into discrete bins.
|
|
21
|
+
It can use uniform width bins, quantile-based bins, or k-means based bins.
|
|
22
|
+
The resulting bins can be returned as ordinal values or one-hot encoded
|
|
23
|
+
features, which may help models that prefer discrete inputs or benefit
|
|
24
|
+
from reduced sensitivity to outliers.
|
|
25
|
+
''')
|
|
26
|
+
|
|
27
|
+
def __init__(self):
|
|
28
|
+
self.columns: list[str] = []
|
|
29
|
+
self.active_columns: list[str] = []
|
|
30
|
+
self.output_columns: list[str] = []
|
|
31
|
+
self.preprocessor: KBinsDiscretizer | None = None
|
|
32
|
+
|
|
33
|
+
self.configuration = {
|
|
34
|
+
'n_bins': {
|
|
35
|
+
'description': 'Number of bins to use for each numeric feature.',
|
|
36
|
+
'default': 5,
|
|
37
|
+
'range': [2, 50]
|
|
38
|
+
},
|
|
39
|
+
'encode': {
|
|
40
|
+
'description': 'Encoding method for the transformed bins.',
|
|
41
|
+
'default': 'ordinal',
|
|
42
|
+
'categorical': ['ordinal', 'onehot', 'onehot-dense']
|
|
43
|
+
},
|
|
44
|
+
'strategy': {
|
|
45
|
+
'description': 'Strategy used to define the widths of the bins.',
|
|
46
|
+
'default': 'quantile',
|
|
47
|
+
'categorical': ['uniform', 'quantile', 'kmeans']
|
|
48
|
+
},
|
|
49
|
+
'subsample': {
|
|
50
|
+
'description': 'Maximum number of samples used to estimate quantile bin edges.',
|
|
51
|
+
'default': 200000,
|
|
52
|
+
'range': [1000, 500000]
|
|
53
|
+
},
|
|
54
|
+
'random_state': {
|
|
55
|
+
'description': 'Random state used when strategy is kmeans.',
|
|
56
|
+
'default': 42
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
self.optimizable: bool = True
|
|
61
|
+
|
|
62
|
+
@staticmethod
|
|
63
|
+
def _coerce_int(value: object, default: int) -> int:
|
|
64
|
+
try:
|
|
65
|
+
return int(value)
|
|
66
|
+
except (TypeError, ValueError):
|
|
67
|
+
return default
|
|
68
|
+
|
|
69
|
+
@staticmethod
|
|
70
|
+
def _select_active_columns(values: pd.DataFrame) -> list[str]:
|
|
71
|
+
if values.empty:
|
|
72
|
+
return []
|
|
73
|
+
unique_counts = values.nunique(dropna=True)
|
|
74
|
+
return [col for col in values.columns if unique_counts.get(col, 0) > 1]
|
|
75
|
+
|
|
76
|
+
@staticmethod
|
|
77
|
+
def _resolve_n_bins(values: pd.DataFrame, base_bins: int) -> int | list[int]:
|
|
78
|
+
if values.empty:
|
|
79
|
+
return base_bins
|
|
80
|
+
unique_counts = values.nunique(dropna=True).astype(int)
|
|
81
|
+
per_feature_bins = [min(base_bins, max(2, count)) for count in unique_counts]
|
|
82
|
+
if not per_feature_bins:
|
|
83
|
+
return base_bins
|
|
84
|
+
if all(bins == per_feature_bins[0] for bins in per_feature_bins):
|
|
85
|
+
return per_feature_bins[0]
|
|
86
|
+
return per_feature_bins
|
|
87
|
+
|
|
88
|
+
def _feature_names(self) -> list[str]:
|
|
89
|
+
if self.output_columns:
|
|
90
|
+
return self.output_columns
|
|
91
|
+
if self.preprocessor is None:
|
|
92
|
+
return []
|
|
93
|
+
if hasattr(self.preprocessor, "get_feature_names_out"):
|
|
94
|
+
try:
|
|
95
|
+
return list(self.preprocessor.get_feature_names_out(self.active_columns))
|
|
96
|
+
except ValueError:
|
|
97
|
+
return []
|
|
98
|
+
if hasattr(self.preprocessor, "n_bins_"):
|
|
99
|
+
names: list[str] = []
|
|
100
|
+
for column, n_bins in zip(self.active_columns, self.preprocessor.n_bins_):
|
|
101
|
+
for idx in range(int(n_bins)):
|
|
102
|
+
names.append(f"{column}_bin_{idx}")
|
|
103
|
+
return names
|
|
104
|
+
return []
|
|
105
|
+
|
|
106
|
+
def _build_transformer(self, n_samples: int, n_bins: int | list[int]) -> KBinsDiscretizer:
|
|
107
|
+
params = self.passthrough_parameters()
|
|
108
|
+
params['n_bins'] = n_bins
|
|
109
|
+
if 'subsample' in params:
|
|
110
|
+
subsample = self._coerce_int(params.get('subsample'), n_samples)
|
|
111
|
+
subsample = max(1, min(subsample, n_samples))
|
|
112
|
+
params['subsample'] = subsample
|
|
113
|
+
return KBinsDiscretizer(**params)
|
|
114
|
+
|
|
115
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
116
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
117
|
+
self.active_columns = []
|
|
118
|
+
self.output_columns = []
|
|
119
|
+
self.preprocessor = None
|
|
120
|
+
|
|
121
|
+
if not self.columns or dataset.X.empty:
|
|
122
|
+
return self
|
|
123
|
+
|
|
124
|
+
values = dataset.X[self.columns]
|
|
125
|
+
if values.isna().any().any():
|
|
126
|
+
return self
|
|
127
|
+
self.active_columns = self._select_active_columns(values)
|
|
128
|
+
if not self.active_columns:
|
|
129
|
+
return self
|
|
130
|
+
|
|
131
|
+
active_values = values[self.active_columns]
|
|
132
|
+
base_bins = self._coerce_int(self.get_config('n_bins'), 5)
|
|
133
|
+
base_bins = max(2, base_bins)
|
|
134
|
+
n_bins = self._resolve_n_bins(active_values, base_bins)
|
|
135
|
+
|
|
136
|
+
self.preprocessor = self._build_transformer(active_values.shape[0], n_bins)
|
|
137
|
+
self.preprocessor.fit(active_values)
|
|
138
|
+
|
|
139
|
+
if self.get_config('encode') in ('onehot', 'onehot-dense'):
|
|
140
|
+
self.output_columns = self._feature_names()
|
|
141
|
+
else:
|
|
142
|
+
self.output_columns = self.active_columns.copy()
|
|
143
|
+
|
|
144
|
+
return self
|
|
145
|
+
|
|
146
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
147
|
+
"""Apply KBinsDiscretizer
|
|
148
|
+
|
|
149
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
150
|
+
:return: Transformed dataset
|
|
151
|
+
"""
|
|
152
|
+
if self.preprocessor is None or not self.active_columns:
|
|
153
|
+
return X
|
|
154
|
+
|
|
155
|
+
values = X[self.active_columns]
|
|
156
|
+
transformed = self.preprocessor.transform(values)
|
|
157
|
+
encode = self.get_config('encode')
|
|
158
|
+
|
|
159
|
+
if encode == 'ordinal':
|
|
160
|
+
X[self.active_columns] = transformed
|
|
161
|
+
return X
|
|
162
|
+
|
|
163
|
+
X = X.reset_index(drop=True)
|
|
164
|
+
feature_names = self.output_columns or self._feature_names()
|
|
165
|
+
|
|
166
|
+
if hasattr(transformed, "toarray") and encode == 'onehot':
|
|
167
|
+
encoded_df = pd.DataFrame.sparse.from_spmatrix(
|
|
168
|
+
transformed,
|
|
169
|
+
columns=feature_names,
|
|
170
|
+
index=X.index
|
|
171
|
+
)
|
|
172
|
+
else:
|
|
173
|
+
if hasattr(transformed, "toarray"):
|
|
174
|
+
transformed = transformed.toarray()
|
|
175
|
+
encoded_df = pd.DataFrame(transformed, columns=feature_names, index=X.index)
|
|
176
|
+
|
|
177
|
+
X = X.drop(columns=self.active_columns)
|
|
178
|
+
return pd.concat([X, encoded_df], axis=1)
|
|
179
|
+
|
|
180
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
181
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
182
|
+
if not columns or dataset.X.empty:
|
|
183
|
+
return False
|
|
184
|
+
values = dataset.X[columns]
|
|
185
|
+
if values.isna().any().any():
|
|
186
|
+
return False
|
|
187
|
+
if values.empty:
|
|
188
|
+
return False
|
|
189
|
+
unique_counts = values.nunique(dropna=True)
|
|
190
|
+
return bool((unique_counts > 1).any())
|
|
191
|
+
|
|
192
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
193
|
+
if candidate is None:
|
|
194
|
+
return 0.0
|
|
195
|
+
dataset = candidate.dataset
|
|
196
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
197
|
+
if not columns or dataset.X.empty:
|
|
198
|
+
return 0.0
|
|
199
|
+
values = dataset.X[columns]
|
|
200
|
+
if values.empty:
|
|
201
|
+
return 0.0
|
|
202
|
+
n_rows = max(1, values.shape[0])
|
|
203
|
+
unique_ratio = values.nunique(dropna=True) / n_rows
|
|
204
|
+
if unique_ratio.empty:
|
|
205
|
+
return 0.0
|
|
206
|
+
mean_ratio = float(unique_ratio.mean())
|
|
207
|
+
return min(1.0, mean_ratio * 2.0)
|