PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""[STEP] Reduce dimensions with SparseRandomProjection."""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.random_projection import SparseRandomProjection
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...candidate import Candidate
|
|
7
|
+
from ...dataset import Dataset
|
|
8
|
+
from ...data_type import DataType
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('features_preprocessing')
|
|
13
|
+
class ActSparseRandomProjection(Actionable):
|
|
14
|
+
"""[STEP] Reduce dimensions with SparseRandomProjection."""
|
|
15
|
+
|
|
16
|
+
name: str = "SparseRandomProjection"
|
|
17
|
+
_description: str = "Project numeric features to a lower-dimensional space quickly"
|
|
18
|
+
_usage: str = "Use when you need fast, scalable reduction of many numeric features and can trade interpretability, vs heavier ActKernelPCA or ActFastICA. Applicable to high-dimensional numeric data, including sparse inputs. Avoid when features are few or you need interpretable axes."
|
|
19
|
+
_description_long: str = textwrap.dedent('''\
|
|
20
|
+
Sparse random projection compresses high-dimensional numeric features by
|
|
21
|
+
multiplying them with a sparse random matrix. This preserves distances in
|
|
22
|
+
expectation while keeping computation fast, making it suitable for large
|
|
23
|
+
feature spaces where traditional decompositions are expensive.
|
|
24
|
+
''')
|
|
25
|
+
|
|
26
|
+
def __init__(self):
|
|
27
|
+
self.columns: list[str] = []
|
|
28
|
+
self.component_names: list[str] = []
|
|
29
|
+
self.preprocessor: SparseRandomProjection | None = None
|
|
30
|
+
|
|
31
|
+
self.configuration = {
|
|
32
|
+
'n_components': {
|
|
33
|
+
'description': 'Number of components to keep.',
|
|
34
|
+
'default': 100,
|
|
35
|
+
'range': [2, 2000]
|
|
36
|
+
},
|
|
37
|
+
'density': {
|
|
38
|
+
'description': textwrap.dedent('''\
|
|
39
|
+
Proportion of non-zero elements in the projection matrix.
|
|
40
|
+
Use "auto" to rely on the default 1/sqrt(n_features).'''),
|
|
41
|
+
'default': 'auto'
|
|
42
|
+
},
|
|
43
|
+
'random_state': {
|
|
44
|
+
'description': 'Random State',
|
|
45
|
+
'default': 42
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
self.optimizable: bool = True
|
|
50
|
+
|
|
51
|
+
@staticmethod
|
|
52
|
+
def _coerce_int(value: object, default: int) -> int:
|
|
53
|
+
try:
|
|
54
|
+
return int(value)
|
|
55
|
+
except (TypeError, ValueError):
|
|
56
|
+
return default
|
|
57
|
+
|
|
58
|
+
def _resolve_n_components(self, n_samples: int, n_features: int) -> int | None:
|
|
59
|
+
max_components = min(n_samples, n_features)
|
|
60
|
+
if max_components < 2:
|
|
61
|
+
return None
|
|
62
|
+
n_components = self._coerce_int(self.get_config('n_components'), max_components)
|
|
63
|
+
n_components = max(2, n_components)
|
|
64
|
+
return min(n_components, max_components)
|
|
65
|
+
|
|
66
|
+
def _resolve_density(self) -> float | str:
|
|
67
|
+
value = self.get_config('density')
|
|
68
|
+
if value is None:
|
|
69
|
+
return 'auto'
|
|
70
|
+
if isinstance(value, str):
|
|
71
|
+
stripped = value.strip().lower()
|
|
72
|
+
if stripped in ['', 'auto', 'none']:
|
|
73
|
+
return 'auto'
|
|
74
|
+
try:
|
|
75
|
+
value = float(stripped)
|
|
76
|
+
except ValueError:
|
|
77
|
+
return 'auto'
|
|
78
|
+
try:
|
|
79
|
+
numeric = float(value)
|
|
80
|
+
except (TypeError, ValueError):
|
|
81
|
+
return 'auto'
|
|
82
|
+
if numeric <= 0:
|
|
83
|
+
return 'auto'
|
|
84
|
+
return min(numeric, 1.0)
|
|
85
|
+
|
|
86
|
+
def _build_transformer(self, n_components: int) -> SparseRandomProjection:
|
|
87
|
+
params = self.passthrough_parameters()
|
|
88
|
+
params['n_components'] = int(n_components)
|
|
89
|
+
params['density'] = self._resolve_density()
|
|
90
|
+
return SparseRandomProjection(**params)
|
|
91
|
+
|
|
92
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
93
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
94
|
+
self.preprocessor = None
|
|
95
|
+
self.component_names = []
|
|
96
|
+
|
|
97
|
+
if not self.columns or dataset.X.empty:
|
|
98
|
+
return self
|
|
99
|
+
|
|
100
|
+
values = dataset.X[self.columns]
|
|
101
|
+
if values.isna().any().any():
|
|
102
|
+
return self
|
|
103
|
+
n_components = self._resolve_n_components(*values.shape)
|
|
104
|
+
if n_components is None:
|
|
105
|
+
return self
|
|
106
|
+
|
|
107
|
+
self.configure('n_components', n_components) # pylint: disable=too-many-function-args
|
|
108
|
+
self.preprocessor = self._build_transformer(n_components)
|
|
109
|
+
self.preprocessor.fit(values)
|
|
110
|
+
|
|
111
|
+
self.component_names = [f"srp_{i}" for i in range(n_components)]
|
|
112
|
+
return self
|
|
113
|
+
|
|
114
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
115
|
+
"""Apply SparseRandomProjection
|
|
116
|
+
|
|
117
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
118
|
+
:return: Transformed dataset
|
|
119
|
+
"""
|
|
120
|
+
if self.preprocessor is None or not self.columns:
|
|
121
|
+
return X
|
|
122
|
+
|
|
123
|
+
values = X[self.columns]
|
|
124
|
+
transformed = self.preprocessor.transform(values)
|
|
125
|
+
if hasattr(transformed, "toarray"):
|
|
126
|
+
transformed = transformed.toarray()
|
|
127
|
+
|
|
128
|
+
component_names = self.component_names or [
|
|
129
|
+
f"srp_{i}" for i in range(transformed.shape[1])
|
|
130
|
+
]
|
|
131
|
+
projected_df = pd.DataFrame(transformed, columns=component_names, index=X.index)
|
|
132
|
+
|
|
133
|
+
X = X.drop(columns=self.columns)
|
|
134
|
+
return pd.concat([X, projected_df], axis=1)
|
|
135
|
+
|
|
136
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
137
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
138
|
+
if not columns or dataset.X.empty:
|
|
139
|
+
return False
|
|
140
|
+
values = dataset.X[columns]
|
|
141
|
+
if values.isna().any().any():
|
|
142
|
+
return False
|
|
143
|
+
return min(values.shape) > 1
|
|
144
|
+
|
|
145
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
146
|
+
if candidate is None:
|
|
147
|
+
return 0.0
|
|
148
|
+
dataset = candidate.dataset
|
|
149
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
150
|
+
if not columns or dataset.X.empty:
|
|
151
|
+
return 0.0
|
|
152
|
+
values = dataset.X[columns]
|
|
153
|
+
if min(values.shape) <= 1:
|
|
154
|
+
return 0.0
|
|
155
|
+
n_features = values.shape[1]
|
|
156
|
+
n_samples = values.shape[0]
|
|
157
|
+
return min(1.0, n_features / max(1, n_samples))
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""[STEP] Reduce dimensions with TruncatedSVD"""
|
|
2
|
+
import textwrap
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from sklearn.decomposition import TruncatedSVD
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...candidate import Candidate
|
|
7
|
+
from ...dataset import Dataset
|
|
8
|
+
from ...data_type import DataType
|
|
9
|
+
from ...decorators.all import is_step
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@is_step('features_preprocessing')
|
|
13
|
+
class ActTruncatedSVD(Actionable):
|
|
14
|
+
"""[STEP] Reduce dimensions with TruncatedSVD"""
|
|
15
|
+
|
|
16
|
+
name: str = "TruncatedSVD"
|
|
17
|
+
_description: str = "Reduce dimensionality for sparse or high-dimensional numeric features"
|
|
18
|
+
_description_long: str = textwrap.dedent('''\
|
|
19
|
+
TruncatedSVD performs a low-rank approximation of the feature matrix.
|
|
20
|
+
It is well suited for sparse representations such as TF-IDF or hashing
|
|
21
|
+
vectors, and can reduce the number of features while preserving most
|
|
22
|
+
of the structure of the data.
|
|
23
|
+
''')
|
|
24
|
+
_usage: str = "Use when you need linear reduction for sparse, high-dimensional numeric features; compare ActKernelPCA for nonlinear patterns. Applicable to TF-IDF, hashing, and large numeric feature matrices. Avoid when data is dense with nonlinear structure or when ActFastICA is the goal."
|
|
25
|
+
|
|
26
|
+
def __init__(self):
|
|
27
|
+
self.columns: list[str] = []
|
|
28
|
+
self.preprocessor: TruncatedSVD | None = None
|
|
29
|
+
self.component_names: list[str] = []
|
|
30
|
+
|
|
31
|
+
self.configuration = {
|
|
32
|
+
'n_components': {
|
|
33
|
+
'description': 'Number of components to keep.',
|
|
34
|
+
'default': 100,
|
|
35
|
+
'range': [2, 2000]
|
|
36
|
+
},
|
|
37
|
+
'algorithm': {
|
|
38
|
+
'description': 'SVD solver to use.',
|
|
39
|
+
'default': 'randomized',
|
|
40
|
+
'categorical': ['randomized', 'arpack']
|
|
41
|
+
},
|
|
42
|
+
'n_iter': {
|
|
43
|
+
'description': 'Number of power iterations for randomized SVD.',
|
|
44
|
+
'default': 5,
|
|
45
|
+
'range': [2, 15]
|
|
46
|
+
},
|
|
47
|
+
'tol': {
|
|
48
|
+
'description': 'Tolerance for arpack solver.',
|
|
49
|
+
'default': 0.0,
|
|
50
|
+
'range': [0.0, 0.1]
|
|
51
|
+
},
|
|
52
|
+
'random_state': {
|
|
53
|
+
'description': 'Random State',
|
|
54
|
+
'default': 42
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
self.optimizable: bool = True
|
|
59
|
+
|
|
60
|
+
@staticmethod
|
|
61
|
+
def _coerce_int(value: object, default: int) -> int:
|
|
62
|
+
try:
|
|
63
|
+
return int(value)
|
|
64
|
+
except (TypeError, ValueError):
|
|
65
|
+
return default
|
|
66
|
+
|
|
67
|
+
def _resolve_n_components(self, n_samples: int, n_features: int) -> int | None:
|
|
68
|
+
max_components = min(n_samples - 1, n_features - 1)
|
|
69
|
+
if max_components < 1:
|
|
70
|
+
return None
|
|
71
|
+
n_components = self._coerce_int(self.get_config('n_components'), max_components)
|
|
72
|
+
n_components = max(1, n_components)
|
|
73
|
+
return min(n_components, max_components)
|
|
74
|
+
|
|
75
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
76
|
+
self.columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
77
|
+
self.preprocessor = None
|
|
78
|
+
self.component_names = []
|
|
79
|
+
|
|
80
|
+
if not self.columns or dataset.X.empty:
|
|
81
|
+
return self
|
|
82
|
+
|
|
83
|
+
values = dataset.X[self.columns]
|
|
84
|
+
if values.isna().any().any():
|
|
85
|
+
return self
|
|
86
|
+
n_components = self._resolve_n_components(*values.shape)
|
|
87
|
+
if n_components is None:
|
|
88
|
+
return self
|
|
89
|
+
|
|
90
|
+
self.configure('n_components', n_components) # pylint: disable=too-many-function-args
|
|
91
|
+
params = self.passthrough_parameters()
|
|
92
|
+
self.preprocessor = TruncatedSVD(**params)
|
|
93
|
+
self.preprocessor.fit(values)
|
|
94
|
+
|
|
95
|
+
self.component_names = [f"svd_{i}" for i in range(n_components)]
|
|
96
|
+
return self
|
|
97
|
+
|
|
98
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
99
|
+
"""Apply TruncatedSVD
|
|
100
|
+
|
|
101
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
102
|
+
:return: Transformed dataset
|
|
103
|
+
"""
|
|
104
|
+
if self.preprocessor is None or not self.columns:
|
|
105
|
+
return X
|
|
106
|
+
|
|
107
|
+
values = X[self.columns]
|
|
108
|
+
transformed = self.preprocessor.transform(values)
|
|
109
|
+
component_names = self.component_names or [
|
|
110
|
+
f"svd_{i}" for i in range(transformed.shape[1])
|
|
111
|
+
]
|
|
112
|
+
svd_df = pd.DataFrame(transformed, columns=component_names, index=X.index)
|
|
113
|
+
|
|
114
|
+
X = X.drop(columns=self.columns)
|
|
115
|
+
return pd.concat([X, svd_df], axis=1)
|
|
116
|
+
|
|
117
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
118
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
119
|
+
if not columns or dataset.X.empty:
|
|
120
|
+
return False
|
|
121
|
+
values = dataset.X[columns]
|
|
122
|
+
if values.isna().any().any():
|
|
123
|
+
return False
|
|
124
|
+
return min(values.shape) > 1
|
|
125
|
+
|
|
126
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
127
|
+
if candidate is None:
|
|
128
|
+
return 0.0
|
|
129
|
+
dataset = candidate.dataset
|
|
130
|
+
columns = dataset.get_columns_names_by_type(DataType.NUMERIC)
|
|
131
|
+
if not columns or dataset.X.empty:
|
|
132
|
+
return 0.0
|
|
133
|
+
n_features = len(columns)
|
|
134
|
+
n_samples = dataset.X.shape[0]
|
|
135
|
+
if n_features <= 1 or n_samples <= 1:
|
|
136
|
+
return 0.0
|
|
137
|
+
return min(1.0, n_features / max(1, n_samples))
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""Features selection Actionables"""
|
|
2
|
+
from .act_remove_high_correlated_column import ActRemoveHighCorrelatedColumn
|
|
3
|
+
from .act_remove_low_variance_column import ActRemoveLowVarianceColumn
|
|
4
|
+
from .act_select_k_best import ActSelectKBest
|
|
5
|
+
from .act_rfe import ActRFE
|
|
6
|
+
from .act_select_from_model import ActSelectFromModel
|
|
7
|
+
from .act_vif_selector import ActVIFSelector
|
|
8
|
+
from .act_permutation_importance_selector import ActPermutationImportanceSelector
|