PyIAML 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- iaml/__init__.py +56 -0
- iaml/actionable.py +11 -0
- iaml/actionables/__init__.py +21 -0
- iaml/actionables/boosting/__init__.py +4 -0
- iaml/actionables/boosting/act_adaboost.py +59 -0
- iaml/actionables/cleaning/__init__.py +26 -0
- iaml/actionables/cleaning/act_categorical_imputer.py +124 -0
- iaml/actionables/cleaning/act_count_vectorizer.py +204 -0
- iaml/actionables/cleaning/act_drop_categorical_column.py +51 -0
- iaml/actionables/cleaning/act_drop_date_column.py +48 -0
- iaml/actionables/cleaning/act_drop_high_cardinality_categorical.py +337 -0
- iaml/actionables/cleaning/act_drop_numerical_column.py +75 -0
- iaml/actionables/cleaning/act_drop_textual_column.py +51 -0
- iaml/actionables/cleaning/act_encode_target_column.py +56 -0
- iaml/actionables/cleaning/act_frequency_encoder.py +127 -0
- iaml/actionables/cleaning/act_hashing_vectorizer.py +186 -0
- iaml/actionables/cleaning/act_knn_imputer.py +152 -0
- iaml/actionables/cleaning/act_mean_column.py +79 -0
- iaml/actionables/cleaning/act_mice.py +464 -0
- iaml/actionables/cleaning/act_missing_count_feature.py +109 -0
- iaml/actionables/cleaning/act_missing_indicator.py +124 -0
- iaml/actionables/cleaning/act_onehot.py +65 -0
- iaml/actionables/cleaning/act_ordinal_encoder.py +177 -0
- iaml/actionables/cleaning/act_rare_category_grouper.py +173 -0
- iaml/actionables/cleaning/act_simple_imputer.py +109 -0
- iaml/actionables/cleaning/act_split_date.py +68 -0
- iaml/actionables/cleaning/act_target_encoder.py +274 -0
- iaml/actionables/cleaning/act_text_normalizer.py +241 -0
- iaml/actionables/cleaning/act_tf_idf.py +80 -0
- iaml/actionables/cleaning/act_word2vec.py +150 -0
- iaml/actionables/features_precleaning/__init__.py +12 -0
- iaml/actionables/features_precleaning/act_coerce_numeric_strings.py +194 -0
- iaml/actionables/features_precleaning/act_date_converter.py +99 -0
- iaml/actionables/features_precleaning/act_drop_bad_quality_rows.py +77 -0
- iaml/actionables/features_precleaning/act_drop_duplicate_rows.py +131 -0
- iaml/actionables/features_precleaning/act_drop_high_missing_columns.py +94 -0
- iaml/actionables/features_precleaning/act_drop_id_like_columns.py +294 -0
- iaml/actionables/features_precleaning/act_normalize_column_names.py +157 -0
- iaml/actionables/features_precleaning/act_sentinel_to_na_n.py +270 -0
- iaml/actionables/features_precleaning/act_trim_space.py +79 -0
- iaml/actionables/features_preprocessing/__init__.py +18 -0
- iaml/actionables/features_preprocessing/act_cyclical_date_encoding.py +212 -0
- iaml/actionables/features_preprocessing/act_fast_ica.py +161 -0
- iaml/actionables/features_preprocessing/act_feature_agglomeration.py +90 -0
- iaml/actionables/features_preprocessing/act_k_bins_discretizer.py +207 -0
- iaml/actionables/features_preprocessing/act_k_means_features.py +296 -0
- iaml/actionables/features_preprocessing/act_kernel_pca.py +143 -0
- iaml/actionables/features_preprocessing/act_log_transformer.py +122 -0
- iaml/actionables/features_preprocessing/act_nystroem.py +100 -0
- iaml/actionables/features_preprocessing/act_pca.py +77 -0
- iaml/actionables/features_preprocessing/act_polynomial_features.py +86 -0
- iaml/actionables/features_preprocessing/act_power_transformer.py +106 -0
- iaml/actionables/features_preprocessing/act_quantile_transformer.py +114 -0
- iaml/actionables/features_preprocessing/act_rbf_sampler.py +88 -0
- iaml/actionables/features_preprocessing/act_select_percentile.py +112 -0
- iaml/actionables/features_preprocessing/act_sparse_random_projection.py +157 -0
- iaml/actionables/features_preprocessing/act_truncated_svd.py +137 -0
- iaml/actionables/features_selection/__init__.py +8 -0
- iaml/actionables/features_selection/act_permutation_importance_selector.py +421 -0
- iaml/actionables/features_selection/act_remove_high_correlated_column.py +70 -0
- iaml/actionables/features_selection/act_remove_low_variance_column.py +74 -0
- iaml/actionables/features_selection/act_rfe.py +214 -0
- iaml/actionables/features_selection/act_select_from_model.py +325 -0
- iaml/actionables/features_selection/act_select_k_best.py +181 -0
- iaml/actionables/features_selection/act_vif_selector.py +130 -0
- iaml/actionables/imbalance/__init__.py +10 -0
- iaml/actionables/imbalance/act_adasyn.py +150 -0
- iaml/actionables/imbalance/act_borderline_smote.py +171 -0
- iaml/actionables/imbalance/act_near_miss.py +158 -0
- iaml/actionables/imbalance/act_random_over_sampling.py +60 -0
- iaml/actionables/imbalance/act_random_under_sampler.py +135 -0
- iaml/actionables/imbalance/act_smote.py +162 -0
- iaml/actionables/imbalance/act_smote_tomek.py +182 -0
- iaml/actionables/imbalance/act_smoteenn.py +193 -0
- iaml/actionables/imbalance/act_tomek_links.py +138 -0
- iaml/actionables/normalize/__init__.py +6 -0
- iaml/actionables/normalize/act_max_abs_scaler.py +78 -0
- iaml/actionables/normalize/act_minmax_scaler.py +56 -0
- iaml/actionables/normalize/act_normalizer.py +95 -0
- iaml/actionables/normalize/act_robust_scaler.py +111 -0
- iaml/actionables/normalize/act_standard_scaler.py +55 -0
- iaml/actionables/predictors/__init__.py +6 -0
- iaml/actionables/predictors/_xgboost.py +16 -0
- iaml/actionables/predictors/classifier/__init__.py +26 -0
- iaml/actionables/predictors/classifier/act_bagging_classifier.py +113 -0
- iaml/actionables/predictors/classifier/act_bernoulli_nb.py +89 -0
- iaml/actionables/predictors/classifier/act_catboost_classifier.py +135 -0
- iaml/actionables/predictors/classifier/act_complement_nb.py +106 -0
- iaml/actionables/predictors/classifier/act_decision_tree_classifier.py +117 -0
- iaml/actionables/predictors/classifier/act_extra_trees_classifier.py +115 -0
- iaml/actionables/predictors/classifier/act_gaussian_nb.py +53 -0
- iaml/actionables/predictors/classifier/act_hist_gradient_boosting_classifier.py +144 -0
- iaml/actionables/predictors/classifier/act_knn.py +86 -0
- iaml/actionables/predictors/classifier/act_light_gbm_classifier.py +211 -0
- iaml/actionables/predictors/classifier/act_linear_discriminant_analysis.py +63 -0
- iaml/actionables/predictors/classifier/act_linear_svc.py +134 -0
- iaml/actionables/predictors/classifier/act_logistic_regression.py +92 -0
- iaml/actionables/predictors/classifier/act_mlp_classifier.py +107 -0
- iaml/actionables/predictors/classifier/act_multinomial_nb.py +76 -0
- iaml/actionables/predictors/classifier/act_passive_aggressive_classifier.py +141 -0
- iaml/actionables/predictors/classifier/act_quadratic_discriminant_analysis.py +72 -0
- iaml/actionables/predictors/classifier/act_randomforest.py +113 -0
- iaml/actionables/predictors/classifier/act_ridge_classifier.py +116 -0
- iaml/actionables/predictors/classifier/act_sgd_classifier.py +149 -0
- iaml/actionables/predictors/classifier/act_svm_svc.py +88 -0
- iaml/actionables/predictors/classifier/act_xgboost.py +111 -0
- iaml/actionables/predictors/regressor/__init__.py +27 -0
- iaml/actionables/predictors/regressor/act_ada_boost_regressor.py +75 -0
- iaml/actionables/predictors/regressor/act_ard_regression.py +95 -0
- iaml/actionables/predictors/regressor/act_catboost_regressor.py +134 -0
- iaml/actionables/predictors/regressor/act_decision_tree_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_elastic_net_regressor.py +109 -0
- iaml/actionables/predictors/regressor/act_extra_trees_regressor.py +113 -0
- iaml/actionables/predictors/regressor/act_gaussian_process_regressor.py +55 -0
- iaml/actionables/predictors/regressor/act_gboost_regressor.py +95 -0
- iaml/actionables/predictors/regressor/act_hist_gradient_boosting_regressor.py +105 -0
- iaml/actionables/predictors/regressor/act_huber_regressor.py +101 -0
- iaml/actionables/predictors/regressor/act_knn_regressor.py +86 -0
- iaml/actionables/predictors/regressor/act_lasso_regressor.py +103 -0
- iaml/actionables/predictors/regressor/act_light_gbm_regressor.py +201 -0
- iaml/actionables/predictors/regressor/act_linear_regression.py +43 -0
- iaml/actionables/predictors/regressor/act_mlp_regressor.py +104 -0
- iaml/actionables/predictors/regressor/act_poisson_regressor.py +111 -0
- iaml/actionables/predictors/regressor/act_quantile_regressor.py +87 -0
- iaml/actionables/predictors/regressor/act_randomforest_regressor.py +116 -0
- iaml/actionables/predictors/regressor/act_ransac_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_ridge_regressor.py +107 -0
- iaml/actionables/predictors/regressor/act_sgd_regressor.py +106 -0
- iaml/actionables/predictors/regressor/act_svm_svr.py +81 -0
- iaml/actionables/predictors/regressor/act_xgboost_regressor.py +97 -0
- iaml/actionables/predictors/survival/__init__.py +12 -0
- iaml/actionables/predictors/survival/act_aalen_additive_model.py +83 -0
- iaml/actionables/predictors/survival/act_cox.py +110 -0
- iaml/actionables/predictors/survival/act_coxnet_survival_analysis.py +134 -0
- iaml/actionables/predictors/survival/act_extra_survival_trees.py +101 -0
- iaml/actionables/predictors/survival/act_fast_survival_svm.py +102 -0
- iaml/actionables/predictors/survival/act_gradient_boosting_survival_analysis.py +93 -0
- iaml/actionables/predictors/survival/act_random_survival_forest.py +91 -0
- iaml/actionables/predictors/survival/act_survival_component_wise_gboost.py +80 -0
- iaml/actionables/predictors/survival/act_survival_tree.py +120 -0
- iaml/actionables/predictors/survival/act_survival_xgboost.py +9 -0
- iaml/actionables/predictors/survival/act_weibull_aft.py +230 -0
- iaml/cache.py +61 -0
- iaml/cache_keys.py +57 -0
- iaml/candidate.py +736 -0
- iaml/core_dispatcher.py +125 -0
- iaml/data_type.py +11 -0
- iaml/dataset.py +506 -0
- iaml/decorators/__init__.py +3 -0
- iaml/decorators/all.py +4 -0
- iaml/decorators/is_step.py +45 -0
- iaml/decorators/runner.py +100 -0
- iaml/explanation.py +112 -0
- iaml/iaml.py +1072 -0
- iaml/iaml_pipeline.py +600 -0
- iaml/logger.py +138 -0
- iaml/meta_explorer_step.py +62 -0
- iaml/meta_ordered_step.py +28 -0
- iaml/meta_partial_explorer_step.py +34 -0
- iaml/meta_singleton.py +24 -0
- iaml/metastep.py +211 -0
- iaml/metric.py +111 -0
- iaml/metric_plot.py +82 -0
- iaml/metrics/__init__.py +21 -0
- iaml/metrics/_classification.py +28 -0
- iaml/metrics/_survival_times.py +22 -0
- iaml/metrics/accuracy_metric.py +59 -0
- iaml/metrics/balanced_accuracy_metric.py +67 -0
- iaml/metrics/brier_score.py +90 -0
- iaml/metrics/classification_error_metric.py +66 -0
- iaml/metrics/concordance_index_ipcw.py +84 -0
- iaml/metrics/concordance_index_metric.py +67 -0
- iaml/metrics/cumulative_dynamic_auc.py +119 -0
- iaml/metrics/f1_score_metric.py +71 -0
- iaml/metrics/integrated_brier_score.py +98 -0
- iaml/metrics/integrated_brier_score_loss.py +41 -0
- iaml/metrics/mean_absolute_error_metric.py +46 -0
- iaml/metrics/mean_squared_error_metric.py +46 -0
- iaml/metrics/mean_squared_log_error_metric.py +49 -0
- iaml/metrics/median_absolute_error_metric.py +48 -0
- iaml/metrics/precision_metric.py +63 -0
- iaml/metrics/r2_score_metric.py +45 -0
- iaml/metrics/recall_metric.py +65 -0
- iaml/metrics/roc_auc_metric.py +50 -0
- iaml/metrics/specificity_metric.py +44 -0
- iaml/metrics/specificity_multiclass_metric.py +55 -0
- iaml/metrics/specificity_multilabel_metric.py +60 -0
- iaml/optimizers/__init__.py +5 -0
- iaml/optimizers/bayesian_optimizer.py +193 -0
- iaml/optimizers/genetic_optimizer.py +284 -0
- iaml/optimizers/optimizer.py +31 -0
- iaml/optimizers/random_optimizer.py +101 -0
- iaml/plot.py +138 -0
- iaml/plots/__init__.py +32 -0
- iaml/plots/bar_plot.py +141 -0
- iaml/plots/box_plot.py +166 -0
- iaml/plots/class_prediction_error_plot.py +37 -0
- iaml/plots/classification_report_plot.py +35 -0
- iaml/plots/confusion_matrix_plot.py +34 -0
- iaml/plots/correlation_heatmap_plot.py +201 -0
- iaml/plots/cumulative_hazard_plot.py +72 -0
- iaml/plots/density_plot.py +210 -0
- iaml/plots/histogram_plot.py +179 -0
- iaml/plots/kaplan_meier_comparison_plot.py +89 -0
- iaml/plots/line_plot.py +70 -0
- iaml/plots/missingness_heatmap_plot.py +203 -0
- iaml/plots/outlier_plot.py +217 -0
- iaml/plots/pair_plot.py +228 -0
- iaml/plots/precision_recall_curve_plot.py +86 -0
- iaml/plots/prediction_error_plot.py +34 -0
- iaml/plots/qq_plot.py +220 -0
- iaml/plots/residual_plot.py +38 -0
- iaml/plots/roc_dynamique_curve_plot.py +79 -0
- iaml/plots/rocauc_plot.py +96 -0
- iaml/plots/shap_plot.py +187 -0
- iaml/plots/target_distribution_plot.py +241 -0
- iaml/plots/violin_plot.py +206 -0
- iaml/predictor.py +139 -0
- iaml/reference.py +65 -0
- iaml/shared_cache.py +90 -0
- iaml/sklearn_preprocessor.py +74 -0
- iaml/splitters/__init__.py +3 -0
- iaml/splitters/kfold_splitter.py +32 -0
- iaml/splitters/random_splitter.py +26 -0
- iaml/stack.py +39 -0
- iaml/statistic.py +66 -0
- iaml/statistics/__init__.py +77 -0
- iaml/statistics/anova_statistic.py +80 -0
- iaml/statistics/cardinality_ratio_statistic.py +63 -0
- iaml/statistics/category_cooccurrence_statistic.py +79 -0
- iaml/statistics/chi_square_statistic.py +81 -0
- iaml/statistics/coef_variation_statistic.py +72 -0
- iaml/statistics/correlation_with_target.py +105 -0
- iaml/statistics/count.py +72 -0
- iaml/statistics/data_type_summary_statistic.py +74 -0
- iaml/statistics/duplicate_row_statistic.py +56 -0
- iaml/statistics/effect_size_statistic.py +129 -0
- iaml/statistics/entropy_statistic.py +69 -0
- iaml/statistics/event_rate_statistic.py +52 -0
- iaml/statistics/grouped_mean_statistic.py +60 -0
- iaml/statistics/iqr_statistic.py +66 -0
- iaml/statistics/kurtosis.py +50 -0
- iaml/statistics/mad_statistic.py +66 -0
- iaml/statistics/mean.py +61 -0
- iaml/statistics/median_statistic.py +61 -0
- iaml/statistics/minmax.py +60 -0
- iaml/statistics/missing_rate_statistic.py +62 -0
- iaml/statistics/mode.py +47 -0
- iaml/statistics/most_frequent_ratio.py +81 -0
- iaml/statistics/outlier_count_iqr_statistic.py +76 -0
- iaml/statistics/quantile.py +59 -0
- iaml/statistics/range.py +53 -0
- iaml/statistics/rare_category_rate.py +92 -0
- iaml/statistics/skewness.py +53 -0
- iaml/statistics/stdev.py +50 -0
- iaml/statistics/summary_table_statistic.py +60 -0
- iaml/statistics/time_by_group_statistic.py +83 -0
- iaml/statistics/time_summary_statistic.py +56 -0
- iaml/statistics/top_k_value_counts.py +68 -0
- iaml/statistics/unique_count_statistic.py +57 -0
- iaml/statistics/value_counts.py +63 -0
- iaml/statistics/variance.py +51 -0
- iaml/statistics/violin.py +63 -0
- iaml/step.py +600 -0
- iaml/step_cache.py +87 -0
- iaml/step_wrapper.py +79 -0
- iaml/timed_pool_executor.py +492 -0
- iaml/type_of_target.py +68 -0
- iaml/void_step.py +101 -0
- iaml/worker_manager.py +169 -0
- iaml/wrapper/__init__.py +4 -0
- iaml/wrapper/wrap_basic_gridsearch.py +68 -0
- iaml/wrapper/wrap_genetic_gridsearch.py +293 -0
- iaml/wrapper/wrap_iterative_gridsearch.py +399 -0
- pyiaml-1.0.0.dist-info/METADATA +802 -0
- pyiaml-1.0.0.dist-info/RECORD +279 -0
- pyiaml-1.0.0.dist-info/WHEEL +5 -0
- pyiaml-1.0.0.dist-info/licenses/LICENSE +674 -0
- pyiaml-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,270 @@
|
|
|
1
|
+
"""[STEP] Replace sentinel values with NaN"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from typing import Any
|
|
4
|
+
import numpy as np
|
|
5
|
+
import pandas as pd
|
|
6
|
+
from pandas.api.types import is_numeric_dtype, is_object_dtype, is_string_dtype
|
|
7
|
+
from ...actionable import Actionable
|
|
8
|
+
from ...candidate import Candidate
|
|
9
|
+
from ...data_type import DataType
|
|
10
|
+
from ...dataset import Dataset
|
|
11
|
+
from ...decorators.all import is_step
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@is_step('features_precleaning')
|
|
15
|
+
class ActSentinelToNaN(Actionable):
|
|
16
|
+
"""[STEP] Replace sentinel values with NaN"""
|
|
17
|
+
|
|
18
|
+
name: str = 'Replace sentinel values'
|
|
19
|
+
_description: str = textwrap.dedent('''\
|
|
20
|
+
Replace configured sentinel values (e.g., -999, "NA", "unknown") with NaN.''')
|
|
21
|
+
_description_long: str = textwrap.dedent('''\
|
|
22
|
+
Sentinel values are placeholders for missing data.
|
|
23
|
+
This step replaces common numeric and text sentinels with NaN so downstream
|
|
24
|
+
steps can treat missing values consistently.''')
|
|
25
|
+
_usage: str = "Use when numeric or text columns contain sentinel placeholders (e.g., -999, 'NA') and you want them treated as missing; run before ActCoerceNumericStrings. Applicable to datasets with explicit sentinel codes or empty-string markers across numeric and categorical fields. Avoid when sentinel values are meaningful domain codes or you should remove sparse fields via ActDropHighMissingColumns."
|
|
26
|
+
refs: list[dict[str, Any]] = []
|
|
27
|
+
|
|
28
|
+
def __init__(self) -> None:
|
|
29
|
+
self.configuration = {
|
|
30
|
+
'numeric_sentinels': {
|
|
31
|
+
'description': textwrap.dedent('''\
|
|
32
|
+
Numeric sentinel values to replace with NaN.'''),
|
|
33
|
+
'default': [-999, -9999, -99999]
|
|
34
|
+
},
|
|
35
|
+
'text_sentinels': {
|
|
36
|
+
'description': textwrap.dedent('''\
|
|
37
|
+
Text sentinel values to replace with NaN.'''),
|
|
38
|
+
'default': ['NA', 'N/A', 'NULL', 'NONE', 'UNKNOWN', 'MISSING', 'NAN']
|
|
39
|
+
},
|
|
40
|
+
'case_insensitive': {
|
|
41
|
+
'description': 'Match text sentinels ignoring case.',
|
|
42
|
+
'default': True
|
|
43
|
+
},
|
|
44
|
+
'strip_whitespace': {
|
|
45
|
+
'description': 'Trim whitespace before matching text sentinels.',
|
|
46
|
+
'default': True
|
|
47
|
+
},
|
|
48
|
+
'include_empty_string': {
|
|
49
|
+
'description': 'Treat empty strings as missing values.',
|
|
50
|
+
'default': True
|
|
51
|
+
},
|
|
52
|
+
'numeric_in_text': {
|
|
53
|
+
'description': 'Also match numeric sentinels stored as text.',
|
|
54
|
+
'default': True
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
self.numeric_columns: list[str] = []
|
|
58
|
+
self.text_columns: list[str] = []
|
|
59
|
+
self.numeric_sentinels: list[float] = []
|
|
60
|
+
self.text_sentinels: list[str] = []
|
|
61
|
+
|
|
62
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
63
|
+
self.numeric_columns, self.text_columns = self.__candidate_columns(dataset)
|
|
64
|
+
self.numeric_sentinels, self.text_sentinels = self.__normalized_sentinels()
|
|
65
|
+
|
|
66
|
+
self.explanations = []
|
|
67
|
+
if dataset.X.empty:
|
|
68
|
+
return self
|
|
69
|
+
|
|
70
|
+
if self.numeric_sentinels:
|
|
71
|
+
for column in self.numeric_columns:
|
|
72
|
+
count = int(dataset.X[column].isin(self.numeric_sentinels).sum())
|
|
73
|
+
if count:
|
|
74
|
+
self.explanations.append(
|
|
75
|
+
f'Replaced {count} numeric sentinel values in **`{column}`**.'
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
if self.text_sentinels:
|
|
79
|
+
for column in self.text_columns:
|
|
80
|
+
mask = self.__text_sentinel_mask(dataset.X[column], self.text_sentinels)
|
|
81
|
+
count = int(mask.sum())
|
|
82
|
+
if count:
|
|
83
|
+
self.explanations.append(
|
|
84
|
+
f'Replaced {count} text sentinel values in **`{column}`**.'
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
return self
|
|
88
|
+
|
|
89
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
90
|
+
"""Replace sentinel values with NaN.
|
|
91
|
+
|
|
92
|
+
:param pd.DataFrame X: DataFrame to transform.
|
|
93
|
+
:return: Transformed DataFrame.
|
|
94
|
+
"""
|
|
95
|
+
if X.empty:
|
|
96
|
+
return X
|
|
97
|
+
|
|
98
|
+
if self.numeric_sentinels and self.numeric_columns:
|
|
99
|
+
columns = [column for column in self.numeric_columns if column in X.columns]
|
|
100
|
+
if columns:
|
|
101
|
+
X[columns] = X[columns].replace(self.numeric_sentinels, np.nan)
|
|
102
|
+
|
|
103
|
+
if self.text_sentinels and self.text_columns:
|
|
104
|
+
for column in self.text_columns:
|
|
105
|
+
if column not in X.columns:
|
|
106
|
+
continue
|
|
107
|
+
mask = self.__text_sentinel_mask(X[column], self.text_sentinels)
|
|
108
|
+
if mask.any():
|
|
109
|
+
X[column] = X[column].mask(mask, np.nan)
|
|
110
|
+
|
|
111
|
+
return X
|
|
112
|
+
|
|
113
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
114
|
+
if candidate is None or candidate.dataset.X.empty:
|
|
115
|
+
return 0.0
|
|
116
|
+
|
|
117
|
+
numeric_sentinels, text_sentinels = self.__normalized_sentinels()
|
|
118
|
+
if not numeric_sentinels and not text_sentinels:
|
|
119
|
+
return 0.0
|
|
120
|
+
|
|
121
|
+
numeric_columns, text_columns = self.__candidate_columns(candidate.dataset)
|
|
122
|
+
total = candidate.dataset.X.size or 1
|
|
123
|
+
count = self.__count_sentinels(
|
|
124
|
+
candidate.dataset,
|
|
125
|
+
numeric_columns,
|
|
126
|
+
text_columns,
|
|
127
|
+
numeric_sentinels,
|
|
128
|
+
text_sentinels
|
|
129
|
+
)
|
|
130
|
+
return min(1.5, 0.5 + count / total)
|
|
131
|
+
|
|
132
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
133
|
+
if dataset.X.empty:
|
|
134
|
+
return False
|
|
135
|
+
|
|
136
|
+
numeric_sentinels, text_sentinels = self.__normalized_sentinels()
|
|
137
|
+
if not numeric_sentinels and not text_sentinels:
|
|
138
|
+
return False
|
|
139
|
+
|
|
140
|
+
numeric_columns, text_columns = self.__candidate_columns(dataset)
|
|
141
|
+
if numeric_sentinels:
|
|
142
|
+
for column in numeric_columns:
|
|
143
|
+
if dataset.X[column].isin(numeric_sentinels).any():
|
|
144
|
+
return True
|
|
145
|
+
|
|
146
|
+
if text_sentinels:
|
|
147
|
+
for column in text_columns:
|
|
148
|
+
if self.__text_sentinel_mask(dataset.X[column], text_sentinels).any():
|
|
149
|
+
return True
|
|
150
|
+
|
|
151
|
+
return False
|
|
152
|
+
|
|
153
|
+
def __candidate_columns(self, dataset: Dataset) -> tuple[list[str], list[str]]:
|
|
154
|
+
numeric_candidates = set(dataset.get_columns_names_by_type(DataType.NUMERIC))
|
|
155
|
+
text_candidates = set(dataset.get_columns_names_by_type(
|
|
156
|
+
[DataType.CATEGORICAL, DataType.TEXT, DataType.SHORT_TEXT]
|
|
157
|
+
))
|
|
158
|
+
|
|
159
|
+
numeric_columns = []
|
|
160
|
+
text_columns = []
|
|
161
|
+
|
|
162
|
+
for column in dataset.X.columns:
|
|
163
|
+
if column in numeric_candidates:
|
|
164
|
+
numeric_columns.append(column)
|
|
165
|
+
continue
|
|
166
|
+
if column in text_candidates:
|
|
167
|
+
text_columns.append(column)
|
|
168
|
+
continue
|
|
169
|
+
|
|
170
|
+
series = dataset.X[column]
|
|
171
|
+
if is_numeric_dtype(series):
|
|
172
|
+
numeric_columns.append(column)
|
|
173
|
+
elif is_string_dtype(series) or is_object_dtype(series):
|
|
174
|
+
text_columns.append(column)
|
|
175
|
+
|
|
176
|
+
return numeric_columns, text_columns
|
|
177
|
+
|
|
178
|
+
def __normalized_sentinels(self) -> tuple[list[float], list[str]]:
|
|
179
|
+
numeric_sentinels = self.__normalize_numeric_sentinels()
|
|
180
|
+
text_sentinels = self.__normalize_text_sentinels(numeric_sentinels)
|
|
181
|
+
return numeric_sentinels, text_sentinels
|
|
182
|
+
|
|
183
|
+
def __normalize_numeric_sentinels(self) -> list[float]:
|
|
184
|
+
values = self.get_config('numeric_sentinels') or []
|
|
185
|
+
cleaned: list[float] = []
|
|
186
|
+
seen: set[float] = set()
|
|
187
|
+
|
|
188
|
+
for value in values:
|
|
189
|
+
if value is None or isinstance(value, bool):
|
|
190
|
+
continue
|
|
191
|
+
try:
|
|
192
|
+
num = float(value)
|
|
193
|
+
except (TypeError, ValueError):
|
|
194
|
+
continue
|
|
195
|
+
if np.isnan(num) or num in seen:
|
|
196
|
+
continue
|
|
197
|
+
cleaned.append(num)
|
|
198
|
+
seen.add(num)
|
|
199
|
+
|
|
200
|
+
return cleaned
|
|
201
|
+
|
|
202
|
+
def __normalize_text_sentinels(self, numeric_sentinels: list[float]) -> list[str]:
|
|
203
|
+
values = list(self.get_config('text_sentinels') or [])
|
|
204
|
+
if self.get_config('include_empty_string'):
|
|
205
|
+
values.append('')
|
|
206
|
+
if self.get_config('numeric_in_text'):
|
|
207
|
+
values.extend(self.__numeric_sentinels_as_text(numeric_sentinels))
|
|
208
|
+
|
|
209
|
+
cleaned: list[str] = []
|
|
210
|
+
seen: set[str] = set()
|
|
211
|
+
|
|
212
|
+
for value in values:
|
|
213
|
+
if value is None:
|
|
214
|
+
continue
|
|
215
|
+
text = str(value)
|
|
216
|
+
if self.get_config('strip_whitespace'):
|
|
217
|
+
text = text.strip()
|
|
218
|
+
if self.get_config('case_insensitive'):
|
|
219
|
+
text = text.lower()
|
|
220
|
+
if text == '' and not self.get_config('include_empty_string'):
|
|
221
|
+
continue
|
|
222
|
+
if text not in seen:
|
|
223
|
+
cleaned.append(text)
|
|
224
|
+
seen.add(text)
|
|
225
|
+
|
|
226
|
+
return cleaned
|
|
227
|
+
|
|
228
|
+
def __numeric_sentinels_as_text(self, numeric_sentinels: list[float]) -> list[str]:
|
|
229
|
+
values: list[str] = []
|
|
230
|
+
for value in numeric_sentinels:
|
|
231
|
+
if np.isnan(value):
|
|
232
|
+
continue
|
|
233
|
+
if float(value).is_integer():
|
|
234
|
+
int_value = int(value)
|
|
235
|
+
values.append(str(int_value))
|
|
236
|
+
values.append(str(float(int_value)))
|
|
237
|
+
else:
|
|
238
|
+
values.append(str(value))
|
|
239
|
+
return values
|
|
240
|
+
|
|
241
|
+
def __text_sentinel_mask(self, series: pd.Series, sentinels: list[str]) -> pd.Series:
|
|
242
|
+
if not sentinels or series.empty:
|
|
243
|
+
return pd.Series(False, index=series.index)
|
|
244
|
+
|
|
245
|
+
values = series.astype('string')
|
|
246
|
+
if self.get_config('strip_whitespace'):
|
|
247
|
+
values = values.str.strip()
|
|
248
|
+
if self.get_config('case_insensitive'):
|
|
249
|
+
values = values.str.lower()
|
|
250
|
+
mask = values.isin(sentinels)
|
|
251
|
+
return mask.fillna(False)
|
|
252
|
+
|
|
253
|
+
def __count_sentinels(self,
|
|
254
|
+
dataset: Dataset,
|
|
255
|
+
numeric_columns: list[str],
|
|
256
|
+
text_columns: list[str],
|
|
257
|
+
numeric_sentinels: list[float],
|
|
258
|
+
text_sentinels: list[str]
|
|
259
|
+
) -> int:
|
|
260
|
+
count = 0
|
|
261
|
+
if numeric_sentinels:
|
|
262
|
+
for column in numeric_columns:
|
|
263
|
+
count += int(dataset.X[column].isin(numeric_sentinels).sum())
|
|
264
|
+
if text_sentinels:
|
|
265
|
+
for column in text_columns:
|
|
266
|
+
count += int(self.__text_sentinel_mask(
|
|
267
|
+
dataset.X[column],
|
|
268
|
+
text_sentinels
|
|
269
|
+
).sum())
|
|
270
|
+
return count
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""[STEP] Trim spaces on each columns"""
|
|
2
|
+
import textwrap
|
|
3
|
+
from pandas.api.types import is_object_dtype
|
|
4
|
+
import pandas as pd
|
|
5
|
+
from ...actionable import Actionable
|
|
6
|
+
from ...dataset import Dataset
|
|
7
|
+
from ...candidate import Candidate
|
|
8
|
+
from ...decorators.all import is_step
|
|
9
|
+
|
|
10
|
+
@is_step('features_precleaning')
|
|
11
|
+
class ActTrimSpaces(Actionable):
|
|
12
|
+
"""
|
|
13
|
+
[STEP] Trim spaces on each columns
|
|
14
|
+
"""
|
|
15
|
+
name = "Trim columns spaces"
|
|
16
|
+
_description = textwrap.dedent('''\
|
|
17
|
+
This step trim spaces on each columns.
|
|
18
|
+
It helps preventing errors on the dataset when casting columns with spaces''')
|
|
19
|
+
_description_long = textwrap.dedent('''\
|
|
20
|
+
In datasets, columns with spaces can be an issue.
|
|
21
|
+
This step remove spaces in front and back of columns values.
|
|
22
|
+
This ensures that columns can be casted correctly with having spaces throwing
|
|
23
|
+
an error.''')
|
|
24
|
+
_usage = "Use when columns or string values have leading/trailing spaces; use before ActCoerceNumericStrings or ActNormalizeColumnNames. Applicable to object/string columns and column names. Avoid when spaces are meaningful or data is already clean."
|
|
25
|
+
|
|
26
|
+
refs = []
|
|
27
|
+
|
|
28
|
+
def __init__(self):
|
|
29
|
+
self.configuration = {
|
|
30
|
+
'left_trim': {
|
|
31
|
+
'description': "Trim all columns leading spaces",
|
|
32
|
+
'default': True
|
|
33
|
+
},
|
|
34
|
+
'right_trim': {
|
|
35
|
+
'description': "Trim all columns ending spaces",
|
|
36
|
+
'default': True
|
|
37
|
+
},
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
def fit(self, dataset: Dataset): # pylint: disable=unused-argument
|
|
41
|
+
return self
|
|
42
|
+
|
|
43
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
44
|
+
"""Apply the dropping of rows with at least 40% empty columns.
|
|
45
|
+
|
|
46
|
+
:param pd.DataFrame X: The dataframe to clean.
|
|
47
|
+
:return: The cleaned dataframe.
|
|
48
|
+
"""
|
|
49
|
+
if self.get_config('left_trim'):
|
|
50
|
+
X = X.rename(columns=lambda x: x.lstrip())
|
|
51
|
+
for col in X.columns:
|
|
52
|
+
if isinstance(X[col].dtype, str) or is_object_dtype(X[col]):
|
|
53
|
+
X[col] = X[col].apply(
|
|
54
|
+
lambda x: x.lstrip() if isinstance(x, str) else x,
|
|
55
|
+
convert_dtype=False)
|
|
56
|
+
|
|
57
|
+
if self.get_config('right_trim'):
|
|
58
|
+
X = X.rename(columns=lambda x: x.rstrip())
|
|
59
|
+
for col in X.columns:
|
|
60
|
+
if isinstance(X[col].dtype, str) or is_object_dtype(X[col]):
|
|
61
|
+
X[col] = X[col].astype(object).apply(
|
|
62
|
+
lambda x: x.rstrip() if isinstance(x, str) else x
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
return X
|
|
66
|
+
|
|
67
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
68
|
+
return 1.5
|
|
69
|
+
|
|
70
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
71
|
+
cond = (
|
|
72
|
+
dataset.X.apply(
|
|
73
|
+
lambda x: (
|
|
74
|
+
x.astype(str).str.strip() if (isinstance(x, str) or is_object_dtype(x)) else x
|
|
75
|
+
) != x
|
|
76
|
+
).stack().any()
|
|
77
|
+
or (dataset.X.columns.str.strip() != dataset.X.columns).any()
|
|
78
|
+
)
|
|
79
|
+
return cond
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Usually last step before predictor, try features decomposition or aggregation.
|
|
3
|
+
Example : PCA
|
|
4
|
+
"""
|
|
5
|
+
from .act_pca import ActPCA
|
|
6
|
+
from .act_kernel_pca import ActKernelPCA
|
|
7
|
+
from .act_feature_agglomeration import ActFeatureAgglomeration
|
|
8
|
+
from .act_nystroem import ActNystroem
|
|
9
|
+
from .act_rbf_sampler import ActRBFSampler
|
|
10
|
+
from .act_select_percentile import ActSelectPercentile
|
|
11
|
+
from .act_power_transformer import ActPowerTransformer
|
|
12
|
+
from .act_quantile_transformer import ActQuantileTransformer
|
|
13
|
+
from .act_truncated_svd import ActTruncatedSVD
|
|
14
|
+
from .act_fast_ica import ActFastICA
|
|
15
|
+
from .act_sparse_random_projection import ActSparseRandomProjection
|
|
16
|
+
from .act_k_bins_discretizer import ActKBinsDiscretizer
|
|
17
|
+
from .act_log_transformer import ActLogTransformer
|
|
18
|
+
from .act_k_means_features import ActKMeansFeatures
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
"""Manual cyclical date encoding, available only through an explicit module import.
|
|
2
|
+
|
|
3
|
+
Date conversion and cleaning precede automatic feature preprocessing, so this
|
|
4
|
+
component needs an explicit position while datetime columns are still available.
|
|
5
|
+
See docs/component_status.rst.
|
|
6
|
+
"""
|
|
7
|
+
import textwrap
|
|
8
|
+
import numpy as np
|
|
9
|
+
import pandas as pd
|
|
10
|
+
from ...actionable import Actionable
|
|
11
|
+
from ...candidate import Candidate
|
|
12
|
+
from ...dataset import Dataset
|
|
13
|
+
from ...data_type import DataType
|
|
14
|
+
from ...decorators.all import is_step
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@is_step('experimental')
|
|
18
|
+
class ActCyclicalDateEncoding(Actionable):
|
|
19
|
+
"""[STEP] Encode date columns with cyclical sine/cosine features."""
|
|
20
|
+
|
|
21
|
+
name: str = "Cyclical Date Encoding"
|
|
22
|
+
_description: str = "Encode date columns into sine/cosine features to capture cyclicity"
|
|
23
|
+
_usage: str = "Use when datetime columns have cyclic parts (month/weekday/hour) and models need numeric features. Applicable to datetime columns with clear periodicity. Avoid when dates are non-cyclic or already encoded; consider ActKBinsDiscretizer or ActLogTransformer instead."
|
|
24
|
+
_description_long: str = textwrap.dedent('''\
|
|
25
|
+
Cyclical encoding turns calendar components (month, weekday, hour, etc.)
|
|
26
|
+
into sine and cosine values. This preserves the circular nature of time,
|
|
27
|
+
so end and start points on a cycle stay close in feature space while
|
|
28
|
+
providing numeric values suitable for ML models.
|
|
29
|
+
''')
|
|
30
|
+
|
|
31
|
+
_COMPONENTS: dict[str, tuple[str, int]] = {
|
|
32
|
+
'month': ('month', 12),
|
|
33
|
+
'dayofweek': ('dayofweek', 7),
|
|
34
|
+
'day': ('day', 31),
|
|
35
|
+
'hour': ('hour', 24),
|
|
36
|
+
'minute': ('minute', 60),
|
|
37
|
+
'second': ('second', 60),
|
|
38
|
+
}
|
|
39
|
+
_ALIASES: dict[str, str] = {
|
|
40
|
+
'weekday': 'dayofweek',
|
|
41
|
+
'dow': 'dayofweek',
|
|
42
|
+
'dayofmonth': 'day',
|
|
43
|
+
'dom': 'day',
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
def __init__(self) -> None:
|
|
47
|
+
self.configuration = {
|
|
48
|
+
'components': {
|
|
49
|
+
'description': 'Date components to encode (month, dayofweek, day, hour, minute, second).',
|
|
50
|
+
'default': ('month', 'dayofweek', 'day', 'hour')
|
|
51
|
+
},
|
|
52
|
+
'drop_original': {
|
|
53
|
+
'description': 'Drop original date columns after encoding.',
|
|
54
|
+
'default': True
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
self.columns: list[str] = []
|
|
58
|
+
self.encoding_plan: dict[str, list[tuple[str, str, int, str, str]]] = {}
|
|
59
|
+
self.drop_original: bool = True
|
|
60
|
+
|
|
61
|
+
@staticmethod
|
|
62
|
+
def _is_datetime(series: pd.Series) -> bool:
|
|
63
|
+
return pd.api.types.is_datetime64_any_dtype(series)
|
|
64
|
+
|
|
65
|
+
@staticmethod
|
|
66
|
+
def _coerce_bool(value: object, default: bool) -> bool:
|
|
67
|
+
if isinstance(value, bool):
|
|
68
|
+
return value
|
|
69
|
+
if isinstance(value, str):
|
|
70
|
+
lowered = value.strip().lower()
|
|
71
|
+
if lowered in ('1', 'true', 'yes', 'y'):
|
|
72
|
+
return True
|
|
73
|
+
if lowered in ('0', 'false', 'no', 'n'):
|
|
74
|
+
return False
|
|
75
|
+
return default
|
|
76
|
+
|
|
77
|
+
@staticmethod
|
|
78
|
+
def _unique_name(name: str, reserved: set[str]) -> str:
|
|
79
|
+
if name not in reserved:
|
|
80
|
+
return name
|
|
81
|
+
idx = 1
|
|
82
|
+
candidate = f"{name}_{idx}"
|
|
83
|
+
while candidate in reserved:
|
|
84
|
+
idx += 1
|
|
85
|
+
candidate = f"{name}_{idx}"
|
|
86
|
+
return candidate
|
|
87
|
+
|
|
88
|
+
def _resolve_components(self) -> list[str]:
|
|
89
|
+
raw = self.get_config('components')
|
|
90
|
+
if raw is None:
|
|
91
|
+
raw_components = list(self._COMPONENTS.keys())
|
|
92
|
+
elif isinstance(raw, str):
|
|
93
|
+
raw_components = [raw]
|
|
94
|
+
else:
|
|
95
|
+
try:
|
|
96
|
+
raw_components = list(raw)
|
|
97
|
+
except TypeError:
|
|
98
|
+
raw_components = [str(raw)]
|
|
99
|
+
|
|
100
|
+
components: list[str] = []
|
|
101
|
+
seen: set[str] = set()
|
|
102
|
+
for component in raw_components:
|
|
103
|
+
if component is None:
|
|
104
|
+
continue
|
|
105
|
+
component = str(component).strip().lower()
|
|
106
|
+
if not component:
|
|
107
|
+
continue
|
|
108
|
+
component = self._ALIASES.get(component, component)
|
|
109
|
+
if component not in self._COMPONENTS:
|
|
110
|
+
continue
|
|
111
|
+
if component in seen:
|
|
112
|
+
continue
|
|
113
|
+
seen.add(component)
|
|
114
|
+
components.append(component)
|
|
115
|
+
|
|
116
|
+
if not components:
|
|
117
|
+
components = list(self._COMPONENTS.keys())
|
|
118
|
+
|
|
119
|
+
return components
|
|
120
|
+
|
|
121
|
+
def _build_plan(self, dataset: Dataset) -> dict[str, list[tuple[str, str, int, str, str]]]:
|
|
122
|
+
columns = dataset.get_columns_names_by_type(DataType.DATE)
|
|
123
|
+
if not columns or dataset.X.empty:
|
|
124
|
+
return {}
|
|
125
|
+
|
|
126
|
+
components = self._resolve_components()
|
|
127
|
+
if not components:
|
|
128
|
+
return {}
|
|
129
|
+
|
|
130
|
+
reserved = set(dataset.X.columns)
|
|
131
|
+
plan: dict[str, list[tuple[str, str, int, str, str]]] = {}
|
|
132
|
+
|
|
133
|
+
for column in columns:
|
|
134
|
+
if column not in dataset.X.columns:
|
|
135
|
+
continue
|
|
136
|
+
series = dataset.X[column]
|
|
137
|
+
if not self._is_datetime(series):
|
|
138
|
+
continue
|
|
139
|
+
|
|
140
|
+
column_plan: list[tuple[str, str, int, str, str]] = []
|
|
141
|
+
for component in components:
|
|
142
|
+
accessor, period = self._COMPONENTS[component]
|
|
143
|
+
values = getattr(series.dt, accessor)
|
|
144
|
+
if values.nunique(dropna=True) <= 1:
|
|
145
|
+
continue
|
|
146
|
+
base = f"{column}_{component}"
|
|
147
|
+
sin_name = self._unique_name(f"{base}_sin", reserved)
|
|
148
|
+
reserved.add(sin_name)
|
|
149
|
+
cos_name = self._unique_name(f"{base}_cos", reserved)
|
|
150
|
+
reserved.add(cos_name)
|
|
151
|
+
column_plan.append((component, accessor, period, sin_name, cos_name))
|
|
152
|
+
|
|
153
|
+
if column_plan:
|
|
154
|
+
plan[column] = column_plan
|
|
155
|
+
|
|
156
|
+
return plan
|
|
157
|
+
|
|
158
|
+
def fit(self, dataset: Dataset) -> Actionable:
|
|
159
|
+
self.encoding_plan = self._build_plan(dataset)
|
|
160
|
+
self.columns = list(self.encoding_plan.keys())
|
|
161
|
+
self.drop_original = self._coerce_bool(self.get_config('drop_original'), True)
|
|
162
|
+
self.explanations = []
|
|
163
|
+
|
|
164
|
+
for column, parts in self.encoding_plan.items():
|
|
165
|
+
components = ", ".join([part[0] for part in parts])
|
|
166
|
+
self.explanations.append(
|
|
167
|
+
f'Encoded date column **`{column}`** into cyclical features ({components}).'
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
return self
|
|
171
|
+
|
|
172
|
+
def transform(self, X: pd.DataFrame) -> pd.DataFrame:
|
|
173
|
+
"""Apply cyclical encoding to date columns.
|
|
174
|
+
|
|
175
|
+
:param pd.DataFrame X: DataFrame to transform
|
|
176
|
+
:return: Transformed dataset
|
|
177
|
+
"""
|
|
178
|
+
if not self.encoding_plan:
|
|
179
|
+
return X
|
|
180
|
+
|
|
181
|
+
for column, parts in self.encoding_plan.items():
|
|
182
|
+
if column not in X.columns:
|
|
183
|
+
continue
|
|
184
|
+
series = X[column]
|
|
185
|
+
if not self._is_datetime(series):
|
|
186
|
+
series = pd.to_datetime(series, errors='coerce')
|
|
187
|
+
|
|
188
|
+
for _, accessor, period, sin_name, cos_name in parts:
|
|
189
|
+
values = getattr(series.dt, accessor).astype(float)
|
|
190
|
+
angles = (2.0 * np.pi * values) / float(period)
|
|
191
|
+
X[sin_name] = np.sin(angles)
|
|
192
|
+
X[cos_name] = np.cos(angles)
|
|
193
|
+
|
|
194
|
+
if self.drop_original and self.columns:
|
|
195
|
+
to_drop = [column for column in self.columns if column in X.columns]
|
|
196
|
+
if to_drop:
|
|
197
|
+
X = X.drop(columns=to_drop)
|
|
198
|
+
|
|
199
|
+
return X
|
|
200
|
+
|
|
201
|
+
def suitable(self, dataset: Dataset) -> bool:
|
|
202
|
+
if dataset.X.empty:
|
|
203
|
+
return False
|
|
204
|
+
return bool(self._build_plan(dataset))
|
|
205
|
+
|
|
206
|
+
def priorize(self, candidate: Candidate = None) -> float:
|
|
207
|
+
if candidate is None:
|
|
208
|
+
return 0.0
|
|
209
|
+
dataset = candidate.dataset
|
|
210
|
+
if dataset.X.empty:
|
|
211
|
+
return 0.0
|
|
212
|
+
return 0.5 if self._build_plan(dataset) else 0.0
|