SuperModelingFactory 0.7.1__tar.gz → 0.7.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Insights.py +23 -1
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Screen.py +36 -16
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Screen_Gates.py +26 -4
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/WOE_Engine_Feature_Patch.py +38 -4
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Weighted_Screen.py +164 -20
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/LRM_Tool.py +36 -2
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/LRM_Tool.pyi +1 -1
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/feature_validation.py +61 -7
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Monotone_Binner.py +566 -86
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/__init__.py +1 -1
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/PKG-INFO +2 -2
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/README.md +1 -1
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/PKG-INFO +2 -2
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/pyproject.toml +1 -1
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/setup.py +1 -1
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/ExcelMaster/ExcelFormatTool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/ExcelMaster/ExcelMaster.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/ExcelMaster/Template.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/ExcelMaster/Utility.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/ExcelMaster/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/LICENSE +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/MANIFEST.in +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Binning_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Binning_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Check_DuckDB_Compatibility.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Json_Data_Converter.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Model_Registry_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/ODPS_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Parallel_Engine.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Parallel_ODPS_Manager.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Proc_Compare.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Slope_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Slope_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/XOR_Encryptor.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/XOR_Encryptor.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/kDataFrame.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/kDataFrame.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/sample_weight_utils.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/utils.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Evaluation_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Evaluation_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Model_Eval_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Model_Eval_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/evaluate_model.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/evaluate_model.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/weighted_eval_utils.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Explainability/Coalition_Structure.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Explainability/Model_Explainer.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Explainability/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Distribution_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Distribution_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Insights.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/ODPS_Distribution_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/ODPS_Distribution_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/PSI_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/PSI_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/Backward_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/Backward_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/GBM_Search_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/GBM_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/GBM_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/_common.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/credit_model.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/field_meta.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/mock_sample.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/orchestrator.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/reject_inference.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/sample_analysis.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/score_comparison.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/score_consistency_uat.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/screening_artifact.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Distribution_Adaptation.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Distribution_Adaptation.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Reject_Infer.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Reject_Infer.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Sample_Split.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Sample_Split.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/UAT/UAT_Consistency_Checker.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/UAT/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Adapter.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Adapter.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Master.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Master.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Monotone_Binner.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Plot_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Plot_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Report_Builder.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Report_Builder.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/plot_woe_tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/plot_woe_tool.pyi +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/nan_guard.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/robust.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/sentinels.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/KaiTi.ttf +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/WeiRuanYaHei.ttf +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/simsun.ttc +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Report/Report_Tool.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Report/__init__.py +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/SOURCES.txt +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/dependency_links.txt +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/not-zip-safe +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/requires.txt +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/top_level.txt +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/requirements.txt +0 -0
- {supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/setup.cfg +0 -0
{supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Insights.py
RENAMED
|
@@ -419,6 +419,7 @@ class CorrelationFilter:
|
|
|
419
419
|
self._corr_matrix_cache = None
|
|
420
420
|
self._corr_matrix_excluded = set()
|
|
421
421
|
self._metric_summary_cache = None
|
|
422
|
+
self._correlation_decision_trace = []
|
|
422
423
|
|
|
423
424
|
def _corr_matrix_frame(self, varlist):
|
|
424
425
|
# 0.7.0-R1: numeric-subset correlation base. A pure-numeric varlist is
|
|
@@ -546,7 +547,27 @@ class CorrelationFilter:
|
|
|
546
547
|
if fnl_selected_var not in selected_varlist:
|
|
547
548
|
selected_varlist.append(fnl_selected_var)
|
|
548
549
|
|
|
549
|
-
|
|
550
|
+
newly_removed = [
|
|
551
|
+
x for x in correlated_list
|
|
552
|
+
if x != fnl_selected_var and x not in removed_varlist
|
|
553
|
+
]
|
|
554
|
+
metric_map = fnl_summary.set_index("var")[name_mapping[base_metric]].to_dict()
|
|
555
|
+
positions = {name: idx for idx, name in enumerate(varlist)}
|
|
556
|
+
for dropped_var in newly_removed:
|
|
557
|
+
decision_pair = [fnl_selected_var, dropped_var]
|
|
558
|
+
decision_pair.sort(key=positions.__getitem__)
|
|
559
|
+
var_a, var_b = decision_pair
|
|
560
|
+
corr_value = self._corr_matrix_cache.loc[var_a, var_b]
|
|
561
|
+
self._correlation_decision_trace.append({
|
|
562
|
+
"var_a": var_a,
|
|
563
|
+
"var_b": var_b,
|
|
564
|
+
"corr": float(corr_value),
|
|
565
|
+
"iv_a": float(metric_map.get(var_a, 0.0)),
|
|
566
|
+
"iv_b": float(metric_map.get(var_b, 0.0)),
|
|
567
|
+
"kept": fnl_selected_var,
|
|
568
|
+
"dropped": dropped_var,
|
|
569
|
+
})
|
|
570
|
+
removed_varlist += newly_removed
|
|
550
571
|
|
|
551
572
|
if var not in correlated_dict:
|
|
552
573
|
correlated_dict[var] = {}
|
|
@@ -585,6 +606,7 @@ class CorrelationFilter:
|
|
|
585
606
|
>>> filter_analyzer = CorrelationFilter(df, 'target')
|
|
586
607
|
>>> keep_vars = filter_analyzer.remove_highly_correlated(['var1', 'var2', 'var3'])
|
|
587
608
|
"""
|
|
609
|
+
self._correlation_decision_trace = []
|
|
588
610
|
self._corr_matrix_cache = self._corr_matrix_frame(varlist).corr(method=self.method)
|
|
589
611
|
self._metric_summary_cache = None
|
|
590
612
|
self._metric_summary(varlist)
|
{supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Screen.py
RENAMED
|
@@ -19,6 +19,7 @@ from .Weighted_Screen import (
|
|
|
19
19
|
_apply_missing_rate_stage,
|
|
20
20
|
_apply_stage_keep,
|
|
21
21
|
_corr_dedup_weighted,
|
|
22
|
+
_corr_filter_dropped_audit,
|
|
22
23
|
_gate_ranking_iv_map,
|
|
23
24
|
_iv_band_keep,
|
|
24
25
|
_legacy_unweighted_screen,
|
|
@@ -701,23 +702,42 @@ def _weighted_woe_bins_screen(
|
|
|
701
702
|
|
|
702
703
|
corr_dropped = pd.DataFrame(columns=["var_a", "var_b", "corr", "iv_a", "iv_b", "kept", "dropped"])
|
|
703
704
|
if config.corr_enabled and len(current) > 1:
|
|
704
|
-
corr = _weighted_corr_for_screen(
|
|
705
|
-
ins, current, w_ins,
|
|
706
|
-
corr_use_woe_bins=config.corr_use_woe_bins,
|
|
707
|
-
corr_nan_policy=config.corr_nan_policy,
|
|
708
|
-
corr_block_size=config.corr_block_size,
|
|
709
|
-
adapter=adapter,
|
|
710
|
-
binner=binner,
|
|
711
|
-
)
|
|
712
|
-
iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
|
|
713
705
|
n_before = len(current)
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
706
|
+
if bool(np.all(w_ins == w_ins[0])) and config.corr_nan_policy == "pairwise":
|
|
707
|
+
# Match the unweighted WOE/mixed-basis decision exactly for
|
|
708
|
+
# constant positive weights, while retaining weighted audit rows.
|
|
709
|
+
from Modeling_Tool import CorrelationFilter
|
|
710
|
+
|
|
711
|
+
corr_input = list(current)
|
|
712
|
+
use_binner = config.corr_use_woe_bins and binner is not None
|
|
713
|
+
cf = CorrelationFilter(
|
|
714
|
+
data=ins[current + [target_col]],
|
|
715
|
+
dep=target_col,
|
|
716
|
+
corr_cutpoint=config.corr_threshold,
|
|
717
|
+
woe_binner=binner if use_binner else None,
|
|
718
|
+
woe_engine="monotone" if use_binner else "master",
|
|
719
|
+
)
|
|
720
|
+
current = cf.remove_highly_correlated(
|
|
721
|
+
current, max_iterations=config.corr_max_iterations
|
|
722
|
+
)
|
|
723
|
+
corr_dropped = _corr_filter_dropped_audit(cf, corr_input, current)
|
|
724
|
+
else:
|
|
725
|
+
corr = _weighted_corr_for_screen(
|
|
726
|
+
ins, current, w_ins,
|
|
727
|
+
corr_use_woe_bins=config.corr_use_woe_bins,
|
|
728
|
+
corr_nan_policy=config.corr_nan_policy,
|
|
729
|
+
corr_block_size=config.corr_block_size,
|
|
730
|
+
adapter=adapter,
|
|
731
|
+
binner=binner,
|
|
732
|
+
)
|
|
733
|
+
iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
|
|
734
|
+
current, corr_dropped = _corr_dedup_weighted(
|
|
735
|
+
current,
|
|
736
|
+
corr,
|
|
737
|
+
iv_map,
|
|
738
|
+
config.corr_threshold,
|
|
739
|
+
config.corr_max_iterations,
|
|
740
|
+
)
|
|
721
741
|
summary_rows.append(_summary_row("corr", n_before, len(current), config.corr_threshold, weight_col))
|
|
722
742
|
|
|
723
743
|
from .Screen_Gates import apply_post_corr_gates
|
{supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Screen_Gates.py
RENAMED
|
@@ -24,6 +24,8 @@ from typing import Any, Callable
|
|
|
24
24
|
import numpy as np
|
|
25
25
|
import pandas as pd
|
|
26
26
|
|
|
27
|
+
from Modeling_Tool.Core.sample_weight_utils import resolve_sample_weight
|
|
28
|
+
|
|
27
29
|
from .Weighted_Screen import _apply_stage_keep, _summary_row
|
|
28
30
|
|
|
29
31
|
|
|
@@ -118,8 +120,18 @@ def apply_vif_stage(
|
|
|
118
120
|
basis excludes non-numeric survivors from the matrix — raw string columns
|
|
119
121
|
used to crash statsmodels — keeping them in the selection untouched.
|
|
120
122
|
"""
|
|
121
|
-
if not getattr(config, "vif_enabled", False)
|
|
122
|
-
|
|
123
|
+
if not getattr(config, "vif_enabled", False):
|
|
124
|
+
return current
|
|
125
|
+
|
|
126
|
+
tie_metric = str(getattr(config, "vif_tie_break_metric", "iv"))
|
|
127
|
+
if tie_metric != "iv":
|
|
128
|
+
raise ValueError(
|
|
129
|
+
"vif_tie_break_metric currently supports only 'iv'; "
|
|
130
|
+
f"got {tie_metric!r}"
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
if len(current) <= max(2, int(config.vif_min_features)):
|
|
134
|
+
if len(current) <= int(config.vif_min_features):
|
|
123
135
|
summary_rows.append(_summary_row(
|
|
124
136
|
"vif", len(current), len(current), config.vif_threshold, weight_col,
|
|
125
137
|
note="skipped_at_floor",
|
|
@@ -139,7 +151,11 @@ def apply_vif_stage(
|
|
|
139
151
|
analyzer = FeatureSelectionAnalyzer()
|
|
140
152
|
threshold = float(config.vif_threshold)
|
|
141
153
|
floor = int(config.vif_min_features)
|
|
142
|
-
|
|
154
|
+
sample_weight = (
|
|
155
|
+
resolve_sample_weight(data=ins, weight_col=weight_col, expected_len=len(ins))
|
|
156
|
+
if weight_col is not None
|
|
157
|
+
else None
|
|
158
|
+
)
|
|
143
159
|
excluded: list[str] = []
|
|
144
160
|
raw_vif_cast_columns: list[str] = []
|
|
145
161
|
if bool(getattr(config, "vif_use_woe_bins", False)):
|
|
@@ -209,7 +225,13 @@ def apply_vif_stage(
|
|
|
209
225
|
for iteration in range(len(current)):
|
|
210
226
|
if len(survivors) <= floor:
|
|
211
227
|
break
|
|
212
|
-
|
|
228
|
+
if sample_weight is None:
|
|
229
|
+
# Preserve the historical unweighted call path byte-for-byte.
|
|
230
|
+
vif_table = analyzer.compute_vif(base[survivors])
|
|
231
|
+
else:
|
|
232
|
+
vif_table = analyzer.compute_vif(
|
|
233
|
+
base[survivors], sample_weight=sample_weight,
|
|
234
|
+
)
|
|
213
235
|
vif_table = vif_table.sort_values("VIF", ascending=False).reset_index(drop=True)
|
|
214
236
|
worst = vif_table.iloc[0]
|
|
215
237
|
if not np.isfinite(worst["VIF"]) or worst["VIF"] > threshold:
|
|
@@ -469,6 +469,7 @@ class CorrelationFilter:
|
|
|
469
469
|
self._corr_matrix_cache = None
|
|
470
470
|
self._corr_matrix_excluded = set()
|
|
471
471
|
self._metric_summary_cache = None
|
|
472
|
+
self._correlation_decision_trace = []
|
|
472
473
|
|
|
473
474
|
def _corr_matrix_frame(self, varlist):
|
|
474
475
|
# 0.7.0-R1: mixed correlation base. Numeric cols stay raw so a
|
|
@@ -594,9 +595,21 @@ class CorrelationFilter:
|
|
|
594
595
|
def calculate_vif(df):
|
|
595
596
|
return _BaseCorrelationFilter.calculate_vif(df)
|
|
596
597
|
|
|
598
|
+
def _sync_base_state(self):
|
|
599
|
+
self.correlated_dict = getattr(self._base, "correlated_dict", {})
|
|
600
|
+
self.filtered_varlist = getattr(self._base, "filtered_varlist", [])
|
|
601
|
+
self._corr_matrix_cache = getattr(self._base, "_corr_matrix_cache", None)
|
|
602
|
+
self._corr_matrix_excluded = getattr(self._base, "_corr_matrix_excluded", set())
|
|
603
|
+
self._metric_summary_cache = getattr(self._base, "_metric_summary_cache", None)
|
|
604
|
+
self._correlation_decision_trace = getattr(
|
|
605
|
+
self._base, "_correlation_decision_trace", [],
|
|
606
|
+
)
|
|
607
|
+
|
|
597
608
|
def filter_single_iteration(self, varlist):
|
|
598
609
|
if self.woe_binner is None and self.woe_engine == "master":
|
|
599
|
-
|
|
610
|
+
result = self._base.filter_single_iteration(varlist)
|
|
611
|
+
self._sync_base_state()
|
|
612
|
+
return result
|
|
600
613
|
|
|
601
614
|
name_mapping = {"iv": "iv", "ks": "ks_in_gains"}
|
|
602
615
|
high_corr_var = self._high_corr_pairs(varlist)
|
|
@@ -617,7 +630,28 @@ class CorrelationFilter:
|
|
|
617
630
|
selected = summary.sort_values([name_mapping[self.base_metric.lower()]], ascending=False)["var"].iloc[0]
|
|
618
631
|
if selected not in selected_varlist:
|
|
619
632
|
selected_varlist.append(selected)
|
|
620
|
-
|
|
633
|
+
newly_removed = [
|
|
634
|
+
x for x in correlated_list
|
|
635
|
+
if x != selected and x not in removed_varlist
|
|
636
|
+
]
|
|
637
|
+
metric_name = name_mapping[self.base_metric.lower()]
|
|
638
|
+
metric_map = summary.set_index("var")[metric_name].to_dict()
|
|
639
|
+
positions = {name: idx for idx, name in enumerate(varlist)}
|
|
640
|
+
for dropped_var in newly_removed:
|
|
641
|
+
decision_pair = [selected, dropped_var]
|
|
642
|
+
decision_pair.sort(key=positions.__getitem__)
|
|
643
|
+
var_a, var_b = decision_pair
|
|
644
|
+
corr_value = self._corr_matrix_cache.loc[var_a, var_b]
|
|
645
|
+
self._correlation_decision_trace.append({
|
|
646
|
+
"var_a": var_a,
|
|
647
|
+
"var_b": var_b,
|
|
648
|
+
"corr": float(corr_value),
|
|
649
|
+
"iv_a": float(metric_map.get(var_a, 0.0)),
|
|
650
|
+
"iv_b": float(metric_map.get(var_b, 0.0)),
|
|
651
|
+
"kept": selected,
|
|
652
|
+
"dropped": dropped_var,
|
|
653
|
+
})
|
|
654
|
+
removed_varlist += newly_removed
|
|
621
655
|
self.correlated_dict[var] = {"corr": single_var_corr, "gains": summary}
|
|
622
656
|
|
|
623
657
|
return selected_varlist + [x for x in varlist if x not in (selected_varlist + removed_varlist)]
|
|
@@ -625,10 +659,10 @@ class CorrelationFilter:
|
|
|
625
659
|
def remove_highly_correlated(self, varlist, max_iterations=10):
|
|
626
660
|
if self.woe_binner is None and self.woe_engine == "master":
|
|
627
661
|
result = self._base.remove_highly_correlated(varlist, max_iterations)
|
|
628
|
-
self.
|
|
629
|
-
self.filtered_varlist = getattr(self._base, "filtered_varlist", [])
|
|
662
|
+
self._sync_base_state()
|
|
630
663
|
return result
|
|
631
664
|
|
|
665
|
+
self._correlation_decision_trace = []
|
|
632
666
|
self._corr_matrix_cache = self._corr_matrix_frame(varlist).corr(method=self.method)
|
|
633
667
|
self._metric_summary_cache = None
|
|
634
668
|
self._metric_summary(varlist)
|
{supermodelingfactory-0.7.1 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Weighted_Screen.py
RENAMED
|
@@ -578,21 +578,52 @@ def _weighted_corr_for_screen(
|
|
|
578
578
|
"""Build weighted correlation matrix for screening (WOE or raw-value path)."""
|
|
579
579
|
from Modeling_Tool.WOE.WOE_Adapter import as_woe_engine
|
|
580
580
|
|
|
581
|
-
if
|
|
581
|
+
non_numeric = [v for v in current if not pd.api.types.is_numeric_dtype(ins[v])]
|
|
582
|
+
if corr_use_woe_bins and non_numeric:
|
|
582
583
|
eng = adapter if adapter is not None else (as_woe_engine(binner) if binner is not None else None)
|
|
583
584
|
if eng is not None:
|
|
584
585
|
suffix = getattr(eng, "woe_suffix", "_woe")
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
586
|
+
# Match the unweighted mixed basis: numeric columns stay raw and
|
|
587
|
+
# only categorical columns are WOE encoded.
|
|
588
|
+
woe_ins = eng.transform(
|
|
589
|
+
ins.copy(), varlist=non_numeric, suffix=suffix
|
|
590
|
+
)
|
|
591
|
+
encoded: dict[str, np.ndarray] = {}
|
|
592
|
+
excluded: list[str] = []
|
|
593
|
+
for name in non_numeric:
|
|
594
|
+
column = f"{name}{suffix}"
|
|
595
|
+
if column in woe_ins.columns and pd.api.types.is_numeric_dtype(woe_ins[column]):
|
|
596
|
+
encoded[name] = woe_ins[column].to_numpy()
|
|
597
|
+
else:
|
|
598
|
+
excluded.append(name)
|
|
599
|
+
if excluded:
|
|
600
|
+
warnings.warn(
|
|
601
|
+
f"corr_use_woe_bins=True could not WOE-encode {len(excluded)} "
|
|
602
|
+
f"feature(s) {excluded[:5]}; they are excluded from the "
|
|
603
|
+
f"correlation matrix and kept through the corr stage. Refit the "
|
|
604
|
+
f"screening WOE engine to cover them.",
|
|
605
|
+
UserWarning,
|
|
606
|
+
stacklevel=2,
|
|
607
|
+
)
|
|
608
|
+
|
|
609
|
+
# Return a matrix on the full current vocabulary. Excluded
|
|
610
|
+
# categories remain NaN rows/columns so dedup keeps them.
|
|
611
|
+
corr_full = np.full((len(current), len(current)), np.nan)
|
|
612
|
+
excluded_set = set(excluded)
|
|
613
|
+
active = [name for name in current if name not in excluded_set]
|
|
614
|
+
if active:
|
|
615
|
+
frame = ins[active].copy()
|
|
616
|
+
for name, values in encoded.items():
|
|
617
|
+
frame[name] = values
|
|
618
|
+
corr_active = _weighted_pearson_corr_matrix(
|
|
619
|
+
frame[active].to_numpy(dtype=float),
|
|
591
620
|
w_ins,
|
|
592
|
-
nan_policy=
|
|
621
|
+
nan_policy=corr_nan_policy,
|
|
593
622
|
corr_block_size=corr_block_size,
|
|
594
623
|
)
|
|
595
|
-
|
|
624
|
+
active_idx = [current.index(name) for name in active]
|
|
625
|
+
corr_full[np.ix_(active_idx, active_idx)] = corr_active
|
|
626
|
+
return corr_full
|
|
596
627
|
if non_numeric:
|
|
597
628
|
# Raw-value Pearson correlation is undefined for categorical/object
|
|
598
629
|
# features. Exclude them from the correlation computation but keep
|
|
@@ -804,6 +835,99 @@ def _corr_dedup_weighted(
|
|
|
804
835
|
return current, pd.DataFrame(dropped_rows)
|
|
805
836
|
|
|
806
837
|
|
|
838
|
+
def _corr_filter_dropped_audit(
|
|
839
|
+
corr_filter: Any,
|
|
840
|
+
varlist: list[str],
|
|
841
|
+
kept: list[str],
|
|
842
|
+
) -> pd.DataFrame:
|
|
843
|
+
"""Convert an unweighted CorrelationFilter decision to weighted audit rows.
|
|
844
|
+
|
|
845
|
+
Constant positive weights reuse CorrelationFilter for byte-compatible
|
|
846
|
+
feature ordering and IV tie-breaking. This adapter preserves the existing
|
|
847
|
+
seven-column ``corr_dropped`` evidence contract for that weighted call.
|
|
848
|
+
Each row's ``var_a/var_b`` endpoints are exactly its ``kept/dropped`` pair.
|
|
849
|
+
For star-shaped groups, ``corr`` may be below the cutoff because it records
|
|
850
|
+
the group winner versus the indirectly dropped member; another group edge
|
|
851
|
+
triggered the decision.
|
|
852
|
+
"""
|
|
853
|
+
columns = ["var_a", "var_b", "corr", "iv_a", "iv_b", "kept", "dropped"]
|
|
854
|
+
kept_set = set(kept)
|
|
855
|
+
dropped = [name for name in varlist if name not in kept_set]
|
|
856
|
+
if not dropped:
|
|
857
|
+
return pd.DataFrame(columns=columns)
|
|
858
|
+
|
|
859
|
+
def _state(name: str):
|
|
860
|
+
value = getattr(corr_filter, name, None)
|
|
861
|
+
if value is None:
|
|
862
|
+
value = getattr(getattr(corr_filter, "_base", None), name, None)
|
|
863
|
+
return value
|
|
864
|
+
|
|
865
|
+
trace = _state("_correlation_decision_trace")
|
|
866
|
+
rows = []
|
|
867
|
+
recorded = set()
|
|
868
|
+
if isinstance(trace, list):
|
|
869
|
+
for row in trace:
|
|
870
|
+
if not isinstance(row, dict) or not set(columns).issubset(row):
|
|
871
|
+
continue
|
|
872
|
+
dropped_name = row["dropped"]
|
|
873
|
+
if dropped_name in dropped and dropped_name not in recorded:
|
|
874
|
+
rows.append({column: row[column] for column in columns})
|
|
875
|
+
recorded.add(dropped_name)
|
|
876
|
+
dropped = [name for name in dropped if name not in recorded]
|
|
877
|
+
if not dropped:
|
|
878
|
+
return pd.DataFrame(rows, columns=columns)
|
|
879
|
+
|
|
880
|
+
# Defensive fallback for third-party/future filters without trace support.
|
|
881
|
+
matrix = _state("_corr_matrix_cache")
|
|
882
|
+
if not isinstance(matrix, pd.DataFrame):
|
|
883
|
+
return pd.DataFrame(rows, columns=columns)
|
|
884
|
+
matrix = matrix.reindex(index=varlist, columns=varlist)
|
|
885
|
+
|
|
886
|
+
metric = _state("_metric_summary_cache")
|
|
887
|
+
iv_map: dict[str, float] = {}
|
|
888
|
+
if isinstance(metric, pd.DataFrame) and {"var", "iv"}.issubset(metric.columns):
|
|
889
|
+
for name, value in zip(metric["var"], metric["iv"]):
|
|
890
|
+
try:
|
|
891
|
+
iv_map[str(name)] = float(value)
|
|
892
|
+
except (TypeError, ValueError):
|
|
893
|
+
iv_map[str(name)] = 0.0
|
|
894
|
+
|
|
895
|
+
positions = {name: idx for idx, name in enumerate(varlist)}
|
|
896
|
+
threshold = float(getattr(corr_filter, "corr_cutpoint"))
|
|
897
|
+
for dropped_name in dropped:
|
|
898
|
+
candidates = []
|
|
899
|
+
for partner in varlist:
|
|
900
|
+
if partner == dropped_name:
|
|
901
|
+
continue
|
|
902
|
+
value = matrix.loc[dropped_name, partner]
|
|
903
|
+
if pd.notna(value) and abs(float(value)) > threshold:
|
|
904
|
+
candidates.append((partner, float(value)))
|
|
905
|
+
if not candidates:
|
|
906
|
+
continue
|
|
907
|
+
partner, corr_value = min(
|
|
908
|
+
candidates,
|
|
909
|
+
key=lambda item: (
|
|
910
|
+
0 if item[0] in kept_set else 1,
|
|
911
|
+
-abs(item[1]),
|
|
912
|
+
positions[item[0]],
|
|
913
|
+
),
|
|
914
|
+
)
|
|
915
|
+
if positions[dropped_name] < positions[partner]:
|
|
916
|
+
var_a, var_b = dropped_name, partner
|
|
917
|
+
else:
|
|
918
|
+
var_a, var_b = partner, dropped_name
|
|
919
|
+
rows.append({
|
|
920
|
+
"var_a": var_a,
|
|
921
|
+
"var_b": var_b,
|
|
922
|
+
"corr": corr_value,
|
|
923
|
+
"iv_a": iv_map.get(var_a, 0.0),
|
|
924
|
+
"iv_b": iv_map.get(var_b, 0.0),
|
|
925
|
+
"kept": partner,
|
|
926
|
+
"dropped": dropped_name,
|
|
927
|
+
})
|
|
928
|
+
return pd.DataFrame(rows, columns=columns)
|
|
929
|
+
|
|
930
|
+
|
|
807
931
|
def _legacy_unweighted_screen(
|
|
808
932
|
splits: dict[str, pd.DataFrame],
|
|
809
933
|
feature_cols: list[str],
|
|
@@ -1110,18 +1234,38 @@ def _weighted_screen_impl(
|
|
|
1110
1234
|
|
|
1111
1235
|
corr_dropped = pd.DataFrame(columns=["var_a", "var_b", "corr", "iv_a", "iv_b", "kept", "dropped"])
|
|
1112
1236
|
if corr_enabled and len(current) > 1:
|
|
1113
|
-
corr = _weighted_corr_for_screen(
|
|
1114
|
-
ins, current, w_ins,
|
|
1115
|
-
corr_use_woe_bins=corr_use_woe_bins,
|
|
1116
|
-
corr_nan_policy=corr_nan_policy,
|
|
1117
|
-
corr_block_size=corr_block_size,
|
|
1118
|
-
binner=prefit_woe_engine,
|
|
1119
|
-
)
|
|
1120
|
-
iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
|
|
1121
1237
|
n_before = len(current)
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
|
|
1238
|
+
if bool(np.all(w_ins == w_ins[0])) and corr_nan_policy == "pairwise":
|
|
1239
|
+
# Constant positive weights follow the exact unweighted decision
|
|
1240
|
+
# contract. Explicit non-default NaN policies stay on the weighted
|
|
1241
|
+
# implementation so they cannot be silently ignored.
|
|
1242
|
+
from Modeling_Tool import CorrelationFilter
|
|
1243
|
+
|
|
1244
|
+
corr_input = list(current)
|
|
1245
|
+
use_binner = corr_use_woe_bins and prefit_woe_engine is not None
|
|
1246
|
+
cf = CorrelationFilter(
|
|
1247
|
+
data=ins[current + [target_col]],
|
|
1248
|
+
dep=target_col,
|
|
1249
|
+
corr_cutpoint=corr_threshold,
|
|
1250
|
+
woe_binner=prefit_woe_engine if use_binner else None,
|
|
1251
|
+
woe_engine="monotone" if use_binner else "master",
|
|
1252
|
+
)
|
|
1253
|
+
current = cf.remove_highly_correlated(
|
|
1254
|
+
current, max_iterations=corr_max_iterations
|
|
1255
|
+
)
|
|
1256
|
+
corr_dropped = _corr_filter_dropped_audit(cf, corr_input, current)
|
|
1257
|
+
else:
|
|
1258
|
+
corr = _weighted_corr_for_screen(
|
|
1259
|
+
ins, current, w_ins,
|
|
1260
|
+
corr_use_woe_bins=corr_use_woe_bins,
|
|
1261
|
+
corr_nan_policy=corr_nan_policy,
|
|
1262
|
+
corr_block_size=corr_block_size,
|
|
1263
|
+
binner=prefit_woe_engine,
|
|
1264
|
+
)
|
|
1265
|
+
iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
|
|
1266
|
+
current, corr_dropped = _corr_dedup_weighted(
|
|
1267
|
+
current, corr, iv_map, corr_threshold, corr_max_iterations,
|
|
1268
|
+
)
|
|
1125
1269
|
summary_rows.append(_summary_row("corr", n_before, len(current), corr_threshold, weight_col))
|
|
1126
1270
|
|
|
1127
1271
|
if gates_config is not None:
|
|
@@ -381,7 +381,13 @@ class FeatureSelectionAnalyzer:
|
|
|
381
381
|
self.selected_features_ = results.loc[results['selected'], 'feature'].tolist()
|
|
382
382
|
return results
|
|
383
383
|
|
|
384
|
-
def compute_vif(
|
|
384
|
+
def compute_vif(
|
|
385
|
+
self,
|
|
386
|
+
data,
|
|
387
|
+
nan_handling="fillna_median",
|
|
388
|
+
nan_warn_threshold=0.05,
|
|
389
|
+
sample_weight=None,
|
|
390
|
+
):
|
|
385
391
|
"""
|
|
386
392
|
Compute Variance Inflation Factor (VIF) for multicollinearity detection.
|
|
387
393
|
|
|
@@ -389,6 +395,11 @@ class FeatureSelectionAnalyzer:
|
|
|
389
395
|
----------
|
|
390
396
|
data : pd.DataFrame
|
|
391
397
|
Feature matrix (should not include target variable)
|
|
398
|
+
sample_weight : array-like, optional
|
|
399
|
+
Per-row frequency/sample weights. Constant weights deliberately
|
|
400
|
+
use the legacy OLS implementation for strict parity. Non-constant
|
|
401
|
+
weights use WLS auxiliary regressions with the same no-intercept
|
|
402
|
+
design as variance_inflation_factor.
|
|
392
403
|
|
|
393
404
|
Returns
|
|
394
405
|
-------
|
|
@@ -403,6 +414,11 @@ class FeatureSelectionAnalyzer:
|
|
|
403
414
|
"with: pip install \"SuperModelingFactory[stats]\""
|
|
404
415
|
) from exc
|
|
405
416
|
|
|
417
|
+
weight = (
|
|
418
|
+
resolve_sample_weight(sample_weight=sample_weight, expected_len=len(data))
|
|
419
|
+
if sample_weight is not None
|
|
420
|
+
else None
|
|
421
|
+
)
|
|
406
422
|
work = _prepare_nan_handled_frame(
|
|
407
423
|
data,
|
|
408
424
|
nan_handling=nan_handling,
|
|
@@ -410,9 +426,27 @@ class FeatureSelectionAnalyzer:
|
|
|
410
426
|
context="FeatureSelectionAnalyzer.compute_vif",
|
|
411
427
|
)
|
|
412
428
|
x = work.values
|
|
429
|
+
if weight is not None and nan_handling == "drop_rows":
|
|
430
|
+
weight = weight[~data.isna().any(axis=1).to_numpy()]
|
|
431
|
+
weight = resolve_sample_weight(
|
|
432
|
+
sample_weight=weight, expected_len=len(work)
|
|
433
|
+
)
|
|
434
|
+
|
|
435
|
+
# Preserve the exact legacy call path for no weight and every
|
|
436
|
+
# constant-weight vector.
|
|
437
|
+
if weight is None or bool(np.all(weight == weight[0])):
|
|
438
|
+
vif_values = [variance_inflation_factor(x, i) for i in range(x.shape[1])]
|
|
439
|
+
else:
|
|
440
|
+
from statsmodels.regression.linear_model import WLS
|
|
441
|
+
|
|
442
|
+
vif_values = []
|
|
443
|
+
for i in range(x.shape[1]):
|
|
444
|
+
others = np.arange(x.shape[1]) != i
|
|
445
|
+
r_squared = WLS(x[:, i], x[:, others], weights=weight).fit().rsquared
|
|
446
|
+
vif_values.append(1.0 / (1.0 - r_squared))
|
|
413
447
|
vif_data = pd.DataFrame({
|
|
414
448
|
'feature': work.columns,
|
|
415
|
-
'VIF':
|
|
449
|
+
'VIF': vif_values,
|
|
416
450
|
}).sort_values('VIF', ascending=False).reset_index(drop=True)
|
|
417
451
|
|
|
418
452
|
return vif_data
|
|
@@ -28,7 +28,7 @@ def compute_bic(model, x, y): ...
|
|
|
28
28
|
class FeatureSelectionAnalyzer:
|
|
29
29
|
def __init__(self, significance_level = 0.05): ...
|
|
30
30
|
def chi2_selection(self, data, feature_cols, target_col, nan_handling = "fillna_median", nan_warn_threshold = 0.05): ...
|
|
31
|
-
def compute_vif(self, data, nan_handling = "fillna_median", nan_warn_threshold = 0.05): ...
|
|
31
|
+
def compute_vif(self, data, nan_handling = "fillna_median", nan_warn_threshold = 0.05, sample_weight = None): ...
|
|
32
32
|
def correlation_filter(self, data, threshold = 0.8): ...
|
|
33
33
|
|
|
34
34
|
class LRMaster:
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import copy
|
|
3
4
|
import gc
|
|
4
5
|
import logging
|
|
5
6
|
import warnings
|
|
@@ -470,6 +471,11 @@ class FeatureValidationPipeline:
|
|
|
470
471
|
"""Drop per-batch raw/WOE dataframes that are not needed for final merge."""
|
|
471
472
|
woe_artifacts = dict(result.woe_artifacts or {})
|
|
472
473
|
woe_artifacts["by_target"] = {}
|
|
474
|
+
for key in (
|
|
475
|
+
"categorical_transform_stats_by_target",
|
|
476
|
+
"unseen_category_stats_by_target",
|
|
477
|
+
):
|
|
478
|
+
woe_artifacts[key] = copy.deepcopy(woe_artifacts.get(key, {}))
|
|
473
479
|
return FeatureValidationPipelineResult(
|
|
474
480
|
splits={},
|
|
475
481
|
distribution_summary=result.distribution_summary,
|
|
@@ -487,7 +493,7 @@ class FeatureValidationPipeline:
|
|
|
487
493
|
batch_results=result.batch_results,
|
|
488
494
|
selected_features=list(result.selected_features),
|
|
489
495
|
selection_summary=dict(result.selection_summary or {}),
|
|
490
|
-
screening_artifact=
|
|
496
|
+
screening_artifact=None,
|
|
491
497
|
config_snapshot=dict(result.config_snapshot or {}),
|
|
492
498
|
)
|
|
493
499
|
|
|
@@ -730,13 +736,42 @@ class FeatureValidationPipeline:
|
|
|
730
736
|
artifacts: list[dict[str, Any]],
|
|
731
737
|
batch_metadata: pd.DataFrame,
|
|
732
738
|
) -> dict[str, Any]:
|
|
739
|
+
valid = [item for item in artifacts if item]
|
|
733
740
|
return {
|
|
734
741
|
"by_target": {},
|
|
735
|
-
"woe_table": self._concat_frames([item.get("woe_table", pd.DataFrame()) for item in
|
|
736
|
-
"refine_summary": self._concat_frames([item.get("refine_summary", pd.DataFrame()) for item in
|
|
742
|
+
"woe_table": self._concat_frames([item.get("woe_table", pd.DataFrame()) for item in valid]),
|
|
743
|
+
"refine_summary": self._concat_frames([item.get("refine_summary", pd.DataFrame()) for item in valid]),
|
|
744
|
+
"categorical_transform_stats_by_target": self._merge_transform_stats(
|
|
745
|
+
valid, "categorical_transform_stats_by_target"
|
|
746
|
+
),
|
|
747
|
+
"unseen_category_stats_by_target": self._merge_transform_stats(
|
|
748
|
+
valid, "unseen_category_stats_by_target"
|
|
749
|
+
),
|
|
737
750
|
"batch_metadata": batch_metadata,
|
|
738
751
|
}
|
|
739
752
|
|
|
753
|
+
@staticmethod
|
|
754
|
+
def _merge_transform_stats(
|
|
755
|
+
artifacts: list[dict[str, Any]], key: str
|
|
756
|
+
) -> dict[str, Any]:
|
|
757
|
+
"""Deep-merge target/split/feature audit stats without silent overwrite."""
|
|
758
|
+
merged: dict[str, Any] = {}
|
|
759
|
+
for artifact in artifacts:
|
|
760
|
+
payload = artifact.get(key, {}) or {}
|
|
761
|
+
for target, split_stats in payload.items():
|
|
762
|
+
target_out = merged.setdefault(target, {})
|
|
763
|
+
for split, feature_stats in split_stats.items():
|
|
764
|
+
split_out = target_out.setdefault(split, {})
|
|
765
|
+
for feature, stats in feature_stats.items():
|
|
766
|
+
value = copy.deepcopy(stats)
|
|
767
|
+
if feature in split_out and split_out[feature] != value:
|
|
768
|
+
raise ValueError(
|
|
769
|
+
f"conflicting {key} stats for target={target!r}, "
|
|
770
|
+
f"split={split!r}, feature={feature!r}"
|
|
771
|
+
)
|
|
772
|
+
split_out[feature] = value
|
|
773
|
+
return merged
|
|
774
|
+
|
|
740
775
|
@staticmethod
|
|
741
776
|
def _concat_frames(frames: list[pd.DataFrame]) -> pd.DataFrame:
|
|
742
777
|
valid = [df for df in frames if isinstance(df, pd.DataFrame) and not df.empty]
|
|
@@ -1247,16 +1282,27 @@ class FeatureValidationPipeline:
|
|
|
1247
1282
|
table = adapter.get_woe_table(fit_features)
|
|
1248
1283
|
table.insert(0, "target", target)
|
|
1249
1284
|
woe_tables.append(table)
|
|
1250
|
-
woe_splits = {
|
|
1251
|
-
|
|
1252
|
-
|
|
1253
|
-
|
|
1285
|
+
woe_splits = {}
|
|
1286
|
+
categorical_transform_stats_by_split = {}
|
|
1287
|
+
unseen_category_stats_by_split = {}
|
|
1288
|
+
for name, df in splits.items():
|
|
1289
|
+
woe_splits[name] = adapter.transform(df, varlist=fit_features)
|
|
1290
|
+
categorical_transform_stats_by_split[name] = copy.deepcopy(
|
|
1291
|
+
getattr(engine, "_categorical_transform_stats", {})
|
|
1292
|
+
)
|
|
1293
|
+
unseen_category_stats_by_split[name] = copy.deepcopy(
|
|
1294
|
+
getattr(engine, "_unseen_category_stats", {})
|
|
1295
|
+
)
|
|
1254
1296
|
self._plot_woe(engine, adapter, train, fit_features, target)
|
|
1255
1297
|
by_target[target] = {
|
|
1256
1298
|
"engine": engine,
|
|
1257
1299
|
"adapter": adapter,
|
|
1258
1300
|
"woe_splits": woe_splits,
|
|
1259
1301
|
"features": fit_features,
|
|
1302
|
+
"categorical_transform_stats_by_split": (
|
|
1303
|
+
categorical_transform_stats_by_split
|
|
1304
|
+
),
|
|
1305
|
+
"unseen_category_stats_by_split": unseen_category_stats_by_split,
|
|
1260
1306
|
}
|
|
1261
1307
|
except Exception as exc:
|
|
1262
1308
|
refine_rows.append({"target": target, "step": "fit", "status": "error", "error": repr(exc)})
|
|
@@ -1264,6 +1310,14 @@ class FeatureValidationPipeline:
|
|
|
1264
1310
|
"by_target": by_target,
|
|
1265
1311
|
"woe_table": pd.concat(woe_tables, ignore_index=True) if woe_tables else pd.DataFrame(),
|
|
1266
1312
|
"refine_summary": pd.DataFrame(refine_rows),
|
|
1313
|
+
"categorical_transform_stats_by_target": {
|
|
1314
|
+
target: copy.deepcopy(item["categorical_transform_stats_by_split"])
|
|
1315
|
+
for target, item in by_target.items()
|
|
1316
|
+
},
|
|
1317
|
+
"unseen_category_stats_by_target": {
|
|
1318
|
+
target: copy.deepcopy(item["unseen_category_stats_by_split"])
|
|
1319
|
+
for target, item in by_target.items()
|
|
1320
|
+
},
|
|
1267
1321
|
}
|
|
1268
1322
|
|
|
1269
1323
|
def _fit_monotone_binner(
|