SuperModelingFactory 0.7.0__tar.gz → 0.7.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Insights.py +50 -4
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Screen.py +36 -16
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Screen_Gates.py +26 -4
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/WOE_Engine_Feature_Patch.py +108 -7
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Weighted_Screen.py +164 -20
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/LRM_Tool.py +36 -2
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/LRM_Tool.pyi +1 -1
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/feature_validation.py +61 -7
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Monotone_Binner.py +566 -86
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/__init__.py +1 -1
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/PKG-INFO +2 -2
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/README.md +1 -1
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/PKG-INFO +2 -2
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/pyproject.toml +1 -1
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/setup.py +1 -1
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/ExcelMaster/ExcelFormatTool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/ExcelMaster/ExcelMaster.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/ExcelMaster/Template.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/ExcelMaster/Utility.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/ExcelMaster/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/LICENSE +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/MANIFEST.in +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Binning_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Binning_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Check_DuckDB_Compatibility.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Json_Data_Converter.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Model_Registry_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/ODPS_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Parallel_Engine.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Parallel_ODPS_Manager.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Proc_Compare.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Slope_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Slope_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/XOR_Encryptor.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/XOR_Encryptor.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/kDataFrame.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/kDataFrame.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/sample_weight_utils.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/utils.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Evaluation_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Evaluation_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Model_Eval_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Model_Eval_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/evaluate_model.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/evaluate_model.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/weighted_eval_utils.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Explainability/Coalition_Structure.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Explainability/Model_Explainer.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Explainability/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Distribution_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Distribution_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Insights.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/ODPS_Distribution_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/ODPS_Distribution_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/PSI_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/PSI_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/Backward_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/Backward_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/GBM_Search_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/GBM_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/GBM_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/_common.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/credit_model.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/field_meta.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/mock_sample.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/orchestrator.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/reject_inference.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/sample_analysis.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/score_comparison.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/score_consistency_uat.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/screening_artifact.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Distribution_Adaptation.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Distribution_Adaptation.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Reject_Infer.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Reject_Infer.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Sample_Split.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Sample_Split.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/UAT/UAT_Consistency_Checker.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/UAT/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Adapter.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Adapter.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Master.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Master.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Monotone_Binner.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Plot_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Plot_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Report_Builder.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Report_Builder.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/plot_woe_tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/plot_woe_tool.pyi +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/nan_guard.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/robust.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/sentinels.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/KaiTi.ttf +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/WeiRuanYaHei.ttf +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/simsun.ttc +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Report/Report_Tool.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Report/__init__.py +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/SOURCES.txt +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/dependency_links.txt +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/not-zip-safe +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/requires.txt +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/top_level.txt +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/requirements.txt +0 -0
- {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/setup.cfg +0 -0
{supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Insights.py
RENAMED
|
@@ -417,14 +417,39 @@ class CorrelationFilter:
|
|
|
417
417
|
self.correlated_dict = {}
|
|
418
418
|
self.filtered_varlist = []
|
|
419
419
|
self._corr_matrix_cache = None
|
|
420
|
+
self._corr_matrix_excluded = set()
|
|
420
421
|
self._metric_summary_cache = None
|
|
422
|
+
self._correlation_decision_trace = []
|
|
423
|
+
|
|
424
|
+
def _corr_matrix_frame(self, varlist):
|
|
425
|
+
# 0.7.0-R1: numeric-subset correlation base. A pure-numeric varlist is
|
|
426
|
+
# byte-identical to the legacy self.data[varlist].corr() path. Non-numeric
|
|
427
|
+
# cols (raw Pearson is undefined) are excluded and tracked in
|
|
428
|
+
# self._corr_matrix_excluded; _high_corr_pairs reindexes them back as NaN
|
|
429
|
+
# rows/cols so they survive the corr stage instead of crashing the float
|
|
430
|
+
# cast. At most one raw-value UserWarning per matrix build (no WOE binner).
|
|
431
|
+
non_numeric = [c for c in varlist if not pd.api.types.is_numeric_dtype(self.data[c])]
|
|
432
|
+
if not non_numeric:
|
|
433
|
+
self._corr_matrix_excluded = set()
|
|
434
|
+
return self.data[varlist]
|
|
435
|
+
self._corr_matrix_excluded = set(non_numeric)
|
|
436
|
+
warnings.warn(
|
|
437
|
+
f"raw-value correlation skips {len(non_numeric)} non-numeric "
|
|
438
|
+
f"feature(s) {non_numeric[:5]}; they are kept through the corr "
|
|
439
|
+
f"stage. Set corr_use_woe_bins=True to correlate categorical "
|
|
440
|
+
f"features via their WOE encoding.",
|
|
441
|
+
UserWarning,
|
|
442
|
+
stacklevel=2,
|
|
443
|
+
)
|
|
444
|
+
selected = [c for c in varlist if c not in self._corr_matrix_excluded]
|
|
445
|
+
return self.data[selected]
|
|
421
446
|
|
|
422
447
|
def _high_corr_pairs(self, varlist):
|
|
423
448
|
if (
|
|
424
449
|
self._corr_matrix_cache is None
|
|
425
|
-
or not set(varlist)
|
|
450
|
+
or not (set(varlist) <= set(self._corr_matrix_cache.columns) | self._corr_matrix_excluded)
|
|
426
451
|
):
|
|
427
|
-
self._corr_matrix_cache = self.
|
|
452
|
+
self._corr_matrix_cache = self._corr_matrix_frame(varlist).corr(method=self.method)
|
|
428
453
|
matrix = self._corr_matrix_cache.reindex(index=varlist, columns=varlist)
|
|
429
454
|
values = matrix.to_numpy(dtype=float)
|
|
430
455
|
row_idx, col_idx = np.triu_indices(len(varlist), k=1)
|
|
@@ -522,7 +547,27 @@ class CorrelationFilter:
|
|
|
522
547
|
if fnl_selected_var not in selected_varlist:
|
|
523
548
|
selected_varlist.append(fnl_selected_var)
|
|
524
549
|
|
|
525
|
-
|
|
550
|
+
newly_removed = [
|
|
551
|
+
x for x in correlated_list
|
|
552
|
+
if x != fnl_selected_var and x not in removed_varlist
|
|
553
|
+
]
|
|
554
|
+
metric_map = fnl_summary.set_index("var")[name_mapping[base_metric]].to_dict()
|
|
555
|
+
positions = {name: idx for idx, name in enumerate(varlist)}
|
|
556
|
+
for dropped_var in newly_removed:
|
|
557
|
+
decision_pair = [fnl_selected_var, dropped_var]
|
|
558
|
+
decision_pair.sort(key=positions.__getitem__)
|
|
559
|
+
var_a, var_b = decision_pair
|
|
560
|
+
corr_value = self._corr_matrix_cache.loc[var_a, var_b]
|
|
561
|
+
self._correlation_decision_trace.append({
|
|
562
|
+
"var_a": var_a,
|
|
563
|
+
"var_b": var_b,
|
|
564
|
+
"corr": float(corr_value),
|
|
565
|
+
"iv_a": float(metric_map.get(var_a, 0.0)),
|
|
566
|
+
"iv_b": float(metric_map.get(var_b, 0.0)),
|
|
567
|
+
"kept": fnl_selected_var,
|
|
568
|
+
"dropped": dropped_var,
|
|
569
|
+
})
|
|
570
|
+
removed_varlist += newly_removed
|
|
526
571
|
|
|
527
572
|
if var not in correlated_dict:
|
|
528
573
|
correlated_dict[var] = {}
|
|
@@ -561,7 +606,8 @@ class CorrelationFilter:
|
|
|
561
606
|
>>> filter_analyzer = CorrelationFilter(df, 'target')
|
|
562
607
|
>>> keep_vars = filter_analyzer.remove_highly_correlated(['var1', 'var2', 'var3'])
|
|
563
608
|
"""
|
|
564
|
-
self.
|
|
609
|
+
self._correlation_decision_trace = []
|
|
610
|
+
self._corr_matrix_cache = self._corr_matrix_frame(varlist).corr(method=self.method)
|
|
565
611
|
self._metric_summary_cache = None
|
|
566
612
|
self._metric_summary(varlist)
|
|
567
613
|
last_keep_list = self.filter_single_iteration(varlist)
|
{supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Screen.py
RENAMED
|
@@ -19,6 +19,7 @@ from .Weighted_Screen import (
|
|
|
19
19
|
_apply_missing_rate_stage,
|
|
20
20
|
_apply_stage_keep,
|
|
21
21
|
_corr_dedup_weighted,
|
|
22
|
+
_corr_filter_dropped_audit,
|
|
22
23
|
_gate_ranking_iv_map,
|
|
23
24
|
_iv_band_keep,
|
|
24
25
|
_legacy_unweighted_screen,
|
|
@@ -701,23 +702,42 @@ def _weighted_woe_bins_screen(
|
|
|
701
702
|
|
|
702
703
|
corr_dropped = pd.DataFrame(columns=["var_a", "var_b", "corr", "iv_a", "iv_b", "kept", "dropped"])
|
|
703
704
|
if config.corr_enabled and len(current) > 1:
|
|
704
|
-
corr = _weighted_corr_for_screen(
|
|
705
|
-
ins, current, w_ins,
|
|
706
|
-
corr_use_woe_bins=config.corr_use_woe_bins,
|
|
707
|
-
corr_nan_policy=config.corr_nan_policy,
|
|
708
|
-
corr_block_size=config.corr_block_size,
|
|
709
|
-
adapter=adapter,
|
|
710
|
-
binner=binner,
|
|
711
|
-
)
|
|
712
|
-
iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
|
|
713
705
|
n_before = len(current)
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
|
|
706
|
+
if bool(np.all(w_ins == w_ins[0])) and config.corr_nan_policy == "pairwise":
|
|
707
|
+
# Match the unweighted WOE/mixed-basis decision exactly for
|
|
708
|
+
# constant positive weights, while retaining weighted audit rows.
|
|
709
|
+
from Modeling_Tool import CorrelationFilter
|
|
710
|
+
|
|
711
|
+
corr_input = list(current)
|
|
712
|
+
use_binner = config.corr_use_woe_bins and binner is not None
|
|
713
|
+
cf = CorrelationFilter(
|
|
714
|
+
data=ins[current + [target_col]],
|
|
715
|
+
dep=target_col,
|
|
716
|
+
corr_cutpoint=config.corr_threshold,
|
|
717
|
+
woe_binner=binner if use_binner else None,
|
|
718
|
+
woe_engine="monotone" if use_binner else "master",
|
|
719
|
+
)
|
|
720
|
+
current = cf.remove_highly_correlated(
|
|
721
|
+
current, max_iterations=config.corr_max_iterations
|
|
722
|
+
)
|
|
723
|
+
corr_dropped = _corr_filter_dropped_audit(cf, corr_input, current)
|
|
724
|
+
else:
|
|
725
|
+
corr = _weighted_corr_for_screen(
|
|
726
|
+
ins, current, w_ins,
|
|
727
|
+
corr_use_woe_bins=config.corr_use_woe_bins,
|
|
728
|
+
corr_nan_policy=config.corr_nan_policy,
|
|
729
|
+
corr_block_size=config.corr_block_size,
|
|
730
|
+
adapter=adapter,
|
|
731
|
+
binner=binner,
|
|
732
|
+
)
|
|
733
|
+
iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
|
|
734
|
+
current, corr_dropped = _corr_dedup_weighted(
|
|
735
|
+
current,
|
|
736
|
+
corr,
|
|
737
|
+
iv_map,
|
|
738
|
+
config.corr_threshold,
|
|
739
|
+
config.corr_max_iterations,
|
|
740
|
+
)
|
|
721
741
|
summary_rows.append(_summary_row("corr", n_before, len(current), config.corr_threshold, weight_col))
|
|
722
742
|
|
|
723
743
|
from .Screen_Gates import apply_post_corr_gates
|
{supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Screen_Gates.py
RENAMED
|
@@ -24,6 +24,8 @@ from typing import Any, Callable
|
|
|
24
24
|
import numpy as np
|
|
25
25
|
import pandas as pd
|
|
26
26
|
|
|
27
|
+
from Modeling_Tool.Core.sample_weight_utils import resolve_sample_weight
|
|
28
|
+
|
|
27
29
|
from .Weighted_Screen import _apply_stage_keep, _summary_row
|
|
28
30
|
|
|
29
31
|
|
|
@@ -118,8 +120,18 @@ def apply_vif_stage(
|
|
|
118
120
|
basis excludes non-numeric survivors from the matrix — raw string columns
|
|
119
121
|
used to crash statsmodels — keeping them in the selection untouched.
|
|
120
122
|
"""
|
|
121
|
-
if not getattr(config, "vif_enabled", False)
|
|
122
|
-
|
|
123
|
+
if not getattr(config, "vif_enabled", False):
|
|
124
|
+
return current
|
|
125
|
+
|
|
126
|
+
tie_metric = str(getattr(config, "vif_tie_break_metric", "iv"))
|
|
127
|
+
if tie_metric != "iv":
|
|
128
|
+
raise ValueError(
|
|
129
|
+
"vif_tie_break_metric currently supports only 'iv'; "
|
|
130
|
+
f"got {tie_metric!r}"
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
if len(current) <= max(2, int(config.vif_min_features)):
|
|
134
|
+
if len(current) <= int(config.vif_min_features):
|
|
123
135
|
summary_rows.append(_summary_row(
|
|
124
136
|
"vif", len(current), len(current), config.vif_threshold, weight_col,
|
|
125
137
|
note="skipped_at_floor",
|
|
@@ -139,7 +151,11 @@ def apply_vif_stage(
|
|
|
139
151
|
analyzer = FeatureSelectionAnalyzer()
|
|
140
152
|
threshold = float(config.vif_threshold)
|
|
141
153
|
floor = int(config.vif_min_features)
|
|
142
|
-
|
|
154
|
+
sample_weight = (
|
|
155
|
+
resolve_sample_weight(data=ins, weight_col=weight_col, expected_len=len(ins))
|
|
156
|
+
if weight_col is not None
|
|
157
|
+
else None
|
|
158
|
+
)
|
|
143
159
|
excluded: list[str] = []
|
|
144
160
|
raw_vif_cast_columns: list[str] = []
|
|
145
161
|
if bool(getattr(config, "vif_use_woe_bins", False)):
|
|
@@ -209,7 +225,13 @@ def apply_vif_stage(
|
|
|
209
225
|
for iteration in range(len(current)):
|
|
210
226
|
if len(survivors) <= floor:
|
|
211
227
|
break
|
|
212
|
-
|
|
228
|
+
if sample_weight is None:
|
|
229
|
+
# Preserve the historical unweighted call path byte-for-byte.
|
|
230
|
+
vif_table = analyzer.compute_vif(base[survivors])
|
|
231
|
+
else:
|
|
232
|
+
vif_table = analyzer.compute_vif(
|
|
233
|
+
base[survivors], sample_weight=sample_weight,
|
|
234
|
+
)
|
|
213
235
|
vif_table = vif_table.sort_values("VIF", ascending=False).reset_index(drop=True)
|
|
214
236
|
worst = vif_table.iloc[0]
|
|
215
237
|
if not np.isfinite(worst["VIF"]) or worst["VIF"] > threshold:
|
|
@@ -467,14 +467,82 @@ class CorrelationFilter:
|
|
|
467
467
|
self.correlated_dict = {}
|
|
468
468
|
self.filtered_varlist = []
|
|
469
469
|
self._corr_matrix_cache = None
|
|
470
|
+
self._corr_matrix_excluded = set()
|
|
470
471
|
self._metric_summary_cache = None
|
|
472
|
+
self._correlation_decision_trace = []
|
|
473
|
+
|
|
474
|
+
def _corr_matrix_frame(self, varlist):
|
|
475
|
+
# 0.7.0-R1: mixed correlation base. Numeric cols stay raw so a
|
|
476
|
+
# pure-numeric varlist is byte-identical to the legacy .corr() path;
|
|
477
|
+
# non-numeric cols are WOE-encoded via the screening binner
|
|
478
|
+
# (corr_use_woe_bins semantics) and renamed back to raw names. Cols that
|
|
479
|
+
# cannot be encoded go to self._corr_matrix_excluded: they leave the
|
|
480
|
+
# matrix and survive the corr stage as NaN rows/cols once _high_corr_pairs
|
|
481
|
+
# reindexes them back in (same as the weighted raw precedent). At most one
|
|
482
|
+
# UserWarning per matrix build.
|
|
483
|
+
non_numeric = [c for c in varlist if not pd.api.types.is_numeric_dtype(self.data[c])]
|
|
484
|
+
if not non_numeric:
|
|
485
|
+
self._corr_matrix_excluded = set()
|
|
486
|
+
return self.data[varlist]
|
|
487
|
+
|
|
488
|
+
encoded = {}
|
|
489
|
+
excluded = set()
|
|
490
|
+
if self.woe_binner is not None:
|
|
491
|
+
suffix = str(getattr(self.woe_binner, "woe_suffix", "_woe"))
|
|
492
|
+
try:
|
|
493
|
+
adapter = as_woe_engine(self.woe_binner, woe_suffix=suffix)
|
|
494
|
+
except TypeError as exc:
|
|
495
|
+
raise TypeError(
|
|
496
|
+
"corr_use_woe_bins=True routed categorical feature(s) through "
|
|
497
|
+
"the screening WOE engine at the corr stage, but the supplied "
|
|
498
|
+
"woe_binner is an Unsupported WOE engine. Expected WOE_Master, "
|
|
499
|
+
"MonotoneWOEBinner, or WOEEngineAdapter."
|
|
500
|
+
) from exc
|
|
501
|
+
suffix = adapter.woe_suffix
|
|
502
|
+
try:
|
|
503
|
+
tx = adapter.transform(self.data.copy(), varlist=non_numeric, suffix=suffix)
|
|
504
|
+
except (TypeError, ValueError, KeyError, AttributeError, np.linalg.LinAlgError):
|
|
505
|
+
tx = None
|
|
506
|
+
for name in non_numeric:
|
|
507
|
+
column = f"{name}{suffix}"
|
|
508
|
+
if tx is not None and column in tx.columns and pd.api.types.is_numeric_dtype(tx[column]):
|
|
509
|
+
encoded[name] = tx[column].to_numpy()
|
|
510
|
+
else:
|
|
511
|
+
excluded.add(name)
|
|
512
|
+
if excluded:
|
|
513
|
+
excluded_list = [c for c in non_numeric if c in excluded]
|
|
514
|
+
warnings.warn(
|
|
515
|
+
f"corr_use_woe_bins=True could not WOE-encode {len(excluded)} "
|
|
516
|
+
f"feature(s) {excluded_list[:5]}; they are excluded from the "
|
|
517
|
+
f"correlation matrix and kept through the corr stage. Refit the "
|
|
518
|
+
f"screening WOE engine to cover them.",
|
|
519
|
+
UserWarning,
|
|
520
|
+
stacklevel=2,
|
|
521
|
+
)
|
|
522
|
+
else:
|
|
523
|
+
excluded = set(non_numeric)
|
|
524
|
+
warnings.warn(
|
|
525
|
+
f"raw-value correlation skips {len(non_numeric)} non-numeric "
|
|
526
|
+
f"feature(s) {non_numeric[:5]}; they are kept through the corr "
|
|
527
|
+
f"stage. Set corr_use_woe_bins=True to correlate categorical "
|
|
528
|
+
f"features via their WOE encoding.",
|
|
529
|
+
UserWarning,
|
|
530
|
+
stacklevel=2,
|
|
531
|
+
)
|
|
532
|
+
|
|
533
|
+
self._corr_matrix_excluded = excluded
|
|
534
|
+
selected = [c for c in varlist if c not in excluded]
|
|
535
|
+
frame = self.data[selected].copy()
|
|
536
|
+
for name, values in encoded.items():
|
|
537
|
+
frame[name] = values
|
|
538
|
+
return frame
|
|
471
539
|
|
|
472
540
|
def _high_corr_pairs(self, varlist):
|
|
473
541
|
if (
|
|
474
542
|
self._corr_matrix_cache is None
|
|
475
|
-
or not set(varlist)
|
|
543
|
+
or not (set(varlist) <= set(self._corr_matrix_cache.columns) | self._corr_matrix_excluded)
|
|
476
544
|
):
|
|
477
|
-
self._corr_matrix_cache = self.
|
|
545
|
+
self._corr_matrix_cache = self._corr_matrix_frame(varlist).corr(method=self.method)
|
|
478
546
|
matrix = self._corr_matrix_cache.reindex(index=varlist, columns=varlist)
|
|
479
547
|
values = matrix.to_numpy(dtype=float)
|
|
480
548
|
row_idx, col_idx = np.triu_indices(len(varlist), k=1)
|
|
@@ -527,9 +595,21 @@ class CorrelationFilter:
|
|
|
527
595
|
def calculate_vif(df):
|
|
528
596
|
return _BaseCorrelationFilter.calculate_vif(df)
|
|
529
597
|
|
|
598
|
+
def _sync_base_state(self):
|
|
599
|
+
self.correlated_dict = getattr(self._base, "correlated_dict", {})
|
|
600
|
+
self.filtered_varlist = getattr(self._base, "filtered_varlist", [])
|
|
601
|
+
self._corr_matrix_cache = getattr(self._base, "_corr_matrix_cache", None)
|
|
602
|
+
self._corr_matrix_excluded = getattr(self._base, "_corr_matrix_excluded", set())
|
|
603
|
+
self._metric_summary_cache = getattr(self._base, "_metric_summary_cache", None)
|
|
604
|
+
self._correlation_decision_trace = getattr(
|
|
605
|
+
self._base, "_correlation_decision_trace", [],
|
|
606
|
+
)
|
|
607
|
+
|
|
530
608
|
def filter_single_iteration(self, varlist):
|
|
531
609
|
if self.woe_binner is None and self.woe_engine == "master":
|
|
532
|
-
|
|
610
|
+
result = self._base.filter_single_iteration(varlist)
|
|
611
|
+
self._sync_base_state()
|
|
612
|
+
return result
|
|
533
613
|
|
|
534
614
|
name_mapping = {"iv": "iv", "ks": "ks_in_gains"}
|
|
535
615
|
high_corr_var = self._high_corr_pairs(varlist)
|
|
@@ -550,7 +630,28 @@ class CorrelationFilter:
|
|
|
550
630
|
selected = summary.sort_values([name_mapping[self.base_metric.lower()]], ascending=False)["var"].iloc[0]
|
|
551
631
|
if selected not in selected_varlist:
|
|
552
632
|
selected_varlist.append(selected)
|
|
553
|
-
|
|
633
|
+
newly_removed = [
|
|
634
|
+
x for x in correlated_list
|
|
635
|
+
if x != selected and x not in removed_varlist
|
|
636
|
+
]
|
|
637
|
+
metric_name = name_mapping[self.base_metric.lower()]
|
|
638
|
+
metric_map = summary.set_index("var")[metric_name].to_dict()
|
|
639
|
+
positions = {name: idx for idx, name in enumerate(varlist)}
|
|
640
|
+
for dropped_var in newly_removed:
|
|
641
|
+
decision_pair = [selected, dropped_var]
|
|
642
|
+
decision_pair.sort(key=positions.__getitem__)
|
|
643
|
+
var_a, var_b = decision_pair
|
|
644
|
+
corr_value = self._corr_matrix_cache.loc[var_a, var_b]
|
|
645
|
+
self._correlation_decision_trace.append({
|
|
646
|
+
"var_a": var_a,
|
|
647
|
+
"var_b": var_b,
|
|
648
|
+
"corr": float(corr_value),
|
|
649
|
+
"iv_a": float(metric_map.get(var_a, 0.0)),
|
|
650
|
+
"iv_b": float(metric_map.get(var_b, 0.0)),
|
|
651
|
+
"kept": selected,
|
|
652
|
+
"dropped": dropped_var,
|
|
653
|
+
})
|
|
654
|
+
removed_varlist += newly_removed
|
|
554
655
|
self.correlated_dict[var] = {"corr": single_var_corr, "gains": summary}
|
|
555
656
|
|
|
556
657
|
return selected_varlist + [x for x in varlist if x not in (selected_varlist + removed_varlist)]
|
|
@@ -558,11 +659,11 @@ class CorrelationFilter:
|
|
|
558
659
|
def remove_highly_correlated(self, varlist, max_iterations=10):
|
|
559
660
|
if self.woe_binner is None and self.woe_engine == "master":
|
|
560
661
|
result = self._base.remove_highly_correlated(varlist, max_iterations)
|
|
561
|
-
self.
|
|
562
|
-
self.filtered_varlist = getattr(self._base, "filtered_varlist", [])
|
|
662
|
+
self._sync_base_state()
|
|
563
663
|
return result
|
|
564
664
|
|
|
565
|
-
self.
|
|
665
|
+
self._correlation_decision_trace = []
|
|
666
|
+
self._corr_matrix_cache = self._corr_matrix_frame(varlist).corr(method=self.method)
|
|
566
667
|
self._metric_summary_cache = None
|
|
567
668
|
self._metric_summary(varlist)
|
|
568
669
|
last_keep_list = self.filter_single_iteration(varlist)
|
{supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Weighted_Screen.py
RENAMED
|
@@ -578,21 +578,52 @@ def _weighted_corr_for_screen(
|
|
|
578
578
|
"""Build weighted correlation matrix for screening (WOE or raw-value path)."""
|
|
579
579
|
from Modeling_Tool.WOE.WOE_Adapter import as_woe_engine
|
|
580
580
|
|
|
581
|
-
if
|
|
581
|
+
non_numeric = [v for v in current if not pd.api.types.is_numeric_dtype(ins[v])]
|
|
582
|
+
if corr_use_woe_bins and non_numeric:
|
|
582
583
|
eng = adapter if adapter is not None else (as_woe_engine(binner) if binner is not None else None)
|
|
583
584
|
if eng is not None:
|
|
584
585
|
suffix = getattr(eng, "woe_suffix", "_woe")
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
586
|
+
# Match the unweighted mixed basis: numeric columns stay raw and
|
|
587
|
+
# only categorical columns are WOE encoded.
|
|
588
|
+
woe_ins = eng.transform(
|
|
589
|
+
ins.copy(), varlist=non_numeric, suffix=suffix
|
|
590
|
+
)
|
|
591
|
+
encoded: dict[str, np.ndarray] = {}
|
|
592
|
+
excluded: list[str] = []
|
|
593
|
+
for name in non_numeric:
|
|
594
|
+
column = f"{name}{suffix}"
|
|
595
|
+
if column in woe_ins.columns and pd.api.types.is_numeric_dtype(woe_ins[column]):
|
|
596
|
+
encoded[name] = woe_ins[column].to_numpy()
|
|
597
|
+
else:
|
|
598
|
+
excluded.append(name)
|
|
599
|
+
if excluded:
|
|
600
|
+
warnings.warn(
|
|
601
|
+
f"corr_use_woe_bins=True could not WOE-encode {len(excluded)} "
|
|
602
|
+
f"feature(s) {excluded[:5]}; they are excluded from the "
|
|
603
|
+
f"correlation matrix and kept through the corr stage. Refit the "
|
|
604
|
+
f"screening WOE engine to cover them.",
|
|
605
|
+
UserWarning,
|
|
606
|
+
stacklevel=2,
|
|
607
|
+
)
|
|
608
|
+
|
|
609
|
+
# Return a matrix on the full current vocabulary. Excluded
|
|
610
|
+
# categories remain NaN rows/columns so dedup keeps them.
|
|
611
|
+
corr_full = np.full((len(current), len(current)), np.nan)
|
|
612
|
+
excluded_set = set(excluded)
|
|
613
|
+
active = [name for name in current if name not in excluded_set]
|
|
614
|
+
if active:
|
|
615
|
+
frame = ins[active].copy()
|
|
616
|
+
for name, values in encoded.items():
|
|
617
|
+
frame[name] = values
|
|
618
|
+
corr_active = _weighted_pearson_corr_matrix(
|
|
619
|
+
frame[active].to_numpy(dtype=float),
|
|
591
620
|
w_ins,
|
|
592
|
-
nan_policy=
|
|
621
|
+
nan_policy=corr_nan_policy,
|
|
593
622
|
corr_block_size=corr_block_size,
|
|
594
623
|
)
|
|
595
|
-
|
|
624
|
+
active_idx = [current.index(name) for name in active]
|
|
625
|
+
corr_full[np.ix_(active_idx, active_idx)] = corr_active
|
|
626
|
+
return corr_full
|
|
596
627
|
if non_numeric:
|
|
597
628
|
# Raw-value Pearson correlation is undefined for categorical/object
|
|
598
629
|
# features. Exclude them from the correlation computation but keep
|
|
@@ -804,6 +835,99 @@ def _corr_dedup_weighted(
|
|
|
804
835
|
return current, pd.DataFrame(dropped_rows)
|
|
805
836
|
|
|
806
837
|
|
|
838
|
+
def _corr_filter_dropped_audit(
|
|
839
|
+
corr_filter: Any,
|
|
840
|
+
varlist: list[str],
|
|
841
|
+
kept: list[str],
|
|
842
|
+
) -> pd.DataFrame:
|
|
843
|
+
"""Convert an unweighted CorrelationFilter decision to weighted audit rows.
|
|
844
|
+
|
|
845
|
+
Constant positive weights reuse CorrelationFilter for byte-compatible
|
|
846
|
+
feature ordering and IV tie-breaking. This adapter preserves the existing
|
|
847
|
+
seven-column ``corr_dropped`` evidence contract for that weighted call.
|
|
848
|
+
Each row's ``var_a/var_b`` endpoints are exactly its ``kept/dropped`` pair.
|
|
849
|
+
For star-shaped groups, ``corr`` may be below the cutoff because it records
|
|
850
|
+
the group winner versus the indirectly dropped member; another group edge
|
|
851
|
+
triggered the decision.
|
|
852
|
+
"""
|
|
853
|
+
columns = ["var_a", "var_b", "corr", "iv_a", "iv_b", "kept", "dropped"]
|
|
854
|
+
kept_set = set(kept)
|
|
855
|
+
dropped = [name for name in varlist if name not in kept_set]
|
|
856
|
+
if not dropped:
|
|
857
|
+
return pd.DataFrame(columns=columns)
|
|
858
|
+
|
|
859
|
+
def _state(name: str):
|
|
860
|
+
value = getattr(corr_filter, name, None)
|
|
861
|
+
if value is None:
|
|
862
|
+
value = getattr(getattr(corr_filter, "_base", None), name, None)
|
|
863
|
+
return value
|
|
864
|
+
|
|
865
|
+
trace = _state("_correlation_decision_trace")
|
|
866
|
+
rows = []
|
|
867
|
+
recorded = set()
|
|
868
|
+
if isinstance(trace, list):
|
|
869
|
+
for row in trace:
|
|
870
|
+
if not isinstance(row, dict) or not set(columns).issubset(row):
|
|
871
|
+
continue
|
|
872
|
+
dropped_name = row["dropped"]
|
|
873
|
+
if dropped_name in dropped and dropped_name not in recorded:
|
|
874
|
+
rows.append({column: row[column] for column in columns})
|
|
875
|
+
recorded.add(dropped_name)
|
|
876
|
+
dropped = [name for name in dropped if name not in recorded]
|
|
877
|
+
if not dropped:
|
|
878
|
+
return pd.DataFrame(rows, columns=columns)
|
|
879
|
+
|
|
880
|
+
# Defensive fallback for third-party/future filters without trace support.
|
|
881
|
+
matrix = _state("_corr_matrix_cache")
|
|
882
|
+
if not isinstance(matrix, pd.DataFrame):
|
|
883
|
+
return pd.DataFrame(rows, columns=columns)
|
|
884
|
+
matrix = matrix.reindex(index=varlist, columns=varlist)
|
|
885
|
+
|
|
886
|
+
metric = _state("_metric_summary_cache")
|
|
887
|
+
iv_map: dict[str, float] = {}
|
|
888
|
+
if isinstance(metric, pd.DataFrame) and {"var", "iv"}.issubset(metric.columns):
|
|
889
|
+
for name, value in zip(metric["var"], metric["iv"]):
|
|
890
|
+
try:
|
|
891
|
+
iv_map[str(name)] = float(value)
|
|
892
|
+
except (TypeError, ValueError):
|
|
893
|
+
iv_map[str(name)] = 0.0
|
|
894
|
+
|
|
895
|
+
positions = {name: idx for idx, name in enumerate(varlist)}
|
|
896
|
+
threshold = float(getattr(corr_filter, "corr_cutpoint"))
|
|
897
|
+
for dropped_name in dropped:
|
|
898
|
+
candidates = []
|
|
899
|
+
for partner in varlist:
|
|
900
|
+
if partner == dropped_name:
|
|
901
|
+
continue
|
|
902
|
+
value = matrix.loc[dropped_name, partner]
|
|
903
|
+
if pd.notna(value) and abs(float(value)) > threshold:
|
|
904
|
+
candidates.append((partner, float(value)))
|
|
905
|
+
if not candidates:
|
|
906
|
+
continue
|
|
907
|
+
partner, corr_value = min(
|
|
908
|
+
candidates,
|
|
909
|
+
key=lambda item: (
|
|
910
|
+
0 if item[0] in kept_set else 1,
|
|
911
|
+
-abs(item[1]),
|
|
912
|
+
positions[item[0]],
|
|
913
|
+
),
|
|
914
|
+
)
|
|
915
|
+
if positions[dropped_name] < positions[partner]:
|
|
916
|
+
var_a, var_b = dropped_name, partner
|
|
917
|
+
else:
|
|
918
|
+
var_a, var_b = partner, dropped_name
|
|
919
|
+
rows.append({
|
|
920
|
+
"var_a": var_a,
|
|
921
|
+
"var_b": var_b,
|
|
922
|
+
"corr": corr_value,
|
|
923
|
+
"iv_a": iv_map.get(var_a, 0.0),
|
|
924
|
+
"iv_b": iv_map.get(var_b, 0.0),
|
|
925
|
+
"kept": partner,
|
|
926
|
+
"dropped": dropped_name,
|
|
927
|
+
})
|
|
928
|
+
return pd.DataFrame(rows, columns=columns)
|
|
929
|
+
|
|
930
|
+
|
|
807
931
|
def _legacy_unweighted_screen(
|
|
808
932
|
splits: dict[str, pd.DataFrame],
|
|
809
933
|
feature_cols: list[str],
|
|
@@ -1110,18 +1234,38 @@ def _weighted_screen_impl(
|
|
|
1110
1234
|
|
|
1111
1235
|
corr_dropped = pd.DataFrame(columns=["var_a", "var_b", "corr", "iv_a", "iv_b", "kept", "dropped"])
|
|
1112
1236
|
if corr_enabled and len(current) > 1:
|
|
1113
|
-
corr = _weighted_corr_for_screen(
|
|
1114
|
-
ins, current, w_ins,
|
|
1115
|
-
corr_use_woe_bins=corr_use_woe_bins,
|
|
1116
|
-
corr_nan_policy=corr_nan_policy,
|
|
1117
|
-
corr_block_size=corr_block_size,
|
|
1118
|
-
binner=prefit_woe_engine,
|
|
1119
|
-
)
|
|
1120
|
-
iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
|
|
1121
1237
|
n_before = len(current)
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
|
|
1238
|
+
if bool(np.all(w_ins == w_ins[0])) and corr_nan_policy == "pairwise":
|
|
1239
|
+
# Constant positive weights follow the exact unweighted decision
|
|
1240
|
+
# contract. Explicit non-default NaN policies stay on the weighted
|
|
1241
|
+
# implementation so they cannot be silently ignored.
|
|
1242
|
+
from Modeling_Tool import CorrelationFilter
|
|
1243
|
+
|
|
1244
|
+
corr_input = list(current)
|
|
1245
|
+
use_binner = corr_use_woe_bins and prefit_woe_engine is not None
|
|
1246
|
+
cf = CorrelationFilter(
|
|
1247
|
+
data=ins[current + [target_col]],
|
|
1248
|
+
dep=target_col,
|
|
1249
|
+
corr_cutpoint=corr_threshold,
|
|
1250
|
+
woe_binner=prefit_woe_engine if use_binner else None,
|
|
1251
|
+
woe_engine="monotone" if use_binner else "master",
|
|
1252
|
+
)
|
|
1253
|
+
current = cf.remove_highly_correlated(
|
|
1254
|
+
current, max_iterations=corr_max_iterations
|
|
1255
|
+
)
|
|
1256
|
+
corr_dropped = _corr_filter_dropped_audit(cf, corr_input, current)
|
|
1257
|
+
else:
|
|
1258
|
+
corr = _weighted_corr_for_screen(
|
|
1259
|
+
ins, current, w_ins,
|
|
1260
|
+
corr_use_woe_bins=corr_use_woe_bins,
|
|
1261
|
+
corr_nan_policy=corr_nan_policy,
|
|
1262
|
+
corr_block_size=corr_block_size,
|
|
1263
|
+
binner=prefit_woe_engine,
|
|
1264
|
+
)
|
|
1265
|
+
iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
|
|
1266
|
+
current, corr_dropped = _corr_dedup_weighted(
|
|
1267
|
+
current, corr, iv_map, corr_threshold, corr_max_iterations,
|
|
1268
|
+
)
|
|
1125
1269
|
summary_rows.append(_summary_row("corr", n_before, len(current), corr_threshold, weight_col))
|
|
1126
1270
|
|
|
1127
1271
|
if gates_config is not None:
|
|
@@ -381,7 +381,13 @@ class FeatureSelectionAnalyzer:
|
|
|
381
381
|
self.selected_features_ = results.loc[results['selected'], 'feature'].tolist()
|
|
382
382
|
return results
|
|
383
383
|
|
|
384
|
-
def compute_vif(
|
|
384
|
+
def compute_vif(
|
|
385
|
+
self,
|
|
386
|
+
data,
|
|
387
|
+
nan_handling="fillna_median",
|
|
388
|
+
nan_warn_threshold=0.05,
|
|
389
|
+
sample_weight=None,
|
|
390
|
+
):
|
|
385
391
|
"""
|
|
386
392
|
Compute Variance Inflation Factor (VIF) for multicollinearity detection.
|
|
387
393
|
|
|
@@ -389,6 +395,11 @@ class FeatureSelectionAnalyzer:
|
|
|
389
395
|
----------
|
|
390
396
|
data : pd.DataFrame
|
|
391
397
|
Feature matrix (should not include target variable)
|
|
398
|
+
sample_weight : array-like, optional
|
|
399
|
+
Per-row frequency/sample weights. Constant weights deliberately
|
|
400
|
+
use the legacy OLS implementation for strict parity. Non-constant
|
|
401
|
+
weights use WLS auxiliary regressions with the same no-intercept
|
|
402
|
+
design as variance_inflation_factor.
|
|
392
403
|
|
|
393
404
|
Returns
|
|
394
405
|
-------
|
|
@@ -403,6 +414,11 @@ class FeatureSelectionAnalyzer:
|
|
|
403
414
|
"with: pip install \"SuperModelingFactory[stats]\""
|
|
404
415
|
) from exc
|
|
405
416
|
|
|
417
|
+
weight = (
|
|
418
|
+
resolve_sample_weight(sample_weight=sample_weight, expected_len=len(data))
|
|
419
|
+
if sample_weight is not None
|
|
420
|
+
else None
|
|
421
|
+
)
|
|
406
422
|
work = _prepare_nan_handled_frame(
|
|
407
423
|
data,
|
|
408
424
|
nan_handling=nan_handling,
|
|
@@ -410,9 +426,27 @@ class FeatureSelectionAnalyzer:
|
|
|
410
426
|
context="FeatureSelectionAnalyzer.compute_vif",
|
|
411
427
|
)
|
|
412
428
|
x = work.values
|
|
429
|
+
if weight is not None and nan_handling == "drop_rows":
|
|
430
|
+
weight = weight[~data.isna().any(axis=1).to_numpy()]
|
|
431
|
+
weight = resolve_sample_weight(
|
|
432
|
+
sample_weight=weight, expected_len=len(work)
|
|
433
|
+
)
|
|
434
|
+
|
|
435
|
+
# Preserve the exact legacy call path for no weight and every
|
|
436
|
+
# constant-weight vector.
|
|
437
|
+
if weight is None or bool(np.all(weight == weight[0])):
|
|
438
|
+
vif_values = [variance_inflation_factor(x, i) for i in range(x.shape[1])]
|
|
439
|
+
else:
|
|
440
|
+
from statsmodels.regression.linear_model import WLS
|
|
441
|
+
|
|
442
|
+
vif_values = []
|
|
443
|
+
for i in range(x.shape[1]):
|
|
444
|
+
others = np.arange(x.shape[1]) != i
|
|
445
|
+
r_squared = WLS(x[:, i], x[:, others], weights=weight).fit().rsquared
|
|
446
|
+
vif_values.append(1.0 / (1.0 - r_squared))
|
|
413
447
|
vif_data = pd.DataFrame({
|
|
414
448
|
'feature': work.columns,
|
|
415
|
-
'VIF':
|
|
449
|
+
'VIF': vif_values,
|
|
416
450
|
}).sort_values('VIF', ascending=False).reset_index(drop=True)
|
|
417
451
|
|
|
418
452
|
return vif_data
|
|
@@ -28,7 +28,7 @@ def compute_bic(model, x, y): ...
|
|
|
28
28
|
class FeatureSelectionAnalyzer:
|
|
29
29
|
def __init__(self, significance_level = 0.05): ...
|
|
30
30
|
def chi2_selection(self, data, feature_cols, target_col, nan_handling = "fillna_median", nan_warn_threshold = 0.05): ...
|
|
31
|
-
def compute_vif(self, data, nan_handling = "fillna_median", nan_warn_threshold = 0.05): ...
|
|
31
|
+
def compute_vif(self, data, nan_handling = "fillna_median", nan_warn_threshold = 0.05, sample_weight = None): ...
|
|
32
32
|
def correlation_filter(self, data, threshold = 0.8): ...
|
|
33
33
|
|
|
34
34
|
class LRMaster:
|