SuperModelingFactory 0.7.0__tar.gz → 0.7.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Insights.py +50 -4
  2. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Screen.py +36 -16
  3. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Screen_Gates.py +26 -4
  4. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/WOE_Engine_Feature_Patch.py +108 -7
  5. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Weighted_Screen.py +164 -20
  6. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/LRM_Tool.py +36 -2
  7. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/LRM_Tool.pyi +1 -1
  8. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/feature_validation.py +61 -7
  9. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Monotone_Binner.py +566 -86
  10. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/__init__.py +1 -1
  11. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/PKG-INFO +2 -2
  12. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/README.md +1 -1
  13. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/PKG-INFO +2 -2
  14. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/pyproject.toml +1 -1
  15. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/setup.py +1 -1
  16. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/ExcelMaster/ExcelFormatTool.py +0 -0
  17. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/ExcelMaster/ExcelMaster.py +0 -0
  18. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/ExcelMaster/Template.py +0 -0
  19. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/ExcelMaster/Utility.py +0 -0
  20. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/ExcelMaster/__init__.py +0 -0
  21. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/LICENSE +0 -0
  22. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/MANIFEST.in +0 -0
  23. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Binning_Tool.py +0 -0
  24. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Binning_Tool.pyi +0 -0
  25. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Check_DuckDB_Compatibility.py +0 -0
  26. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Json_Data_Converter.py +0 -0
  27. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Model_Registry_Tool.py +0 -0
  28. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/ODPS_Tool.py +0 -0
  29. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Parallel_Engine.py +0 -0
  30. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Parallel_ODPS_Manager.py +0 -0
  31. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Proc_Compare.py +0 -0
  32. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Slope_Tool.py +0 -0
  33. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/Slope_Tool.pyi +0 -0
  34. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/XOR_Encryptor.py +0 -0
  35. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/XOR_Encryptor.pyi +0 -0
  36. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/__init__.py +0 -0
  37. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/kDataFrame.py +0 -0
  38. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/kDataFrame.pyi +0 -0
  39. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/sample_weight_utils.py +0 -0
  40. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Core/utils.py +0 -0
  41. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Evaluation_Tool.py +0 -0
  42. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Evaluation_Tool.pyi +0 -0
  43. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Model_Eval_Tool.py +0 -0
  44. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/Model_Eval_Tool.pyi +0 -0
  45. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/__init__.py +0 -0
  46. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/evaluate_model.py +0 -0
  47. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/evaluate_model.pyi +0 -0
  48. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Eval/weighted_eval_utils.py +0 -0
  49. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Explainability/Coalition_Structure.py +0 -0
  50. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Explainability/Model_Explainer.py +0 -0
  51. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Explainability/__init__.py +0 -0
  52. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Distribution_Tool.py +0 -0
  53. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Distribution_Tool.pyi +0 -0
  54. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/Feature_Insights.pyi +0 -0
  55. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/ODPS_Distribution_Tool.py +0 -0
  56. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/ODPS_Distribution_Tool.pyi +0 -0
  57. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/PSI_Tool.py +0 -0
  58. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/PSI_Tool.pyi +0 -0
  59. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Feature/__init__.py +0 -0
  60. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/Backward_Tool.py +0 -0
  61. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/Backward_Tool.pyi +0 -0
  62. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/GBM_Search_Tool.py +0 -0
  63. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/GBM_Tool.py +0 -0
  64. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/GBM_Tool.pyi +0 -0
  65. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Model/__init__.py +0 -0
  66. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/__init__.py +0 -0
  67. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/_common.py +0 -0
  68. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/credit_model.py +0 -0
  69. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/field_meta.py +0 -0
  70. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/mock_sample.py +0 -0
  71. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/orchestrator.py +0 -0
  72. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/reject_inference.py +0 -0
  73. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/sample_analysis.py +0 -0
  74. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/score_comparison.py +0 -0
  75. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/score_consistency_uat.py +0 -0
  76. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Pipeline/screening_artifact.py +0 -0
  77. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Distribution_Adaptation.py +0 -0
  78. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Distribution_Adaptation.pyi +0 -0
  79. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Reject_Infer.py +0 -0
  80. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Reject_Infer.pyi +0 -0
  81. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Sample_Split.py +0 -0
  82. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/Sample_Split.pyi +0 -0
  83. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/Sample/__init__.py +0 -0
  84. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/UAT/UAT_Consistency_Checker.py +0 -0
  85. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/UAT/__init__.py +0 -0
  86. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Adapter.py +0 -0
  87. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Adapter.pyi +0 -0
  88. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Master.py +0 -0
  89. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Master.pyi +0 -0
  90. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Monotone_Binner.pyi +0 -0
  91. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Plot_Tool.py +0 -0
  92. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Plot_Tool.pyi +0 -0
  93. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Report_Builder.py +0 -0
  94. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Report_Builder.pyi +0 -0
  95. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Tool.py +0 -0
  96. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/WOE_Tool.pyi +0 -0
  97. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/__init__.py +0 -0
  98. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/plot_woe_tool.py +0 -0
  99. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/WOE/plot_woe_tool.pyi +0 -0
  100. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/__init__.py +0 -0
  101. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/nan_guard.py +0 -0
  102. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/robust.py +0 -0
  103. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/_utils/sentinels.py +0 -0
  104. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/KaiTi.ttf +0 -0
  105. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/WeiRuanYaHei.ttf +0 -0
  106. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/__init__.py +0 -0
  107. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Modeling_Tool/ref_font/simsun.ttc +0 -0
  108. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Report/Report_Tool.py +0 -0
  109. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/Report/__init__.py +0 -0
  110. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/SOURCES.txt +0 -0
  111. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/dependency_links.txt +0 -0
  112. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/not-zip-safe +0 -0
  113. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/requires.txt +0 -0
  114. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/SuperModelingFactory.egg-info/top_level.txt +0 -0
  115. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/requirements.txt +0 -0
  116. {supermodelingfactory-0.7.0 → supermodelingfactory-0.7.2}/setup.cfg +0 -0
@@ -417,14 +417,39 @@ class CorrelationFilter:
417
417
  self.correlated_dict = {}
418
418
  self.filtered_varlist = []
419
419
  self._corr_matrix_cache = None
420
+ self._corr_matrix_excluded = set()
420
421
  self._metric_summary_cache = None
422
+ self._correlation_decision_trace = []
423
+
424
+ def _corr_matrix_frame(self, varlist):
425
+ # 0.7.0-R1: numeric-subset correlation base. A pure-numeric varlist is
426
+ # byte-identical to the legacy self.data[varlist].corr() path. Non-numeric
427
+ # cols (raw Pearson is undefined) are excluded and tracked in
428
+ # self._corr_matrix_excluded; _high_corr_pairs reindexes them back as NaN
429
+ # rows/cols so they survive the corr stage instead of crashing the float
430
+ # cast. At most one raw-value UserWarning per matrix build (no WOE binner).
431
+ non_numeric = [c for c in varlist if not pd.api.types.is_numeric_dtype(self.data[c])]
432
+ if not non_numeric:
433
+ self._corr_matrix_excluded = set()
434
+ return self.data[varlist]
435
+ self._corr_matrix_excluded = set(non_numeric)
436
+ warnings.warn(
437
+ f"raw-value correlation skips {len(non_numeric)} non-numeric "
438
+ f"feature(s) {non_numeric[:5]}; they are kept through the corr "
439
+ f"stage. Set corr_use_woe_bins=True to correlate categorical "
440
+ f"features via their WOE encoding.",
441
+ UserWarning,
442
+ stacklevel=2,
443
+ )
444
+ selected = [c for c in varlist if c not in self._corr_matrix_excluded]
445
+ return self.data[selected]
421
446
 
422
447
  def _high_corr_pairs(self, varlist):
423
448
  if (
424
449
  self._corr_matrix_cache is None
425
- or not set(varlist).issubset(self._corr_matrix_cache.columns)
450
+ or not (set(varlist) <= set(self._corr_matrix_cache.columns) | self._corr_matrix_excluded)
426
451
  ):
427
- self._corr_matrix_cache = self.data[varlist].corr(method=self.method)
452
+ self._corr_matrix_cache = self._corr_matrix_frame(varlist).corr(method=self.method)
428
453
  matrix = self._corr_matrix_cache.reindex(index=varlist, columns=varlist)
429
454
  values = matrix.to_numpy(dtype=float)
430
455
  row_idx, col_idx = np.triu_indices(len(varlist), k=1)
@@ -522,7 +547,27 @@ class CorrelationFilter:
522
547
  if fnl_selected_var not in selected_varlist:
523
548
  selected_varlist.append(fnl_selected_var)
524
549
 
525
- removed_varlist += [x for x in correlated_list if x != fnl_selected_var and x not in removed_varlist]
550
+ newly_removed = [
551
+ x for x in correlated_list
552
+ if x != fnl_selected_var and x not in removed_varlist
553
+ ]
554
+ metric_map = fnl_summary.set_index("var")[name_mapping[base_metric]].to_dict()
555
+ positions = {name: idx for idx, name in enumerate(varlist)}
556
+ for dropped_var in newly_removed:
557
+ decision_pair = [fnl_selected_var, dropped_var]
558
+ decision_pair.sort(key=positions.__getitem__)
559
+ var_a, var_b = decision_pair
560
+ corr_value = self._corr_matrix_cache.loc[var_a, var_b]
561
+ self._correlation_decision_trace.append({
562
+ "var_a": var_a,
563
+ "var_b": var_b,
564
+ "corr": float(corr_value),
565
+ "iv_a": float(metric_map.get(var_a, 0.0)),
566
+ "iv_b": float(metric_map.get(var_b, 0.0)),
567
+ "kept": fnl_selected_var,
568
+ "dropped": dropped_var,
569
+ })
570
+ removed_varlist += newly_removed
526
571
 
527
572
  if var not in correlated_dict:
528
573
  correlated_dict[var] = {}
@@ -561,7 +606,8 @@ class CorrelationFilter:
561
606
  >>> filter_analyzer = CorrelationFilter(df, 'target')
562
607
  >>> keep_vars = filter_analyzer.remove_highly_correlated(['var1', 'var2', 'var3'])
563
608
  """
564
- self._corr_matrix_cache = self.data[varlist].corr(method=self.method)
609
+ self._correlation_decision_trace = []
610
+ self._corr_matrix_cache = self._corr_matrix_frame(varlist).corr(method=self.method)
565
611
  self._metric_summary_cache = None
566
612
  self._metric_summary(varlist)
567
613
  last_keep_list = self.filter_single_iteration(varlist)
@@ -19,6 +19,7 @@ from .Weighted_Screen import (
19
19
  _apply_missing_rate_stage,
20
20
  _apply_stage_keep,
21
21
  _corr_dedup_weighted,
22
+ _corr_filter_dropped_audit,
22
23
  _gate_ranking_iv_map,
23
24
  _iv_band_keep,
24
25
  _legacy_unweighted_screen,
@@ -701,23 +702,42 @@ def _weighted_woe_bins_screen(
701
702
 
702
703
  corr_dropped = pd.DataFrame(columns=["var_a", "var_b", "corr", "iv_a", "iv_b", "kept", "dropped"])
703
704
  if config.corr_enabled and len(current) > 1:
704
- corr = _weighted_corr_for_screen(
705
- ins, current, w_ins,
706
- corr_use_woe_bins=config.corr_use_woe_bins,
707
- corr_nan_policy=config.corr_nan_policy,
708
- corr_block_size=config.corr_block_size,
709
- adapter=adapter,
710
- binner=binner,
711
- )
712
- iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
713
705
  n_before = len(current)
714
- current, corr_dropped = _corr_dedup_weighted(
715
- current,
716
- corr,
717
- iv_map,
718
- config.corr_threshold,
719
- config.corr_max_iterations,
720
- )
706
+ if bool(np.all(w_ins == w_ins[0])) and config.corr_nan_policy == "pairwise":
707
+ # Match the unweighted WOE/mixed-basis decision exactly for
708
+ # constant positive weights, while retaining weighted audit rows.
709
+ from Modeling_Tool import CorrelationFilter
710
+
711
+ corr_input = list(current)
712
+ use_binner = config.corr_use_woe_bins and binner is not None
713
+ cf = CorrelationFilter(
714
+ data=ins[current + [target_col]],
715
+ dep=target_col,
716
+ corr_cutpoint=config.corr_threshold,
717
+ woe_binner=binner if use_binner else None,
718
+ woe_engine="monotone" if use_binner else "master",
719
+ )
720
+ current = cf.remove_highly_correlated(
721
+ current, max_iterations=config.corr_max_iterations
722
+ )
723
+ corr_dropped = _corr_filter_dropped_audit(cf, corr_input, current)
724
+ else:
725
+ corr = _weighted_corr_for_screen(
726
+ ins, current, w_ins,
727
+ corr_use_woe_bins=config.corr_use_woe_bins,
728
+ corr_nan_policy=config.corr_nan_policy,
729
+ corr_block_size=config.corr_block_size,
730
+ adapter=adapter,
731
+ binner=binner,
732
+ )
733
+ iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
734
+ current, corr_dropped = _corr_dedup_weighted(
735
+ current,
736
+ corr,
737
+ iv_map,
738
+ config.corr_threshold,
739
+ config.corr_max_iterations,
740
+ )
721
741
  summary_rows.append(_summary_row("corr", n_before, len(current), config.corr_threshold, weight_col))
722
742
 
723
743
  from .Screen_Gates import apply_post_corr_gates
@@ -24,6 +24,8 @@ from typing import Any, Callable
24
24
  import numpy as np
25
25
  import pandas as pd
26
26
 
27
+ from Modeling_Tool.Core.sample_weight_utils import resolve_sample_weight
28
+
27
29
  from .Weighted_Screen import _apply_stage_keep, _summary_row
28
30
 
29
31
 
@@ -118,8 +120,18 @@ def apply_vif_stage(
118
120
  basis excludes non-numeric survivors from the matrix — raw string columns
119
121
  used to crash statsmodels — keeping them in the selection untouched.
120
122
  """
121
- if not getattr(config, "vif_enabled", False) or len(current) <= max(2, int(config.vif_min_features)):
122
- if getattr(config, "vif_enabled", False) and len(current) <= int(config.vif_min_features):
123
+ if not getattr(config, "vif_enabled", False):
124
+ return current
125
+
126
+ tie_metric = str(getattr(config, "vif_tie_break_metric", "iv"))
127
+ if tie_metric != "iv":
128
+ raise ValueError(
129
+ "vif_tie_break_metric currently supports only 'iv'; "
130
+ f"got {tie_metric!r}"
131
+ )
132
+
133
+ if len(current) <= max(2, int(config.vif_min_features)):
134
+ if len(current) <= int(config.vif_min_features):
123
135
  summary_rows.append(_summary_row(
124
136
  "vif", len(current), len(current), config.vif_threshold, weight_col,
125
137
  note="skipped_at_floor",
@@ -139,7 +151,11 @@ def apply_vif_stage(
139
151
  analyzer = FeatureSelectionAnalyzer()
140
152
  threshold = float(config.vif_threshold)
141
153
  floor = int(config.vif_min_features)
142
- tie_metric = str(getattr(config, "vif_tie_break_metric", "iv"))
154
+ sample_weight = (
155
+ resolve_sample_weight(data=ins, weight_col=weight_col, expected_len=len(ins))
156
+ if weight_col is not None
157
+ else None
158
+ )
143
159
  excluded: list[str] = []
144
160
  raw_vif_cast_columns: list[str] = []
145
161
  if bool(getattr(config, "vif_use_woe_bins", False)):
@@ -209,7 +225,13 @@ def apply_vif_stage(
209
225
  for iteration in range(len(current)):
210
226
  if len(survivors) <= floor:
211
227
  break
212
- vif_table = analyzer.compute_vif(base[survivors])
228
+ if sample_weight is None:
229
+ # Preserve the historical unweighted call path byte-for-byte.
230
+ vif_table = analyzer.compute_vif(base[survivors])
231
+ else:
232
+ vif_table = analyzer.compute_vif(
233
+ base[survivors], sample_weight=sample_weight,
234
+ )
213
235
  vif_table = vif_table.sort_values("VIF", ascending=False).reset_index(drop=True)
214
236
  worst = vif_table.iloc[0]
215
237
  if not np.isfinite(worst["VIF"]) or worst["VIF"] > threshold:
@@ -467,14 +467,82 @@ class CorrelationFilter:
467
467
  self.correlated_dict = {}
468
468
  self.filtered_varlist = []
469
469
  self._corr_matrix_cache = None
470
+ self._corr_matrix_excluded = set()
470
471
  self._metric_summary_cache = None
472
+ self._correlation_decision_trace = []
473
+
474
+ def _corr_matrix_frame(self, varlist):
475
+ # 0.7.0-R1: mixed correlation base. Numeric cols stay raw so a
476
+ # pure-numeric varlist is byte-identical to the legacy .corr() path;
477
+ # non-numeric cols are WOE-encoded via the screening binner
478
+ # (corr_use_woe_bins semantics) and renamed back to raw names. Cols that
479
+ # cannot be encoded go to self._corr_matrix_excluded: they leave the
480
+ # matrix and survive the corr stage as NaN rows/cols once _high_corr_pairs
481
+ # reindexes them back in (same as the weighted raw precedent). At most one
482
+ # UserWarning per matrix build.
483
+ non_numeric = [c for c in varlist if not pd.api.types.is_numeric_dtype(self.data[c])]
484
+ if not non_numeric:
485
+ self._corr_matrix_excluded = set()
486
+ return self.data[varlist]
487
+
488
+ encoded = {}
489
+ excluded = set()
490
+ if self.woe_binner is not None:
491
+ suffix = str(getattr(self.woe_binner, "woe_suffix", "_woe"))
492
+ try:
493
+ adapter = as_woe_engine(self.woe_binner, woe_suffix=suffix)
494
+ except TypeError as exc:
495
+ raise TypeError(
496
+ "corr_use_woe_bins=True routed categorical feature(s) through "
497
+ "the screening WOE engine at the corr stage, but the supplied "
498
+ "woe_binner is an Unsupported WOE engine. Expected WOE_Master, "
499
+ "MonotoneWOEBinner, or WOEEngineAdapter."
500
+ ) from exc
501
+ suffix = adapter.woe_suffix
502
+ try:
503
+ tx = adapter.transform(self.data.copy(), varlist=non_numeric, suffix=suffix)
504
+ except (TypeError, ValueError, KeyError, AttributeError, np.linalg.LinAlgError):
505
+ tx = None
506
+ for name in non_numeric:
507
+ column = f"{name}{suffix}"
508
+ if tx is not None and column in tx.columns and pd.api.types.is_numeric_dtype(tx[column]):
509
+ encoded[name] = tx[column].to_numpy()
510
+ else:
511
+ excluded.add(name)
512
+ if excluded:
513
+ excluded_list = [c for c in non_numeric if c in excluded]
514
+ warnings.warn(
515
+ f"corr_use_woe_bins=True could not WOE-encode {len(excluded)} "
516
+ f"feature(s) {excluded_list[:5]}; they are excluded from the "
517
+ f"correlation matrix and kept through the corr stage. Refit the "
518
+ f"screening WOE engine to cover them.",
519
+ UserWarning,
520
+ stacklevel=2,
521
+ )
522
+ else:
523
+ excluded = set(non_numeric)
524
+ warnings.warn(
525
+ f"raw-value correlation skips {len(non_numeric)} non-numeric "
526
+ f"feature(s) {non_numeric[:5]}; they are kept through the corr "
527
+ f"stage. Set corr_use_woe_bins=True to correlate categorical "
528
+ f"features via their WOE encoding.",
529
+ UserWarning,
530
+ stacklevel=2,
531
+ )
532
+
533
+ self._corr_matrix_excluded = excluded
534
+ selected = [c for c in varlist if c not in excluded]
535
+ frame = self.data[selected].copy()
536
+ for name, values in encoded.items():
537
+ frame[name] = values
538
+ return frame
471
539
 
472
540
  def _high_corr_pairs(self, varlist):
473
541
  if (
474
542
  self._corr_matrix_cache is None
475
- or not set(varlist).issubset(self._corr_matrix_cache.columns)
543
+ or not (set(varlist) <= set(self._corr_matrix_cache.columns) | self._corr_matrix_excluded)
476
544
  ):
477
- self._corr_matrix_cache = self.data[varlist].corr(method=self.method)
545
+ self._corr_matrix_cache = self._corr_matrix_frame(varlist).corr(method=self.method)
478
546
  matrix = self._corr_matrix_cache.reindex(index=varlist, columns=varlist)
479
547
  values = matrix.to_numpy(dtype=float)
480
548
  row_idx, col_idx = np.triu_indices(len(varlist), k=1)
@@ -527,9 +595,21 @@ class CorrelationFilter:
527
595
  def calculate_vif(df):
528
596
  return _BaseCorrelationFilter.calculate_vif(df)
529
597
 
598
+ def _sync_base_state(self):
599
+ self.correlated_dict = getattr(self._base, "correlated_dict", {})
600
+ self.filtered_varlist = getattr(self._base, "filtered_varlist", [])
601
+ self._corr_matrix_cache = getattr(self._base, "_corr_matrix_cache", None)
602
+ self._corr_matrix_excluded = getattr(self._base, "_corr_matrix_excluded", set())
603
+ self._metric_summary_cache = getattr(self._base, "_metric_summary_cache", None)
604
+ self._correlation_decision_trace = getattr(
605
+ self._base, "_correlation_decision_trace", [],
606
+ )
607
+
530
608
  def filter_single_iteration(self, varlist):
531
609
  if self.woe_binner is None and self.woe_engine == "master":
532
- return self._base.filter_single_iteration(varlist)
610
+ result = self._base.filter_single_iteration(varlist)
611
+ self._sync_base_state()
612
+ return result
533
613
 
534
614
  name_mapping = {"iv": "iv", "ks": "ks_in_gains"}
535
615
  high_corr_var = self._high_corr_pairs(varlist)
@@ -550,7 +630,28 @@ class CorrelationFilter:
550
630
  selected = summary.sort_values([name_mapping[self.base_metric.lower()]], ascending=False)["var"].iloc[0]
551
631
  if selected not in selected_varlist:
552
632
  selected_varlist.append(selected)
553
- removed_varlist += [x for x in correlated_list if x != selected and x not in removed_varlist]
633
+ newly_removed = [
634
+ x for x in correlated_list
635
+ if x != selected and x not in removed_varlist
636
+ ]
637
+ metric_name = name_mapping[self.base_metric.lower()]
638
+ metric_map = summary.set_index("var")[metric_name].to_dict()
639
+ positions = {name: idx for idx, name in enumerate(varlist)}
640
+ for dropped_var in newly_removed:
641
+ decision_pair = [selected, dropped_var]
642
+ decision_pair.sort(key=positions.__getitem__)
643
+ var_a, var_b = decision_pair
644
+ corr_value = self._corr_matrix_cache.loc[var_a, var_b]
645
+ self._correlation_decision_trace.append({
646
+ "var_a": var_a,
647
+ "var_b": var_b,
648
+ "corr": float(corr_value),
649
+ "iv_a": float(metric_map.get(var_a, 0.0)),
650
+ "iv_b": float(metric_map.get(var_b, 0.0)),
651
+ "kept": selected,
652
+ "dropped": dropped_var,
653
+ })
654
+ removed_varlist += newly_removed
554
655
  self.correlated_dict[var] = {"corr": single_var_corr, "gains": summary}
555
656
 
556
657
  return selected_varlist + [x for x in varlist if x not in (selected_varlist + removed_varlist)]
@@ -558,11 +659,11 @@ class CorrelationFilter:
558
659
  def remove_highly_correlated(self, varlist, max_iterations=10):
559
660
  if self.woe_binner is None and self.woe_engine == "master":
560
661
  result = self._base.remove_highly_correlated(varlist, max_iterations)
561
- self.correlated_dict = getattr(self._base, "correlated_dict", {})
562
- self.filtered_varlist = getattr(self._base, "filtered_varlist", [])
662
+ self._sync_base_state()
563
663
  return result
564
664
 
565
- self._corr_matrix_cache = self.data[varlist].corr(method=self.method)
665
+ self._correlation_decision_trace = []
666
+ self._corr_matrix_cache = self._corr_matrix_frame(varlist).corr(method=self.method)
566
667
  self._metric_summary_cache = None
567
668
  self._metric_summary(varlist)
568
669
  last_keep_list = self.filter_single_iteration(varlist)
@@ -578,21 +578,52 @@ def _weighted_corr_for_screen(
578
578
  """Build weighted correlation matrix for screening (WOE or raw-value path)."""
579
579
  from Modeling_Tool.WOE.WOE_Adapter import as_woe_engine
580
580
 
581
- if corr_use_woe_bins:
581
+ non_numeric = [v for v in current if not pd.api.types.is_numeric_dtype(ins[v])]
582
+ if corr_use_woe_bins and non_numeric:
582
583
  eng = adapter if adapter is not None else (as_woe_engine(binner) if binner is not None else None)
583
584
  if eng is not None:
584
585
  suffix = getattr(eng, "woe_suffix", "_woe")
585
- woe_ins = eng.transform(ins, varlist=current)
586
- cols = [f"{v}{suffix}" for v in current if f"{v}{suffix}" in woe_ins.columns]
587
- if cols:
588
- X = woe_ins[cols].to_numpy(dtype=float)
589
- return _weighted_pearson_corr_matrix(
590
- X,
586
+ # Match the unweighted mixed basis: numeric columns stay raw and
587
+ # only categorical columns are WOE encoded.
588
+ woe_ins = eng.transform(
589
+ ins.copy(), varlist=non_numeric, suffix=suffix
590
+ )
591
+ encoded: dict[str, np.ndarray] = {}
592
+ excluded: list[str] = []
593
+ for name in non_numeric:
594
+ column = f"{name}{suffix}"
595
+ if column in woe_ins.columns and pd.api.types.is_numeric_dtype(woe_ins[column]):
596
+ encoded[name] = woe_ins[column].to_numpy()
597
+ else:
598
+ excluded.append(name)
599
+ if excluded:
600
+ warnings.warn(
601
+ f"corr_use_woe_bins=True could not WOE-encode {len(excluded)} "
602
+ f"feature(s) {excluded[:5]}; they are excluded from the "
603
+ f"correlation matrix and kept through the corr stage. Refit the "
604
+ f"screening WOE engine to cover them.",
605
+ UserWarning,
606
+ stacklevel=2,
607
+ )
608
+
609
+ # Return a matrix on the full current vocabulary. Excluded
610
+ # categories remain NaN rows/columns so dedup keeps them.
611
+ corr_full = np.full((len(current), len(current)), np.nan)
612
+ excluded_set = set(excluded)
613
+ active = [name for name in current if name not in excluded_set]
614
+ if active:
615
+ frame = ins[active].copy()
616
+ for name, values in encoded.items():
617
+ frame[name] = values
618
+ corr_active = _weighted_pearson_corr_matrix(
619
+ frame[active].to_numpy(dtype=float),
591
620
  w_ins,
592
- nan_policy="pairwise",
621
+ nan_policy=corr_nan_policy,
593
622
  corr_block_size=corr_block_size,
594
623
  )
595
- non_numeric = [v for v in current if not pd.api.types.is_numeric_dtype(ins[v])]
624
+ active_idx = [current.index(name) for name in active]
625
+ corr_full[np.ix_(active_idx, active_idx)] = corr_active
626
+ return corr_full
596
627
  if non_numeric:
597
628
  # Raw-value Pearson correlation is undefined for categorical/object
598
629
  # features. Exclude them from the correlation computation but keep
@@ -804,6 +835,99 @@ def _corr_dedup_weighted(
804
835
  return current, pd.DataFrame(dropped_rows)
805
836
 
806
837
 
838
+ def _corr_filter_dropped_audit(
839
+ corr_filter: Any,
840
+ varlist: list[str],
841
+ kept: list[str],
842
+ ) -> pd.DataFrame:
843
+ """Convert an unweighted CorrelationFilter decision to weighted audit rows.
844
+
845
+ Constant positive weights reuse CorrelationFilter for byte-compatible
846
+ feature ordering and IV tie-breaking. This adapter preserves the existing
847
+ seven-column ``corr_dropped`` evidence contract for that weighted call.
848
+ Each row's ``var_a/var_b`` endpoints are exactly its ``kept/dropped`` pair.
849
+ For star-shaped groups, ``corr`` may be below the cutoff because it records
850
+ the group winner versus the indirectly dropped member; another group edge
851
+ triggered the decision.
852
+ """
853
+ columns = ["var_a", "var_b", "corr", "iv_a", "iv_b", "kept", "dropped"]
854
+ kept_set = set(kept)
855
+ dropped = [name for name in varlist if name not in kept_set]
856
+ if not dropped:
857
+ return pd.DataFrame(columns=columns)
858
+
859
+ def _state(name: str):
860
+ value = getattr(corr_filter, name, None)
861
+ if value is None:
862
+ value = getattr(getattr(corr_filter, "_base", None), name, None)
863
+ return value
864
+
865
+ trace = _state("_correlation_decision_trace")
866
+ rows = []
867
+ recorded = set()
868
+ if isinstance(trace, list):
869
+ for row in trace:
870
+ if not isinstance(row, dict) or not set(columns).issubset(row):
871
+ continue
872
+ dropped_name = row["dropped"]
873
+ if dropped_name in dropped and dropped_name not in recorded:
874
+ rows.append({column: row[column] for column in columns})
875
+ recorded.add(dropped_name)
876
+ dropped = [name for name in dropped if name not in recorded]
877
+ if not dropped:
878
+ return pd.DataFrame(rows, columns=columns)
879
+
880
+ # Defensive fallback for third-party/future filters without trace support.
881
+ matrix = _state("_corr_matrix_cache")
882
+ if not isinstance(matrix, pd.DataFrame):
883
+ return pd.DataFrame(rows, columns=columns)
884
+ matrix = matrix.reindex(index=varlist, columns=varlist)
885
+
886
+ metric = _state("_metric_summary_cache")
887
+ iv_map: dict[str, float] = {}
888
+ if isinstance(metric, pd.DataFrame) and {"var", "iv"}.issubset(metric.columns):
889
+ for name, value in zip(metric["var"], metric["iv"]):
890
+ try:
891
+ iv_map[str(name)] = float(value)
892
+ except (TypeError, ValueError):
893
+ iv_map[str(name)] = 0.0
894
+
895
+ positions = {name: idx for idx, name in enumerate(varlist)}
896
+ threshold = float(getattr(corr_filter, "corr_cutpoint"))
897
+ for dropped_name in dropped:
898
+ candidates = []
899
+ for partner in varlist:
900
+ if partner == dropped_name:
901
+ continue
902
+ value = matrix.loc[dropped_name, partner]
903
+ if pd.notna(value) and abs(float(value)) > threshold:
904
+ candidates.append((partner, float(value)))
905
+ if not candidates:
906
+ continue
907
+ partner, corr_value = min(
908
+ candidates,
909
+ key=lambda item: (
910
+ 0 if item[0] in kept_set else 1,
911
+ -abs(item[1]),
912
+ positions[item[0]],
913
+ ),
914
+ )
915
+ if positions[dropped_name] < positions[partner]:
916
+ var_a, var_b = dropped_name, partner
917
+ else:
918
+ var_a, var_b = partner, dropped_name
919
+ rows.append({
920
+ "var_a": var_a,
921
+ "var_b": var_b,
922
+ "corr": corr_value,
923
+ "iv_a": iv_map.get(var_a, 0.0),
924
+ "iv_b": iv_map.get(var_b, 0.0),
925
+ "kept": partner,
926
+ "dropped": dropped_name,
927
+ })
928
+ return pd.DataFrame(rows, columns=columns)
929
+
930
+
807
931
  def _legacy_unweighted_screen(
808
932
  splits: dict[str, pd.DataFrame],
809
933
  feature_cols: list[str],
@@ -1110,18 +1234,38 @@ def _weighted_screen_impl(
1110
1234
 
1111
1235
  corr_dropped = pd.DataFrame(columns=["var_a", "var_b", "corr", "iv_a", "iv_b", "kept", "dropped"])
1112
1236
  if corr_enabled and len(current) > 1:
1113
- corr = _weighted_corr_for_screen(
1114
- ins, current, w_ins,
1115
- corr_use_woe_bins=corr_use_woe_bins,
1116
- corr_nan_policy=corr_nan_policy,
1117
- corr_block_size=corr_block_size,
1118
- binner=prefit_woe_engine,
1119
- )
1120
- iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
1121
1237
  n_before = len(current)
1122
- current, corr_dropped = _corr_dedup_weighted(
1123
- current, corr, iv_map, corr_threshold, corr_max_iterations,
1124
- )
1238
+ if bool(np.all(w_ins == w_ins[0])) and corr_nan_policy == "pairwise":
1239
+ # Constant positive weights follow the exact unweighted decision
1240
+ # contract. Explicit non-default NaN policies stay on the weighted
1241
+ # implementation so they cannot be silently ignored.
1242
+ from Modeling_Tool import CorrelationFilter
1243
+
1244
+ corr_input = list(current)
1245
+ use_binner = corr_use_woe_bins and prefit_woe_engine is not None
1246
+ cf = CorrelationFilter(
1247
+ data=ins[current + [target_col]],
1248
+ dep=target_col,
1249
+ corr_cutpoint=corr_threshold,
1250
+ woe_binner=prefit_woe_engine if use_binner else None,
1251
+ woe_engine="monotone" if use_binner else "master",
1252
+ )
1253
+ current = cf.remove_highly_correlated(
1254
+ current, max_iterations=corr_max_iterations
1255
+ )
1256
+ corr_dropped = _corr_filter_dropped_audit(cf, corr_input, current)
1257
+ else:
1258
+ corr = _weighted_corr_for_screen(
1259
+ ins, current, w_ins,
1260
+ corr_use_woe_bins=corr_use_woe_bins,
1261
+ corr_nan_policy=corr_nan_policy,
1262
+ corr_block_size=corr_block_size,
1263
+ binner=prefit_woe_engine,
1264
+ )
1265
+ iv_map = dict(zip(iv_table["var"], iv_table["iv_weighted"])) if not iv_table.empty else {}
1266
+ current, corr_dropped = _corr_dedup_weighted(
1267
+ current, corr, iv_map, corr_threshold, corr_max_iterations,
1268
+ )
1125
1269
  summary_rows.append(_summary_row("corr", n_before, len(current), corr_threshold, weight_col))
1126
1270
 
1127
1271
  if gates_config is not None:
@@ -381,7 +381,13 @@ class FeatureSelectionAnalyzer:
381
381
  self.selected_features_ = results.loc[results['selected'], 'feature'].tolist()
382
382
  return results
383
383
 
384
- def compute_vif(self, data, nan_handling="fillna_median", nan_warn_threshold=0.05):
384
+ def compute_vif(
385
+ self,
386
+ data,
387
+ nan_handling="fillna_median",
388
+ nan_warn_threshold=0.05,
389
+ sample_weight=None,
390
+ ):
385
391
  """
386
392
  Compute Variance Inflation Factor (VIF) for multicollinearity detection.
387
393
 
@@ -389,6 +395,11 @@ class FeatureSelectionAnalyzer:
389
395
  ----------
390
396
  data : pd.DataFrame
391
397
  Feature matrix (should not include target variable)
398
+ sample_weight : array-like, optional
399
+ Per-row frequency/sample weights. Constant weights deliberately
400
+ use the legacy OLS implementation for strict parity. Non-constant
401
+ weights use WLS auxiliary regressions with the same no-intercept
402
+ design as variance_inflation_factor.
392
403
 
393
404
  Returns
394
405
  -------
@@ -403,6 +414,11 @@ class FeatureSelectionAnalyzer:
403
414
  "with: pip install \"SuperModelingFactory[stats]\""
404
415
  ) from exc
405
416
 
417
+ weight = (
418
+ resolve_sample_weight(sample_weight=sample_weight, expected_len=len(data))
419
+ if sample_weight is not None
420
+ else None
421
+ )
406
422
  work = _prepare_nan_handled_frame(
407
423
  data,
408
424
  nan_handling=nan_handling,
@@ -410,9 +426,27 @@ class FeatureSelectionAnalyzer:
410
426
  context="FeatureSelectionAnalyzer.compute_vif",
411
427
  )
412
428
  x = work.values
429
+ if weight is not None and nan_handling == "drop_rows":
430
+ weight = weight[~data.isna().any(axis=1).to_numpy()]
431
+ weight = resolve_sample_weight(
432
+ sample_weight=weight, expected_len=len(work)
433
+ )
434
+
435
+ # Preserve the exact legacy call path for no weight and every
436
+ # constant-weight vector.
437
+ if weight is None or bool(np.all(weight == weight[0])):
438
+ vif_values = [variance_inflation_factor(x, i) for i in range(x.shape[1])]
439
+ else:
440
+ from statsmodels.regression.linear_model import WLS
441
+
442
+ vif_values = []
443
+ for i in range(x.shape[1]):
444
+ others = np.arange(x.shape[1]) != i
445
+ r_squared = WLS(x[:, i], x[:, others], weights=weight).fit().rsquared
446
+ vif_values.append(1.0 / (1.0 - r_squared))
413
447
  vif_data = pd.DataFrame({
414
448
  'feature': work.columns,
415
- 'VIF': [variance_inflation_factor(x, i) for i in range(x.shape[1])]
449
+ 'VIF': vif_values,
416
450
  }).sort_values('VIF', ascending=False).reset_index(drop=True)
417
451
 
418
452
  return vif_data
@@ -28,7 +28,7 @@ def compute_bic(model, x, y): ...
28
28
  class FeatureSelectionAnalyzer:
29
29
  def __init__(self, significance_level = 0.05): ...
30
30
  def chi2_selection(self, data, feature_cols, target_col, nan_handling = "fillna_median", nan_warn_threshold = 0.05): ...
31
- def compute_vif(self, data, nan_handling = "fillna_median", nan_warn_threshold = 0.05): ...
31
+ def compute_vif(self, data, nan_handling = "fillna_median", nan_warn_threshold = 0.05, sample_weight = None): ...
32
32
  def correlation_filter(self, data, threshold = 0.8): ...
33
33
 
34
34
  class LRMaster: