SuperModelingFactory 0.7.2__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/Feature_Screen.py +4 -0
  2. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/credit_model.py +21 -2
  3. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/feature_validation.py +26 -2
  4. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Master.py +24 -2
  5. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Master.pyi +2 -2
  6. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Monotone_Binner.py +164 -6
  7. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Monotone_Binner.pyi +3 -2
  8. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Tool.py +208 -4
  9. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Tool.pyi +7 -2
  10. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/__init__.py +1 -1
  11. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/PKG-INFO +2 -2
  12. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/README.md +1 -1
  13. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/SuperModelingFactory.egg-info/PKG-INFO +2 -2
  14. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/pyproject.toml +1 -1
  15. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/setup.py +1 -1
  16. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/ExcelMaster/ExcelFormatTool.py +0 -0
  17. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/ExcelMaster/ExcelMaster.py +0 -0
  18. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/ExcelMaster/Template.py +0 -0
  19. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/ExcelMaster/Utility.py +0 -0
  20. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/ExcelMaster/__init__.py +0 -0
  21. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/LICENSE +0 -0
  22. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/MANIFEST.in +0 -0
  23. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/Binning_Tool.py +0 -0
  24. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/Binning_Tool.pyi +0 -0
  25. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/Check_DuckDB_Compatibility.py +0 -0
  26. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/Json_Data_Converter.py +0 -0
  27. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/Model_Registry_Tool.py +0 -0
  28. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/ODPS_Tool.py +0 -0
  29. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/Parallel_Engine.py +0 -0
  30. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/Parallel_ODPS_Manager.py +0 -0
  31. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/Proc_Compare.py +0 -0
  32. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/Slope_Tool.py +0 -0
  33. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/Slope_Tool.pyi +0 -0
  34. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/XOR_Encryptor.py +0 -0
  35. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/XOR_Encryptor.pyi +0 -0
  36. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/__init__.py +0 -0
  37. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/kDataFrame.py +0 -0
  38. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/kDataFrame.pyi +0 -0
  39. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/sample_weight_utils.py +0 -0
  40. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Core/utils.py +0 -0
  41. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Eval/Evaluation_Tool.py +0 -0
  42. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Eval/Evaluation_Tool.pyi +0 -0
  43. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Eval/Model_Eval_Tool.py +0 -0
  44. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Eval/Model_Eval_Tool.pyi +0 -0
  45. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Eval/__init__.py +0 -0
  46. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Eval/evaluate_model.py +0 -0
  47. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Eval/evaluate_model.pyi +0 -0
  48. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Eval/weighted_eval_utils.py +0 -0
  49. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Explainability/Coalition_Structure.py +0 -0
  50. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Explainability/Model_Explainer.py +0 -0
  51. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Explainability/__init__.py +0 -0
  52. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/Distribution_Tool.py +0 -0
  53. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/Distribution_Tool.pyi +0 -0
  54. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/Feature_Insights.py +0 -0
  55. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/Feature_Insights.pyi +0 -0
  56. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/ODPS_Distribution_Tool.py +0 -0
  57. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/ODPS_Distribution_Tool.pyi +0 -0
  58. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/PSI_Tool.py +0 -0
  59. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/PSI_Tool.pyi +0 -0
  60. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/Screen_Gates.py +0 -0
  61. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/WOE_Engine_Feature_Patch.py +0 -0
  62. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/Weighted_Screen.py +0 -0
  63. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Feature/__init__.py +0 -0
  64. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Model/Backward_Tool.py +0 -0
  65. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Model/Backward_Tool.pyi +0 -0
  66. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Model/GBM_Search_Tool.py +0 -0
  67. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Model/GBM_Tool.py +0 -0
  68. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Model/GBM_Tool.pyi +0 -0
  69. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Model/LRM_Tool.py +0 -0
  70. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Model/LRM_Tool.pyi +0 -0
  71. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Model/__init__.py +0 -0
  72. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/__init__.py +0 -0
  73. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/_common.py +0 -0
  74. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/field_meta.py +0 -0
  75. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/mock_sample.py +0 -0
  76. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/orchestrator.py +0 -0
  77. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/reject_inference.py +0 -0
  78. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/sample_analysis.py +0 -0
  79. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/score_comparison.py +0 -0
  80. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/score_consistency_uat.py +0 -0
  81. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Pipeline/screening_artifact.py +0 -0
  82. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Sample/Distribution_Adaptation.py +0 -0
  83. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Sample/Distribution_Adaptation.pyi +0 -0
  84. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Sample/Reject_Infer.py +0 -0
  85. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Sample/Reject_Infer.pyi +0 -0
  86. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Sample/Sample_Split.py +0 -0
  87. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Sample/Sample_Split.pyi +0 -0
  88. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/Sample/__init__.py +0 -0
  89. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/UAT/UAT_Consistency_Checker.py +0 -0
  90. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/UAT/__init__.py +0 -0
  91. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Adapter.py +0 -0
  92. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Adapter.pyi +0 -0
  93. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Plot_Tool.py +0 -0
  94. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Plot_Tool.pyi +0 -0
  95. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Report_Builder.py +0 -0
  96. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/WOE_Report_Builder.pyi +0 -0
  97. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/__init__.py +0 -0
  98. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/plot_woe_tool.py +0 -0
  99. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/WOE/plot_woe_tool.pyi +0 -0
  100. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/_utils/__init__.py +0 -0
  101. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/_utils/nan_guard.py +0 -0
  102. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/_utils/robust.py +0 -0
  103. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/_utils/sentinels.py +0 -0
  104. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/ref_font/KaiTi.ttf +0 -0
  105. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/ref_font/WeiRuanYaHei.ttf +0 -0
  106. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/ref_font/__init__.py +0 -0
  107. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Modeling_Tool/ref_font/simsun.ttc +0 -0
  108. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Report/Report_Tool.py +0 -0
  109. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/Report/__init__.py +0 -0
  110. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/SuperModelingFactory.egg-info/SOURCES.txt +0 -0
  111. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/SuperModelingFactory.egg-info/dependency_links.txt +0 -0
  112. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/SuperModelingFactory.egg-info/not-zip-safe +0 -0
  113. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/SuperModelingFactory.egg-info/requires.txt +0 -0
  114. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/SuperModelingFactory.egg-info/top_level.txt +0 -0
  115. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/requirements.txt +0 -0
  116. {supermodelingfactory-0.7.2 → supermodelingfactory-0.8.0}/setup.cfg +0 -0
@@ -51,6 +51,10 @@ _MONOTONE_INIT_KEYS = frozenset({
51
51
  "direction_conflict_policy",
52
52
  "missing_bin_strategy",
53
53
  "refine_min_n_bins_policy",
54
+ "sv_min_bin_size",
55
+ "sv_small_policy",
56
+ "sv_woe_smoothing",
57
+ "sv_smoothing_alpha",
54
58
  })
55
59
  _MONOTONE_FIT_KEYS = frozenset({"chi2_binning", "chi2_p", "chi2_init_size", "n_jobs"})
56
60
 
@@ -62,11 +62,30 @@ class CreditModelPipelineConfig:
62
62
  woe_engine: str = "equal_freq"
63
63
  woe_fit_query: str | None = None
64
64
  extra_eval_datasets: dict[str, pd.DataFrame] | None = None
65
+ # sv_* govern special-value (SV) WOE bins only; the defaults below reproduce
66
+ # pre-0.8.0 behavior bit-for-bit. They reach WOE_Master via fit(**woe_params)
67
+ # and MonotoneWOEBinner via __init__(**monotone_woe_params).
65
68
  woe_params: dict[str, Any] = field(
66
- default_factory=lambda: {"nbins": 10, "equal_freq": True, "min_bin_prop": 0.05}
69
+ default_factory=lambda: {
70
+ "nbins": 10,
71
+ "equal_freq": True,
72
+ "min_bin_prop": 0.05,
73
+ "sv_min_bin_size": 0.0,
74
+ "sv_small_policy": "keep",
75
+ "sv_woe_smoothing": "none",
76
+ "sv_smoothing_alpha": 0.0,
77
+ }
67
78
  )
68
79
  monotone_woe_params: dict[str, Any] = field(
69
- default_factory=lambda: {"n_init_bins": 20, "min_bin_size": 0.03, "min_n_bins": 2}
80
+ default_factory=lambda: {
81
+ "n_init_bins": 20,
82
+ "min_bin_size": 0.03,
83
+ "min_n_bins": 2,
84
+ "sv_min_bin_size": 0.0,
85
+ "sv_small_policy": "keep",
86
+ "sv_woe_smoothing": "none",
87
+ "sv_smoothing_alpha": 0.0,
88
+ }
70
89
  )
71
90
 
72
91
  train_models: list[str] = field(default_factory=lambda: ["lr", "lgb", "xgb", "cat"])
@@ -66,11 +66,31 @@ class FeatureValidationPipelineConfig:
66
66
  woe_enabled: bool = True
67
67
  woe_fit_query: str | None = None
68
68
  woe_engine: str = "monotone"
69
+ # sv_* govern special-value (SV) WOE bins only; the defaults below reproduce
70
+ # pre-0.8.0 behavior bit-for-bit. Monotone-side sv_* keys only survive because
71
+ # they are listed in _MONOTONE_INIT_KEYS below — unlisted keys are dropped
72
+ # silently, so the two lists must be kept in sync.
69
73
  woe_params: dict[str, Any] = field(
70
- default_factory=lambda: {"nbins": 10, "equal_freq": True, "min_bin_prop": 0.05}
74
+ default_factory=lambda: {
75
+ "nbins": 10,
76
+ "equal_freq": True,
77
+ "min_bin_prop": 0.05,
78
+ "sv_min_bin_size": 0.0,
79
+ "sv_small_policy": "keep",
80
+ "sv_woe_smoothing": "none",
81
+ "sv_smoothing_alpha": 0.0,
82
+ }
71
83
  )
72
84
  monotone_woe_params: dict[str, Any] = field(
73
- default_factory=lambda: {"n_init_bins": 20, "min_bin_size": 0.03, "min_n_bins": 2}
85
+ default_factory=lambda: {
86
+ "n_init_bins": 20,
87
+ "min_bin_size": 0.03,
88
+ "min_n_bins": 2,
89
+ "sv_min_bin_size": 0.0,
90
+ "sv_small_policy": "keep",
91
+ "sv_woe_smoothing": "none",
92
+ "sv_smoothing_alpha": 0.0,
93
+ }
74
94
  )
75
95
  categorical_features: list[str] | None = None
76
96
  monotone_refine_cate_enabled: bool = False
@@ -170,6 +190,10 @@ class FeatureValidationPipeline:
170
190
  "direction_conflict_policy",
171
191
  "missing_bin_strategy",
172
192
  "refine_min_n_bins_policy",
193
+ "sv_min_bin_size",
194
+ "sv_small_policy",
195
+ "sv_woe_smoothing",
196
+ "sv_smoothing_alpha",
173
197
  }
174
198
  _MONOTONE_FIT_KEYS = {"chi2_binning", "chi2_p", "chi2_init_size", "n_jobs"}
175
199
 
@@ -289,7 +289,9 @@ class WOE_Master(object):
289
289
  self.varlist = varlist
290
290
 
291
291
  def fit(self, nbins=10, equal_freq=True, tree_binning_seed=None, chi2_config=None,
292
- precision=5, min_bin_prop=0.05, include_missing=True, fillna=None, spec_values=[]):
292
+ precision=5, min_bin_prop=0.05, include_missing=True, fillna=None, spec_values=[],
293
+ sv_min_bin_size=0.0, sv_small_policy="keep",
294
+ sv_woe_smoothing="none", sv_smoothing_alpha=0.0):
293
295
  """Fit WOE binning for variables in varlist.
294
296
 
295
297
  Args:
@@ -302,6 +304,11 @@ class WOE_Master(object):
302
304
  include_missing: bool, include missing value bin (default True)
303
305
  fillna: int/float, value to fill missing data
304
306
  spec_values: list, special values to handle
307
+ sv_min_bin_size: float, low-frequency special-value bin threshold as a
308
+ share of all samples; 0.0 disables the fallback (legacy behaviour)
309
+ sv_small_policy: str, 'keep' (default) / 'neutral' / 'merge_missing'
310
+ sv_woe_smoothing: str, 'none' (default) / 'laplace'
311
+ sv_smoothing_alpha: float, Laplace smoothing strength alpha (pseudo-counts)
305
312
  Returns:
306
313
  None, updates self.woe_dict attribute
307
314
  """
@@ -323,6 +330,10 @@ class WOE_Master(object):
323
330
  ascending=True,
324
331
  fillna=fillna,
325
332
  spec_values=spec_values,
333
+ sv_min_bin_size=sv_min_bin_size,
334
+ sv_small_policy=sv_small_policy,
335
+ sv_woe_smoothing=sv_woe_smoothing,
336
+ sv_smoothing_alpha=sv_smoothing_alpha,
326
337
  drop_bin_info=True,
327
338
  ret_woe_table=True)
328
339
  woe_table = woe_res[-1]
@@ -381,7 +392,9 @@ class WOE_Master(object):
381
392
  return data_woe
382
393
 
383
394
  def update_woe(self, varlist, nbins=10, equal_freq=True, tree_binning_seed=None, chi2_config=None,
384
- precision=5, min_bin_prop=0.05, include_missing=True, fillna=None, spec_values=[]):
395
+ precision=5, min_bin_prop=0.05, include_missing=True, fillna=None, spec_values=[],
396
+ sv_min_bin_size=0.0, sv_small_policy="keep",
397
+ sv_woe_smoothing="none", sv_smoothing_alpha=0.0):
385
398
  """Update WOE binning for specified variables.
386
399
 
387
400
  Args:
@@ -395,6 +408,11 @@ class WOE_Master(object):
395
408
  include_missing: bool, include missing value bin (default True)
396
409
  fillna: int/float, value to fill missing data
397
410
  spec_values: list, special values to handle
411
+ sv_min_bin_size: float, low-frequency special-value bin threshold as a
412
+ share of all samples; 0.0 disables the fallback (legacy behaviour)
413
+ sv_small_policy: str, 'keep' (default) / 'neutral' / 'merge_missing'
414
+ sv_woe_smoothing: str, 'none' (default) / 'laplace'
415
+ sv_smoothing_alpha: float, Laplace smoothing strength alpha (pseudo-counts)
398
416
  Returns:
399
417
  None, updates self.woe_dict attribute
400
418
  """
@@ -417,6 +435,10 @@ class WOE_Master(object):
417
435
  ascending=True,
418
436
  fillna=fillna,
419
437
  spec_values=spec_values,
438
+ sv_min_bin_size=sv_min_bin_size,
439
+ sv_small_policy=sv_small_policy,
440
+ sv_woe_smoothing=sv_woe_smoothing,
441
+ sv_smoothing_alpha=sv_smoothing_alpha,
420
442
  drop_bin_info=True,
421
443
  ret_woe_table=True)
422
444
  woe_table = woe_res[-1]
@@ -27,11 +27,11 @@ class WOE_Master(object):
27
27
  def __init__(self, train_data, varlist, dep = None, graph_save_dir = '', woe_suffix = '_woe', missing_ref_value = ..., remove_exist_dir = False): ...
28
28
  def remove_folder(file_path): ...
29
29
  def load_mapping_table(self, mapping_table_csv): ...
30
- def fit(self, nbins = 10, equal_freq = True, tree_binning_seed = None, chi2_config = None, precision = 5, min_bin_prop = 0.05, include_missing = True, fillna = None, spec_values = []): ...
30
+ def fit(self, nbins = 10, equal_freq = True, tree_binning_seed = None, chi2_config = None, precision = 5, min_bin_prop = 0.05, include_missing = True, fillna = None, spec_values = [], sv_min_bin_size = 0.0, sv_small_policy = 'keep', sv_woe_smoothing = 'none', sv_smoothing_alpha = 0.0): ...
31
31
  def get_mapping_table(self): ...
32
32
  def save_mapping_table(self, save_dir): ...
33
33
  def transform(self, data = None, varlist = None): ...
34
- def update_woe(self, varlist, nbins = 10, equal_freq = True, tree_binning_seed = None, chi2_config = None, precision = 5, min_bin_prop = 0.05, include_missing = True, fillna = None, spec_values = []): ...
34
+ def update_woe(self, varlist, nbins = 10, equal_freq = True, tree_binning_seed = None, chi2_config = None, precision = 5, min_bin_prop = 0.05, include_missing = True, fillna = None, spec_values = [], sv_min_bin_size = 0.0, sv_small_policy = 'keep', sv_woe_smoothing = 'none', sv_smoothing_alpha = 0.0): ...
35
35
  def plot_bivar_graph(self, data, group = None, dirname = None, varlist = None): ...
36
36
  def load_mapping_table(mapping_table_csv): ...
37
37
  def get_mapping_table(woe_dict): ...
@@ -245,6 +245,18 @@ class MonotoneWOEBinner:
245
245
  显示 N 位小数(:.Nf),例如 N=2 时 1234.5678 → 1234.57。
246
246
  注意:较低的精度会使 load_woe_bins(get_final_bins()) 的
247
247
  round-trip 边界稍有误差,但通常可忽略。
248
+ sv_min_bin_size : 低占比 SV 兜底阈值(SV 箱占**全量**样本的占比),
249
+ 默认 0.0 = 关闭。
250
+ sv_small_policy : 占比 < sv_min_bin_size 的 SV 箱如何处理,
251
+ 'keep'(默认,经验 WOE,零行为变更)/
252
+ 'neutral'(woe=iv=0)/
253
+ 'merge_missing'(bad/good 并入 [Missing] 箱后重算,
254
+ 被合并行的存表 WOE 改写为 [Missing] 的 WOE;
255
+ 无 [Missing] 箱时降级 'neutral' 并告警)。
256
+ sv_woe_smoothing : SV 箱 WOE 是否向全局坏率收缩,'none'(默认)/'laplace'。
257
+ sv_smoothing_alpha : 平滑强度 α(伪计数),默认 0.0(数值等价旧 WOE)。
258
+ 方式1 优先:低占比箱走兜底后**不再**平滑;平滑只作用于
259
+ 占比达标(或 policy='keep')的 SV 箱。
248
260
 
249
261
  fit() 参数(传入 fit() 方法,不在 __init__ 中设置)
250
262
  -------------------------------------------------------
@@ -279,6 +291,10 @@ class MonotoneWOEBinner:
279
291
  direction_conflict_policy: Optional[str] = None,
280
292
  missing_bin_strategy: Optional[str] = None,
281
293
  refine_min_n_bins_policy: Optional[str] = "warn",
294
+ sv_min_bin_size: float = 0.0,
295
+ sv_small_policy: str = "keep",
296
+ sv_woe_smoothing: str = "none",
297
+ sv_smoothing_alpha: float = 0.0,
282
298
  ):
283
299
  self.feature_cols = list(feature_cols)
284
300
  self.target_col = target_col
@@ -344,9 +360,32 @@ class MonotoneWOEBinner:
344
360
  "missing_bin_strategy='fixed_woe' conflicts with NaN in special_values: "
345
361
  "missing rows would get an empirical bin, not the fixed missing_woe constant."
346
362
  )
363
+ # ── SV 箱治理参数(G19;默认 keep/none/0.0 = 旧行为,零行为变更) ──
364
+ if sv_small_policy not in {"keep", "neutral", "merge_missing"}:
365
+ raise ValueError(
366
+ f"sv_small_policy must be one of ['keep', 'neutral', 'merge_missing']; "
367
+ f"got {sv_small_policy!r}"
368
+ )
369
+ if sv_woe_smoothing not in {"none", "laplace"}:
370
+ raise ValueError(
371
+ f"sv_woe_smoothing must be one of ['none', 'laplace']; "
372
+ f"got {sv_woe_smoothing!r}"
373
+ )
374
+ if not (0.0 <= sv_min_bin_size < 1.0):
375
+ raise ValueError(
376
+ f"sv_min_bin_size must be in [0.0, 1.0); got {sv_min_bin_size}"
377
+ )
378
+ if sv_smoothing_alpha < 0.0:
379
+ raise ValueError(
380
+ f"sv_smoothing_alpha must be >= 0.0; got {sv_smoothing_alpha}"
381
+ )
347
382
  self.min_bad_count = min_bad_count
348
383
  self.min_good_count = min_good_count
349
384
  self.small_bin_policy = small_bin_policy
385
+ self.sv_min_bin_size = sv_min_bin_size
386
+ self.sv_small_policy = sv_small_policy
387
+ self.sv_woe_smoothing = sv_woe_smoothing
388
+ self.sv_smoothing_alpha = sv_smoothing_alpha
350
389
  self.monotone_direction = monotone_direction
351
390
  self.reference_target = reference_target
352
391
  self.direction_conflict_policy = direction_conflict_policy
@@ -514,16 +553,31 @@ class MonotoneWOEBinner:
514
553
  return pd.Series(0, index=sub.index)
515
554
 
516
555
  def _compute_woe_single_bin(
517
- self, sub: pd.DataFrame, total_bad: float, total_good: float
556
+ self, sub: pd.DataFrame, total_bad: float, total_good: float,
557
+ smooth: bool = False,
518
558
  ) -> Dict[str, float]:
519
- """计算某子集的 bad/good/woe/iv 等统计量。"""
559
+ """计算某子集的 bad/good/woe/iv 等统计量。
560
+
561
+ ``smooth=True`` 允许 G19 的拉普拉斯平滑生效(仅 SV 箱路径显式开启,
562
+ 普通箱调用保持 ``smooth=False``、口径不变)。
563
+ """
520
564
  eps = self.eps
521
565
  n = len(sub)
522
566
  bad = float(sub[self.target_col].sum())
523
567
  good = float((sub[self.target_col] == 0).sum())
524
568
  bad_rate = bad / (bad + good) if (bad + good) > 0 else 0.0
525
- pct_bad = bad / (total_bad + eps)
526
- pct_good = good / (total_good + eps)
569
+ if smooth and self.sv_woe_smoothing == "laplace" and self.sv_smoothing_alpha > 0.0:
570
+ # 把箱内 bad_rate 向全局基准率 p 收缩,再换算回等效 bad/good 计数。
571
+ # 该式在 alpha→∞ 时 bad_rate→p,WOE→0(严格单调收缩到中性);
572
+ # 直接给 pct_bad/pct_good 加伪计数则会收敛到 logit(p) 而非 0。
573
+ a = self.sv_smoothing_alpha
574
+ p = total_bad / (total_bad + total_good + eps)
575
+ r = (bad + a * p) / (bad + good + a)
576
+ pct_bad = ((bad + good) * r) / (total_bad + eps)
577
+ pct_good = ((bad + good) * (1.0 - r)) / (total_good + eps)
578
+ else:
579
+ pct_bad = bad / (total_bad + eps)
580
+ pct_good = good / (total_good + eps)
527
581
  woe = math.log((pct_bad + eps) / (pct_good + eps))
528
582
  iv = (pct_bad - pct_good) * woe
529
583
  return dict(n=n, bad=int(bad), good=int(good),
@@ -577,16 +631,120 @@ class MonotoneWOEBinner:
577
631
  """
578
632
  计算所有特殊值的独立 WOE 明细,返回 DataFrame。
579
633
  每行对应一个特殊值,bin_label 为 '[sv=xxx]' 或 '[Missing]'。
634
+
635
+ G19:当 sv_small_policy / sv_woe_smoothing 启用时,按固定顺序决策
636
+ (方式1 兜底优先,方式2 平滑仅作用于占比达标的保留箱),并额外产出
637
+ ``sv_policy_applied`` 审计列。
580
638
  """
639
+ governance_on = (
640
+ self.sv_small_policy != "keep" or self.sv_woe_smoothing != "none"
641
+ )
642
+ n_total = total_bad + total_good
581
643
  records = []
644
+ missing_row_idx = None
582
645
  for sv, sv_df in sv_groups.items():
583
646
  if len(sv_df) == 0:
584
647
  continue
585
- stats = self._compute_woe_single_bin(sv_df, total_bad, total_good)
648
+ if not governance_on:
649
+ stats = self._compute_woe_single_bin(sv_df, total_bad, total_good)
650
+ else:
651
+ label = _sv_label(sv)
652
+ is_small = (
653
+ self.sv_small_policy != "keep"
654
+ and self.sv_min_bin_size > 0.0
655
+ and n_total > 0
656
+ and len(sv_df) / n_total < self.sv_min_bin_size
657
+ # [Missing] is the merge *target*, never a merge source.
658
+ and not (self.sv_small_policy == "merge_missing"
659
+ and label == "[Missing]")
660
+ )
661
+ if is_small and self.sv_small_policy == "neutral":
662
+ stats = self._compute_woe_single_bin(sv_df, total_bad, total_good)
663
+ stats["woe"] = 0.0
664
+ stats["iv"] = 0.0
665
+ stats["sv_policy_applied"] = "neutral"
666
+ elif is_small:
667
+ # merge_missing:先留经验值,全部 SV 收集完后再合并
668
+ stats = self._compute_woe_single_bin(sv_df, total_bad, total_good)
669
+ stats["sv_policy_applied"] = "pending_merge"
670
+ else:
671
+ stats = self._compute_woe_single_bin(
672
+ sv_df, total_bad, total_good, smooth=True
673
+ )
674
+ stats["sv_policy_applied"] = "keep"
675
+ if label == "[Missing]":
676
+ missing_row_idx = len(records)
586
677
  stats["bin_label"] = _sv_label(sv)
587
678
  stats["sv"] = sv
588
679
  records.append(stats)
589
- return pd.DataFrame(records) if records else pd.DataFrame()
680
+ sv_table = pd.DataFrame(records) if records else pd.DataFrame()
681
+ if (
682
+ governance_on
683
+ and self.sv_small_policy == "merge_missing"
684
+ and len(sv_table) > 0
685
+ ):
686
+ sv_table = self._merge_small_into_missing(
687
+ sv_table, missing_row_idx, total_bad, total_good
688
+ )
689
+ return sv_table
690
+
691
+ def _merge_small_into_missing(
692
+ self, sv_table: pd.DataFrame, missing_row_idx: Optional[int],
693
+ total_bad: float, total_good: float,
694
+ ) -> pd.DataFrame:
695
+ """把 pending_merge 的低占比 SV 行并入 [Missing] 行并重算 WOE。
696
+
697
+ 被合并行保留自己的行,但其存表 WOE 被改写为 [Missing] 重算后的 WOE,
698
+ iv 置 0(避免与合并目标重复计入总 IV)。这样 apply_woe 的
699
+ ``bin_label -> woe`` 映射天然指向缺失箱口径,transform 侧零改动。
700
+ 无 [Missing] 箱时降级为 neutral 并告警。
701
+
702
+ n/bad/good 是**转移**而非复制:合并目标加上、来源行清零。否则
703
+ ``sv_table["n"].sum()`` 会重复计数,进而污染 pct_n/lift 与
704
+ transform 侧的 fit_missing_rate 漂移基线。
705
+ """
706
+ pend = sv_table["sv_policy_applied"] == "pending_merge"
707
+ if not pend.any():
708
+ return sv_table
709
+ if missing_row_idx is None:
710
+ for i in sv_table.index[pend]:
711
+ warnings.warn(
712
+ f"sv_small_policy='merge_missing' but feature has no [Missing] bin; "
713
+ f"falling back to 'neutral' for SV {sv_table.loc[i, 'sv']!r}.",
714
+ UserWarning, stacklevel=2,
715
+ )
716
+ sv_table.loc[i, "woe"] = 0.0
717
+ sv_table.loc[i, "iv"] = 0.0
718
+ sv_table.loc[i, "sv_policy_applied"] = "neutral(fallback)"
719
+ return sv_table
720
+ m = missing_row_idx
721
+ new_n = int(sv_table.loc[m, "n"]) + int(sv_table.loc[pend, "n"].sum())
722
+ new_bad = float(sv_table.loc[m, "bad"]) + float(sv_table.loc[pend, "bad"].sum())
723
+ new_good = float(sv_table.loc[m, "good"]) + float(sv_table.loc[pend, "good"].sum())
724
+ pct_bad = new_bad / (total_bad + self.eps)
725
+ pct_good = new_good / (total_good + self.eps)
726
+ woe = math.log((pct_bad + self.eps) / (pct_good + self.eps))
727
+ sv_table.loc[m, "n"] = new_n
728
+ sv_table.loc[m, "bad"] = int(new_bad)
729
+ sv_table.loc[m, "good"] = int(new_good)
730
+ sv_table.loc[m, "bad_rate"] = (
731
+ new_bad / (new_bad + new_good) if (new_bad + new_good) > 0 else 0.0
732
+ )
733
+ sv_table.loc[m, "pct_bad"] = pct_bad
734
+ sv_table.loc[m, "pct_good"] = pct_good
735
+ sv_table.loc[m, "woe"] = woe
736
+ sv_table.loc[m, "iv"] = (pct_bad - pct_good) * woe
737
+ sv_table.loc[m, "sv_policy_applied"] = "merge_target"
738
+ sv_table.loc[pend, "n"] = 0
739
+ sv_table.loc[pend, "bad"] = 0
740
+ sv_table.loc[pend, "good"] = 0
741
+ sv_table.loc[pend, "bad_rate"] = 0.0
742
+ sv_table.loc[pend, "pct_bad"] = 0.0
743
+ sv_table.loc[pend, "pct_good"] = 0.0
744
+ sv_table.loc[pend, "woe"] = woe
745
+ sv_table.loc[pend, "iv"] = 0.0
746
+ sv_table.loc[pend, "sv_policy_applied"] = "merged_into_missing"
747
+ return sv_table
590
748
 
591
749
  @staticmethod
592
750
  def _is_monotone(woe_values: np.ndarray) -> bool:
@@ -33,14 +33,15 @@ _SPECIAL_BIN_PREFIX: object
33
33
  _CATE_GROUP_SEP: object
34
34
 
35
35
  class MonotoneWOEBinner:
36
- def __init__(self, feature_cols: List[str], target_col: str, n_init_bins: int = 20, min_bin_size: float = 0.03, min_n_bins: int = 2, eps: float = 1e-06, missing_woe: float = 0.0, special_values: Optional[List] = None, cate_feats: Optional[List[str]] = None, bin_label_decimals: Optional[int] = None, min_bad_count: Optional[int] = None, min_good_count: Optional[int] = None, small_bin_policy: Optional[str] = None, monotone_direction: Any = 'auto', reference_target: Optional[str] = None, direction_conflict_policy: Optional[str] = None, missing_bin_strategy: Optional[str] = None, refine_min_n_bins_policy: Optional[str] = 'warn'): ...
36
+ def __init__(self, feature_cols: List[str], target_col: str, n_init_bins: int = 20, min_bin_size: float = 0.03, min_n_bins: int = 2, eps: float = 1e-06, missing_woe: float = 0.0, special_values: Optional[List] = None, cate_feats: Optional[List[str]] = None, bin_label_decimals: Optional[int] = None, min_bad_count: Optional[int] = None, min_good_count: Optional[int] = None, small_bin_policy: Optional[str] = None, monotone_direction: Any = 'auto', reference_target: Optional[str] = None, direction_conflict_policy: Optional[str] = None, missing_bin_strategy: Optional[str] = None, refine_min_n_bins_policy: Optional[str] = 'warn', sv_min_bin_size: float = 0.0, sv_small_policy: str = 'keep', sv_woe_smoothing: str = 'none', sv_smoothing_alpha: float = 0.0): ...
37
37
  def _split_special(self, df: pd.DataFrame, feat: str): ...
38
38
  def _cat_to_bin_map(vr: Dict) -> Dict: ...
39
39
  def _split_special_for_plot(self, df: pd.DataFrame, feat: str, vr: Dict): ...
40
40
  def _assign_normal_bins(self, sub: pd.DataFrame, feat: str, vr: Dict, fitted_edges: list) -> pd.Series: ...
41
- def _compute_woe_single_bin(self, sub: pd.DataFrame, total_bad: float, total_good: float) -> Dict[str, float]: ...
41
+ def _compute_woe_single_bin(self, sub: pd.DataFrame, total_bad: float, total_good: float, smooth: bool = False) -> Dict[str, float]: ...
42
42
  def _compute_woe_table(self, df: pd.DataFrame, feat: str, edges: list) -> tuple: ...
43
43
  def _compute_sv_table(self, sv_groups: Dict, total_bad: float, total_good: float) -> pd.DataFrame: ...
44
+ def _merge_small_into_missing(self, sv_table: pd.DataFrame, missing_row_idx: Optional[int], total_bad: float, total_good: float) -> pd.DataFrame: ...
44
45
  def _is_monotone(woe_values: np.ndarray) -> bool: ...
45
46
  def _is_monotone_dir(woe_values: np.ndarray, direction: int) -> bool: ...
46
47
  def _direction_of(woe_values: np.ndarray) -> str: ...
@@ -6,6 +6,7 @@ WOE转换与单调性分析工具包
6
6
  import numpy as np
7
7
  import pandas as pd
8
8
  import logging
9
+ import warnings
9
10
 
10
11
  from Modeling_Tool.Core.Binning_Tool import (
11
12
  _parse_bin_range_bounds,
@@ -203,7 +204,9 @@ class WOETransformer:
203
204
 
204
205
  def __init__(self, nbins=10, precision=5, min_bin_prop=0.05, include_missing=False,
205
206
  equal_freq=True, fillna=-999999, chi2_config=None, tree_binning_seed=None,
206
- spec_values=None, drop_bin_info=True, ret_woe_table=True):
207
+ spec_values=None, drop_bin_info=True, ret_woe_table=True,
208
+ sv_min_bin_size=0.0, sv_small_policy="keep",
209
+ sv_woe_smoothing="none", sv_smoothing_alpha=0.0):
207
210
  """初始化WOE转换器。
208
211
 
209
212
  参数:
@@ -230,7 +233,33 @@ class WOETransformer:
230
233
  是否删除中间分箱信息列
231
234
  ret_woe_table : bool, optional
232
235
  是否返回WOE映射表
236
+ sv_min_bin_size : float, optional
237
+ 低占比特殊值箱兜底阈值(占全量样本比例),0.0 = 关闭(保旧行为)
238
+ sv_small_policy : str, optional
239
+ 'keep'(默认)/'neutral'/'merge_missing',低占比 SV 箱的处理方式
240
+ sv_woe_smoothing : str, optional
241
+ 'none'(默认)/'laplace',SV 箱 WOE 是否向全局坏率收缩
242
+ sv_smoothing_alpha : float, optional
243
+ 拉普拉斯平滑强度 α(伪计数),0.0 = 数值等价旧 WOE
233
244
  """
245
+ if sv_small_policy not in {"keep", "neutral", "merge_missing"}:
246
+ raise ValueError(
247
+ f"sv_small_policy must be one of ['keep', 'neutral', 'merge_missing']; "
248
+ f"got {sv_small_policy!r}"
249
+ )
250
+ if sv_woe_smoothing not in {"none", "laplace"}:
251
+ raise ValueError(
252
+ f"sv_woe_smoothing must be one of ['none', 'laplace']; "
253
+ f"got {sv_woe_smoothing!r}"
254
+ )
255
+ if not (0.0 <= sv_min_bin_size < 1.0):
256
+ raise ValueError(
257
+ f"sv_min_bin_size must be in [0.0, 1.0); got {sv_min_bin_size}"
258
+ )
259
+ if sv_smoothing_alpha < 0.0:
260
+ raise ValueError(
261
+ f"sv_smoothing_alpha must be >= 0.0; got {sv_smoothing_alpha}"
262
+ )
234
263
  self.nbins = nbins
235
264
  self.precision = precision
236
265
  self.min_bin_prop = min_bin_prop
@@ -242,6 +271,162 @@ class WOETransformer:
242
271
  self.spec_values = spec_values if spec_values is not None else []
243
272
  self.drop_bin_info = drop_bin_info
244
273
  self.ret_woe_table = ret_woe_table
274
+ self.sv_min_bin_size = sv_min_bin_size
275
+ self.sv_small_policy = sv_small_policy
276
+ self.sv_woe_smoothing = sv_woe_smoothing
277
+ self.sv_smoothing_alpha = sv_smoothing_alpha
278
+
279
+ # G19: SV-bin governance shares MonotoneWOEBinner's eps so both engines
280
+ # produce identical numbers for the same counts.
281
+ _SV_EPS = 1e-6
282
+
283
+ def _sv_row_mask(self, woe_table):
284
+ """识别特殊值箱行:MIN == MAX 且该值属于 spec_values(双条件防误识别)。"""
285
+ spec_values = list(self.spec_values or [])
286
+ if not spec_values:
287
+ return pd.Series(False, index=woe_table.index)
288
+ return (
289
+ woe_table["MIN"].isin(spec_values)
290
+ & (woe_table["MIN"] == woe_table["MAX"])
291
+ & (woe_table["N"] > 0)
292
+ )
293
+
294
+ def _missing_row_label(self, woe_table, sv_mask):
295
+ """识别缺失值箱行标签;无缺失箱返回 None。
296
+
297
+ fit 路径下 MIN/MAX 聚合的是**原始**变量列,缺失箱因此整箱为 NaN;
298
+ 若调用方在分箱前已 fillna(哨兵进入数据),则 MIN == MAX == fillna。
299
+ """
300
+ nan_bin = woe_table["MIN"].isna() & woe_table["MAX"].isna() & (woe_table["N"] > 0)
301
+ sentinel_bin = (
302
+ (woe_table["MIN"] == woe_table["MAX"])
303
+ & (woe_table["MIN"] == self.fillna)
304
+ & (woe_table["N"] > 0)
305
+ & ~sv_mask
306
+ )
307
+ candidates = woe_table.index[nan_bin | sentinel_bin]
308
+ return candidates[0] if len(candidates) else None
309
+
310
+ def _govern_sv_bins(self, woe_table, var):
311
+ """对 spec_values 对应的箱行应用 sv_small_policy / sv_woe_smoothing。
312
+
313
+ 口径与 ``MonotoneWOEBinner._compute_sv_table`` 严格一致:方式1 兜底优先,
314
+ 方式2 平滑只作用于占比达标(或 policy='keep')的 SV 箱。
315
+
316
+ 缺失箱同样是一个受治理的 SV 箱(MonotoneWOEBinner 的 sv_table 里
317
+ ``[Missing]`` 与其它 SV 行同权),因此并入治理行集合;它只是**永远不作为
318
+ merge 来源**(target-only,不能并入自己)。若仅按 ``_sv_row_mask``
319
+ 取行,缺失箱(MIN/MAX 聚合原始列 ⇒ NaN/NaN)会漏出平滑循环,导致同参数
320
+ 下两引擎的 [Missing] WOE 不一致。
321
+ """
322
+ eps = self._SV_EPS
323
+ total_bad = float(woe_table["N_BAD"].sum())
324
+ total_good = float(woe_table["N_GOOD"].sum())
325
+ n_total = total_bad + total_good
326
+ p = total_bad / (n_total + eps)
327
+
328
+ sv_mask = self._sv_row_mask(woe_table)
329
+ missing_label = self._missing_row_label(woe_table, sv_mask)
330
+ governed_idx = list(woe_table.index[sv_mask])
331
+ if missing_label is not None and missing_label not in governed_idx:
332
+ governed_idx.append(missing_label)
333
+ if not governed_idx:
334
+ return woe_table
335
+ pending_merge = []
336
+ for idx in governed_idx:
337
+ is_small = (
338
+ self.sv_small_policy != "keep"
339
+ and self.sv_min_bin_size > 0.0
340
+ and n_total > 0
341
+ and float(woe_table.loc[idx, "N"]) / n_total < self.sv_min_bin_size
342
+ # 缺失箱是 merge 的目标,永远不做来源。
343
+ and not (self.sv_small_policy == "merge_missing"
344
+ and idx == missing_label)
345
+ )
346
+ if is_small and self.sv_small_policy == "neutral":
347
+ woe_table.loc[idx, "WOE"] = 0.0
348
+ woe_table.loc[idx, "IV"] = 0.0
349
+ elif is_small:
350
+ pending_merge.append(idx)
351
+ elif self.sv_woe_smoothing == "laplace" and self.sv_smoothing_alpha > 0.0:
352
+ # 与 MonotoneWOEBinner._compute_woe_single_bin 同式:先把箱内
353
+ # bad_rate 向基准率 p 收缩,再换算回等效 bad/good 计数。
354
+ a = self.sv_smoothing_alpha
355
+ n_bad = float(woe_table.loc[idx, "N_BAD"])
356
+ n_good = float(woe_table.loc[idx, "N_GOOD"])
357
+ r = (n_bad + a * p) / (n_bad + n_good + a)
358
+ pct_bad = ((n_bad + n_good) * r) / (total_bad + eps)
359
+ pct_good = ((n_bad + n_good) * (1.0 - r)) / (total_good + eps)
360
+ woe = float(np.log((pct_bad + eps) / (pct_good + eps)))
361
+ woe_table.loc[idx, "BAD_PCT_PER_BIN"] = pct_bad
362
+ woe_table.loc[idx, "GOOD_PCT_PER_BIN"] = pct_good
363
+ woe_table.loc[idx, "WOE"] = woe
364
+ woe_table.loc[idx, "IV"] = (pct_bad - pct_good) * woe
365
+ if pending_merge:
366
+ woe_table = self._merge_sv_into_missing_master(
367
+ woe_table, pending_merge, missing_label,
368
+ total_bad, total_good, var,
369
+ )
370
+ return woe_table
371
+
372
+ def _merge_sv_into_missing_master(self, woe_table, pending_merge, missing_label,
373
+ total_bad, total_good, var):
374
+ """把低占比 SV 行的 bad/good 并入缺失箱行并重算 WOE。
375
+
376
+ 与 ``MonotoneWOEBinner._merge_small_into_missing`` 一一对应:被合并行保留
377
+ 自己的行,但存表 WOE 改写为缺失箱重算后的 WOE、IV 置 0,因此 transform
378
+ 侧(``mapping_woe`` / ``convert_single_var_woe``)无需任何改动。
379
+ 无缺失箱时降级为 neutral 并告警。
380
+
381
+ N/N_BAD/N_GOOD 是**转移**而非复制(来源行清零),保证 N 列合计不变。
382
+ """
383
+ eps = self._SV_EPS
384
+ if missing_label is None:
385
+ for idx in pending_merge:
386
+ warnings.warn(
387
+ f"sv_small_policy='merge_missing' but {var!r} has no [Missing] bin; "
388
+ f"falling back to 'neutral' for SV "
389
+ f"{woe_table.loc[idx, 'MIN']!r}.",
390
+ UserWarning, stacklevel=2,
391
+ )
392
+ woe_table.loc[idx, "WOE"] = 0.0
393
+ woe_table.loc[idx, "IV"] = 0.0
394
+ return woe_table
395
+ m = missing_label
396
+ add_bad = float(woe_table.loc[pending_merge, "N_BAD"].sum())
397
+ add_good = float(woe_table.loc[pending_merge, "N_GOOD"].sum())
398
+ new_n = int(woe_table.loc[m, "N"]) + int(woe_table.loc[pending_merge, "N"].sum())
399
+ new_bad = float(woe_table.loc[m, "N_BAD"]) + add_bad
400
+ new_good = float(woe_table.loc[m, "N_GOOD"]) + add_good
401
+ pct_bad = new_bad / (total_bad + eps)
402
+ pct_good = new_good / (total_good + eps)
403
+ woe = float(np.log((pct_bad + eps) / (pct_good + eps)))
404
+ woe_table.loc[m, "N"] = new_n
405
+ woe_table.loc[m, "N_BAD"] = int(new_bad)
406
+ woe_table.loc[m, "N_GOOD"] = int(new_good)
407
+ woe_table.loc[m, "AVG_BAD"] = (
408
+ new_bad / (new_bad + new_good) if (new_bad + new_good) > 0 else np.nan
409
+ )
410
+ woe_table.loc[m, "AVG_GOOD"] = (
411
+ new_good / (new_bad + new_good) if (new_bad + new_good) > 0 else np.nan
412
+ )
413
+ woe_table.loc[m, "BAD_PCT_PER_BIN"] = pct_bad
414
+ woe_table.loc[m, "GOOD_PCT_PER_BIN"] = pct_good
415
+ woe_table.loc[m, "WOE"] = woe
416
+ woe_table.loc[m, "IV"] = (pct_bad - pct_good) * woe
417
+ woe_table.loc[pending_merge, "N"] = 0
418
+ woe_table.loc[pending_merge, "N_BAD"] = 0
419
+ woe_table.loc[pending_merge, "N_GOOD"] = 0
420
+ woe_table.loc[pending_merge, "AVG_BAD"] = np.nan
421
+ woe_table.loc[pending_merge, "AVG_GOOD"] = np.nan
422
+ woe_table.loc[pending_merge, "LIFT"] = np.nan
423
+ woe_table.loc[pending_merge, "BAD_PCT_PER_BIN"] = 0.0
424
+ woe_table.loc[pending_merge, "GOOD_PCT_PER_BIN"] = 0.0
425
+ woe_table.loc[pending_merge, "WOE"] = woe
426
+ woe_table.loc[pending_merge, "IV"] = 0.0
427
+ # AVG_BAD 已被合并改写,LIFT 依赖其均值,需整表重算。
428
+ woe_table["LIFT"] = woe_table["AVG_BAD"] / woe_table["AVG_BAD"].mean()
429
+ return woe_table
245
430
 
246
431
  def _get_woe_table(self, binning_res, var, dep):
247
432
  """根据分箱结果计算WOE表。
@@ -288,8 +473,13 @@ class WOETransformer:
288
473
  woe_table, "BAD_PCT_PER_BIN", "GOOD_PCT_PER_BIN"
289
474
  )
290
475
 
291
- # WOE Mapping Dictionary
292
476
  woe_table = woe_table.reset_index(drop=False)
477
+
478
+ # ── G19:低占比 SV 箱治理(与 MonotoneWOEBinner 口径对齐)──
479
+ if self.sv_small_policy != "keep" or self.sv_woe_smoothing != "none":
480
+ woe_table = self._govern_sv_bins(woe_table, var)
481
+
482
+ # WOE Mapping Dictionary
293
483
  woe_mapping_dict = dict(zip(woe_table[f"_bin_range_{var}"], woe_table["WOE"]))
294
484
 
295
485
  return woe_table, woe_mapping_dict
@@ -912,7 +1102,9 @@ def plot_monotonicity_check(data, column, title=None, include_missing=True):
912
1102
  def woe_transform(train_df, var, dep, nbins, oot_df=None, chi2_config=None, tree_binning_seed=None,
913
1103
  precision=5, min_bin_prop=0.05, include_missing=False, equal_freq=True,
914
1104
  ascending=True, fillna=-999999, spec_values=None, drop_bin_info=True,
915
- ret_woe_table=True, check_monotonicity=False):
1105
+ ret_woe_table=True, check_monotonicity=False,
1106
+ sv_min_bin_size=0.0, sv_small_policy="keep",
1107
+ sv_woe_smoothing="none", sv_smoothing_alpha=0.0):
916
1108
  """将变量转换为WOE值。
917
1109
 
918
1110
  对单个变量进行分箱并计算WOE值,支持训练集和验证集的转换。
@@ -954,6 +1146,14 @@ def woe_transform(train_df, var, dep, nbins, oot_df=None, chi2_config=None, tree
954
1146
  是否返回WOE映射表,默认为True
955
1147
  check_monotonicity : bool, optional
956
1148
  是否检查单调性,默认为False
1149
+ sv_min_bin_size : float, optional
1150
+ 低占比特殊值箱兜底阈值(占全量样本比例),默认0.0(关闭)
1151
+ sv_small_policy : str, optional
1152
+ 'keep'(默认)/'neutral'/'merge_missing'
1153
+ sv_woe_smoothing : str, optional
1154
+ 'none'(默认)/'laplace'
1155
+ sv_smoothing_alpha : float, optional
1156
+ 拉普拉斯平滑强度α,默认0.0
957
1157
 
958
1158
  返回:
959
1159
  --------
@@ -982,7 +1182,11 @@ def woe_transform(train_df, var, dep, nbins, oot_df=None, chi2_config=None, tree
982
1182
  tree_binning_seed=tree_binning_seed,
983
1183
  spec_values=spec_values,
984
1184
  drop_bin_info=drop_bin_info,
985
- ret_woe_table=ret_woe_table
1185
+ ret_woe_table=ret_woe_table,
1186
+ sv_min_bin_size=sv_min_bin_size,
1187
+ sv_small_policy=sv_small_policy,
1188
+ sv_woe_smoothing=sv_woe_smoothing,
1189
+ sv_smoothing_alpha=sv_smoothing_alpha,
986
1190
  )
987
1191
  return transformer.transform_single(
988
1192
  train_df=train_df,
@@ -23,7 +23,12 @@ def is_monotonic(data, column, direction = 'auto', strict = False, handle_nan =
23
23
  def check_monotonicity(data, var): ...
24
24
 
25
25
  class WOETransformer:
26
- def __init__(self, nbins = 10, precision = 5, min_bin_prop = 0.05, include_missing = False, equal_freq = True, fillna = -999999, chi2_config = None, tree_binning_seed = None, spec_values = None, drop_bin_info = True, ret_woe_table = True): ...
26
+ _SV_EPS: float
27
+ def __init__(self, nbins = 10, precision = 5, min_bin_prop = 0.05, include_missing = False, equal_freq = True, fillna = -999999, chi2_config = None, tree_binning_seed = None, spec_values = None, drop_bin_info = True, ret_woe_table = True, sv_min_bin_size = 0.0, sv_small_policy = 'keep', sv_woe_smoothing = 'none', sv_smoothing_alpha = 0.0): ...
28
+ def _sv_row_mask(self, woe_table): ...
29
+ def _missing_row_label(self, woe_table, sv_mask): ...
30
+ def _govern_sv_bins(self, woe_table, var): ...
31
+ def _merge_sv_into_missing_master(self, woe_table, pending_merge, missing_label, total_bad, total_good, var): ...
27
32
  def _get_woe_table(self, binning_res, var, dep): ...
28
33
  def transform_single(self, train_df, var, dep, oot_df = None, check_monotonicity_flag = False): ...
29
34
  def transform(self, train_df, varlist, dep, oot_df = None, check_monotonicity_flag = False): ...
@@ -36,6 +41,6 @@ class WOEMappingTransformer:
36
41
  def woe_transform_cdaml(data, varlist, woe_mapping_path, missing_ref = None, ret_bin_no = False, ret_category = False, rename_orig_var = False, suffix = ''): ...
37
42
  def get_woe_table(data, var, dep, grp_name = None, nbins = 10, precision = 5, min_bin_prop = 0.05, include_missing = True, equal_freq = True, fillna = -999999, chi2_config = None, tree_binning_seed = None, spec_values = None): ...
38
43
  def plot_monotonicity_check(data, column, title = None, include_missing = True): ...
39
- def woe_transform(train_df, var, dep, nbins, oot_df = None, chi2_config = None, tree_binning_seed = None, precision = 5, min_bin_prop = 0.05, include_missing = False, equal_freq = True, ascending = True, fillna = -999999, spec_values = None, drop_bin_info = True, ret_woe_table = True, check_monotonicity = False): ...
44
+ def woe_transform(train_df, var, dep, nbins, oot_df = None, chi2_config = None, tree_binning_seed = None, precision = 5, min_bin_prop = 0.05, include_missing = False, equal_freq = True, ascending = True, fillna = -999999, spec_values = None, drop_bin_info = True, ret_woe_table = True, check_monotonicity = False, sv_min_bin_size = 0.0, sv_small_policy = 'keep', sv_woe_smoothing = 'none', sv_smoothing_alpha = 0.0): ...
40
45
  def woe_transformation(train_df, varlist, dep, oot_df = None, nbins = 10, chi2_config = None, tree_binning_seed = None, precision = 5, min_bin_prop = 0.05, include_missing = False, equal_freq = True, fillna = -999999, spec_values = None, drop_bin_info = True, ret_woe_table = True): ...
41
46
  def mapping_woe(data, varlist, woe_mapping_table, suffix = '_woe', drop_bin_info = True, missing_ref_value = -999999): ...
@@ -1,7 +1,7 @@
1
1
  # encoding: utf-8
2
2
 
3
3
  __author__ = "Jingkai Sun"
4
- __version__ = "0.7.2"
4
+ __version__ = "0.8.0"
5
5
 
6
6
  from ._utils import SMF_MISSING_BIN
7
7
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: SuperModelingFactory
3
- Version: 0.7.2
3
+ Version: 0.8.0
4
4
  Summary: Credit risk modeling factory: WOE binning, scorecards, LightGBM, Excel reporting.
5
5
  Home-page: https://github.com/Kyle-J-Sun/SuperModelingFactory
6
6
  Author: Kyle Sun
@@ -265,7 +265,7 @@ em.close_workbook()
265
265
 
266
266
  ## 版本
267
267
 
268
- - **Version**: 0.7.2
268
+ - **Version**: 0.8.0
269
269
  - **Author**: Jingkai Sun
270
270
 
271
271
  ## 许可证
@@ -208,7 +208,7 @@ em.close_workbook()
208
208
 
209
209
  ## 版本
210
210
 
211
- - **Version**: 0.7.2
211
+ - **Version**: 0.8.0
212
212
  - **Author**: Jingkai Sun
213
213
 
214
214
  ## 许可证
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: SuperModelingFactory
3
- Version: 0.7.2
3
+ Version: 0.8.0
4
4
  Summary: Credit risk modeling factory: WOE binning, scorecards, LightGBM, Excel reporting.
5
5
  Home-page: https://github.com/Kyle-J-Sun/SuperModelingFactory
6
6
  Author: Kyle Sun
@@ -265,7 +265,7 @@ em.close_workbook()
265
265
 
266
266
  ## 版本
267
267
 
268
- - **Version**: 0.7.2
268
+ - **Version**: 0.8.0
269
269
  - **Author**: Jingkai Sun
270
270
 
271
271
  ## 许可证
@@ -12,7 +12,7 @@ build-backend = "setuptools.build_meta"
12
12
 
13
13
  [project]
14
14
  name = "SuperModelingFactory"
15
- version = "0.7.2"
15
+ version = "0.8.0"
16
16
  description = "Credit risk modeling factory: WOE binning, scorecards, LightGBM, Excel reporting."
17
17
  readme = "README.md"
18
18
  requires-python = ">=3.10"
@@ -15,7 +15,7 @@ def _read(path: str) -> str:
15
15
 
16
16
  setup(
17
17
  name="SuperModelingFactory",
18
- version=os.environ.get("SMF_VERSION", "0.7.2"),
18
+ version=os.environ.get("SMF_VERSION", "0.8.0"),
19
19
  description="Credit risk modeling factory: WOE binning, scorecards, LightGBM, Excel reporting.",
20
20
  long_description=_read("README.md"),
21
21
  long_description_content_type="text/markdown",