SuperModelingFactory 0.8.0__tar.gz → 0.8.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Parallel_Engine.py +6 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Monotone_Binner.py +211 -27
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Monotone_Binner.pyi +2 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/__init__.py +1 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/PKG-INFO +2 -2
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/README.md +1 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/SuperModelingFactory.egg-info/PKG-INFO +2 -2
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/pyproject.toml +1 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/setup.py +1 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/ExcelMaster/ExcelFormatTool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/ExcelMaster/ExcelMaster.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/ExcelMaster/Template.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/ExcelMaster/Utility.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/ExcelMaster/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/LICENSE +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/MANIFEST.in +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Binning_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Binning_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Check_DuckDB_Compatibility.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Json_Data_Converter.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Model_Registry_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/ODPS_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Parallel_ODPS_Manager.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Proc_Compare.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Slope_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Slope_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/XOR_Encryptor.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/XOR_Encryptor.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/kDataFrame.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/kDataFrame.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/sample_weight_utils.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/utils.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/Evaluation_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/Evaluation_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/Model_Eval_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/Model_Eval_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/evaluate_model.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/evaluate_model.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/weighted_eval_utils.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Explainability/Coalition_Structure.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Explainability/Model_Explainer.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Explainability/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Distribution_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Distribution_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Feature_Insights.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Feature_Insights.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Feature_Screen.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/ODPS_Distribution_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/ODPS_Distribution_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/PSI_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/PSI_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Screen_Gates.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/WOE_Engine_Feature_Patch.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Weighted_Screen.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/Backward_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/Backward_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/GBM_Search_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/GBM_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/GBM_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/LRM_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/LRM_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/_common.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/credit_model.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/feature_validation.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/field_meta.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/mock_sample.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/orchestrator.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/reject_inference.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/sample_analysis.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/score_comparison.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/score_consistency_uat.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/screening_artifact.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/Distribution_Adaptation.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/Distribution_Adaptation.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/Reject_Infer.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/Reject_Infer.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/Sample_Split.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/Sample_Split.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/UAT/UAT_Consistency_Checker.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/UAT/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Adapter.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Adapter.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Master.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Master.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Plot_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Plot_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Report_Builder.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Report_Builder.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/plot_woe_tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/plot_woe_tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/_utils/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/_utils/nan_guard.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/_utils/robust.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/_utils/sentinels.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/ref_font/KaiTi.ttf +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/ref_font/WeiRuanYaHei.ttf +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/ref_font/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/ref_font/simsun.ttc +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Report/Report_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Report/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/SuperModelingFactory.egg-info/SOURCES.txt +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/SuperModelingFactory.egg-info/dependency_links.txt +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/SuperModelingFactory.egg-info/not-zip-safe +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/SuperModelingFactory.egg-info/requires.txt +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/SuperModelingFactory.egg-info/top_level.txt +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/requirements.txt +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/setup.cfg +0 -0
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Parallel_Engine.py
RENAMED
|
@@ -189,9 +189,15 @@ class ParallelApplyEngine:
|
|
|
189
189
|
func_args: tuple[Any, ...],
|
|
190
190
|
func_kwargs: dict[str, Any],
|
|
191
191
|
) -> None:
|
|
192
|
+
# joblib < 1.6 vendors cloudpickle; joblib >= 1.6 dropped the copy and depends on
|
|
193
|
+
# the cloudpickle package. Resolve the serializer outside the check below so an
|
|
194
|
+
# import problem is never reported as a non-serializable callable.
|
|
192
195
|
try:
|
|
193
196
|
from joblib.externals import cloudpickle
|
|
197
|
+
except ImportError:
|
|
198
|
+
import cloudpickle
|
|
194
199
|
|
|
200
|
+
try:
|
|
195
201
|
cloudpickle.dumps((func, func_args, func_kwargs))
|
|
196
202
|
except Exception as exc:
|
|
197
203
|
raise TypeError(
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Monotone_Binner.py
RENAMED
|
@@ -69,6 +69,8 @@ warnings.filterwarnings("ignore")
|
|
|
69
69
|
|
|
70
70
|
_SPECIAL_BIN_PREFIX = "__special__" # 内部用于标记特殊箱的前缀
|
|
71
71
|
_CATE_GROUP_SEP = " | " # refine_cate 合并多个类别后,bin_label 的成员分隔符
|
|
72
|
+
# 拟合后 sv_table["sv_policy_applied"] 的合法取值(pending_merge 只在拟合中途出现)
|
|
73
|
+
_SV_POLICIES = frozenset({"keep", "neutral", "neutral(fallback)", "merged_into_missing", "merge_target"})
|
|
72
74
|
|
|
73
75
|
def _sv_label(sv) -> str:
|
|
74
76
|
"""将特殊值转为分箱标签,nan → '[Missing]',其余 → '[sv=xxx]'"""
|
|
@@ -552,25 +554,103 @@ class MonotoneWOEBinner:
|
|
|
552
554
|
labels=False, right=True)
|
|
553
555
|
return pd.Series(0, index=sub.index)
|
|
554
556
|
|
|
557
|
+
def _group_iv_for_plot(self, grp_df: pd.DataFrame, feat: str, vr: Dict,
|
|
558
|
+
fitted_edges: list) -> tuple:
|
|
559
|
+
"""分组绘图用的组内 IV,返回 (普通箱 IV, 特殊值箱 IV)。
|
|
560
|
+
|
|
561
|
+
与拟合时 vr["iv"] 同口径,只把样本换成该组:
|
|
562
|
+
- 普通箱:按拟合分箱归箱,分母 = 该组落入普通箱的行的 bad/good
|
|
563
|
+
- 特殊值箱:分母 = 该组全部行的 bad/good,沿用拟合时的
|
|
564
|
+
sv_policy_applied 决策、不在组内重判占比:keep → 经验值(拟合启用
|
|
565
|
+
平滑时按拟合时的平滑参数平滑);neutral / neutral(fallback) → 0;
|
|
566
|
+
merged_into_missing → 行并入该组 [Missing];merge_target → 合并后经验值(不平滑)
|
|
567
|
+
- 单类箱:组内 bad 或 good 为 0 且未经平滑的箱(普通箱、merge_target、未启用
|
|
568
|
+
laplace 的特殊值箱)不计入,与筛选 IV 的 iv_guard 口径一致,避免 eps 把个别
|
|
569
|
+
空类箱放大成虚高 IV;laplace 平滑过的特殊值箱 WOE 有限,照常计入
|
|
570
|
+
以整份拟合样本为一组时,两部分之和等于 vr["iv"]——前提是拟合样本中未经平滑的
|
|
571
|
+
箱两类齐全、没有落不进任何箱的取值(如 -inf),且特殊值决策可得:本次 fit 所得,或经
|
|
572
|
+
get_final_bins → load_woe_bins 的 Format-A attrs 恢复。CSV/Excel 回载(attrs
|
|
573
|
+
丢失)与格式 B 不带决策,特殊值箱一律按 keep 经验值计。
|
|
574
|
+
"""
|
|
575
|
+
target = self.target_col
|
|
576
|
+
normal_df, sv_groups = self._split_special_for_plot(grp_df, feat, vr)
|
|
577
|
+
sub = normal_df[[feat, target]].dropna(subset=[feat]).copy()
|
|
578
|
+
sub["_bin"] = self._assign_normal_bins(sub, feat, vr, fitted_edges)
|
|
579
|
+
sub = sub[sub["_bin"].notna()]
|
|
580
|
+
norm_bad = float(sub[target].sum())
|
|
581
|
+
norm_good = float((sub[target] == 0).sum())
|
|
582
|
+
iv_normal = 0.0
|
|
583
|
+
for _, bin_rows in sub.groupby("_bin"):
|
|
584
|
+
stats = self._compute_woe_single_bin(bin_rows, norm_bad, norm_good)
|
|
585
|
+
if stats["bad"] > 0 and stats["good"] > 0:
|
|
586
|
+
iv_normal += stats["iv"]
|
|
587
|
+
|
|
588
|
+
iv_sv = 0.0
|
|
589
|
+
sv_table = vr.get("sv_table", pd.DataFrame())
|
|
590
|
+
if len(sv_table) > 0:
|
|
591
|
+
full_bad = float(grp_df[target].sum())
|
|
592
|
+
full_good = float((grp_df[target] == 0).sum())
|
|
593
|
+
policy_recorded = "sv_policy_applied" in sv_table.columns
|
|
594
|
+
policies = (list(sv_table["sv_policy_applied"]) if policy_recorded
|
|
595
|
+
else ["keep"] * len(sv_table))
|
|
596
|
+
# load_woe_bins 恢复的拟合平滑参数优先;fit 所得的分箱沿用实例参数
|
|
597
|
+
smoothing = vr.get("sv_smoothing") or {}
|
|
598
|
+
method = smoothing.get("woe_smoothing")
|
|
599
|
+
method = self.sv_woe_smoothing if method is None else method
|
|
600
|
+
alpha = smoothing.get("smoothing_alpha")
|
|
601
|
+
alpha = self.sv_smoothing_alpha if alpha is None else alpha
|
|
602
|
+
labels = list(sv_table["bin_label"])
|
|
603
|
+
# 多个特殊值渲染成同一标签时取第一个非空子集(与柱图匹配口径一致)
|
|
604
|
+
rows_by_label: Dict[str, pd.DataFrame] = {}
|
|
605
|
+
for sv, rows in sv_groups.items():
|
|
606
|
+
lb = _sv_label(sv)
|
|
607
|
+
if lb not in rows_by_label or len(rows_by_label[lb]) == 0:
|
|
608
|
+
rows_by_label[lb] = rows
|
|
609
|
+
merged_rows = [rows_by_label[lb] for lb, policy in zip(labels, policies)
|
|
610
|
+
if policy == "merged_into_missing" and lb in rows_by_label]
|
|
611
|
+
for lb, policy in zip(labels, policies):
|
|
612
|
+
if policy in ("neutral", "neutral(fallback)", "merged_into_missing"):
|
|
613
|
+
continue
|
|
614
|
+
rows = rows_by_label.get(lb)
|
|
615
|
+
if policy == "merge_target":
|
|
616
|
+
parts = [r for r in [rows, *merged_rows] if r is not None and len(r) > 0]
|
|
617
|
+
rows = pd.concat(parts) if parts else None
|
|
618
|
+
if rows is None or len(rows) == 0:
|
|
619
|
+
continue
|
|
620
|
+
smooth = policy_recorded and policy != "merge_target"
|
|
621
|
+
stats = self._compute_woe_single_bin(
|
|
622
|
+
rows, full_bad, full_good, smooth=smooth,
|
|
623
|
+
woe_smoothing=method, smoothing_alpha=alpha,
|
|
624
|
+
)
|
|
625
|
+
# 平滑后的单类箱 WOE 有限、不会被 eps 放大,照常计入
|
|
626
|
+
smoothed = smooth and method == "laplace" and alpha > 0.0
|
|
627
|
+
if smoothed or (stats["bad"] > 0 and stats["good"] > 0):
|
|
628
|
+
iv_sv += stats["iv"]
|
|
629
|
+
return iv_normal, iv_sv
|
|
630
|
+
|
|
555
631
|
def _compute_woe_single_bin(
|
|
556
632
|
self, sub: pd.DataFrame, total_bad: float, total_good: float,
|
|
557
|
-
smooth: bool = False,
|
|
633
|
+
smooth: bool = False, *, woe_smoothing: Optional[str] = None,
|
|
634
|
+
smoothing_alpha: Optional[float] = None,
|
|
558
635
|
) -> Dict[str, float]:
|
|
559
636
|
"""计算某子集的 bad/good/woe/iv 等统计量。
|
|
560
637
|
|
|
561
638
|
``smooth=True`` 允许 G19 的拉普拉斯平滑生效(仅 SV 箱路径显式开启,
|
|
562
|
-
普通箱调用保持 ``smooth=False
|
|
639
|
+
普通箱调用保持 ``smooth=False``、口径不变)。``woe_smoothing`` /
|
|
640
|
+
``smoothing_alpha`` 为 None 时取实例参数;分组 IV 借此传入加载时恢复的拟合参数。
|
|
563
641
|
"""
|
|
564
642
|
eps = self.eps
|
|
565
643
|
n = len(sub)
|
|
566
644
|
bad = float(sub[self.target_col].sum())
|
|
567
645
|
good = float((sub[self.target_col] == 0).sum())
|
|
568
646
|
bad_rate = bad / (bad + good) if (bad + good) > 0 else 0.0
|
|
569
|
-
|
|
647
|
+
method = self.sv_woe_smoothing if woe_smoothing is None else woe_smoothing
|
|
648
|
+
alpha = self.sv_smoothing_alpha if smoothing_alpha is None else smoothing_alpha
|
|
649
|
+
if smooth and method == "laplace" and alpha > 0.0:
|
|
570
650
|
# 把箱内 bad_rate 向全局基准率 p 收缩,再换算回等效 bad/good 计数。
|
|
571
651
|
# 该式在 alpha→∞ 时 bad_rate→p,WOE→0(严格单调收缩到中性);
|
|
572
652
|
# 直接给 pct_bad/pct_good 加伪计数则会收敛到 logit(p) 而非 0。
|
|
573
|
-
a =
|
|
653
|
+
a = alpha
|
|
574
654
|
p = total_bad / (total_bad + total_good + eps)
|
|
575
655
|
r = (bad + a * p) / (bad + good + a)
|
|
576
656
|
pct_bad = ((bad + good) * r) / (total_bad + eps)
|
|
@@ -2313,6 +2393,72 @@ class MonotoneWOEBinner:
|
|
|
2313
2393
|
payload = tuple((key, metadata.get(key)) for key in keys)
|
|
2314
2394
|
return hashlib.sha256(pickle.dumps(payload, protocol=4)).hexdigest()
|
|
2315
2395
|
|
|
2396
|
+
@staticmethod
|
|
2397
|
+
def _format_a_sv_decisions_digest(metadata_digest: Any, sv_decisions: Any) -> str:
|
|
2398
|
+
"""Checksum persisted SV decisions, bound to the base digest (and so to the rows)."""
|
|
2399
|
+
import hashlib
|
|
2400
|
+
import pickle
|
|
2401
|
+
|
|
2402
|
+
payload = (metadata_digest, sv_decisions)
|
|
2403
|
+
return hashlib.sha256(pickle.dumps(payload, protocol=4)).hexdigest()
|
|
2404
|
+
|
|
2405
|
+
def _format_a_sv_decisions(self, vr: Dict) -> Optional[Dict[str, Any]]:
|
|
2406
|
+
"""拟合时的 SV 治理决策(逐行 sv_policy_applied + 平滑参数);未启用 SV 治理时为 None。"""
|
|
2407
|
+
sv_table = vr.get("sv_table", pd.DataFrame())
|
|
2408
|
+
if len(sv_table) == 0 or "sv_policy_applied" not in sv_table.columns:
|
|
2409
|
+
return None
|
|
2410
|
+
smoothing = vr.get("sv_smoothing") or {
|
|
2411
|
+
"woe_smoothing": self.sv_woe_smoothing,
|
|
2412
|
+
"smoothing_alpha": self.sv_smoothing_alpha,
|
|
2413
|
+
}
|
|
2414
|
+
try:
|
|
2415
|
+
method = str(smoothing["woe_smoothing"])
|
|
2416
|
+
alpha = float(smoothing["smoothing_alpha"])
|
|
2417
|
+
except (KeyError, TypeError, ValueError, OverflowError):
|
|
2418
|
+
return None
|
|
2419
|
+
if not math.isfinite(alpha):
|
|
2420
|
+
# 无法可靠往返的平滑参数:不写决策(加载后按无决策处理),绝不让导出失败
|
|
2421
|
+
return None
|
|
2422
|
+
return {
|
|
2423
|
+
"policies": [str(policy) for policy in sv_table["sv_policy_applied"]],
|
|
2424
|
+
"woe_smoothing": method,
|
|
2425
|
+
"smoothing_alpha": alpha,
|
|
2426
|
+
}
|
|
2427
|
+
|
|
2428
|
+
def _restore_format_a_sv_decisions(
|
|
2429
|
+
self, metadata: Dict[str, Any], n_sv_rows: int
|
|
2430
|
+
) -> Optional[tuple]:
|
|
2431
|
+
"""校验并取回 Format-A attrs 中的 SV 决策,返回 (policies, 平滑参数);
|
|
2432
|
+
缺失、被改动或取值非法时返回 None(按无决策处理)。"""
|
|
2433
|
+
try:
|
|
2434
|
+
decisions = metadata.get("sv_decisions")
|
|
2435
|
+
if not isinstance(decisions, dict):
|
|
2436
|
+
return None
|
|
2437
|
+
if metadata.get("sv_decisions_digest") != self._format_a_sv_decisions_digest(
|
|
2438
|
+
metadata.get("metadata_digest"), decisions
|
|
2439
|
+
):
|
|
2440
|
+
return None
|
|
2441
|
+
policies = decisions.get("policies")
|
|
2442
|
+
method = decisions.get("woe_smoothing")
|
|
2443
|
+
alpha = decisions.get("smoothing_alpha")
|
|
2444
|
+
valid = (
|
|
2445
|
+
isinstance(policies, (list, tuple))
|
|
2446
|
+
and len(policies) == n_sv_rows
|
|
2447
|
+
and all(isinstance(policy, str) and policy in _SV_POLICIES for policy in policies)
|
|
2448
|
+
and isinstance(method, str)
|
|
2449
|
+
and method in {"none", "laplace"}
|
|
2450
|
+
and isinstance(alpha, (int, float, np.integer, np.floating))
|
|
2451
|
+
and not isinstance(alpha, (bool, np.bool_))
|
|
2452
|
+
and bool(np.isfinite(alpha))
|
|
2453
|
+
and float(alpha) >= 0.0
|
|
2454
|
+
)
|
|
2455
|
+
if not valid:
|
|
2456
|
+
return None
|
|
2457
|
+
return list(policies), {"woe_smoothing": method, "smoothing_alpha": float(alpha)}
|
|
2458
|
+
except Exception:
|
|
2459
|
+
# attrs are advisory: any validation error means "no usable decisions".
|
|
2460
|
+
return None
|
|
2461
|
+
|
|
2316
2462
|
# ── 1. get_final_bins ────────────────────────────────────────────
|
|
2317
2463
|
|
|
2318
2464
|
def get_direction_summary(self) -> pd.DataFrame:
|
|
@@ -2355,7 +2501,8 @@ class MonotoneWOEBinner:
|
|
|
2355
2501
|
pct_bad | pct_good | woe | iv | cumiv
|
|
2356
2502
|
is_special (bool, True=特殊值箱)
|
|
2357
2503
|
|
|
2358
|
-
精确数值边界、稀疏箱号、missing_woe
|
|
2504
|
+
精确数值边界、稀疏箱号、missing_woe、类别成员,以及启用 SV 治理时
|
|
2505
|
+
逐行的 sv_policy_applied 与平滑参数保存在
|
|
2359
2506
|
DataFrame.attrs 中;直接传递或 pickle 往返可精确恢复。CSV/Excel
|
|
2360
2507
|
不保留 attrs,回载时以可见 bin_label(默认 .8g)为准。
|
|
2361
2508
|
|
|
@@ -2495,6 +2642,14 @@ class MonotoneWOEBinner:
|
|
|
2495
2642
|
format_a_meta["metadata_digest"] = self._format_a_metadata_digest(
|
|
2496
2643
|
format_a_meta
|
|
2497
2644
|
)
|
|
2645
|
+
# SV 治理决策单独存放、单独校验:不进 metadata_digest,旧版本加载器
|
|
2646
|
+
# 对新键无感知且仍能验证原有字段;未启用 SV 治理时不写,attrs 与旧版一致
|
|
2647
|
+
sv_decisions = self._format_a_sv_decisions(vr)
|
|
2648
|
+
if sv_decisions is not None:
|
|
2649
|
+
format_a_meta["sv_decisions"] = sv_decisions
|
|
2650
|
+
format_a_meta["sv_decisions_digest"] = self._format_a_sv_decisions_digest(
|
|
2651
|
+
format_a_meta["metadata_digest"], sv_decisions
|
|
2652
|
+
)
|
|
2498
2653
|
final.attrs["smf_woe_format_a"] = format_a_meta
|
|
2499
2654
|
result[feat] = final
|
|
2500
2655
|
return result
|
|
@@ -2553,8 +2708,8 @@ class MonotoneWOEBinner:
|
|
|
2553
2708
|
DataFrame 必须包含列: bin_label | n | bad | woe | iv
|
|
2554
2709
|
(可含 is_special 列;无则假设全为普通箱)
|
|
2555
2710
|
SMF 生成且 checksum/行身份校验通过的 DataFrame.attrs 优先用于
|
|
2556
|
-
|
|
2557
|
-
attrs,因此不能恢复超出可见文本精度的信息。
|
|
2711
|
+
精确恢复(含 SV 治理决策,另有独立 checksum);attrs 缺失或失效时退回
|
|
2712
|
+
可见 bin_label。CSV/Excel 会丢失 attrs,因此不能恢复超出可见文本精度的信息。
|
|
2558
2713
|
类别特征自动识别:若普通箱 bin_label 不是数值区间格式(如 "(-∞, 1.5]"),
|
|
2559
2714
|
则按类别特征加载,apply_woe 时按取值直接查表。
|
|
2560
2715
|
|
|
@@ -2586,13 +2741,20 @@ class MonotoneWOEBinner:
|
|
|
2586
2741
|
# 格式 B(dict with edges / woe_map / bin_df)
|
|
2587
2742
|
fmt = "B"
|
|
2588
2743
|
elif isinstance(payload, dict):
|
|
2589
|
-
# 格式 A 包在 dict
|
|
2744
|
+
# 格式 A 包在 dict 里(不常见,兼容);DataFrame 不能用 `or` 取值(真值有歧义)
|
|
2590
2745
|
fmt = "A"
|
|
2591
|
-
df_bin = payload.get("bin_df")
|
|
2746
|
+
df_bin = payload.get("bin_df")
|
|
2747
|
+
if df_bin is None:
|
|
2748
|
+
df_bin = payload.get("df")
|
|
2592
2749
|
if df_bin is None:
|
|
2593
2750
|
raise ValueError(
|
|
2594
2751
|
f"特征 '{feat}': dict 格式既无 'woe_map' 也无 'bin_df',无法识别格式"
|
|
2595
2752
|
)
|
|
2753
|
+
if not isinstance(df_bin, pd.DataFrame):
|
|
2754
|
+
raise ValueError(
|
|
2755
|
+
f"特征 '{feat}': dict 包装的分箱表必须是 DataFrame,"
|
|
2756
|
+
f"收到 {type(df_bin).__name__}"
|
|
2757
|
+
)
|
|
2596
2758
|
else:
|
|
2597
2759
|
raise ValueError(
|
|
2598
2760
|
f"特征 '{feat}': 不支持的类型 {type(payload)},"
|
|
@@ -2601,6 +2763,8 @@ class MonotoneWOEBinner:
|
|
|
2601
2763
|
|
|
2602
2764
|
# 类别特征标记(格式 A 自动识别;格式 B 暂不支持类别特征)
|
|
2603
2765
|
is_categorical = False
|
|
2766
|
+
# 拟合时的 SV 平滑参数(仅格式 A 且 attrs 校验通过时恢复)
|
|
2767
|
+
sv_smoothing = None
|
|
2604
2768
|
|
|
2605
2769
|
# ════════════════════════════════════════════════════════
|
|
2606
2770
|
# 格式 A 处理路径
|
|
@@ -2839,6 +3003,13 @@ class MonotoneWOEBinner:
|
|
|
2839
3003
|
_norm_labels, edges
|
|
2840
3004
|
)
|
|
2841
3005
|
sv_table = df_sv.copy() if len(df_sv) > 0 else pd.DataFrame()
|
|
3006
|
+
if meta_base_matches and len(sv_table) > 0:
|
|
3007
|
+
restored_sv = self._restore_format_a_sv_decisions(
|
|
3008
|
+
format_a_meta, len(sv_table)
|
|
3009
|
+
)
|
|
3010
|
+
if restored_sv is not None:
|
|
3011
|
+
sv_table["sv_policy_applied"] = restored_sv[0]
|
|
3012
|
+
sv_smoothing = restored_sv[1]
|
|
2842
3013
|
total_iv = float(df_bin["iv"].sum())
|
|
2843
3014
|
n_bins = len(df_normal)
|
|
2844
3015
|
woes = df_normal["woe"].values if len(df_normal) > 0 else np.array([])
|
|
@@ -2970,6 +3141,8 @@ class MonotoneWOEBinner:
|
|
|
2970
3141
|
is_monotonic = self._is_monotone(woes) if len(woes) > 1 else True,
|
|
2971
3142
|
n_bins = n_bins,
|
|
2972
3143
|
)
|
|
3144
|
+
if sv_smoothing is not None:
|
|
3145
|
+
res["sv_smoothing"] = sv_smoothing
|
|
2973
3146
|
if is_categorical:
|
|
2974
3147
|
res["is_categorical"] = True
|
|
2975
3148
|
res["categories"] = (
|
|
@@ -3858,6 +4031,11 @@ class MonotoneWOEBinner:
|
|
|
3858
4031
|
- "small_multiples" : 每个 group 一个子图 panel,各画该组组内占比柱
|
|
3859
4032
|
+ 该组 WOE 线(WOE y 轴跨 panel 统一,便于对比)
|
|
3860
4033
|
- 标题:"{feat}: IV_range={min}−{max}"
|
|
4034
|
+
- 各组 IV(图例 / 子图标题):组内口径——以该组自身 bad/good 为分母,
|
|
4035
|
+
含特殊值 / 缺失箱并沿用拟合时的 SV 治理决策;组内只有单一类别且未经 laplace
|
|
4036
|
+
平滑的箱不计入(iv_guard 口径),详见 _group_iv_for_plot
|
|
4037
|
+
- 各组 WOE 折线:以全量 bad/good 为基准(对两类齐全的箱 = 组内 WOE + 常数
|
|
4038
|
+
ln(该组 bad 占全量 bad 的比例 / 该组 good 占全量 good 的比例)),便于跨组比较水平
|
|
3861
4039
|
|
|
3862
4040
|
Parameters
|
|
3863
4041
|
----------
|
|
@@ -3905,6 +4083,8 @@ class MonotoneWOEBinner:
|
|
|
3905
4083
|
n_sv = len(sv_df)
|
|
3906
4084
|
n_total = n_normal + n_sv
|
|
3907
4085
|
iv_overall = vr["iv"]
|
|
4086
|
+
# 普通箱真实箱号(拟合箱号可能不连续,如 [0, 2, 3]):分组统计按箱号取行、按位置画
|
|
4087
|
+
normal_bin_ids = [int(b) - 1 for b in normal_df["bin_no"]]
|
|
3908
4088
|
|
|
3909
4089
|
x_normal = np.arange(n_normal)
|
|
3910
4090
|
x_sv = np.arange(n_normal, n_total)
|
|
@@ -4071,7 +4251,8 @@ class MonotoneWOEBinner:
|
|
|
4071
4251
|
cmap_colors = plt.cm.tab10(np.linspace(0, 0.9, min(max(n_groups,1), 10)))
|
|
4072
4252
|
group_ivs = []
|
|
4073
4253
|
|
|
4074
|
-
# WOE 基准:全量 total_bad / total_good(各组 WOE
|
|
4254
|
+
# WOE 基准:全量 total_bad / total_good(各组 WOE 相对全量,保证跨组可比;
|
|
4255
|
+
# 组 IV 另按组内口径计算,见 _group_iv_for_plot)
|
|
4075
4256
|
all_normal_df, all_sv_groups = self._split_special_for_plot(_df_for_group, feat, vr)
|
|
4076
4257
|
all_normal_sub = all_normal_df[[feat, self.target_col]].dropna(subset=[feat]).copy()
|
|
4077
4258
|
all_normal_sub["_bin"] = self._assign_normal_bins(
|
|
@@ -4086,12 +4267,12 @@ class MonotoneWOEBinner:
|
|
|
4086
4267
|
all_n_full = len(all_normal_sub)
|
|
4087
4268
|
pct_good_n_grp = np.zeros(n_normal)
|
|
4088
4269
|
pct_bad_n_grp = np.zeros(n_normal)
|
|
4089
|
-
for b in
|
|
4270
|
+
for xi, b in enumerate(normal_bin_ids):
|
|
4090
4271
|
grp_b = all_normal_sub[all_normal_sub["_bin"] == b]
|
|
4091
4272
|
bad_b = float(grp_b[self.target_col].sum())
|
|
4092
4273
|
good_b = float((grp_b[self.target_col] == 0).sum())
|
|
4093
|
-
pct_good_n_grp[
|
|
4094
|
-
pct_bad_n_grp[
|
|
4274
|
+
pct_good_n_grp[xi] = good_b / (all_n_full + eps) if all_n_full > 0 else 0.0
|
|
4275
|
+
pct_bad_n_grp[xi] = bad_b / (all_n_full + eps) if all_n_full > 0 else 0.0
|
|
4095
4276
|
# 全量特殊值箱比例(分母 = 全量行数)
|
|
4096
4277
|
all_sv_n = len(_df_for_group)
|
|
4097
4278
|
pct_good_sv_grp = np.zeros(n_sv)
|
|
@@ -4100,7 +4281,7 @@ class MonotoneWOEBinner:
|
|
|
4100
4281
|
for si, sv_row in enumerate(sv_df.itertuples()):
|
|
4101
4282
|
matched_sv_df = None
|
|
4102
4283
|
for sv_key, sv_sub in all_sv_groups.items():
|
|
4103
|
-
if _sv_label(sv_key) == sv_row.bin_label:
|
|
4284
|
+
if _sv_label(sv_key) == sv_row.bin_label and len(sv_sub) > 0:
|
|
4104
4285
|
matched_sv_df = sv_sub
|
|
4105
4286
|
break
|
|
4106
4287
|
if matched_sv_df is not None and len(matched_sv_df) > 0:
|
|
@@ -4148,20 +4329,18 @@ class MonotoneWOEBinner:
|
|
|
4148
4329
|
pct_good_n_g = np.zeros(n_normal)
|
|
4149
4330
|
pct_bad_n_g = np.zeros(n_normal)
|
|
4150
4331
|
grp_woe = []
|
|
4151
|
-
|
|
4152
|
-
for b in range(n_normal):
|
|
4332
|
+
for xi, b in enumerate(normal_bin_ids):
|
|
4153
4333
|
bin_rows = grp_sub[grp_sub["_bin"] == b]
|
|
4154
4334
|
bad_b = float(bin_rows[self.target_col].sum())
|
|
4155
4335
|
good_b = float((bin_rows[self.target_col] == 0).sum())
|
|
4156
|
-
pct_good_n_g[
|
|
4157
|
-
pct_bad_n_g[
|
|
4336
|
+
pct_good_n_g[xi] = good_b / (n_grp + eps)
|
|
4337
|
+
pct_bad_n_g[xi] = bad_b / (n_grp + eps)
|
|
4158
4338
|
if len(bin_rows) == 0:
|
|
4159
4339
|
grp_woe.append(np.nan)
|
|
4160
4340
|
continue
|
|
4161
4341
|
pct_bad_w = bad_b / (all_total_bad + eps)
|
|
4162
4342
|
pct_good_w = good_b / (all_total_good + eps)
|
|
4163
4343
|
woe_b = math.log((pct_bad_w + eps) / (pct_good_w + eps))
|
|
4164
|
-
grp_iv += (pct_bad_w - pct_good_w) * woe_b
|
|
4165
4344
|
grp_woe.append(woe_b)
|
|
4166
4345
|
|
|
4167
4346
|
# ── clustered:画该组组内占比柱(边框用组色,与 WOE 折线对应)──
|
|
@@ -4176,7 +4355,7 @@ class MonotoneWOEBinner:
|
|
|
4176
4355
|
for si, sv_row in enumerate(sv_df.itertuples()):
|
|
4177
4356
|
matched_sv_df = None
|
|
4178
4357
|
for sv_key, sv_sub in grp_sv_groups.items():
|
|
4179
|
-
if _sv_label(sv_key) == sv_row.bin_label:
|
|
4358
|
+
if _sv_label(sv_key) == sv_row.bin_label and len(sv_sub) > 0:
|
|
4180
4359
|
matched_sv_df = sv_sub
|
|
4181
4360
|
break
|
|
4182
4361
|
if matched_sv_df is not None and len(matched_sv_df) > 0:
|
|
@@ -4206,6 +4385,8 @@ class MonotoneWOEBinner:
|
|
|
4206
4385
|
color=clr, linewidth=1.5, marker="o",
|
|
4207
4386
|
markersize=4, zorder=5, label=lbl)
|
|
4208
4387
|
else:
|
|
4388
|
+
# 组 IV 取组内口径(含特殊值箱);WOE 折线仍相对全量基准
|
|
4389
|
+
grp_iv = sum(self._group_iv_for_plot(grp_df_full, feat, vr, fitted_edges))
|
|
4209
4390
|
group_ivs.append(round(grp_iv, 4))
|
|
4210
4391
|
lbl = f"{grp_val} N={n_grp:,} TR={tr:.1%} IV={grp_iv:.3f}"
|
|
4211
4392
|
ax_woe.plot(x_normal, grp_woe, color=clr,
|
|
@@ -4303,6 +4484,7 @@ class MonotoneWOEBinner:
|
|
|
4303
4484
|
|
|
4304
4485
|
- 柱高 = 该箱样本 / 该组总样本(组内占比),good/bad 堆叠,含特殊值箱
|
|
4305
4486
|
- WOE 相对全量基准计算;WOE y 轴范围跨全部 panel 统一,便于横向对比
|
|
4487
|
+
- 子图标题中的 IV 为组内口径(见 _group_iv_for_plot;组内单一类别且未经平滑的箱不计入)
|
|
4306
4488
|
- 文件名后缀 _by_{group_name},与 pooled / clustered 模式一致
|
|
4307
4489
|
"""
|
|
4308
4490
|
# ── guard(与单图路径一致)──
|
|
@@ -4316,13 +4498,15 @@ class MonotoneWOEBinner:
|
|
|
4316
4498
|
vr = self._results[feat]
|
|
4317
4499
|
fitted_edges = list(vr["edges"])
|
|
4318
4500
|
eps = self.eps
|
|
4501
|
+
# 普通箱真实箱号(可能不连续):按箱号取行、按位置画
|
|
4502
|
+
normal_bin_ids = [int(b) - 1 for b in normal_df["bin_no"]]
|
|
4319
4503
|
|
|
4320
4504
|
all_labels = (
|
|
4321
4505
|
[str(b) for b in normal_df["bin_label"]]
|
|
4322
4506
|
+ ([str(b) for b in sv_df["bin_label"]] if n_sv > 0 else [])
|
|
4323
4507
|
)
|
|
4324
4508
|
|
|
4325
|
-
# WOE 基准:全量 total_bad / total_good
|
|
4509
|
+
# WOE 基准:全量 total_bad / total_good(组 IV 另按组内口径,见 _group_iv_for_plot)
|
|
4326
4510
|
all_normal_df, _ = self._split_special_for_plot(_df_for_group, feat, vr)
|
|
4327
4511
|
all_normal_sub = all_normal_df[[feat, self.target_col]].dropna(subset=[feat]).copy()
|
|
4328
4512
|
all_normal_sub["_bin"] = self._assign_normal_bins(
|
|
@@ -4370,13 +4554,12 @@ class MonotoneWOEBinner:
|
|
|
4370
4554
|
pct_bad_n_g = np.zeros(n_normal)
|
|
4371
4555
|
grp_woe = []
|
|
4372
4556
|
grp_br = [] # 各箱组内 bad_rate(用于数据标签)
|
|
4373
|
-
|
|
4374
|
-
for b in range(n_normal):
|
|
4557
|
+
for xi, b in enumerate(normal_bin_ids):
|
|
4375
4558
|
bin_rows = grp_sub[grp_sub["_bin"] == b]
|
|
4376
4559
|
bad_b = float(bin_rows[self.target_col].sum())
|
|
4377
4560
|
good_b = float((bin_rows[self.target_col] == 0).sum())
|
|
4378
|
-
pct_good_n_g[
|
|
4379
|
-
pct_bad_n_g[
|
|
4561
|
+
pct_good_n_g[xi] = good_b / (n_grp + eps)
|
|
4562
|
+
pct_bad_n_g[xi] = bad_b / (n_grp + eps)
|
|
4380
4563
|
if len(bin_rows) == 0:
|
|
4381
4564
|
grp_woe.append(np.nan)
|
|
4382
4565
|
grp_br.append(np.nan)
|
|
@@ -4385,7 +4568,6 @@ class MonotoneWOEBinner:
|
|
|
4385
4568
|
pct_bad_w = bad_b / (all_total_bad + eps)
|
|
4386
4569
|
pct_good_w = good_b / (all_total_good + eps)
|
|
4387
4570
|
woe_b = math.log((pct_bad_w + eps) / (pct_good_w + eps))
|
|
4388
|
-
grp_iv += (pct_bad_w - pct_good_w) * woe_b
|
|
4389
4571
|
grp_woe.append(woe_b)
|
|
4390
4572
|
|
|
4391
4573
|
# 特殊值箱:组内占比 + WOE(相对全量基准)+ 组内 bad_rate
|
|
@@ -4397,7 +4579,7 @@ class MonotoneWOEBinner:
|
|
|
4397
4579
|
for si, sv_row in enumerate(sv_df.itertuples()):
|
|
4398
4580
|
matched_sv_df = None
|
|
4399
4581
|
for sv_key, sv_sub in grp_sv_groups.items():
|
|
4400
|
-
if _sv_label(sv_key) == sv_row.bin_label:
|
|
4582
|
+
if _sv_label(sv_key) == sv_row.bin_label and len(sv_sub) > 0:
|
|
4401
4583
|
matched_sv_df = sv_sub
|
|
4402
4584
|
break
|
|
4403
4585
|
if matched_sv_df is not None and len(matched_sv_df) > 0:
|
|
@@ -4436,6 +4618,8 @@ class MonotoneWOEBinner:
|
|
|
4436
4618
|
ax_woe.plot(x_normal, [np.nan] * n_normal, color="#2E75B6",
|
|
4437
4619
|
linewidth=1.8, marker="o", markersize=5, zorder=5)
|
|
4438
4620
|
else:
|
|
4621
|
+
# 组 IV 取组内口径(含特殊值箱);WOE 折线仍相对全量基准
|
|
4622
|
+
grp_iv = sum(self._group_iv_for_plot(grp_df_full, feat, vr, fitted_edges))
|
|
4439
4623
|
group_ivs.append(round(grp_iv, 4))
|
|
4440
4624
|
iv_disp = grp_iv
|
|
4441
4625
|
ax_woe.plot(x_normal, grp_woe, color="#2E75B6",
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Monotone_Binner.pyi
RENAMED
|
@@ -38,7 +38,8 @@ class MonotoneWOEBinner:
|
|
|
38
38
|
def _cat_to_bin_map(vr: Dict) -> Dict: ...
|
|
39
39
|
def _split_special_for_plot(self, df: pd.DataFrame, feat: str, vr: Dict): ...
|
|
40
40
|
def _assign_normal_bins(self, sub: pd.DataFrame, feat: str, vr: Dict, fitted_edges: list) -> pd.Series: ...
|
|
41
|
-
def
|
|
41
|
+
def _group_iv_for_plot(self, grp_df: pd.DataFrame, feat: str, vr: Dict, fitted_edges: list) -> tuple: ...
|
|
42
|
+
def _compute_woe_single_bin(self, sub: pd.DataFrame, total_bad: float, total_good: float, smooth: bool = False, *, woe_smoothing: Optional[str] = None, smoothing_alpha: Optional[float] = None) -> Dict[str, float]: ...
|
|
42
43
|
def _compute_woe_table(self, df: pd.DataFrame, feat: str, edges: list) -> tuple: ...
|
|
43
44
|
def _compute_sv_table(self, sv_groups: Dict, total_bad: float, total_good: float) -> pd.DataFrame: ...
|
|
44
45
|
def _merge_small_into_missing(self, sv_table: pd.DataFrame, missing_row_idx: Optional[int], total_bad: float, total_good: float) -> pd.DataFrame: ...
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: SuperModelingFactory
|
|
3
|
-
Version: 0.8.
|
|
3
|
+
Version: 0.8.1
|
|
4
4
|
Summary: Credit risk modeling factory: WOE binning, scorecards, LightGBM, Excel reporting.
|
|
5
5
|
Home-page: https://github.com/Kyle-J-Sun/SuperModelingFactory
|
|
6
6
|
Author: Kyle Sun
|
|
@@ -265,7 +265,7 @@ em.close_workbook()
|
|
|
265
265
|
|
|
266
266
|
## 版本
|
|
267
267
|
|
|
268
|
-
- **Version**: 0.8.
|
|
268
|
+
- **Version**: 0.8.1
|
|
269
269
|
- **Author**: Jingkai Sun
|
|
270
270
|
|
|
271
271
|
## 许可证
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/SuperModelingFactory.egg-info/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: SuperModelingFactory
|
|
3
|
-
Version: 0.8.
|
|
3
|
+
Version: 0.8.1
|
|
4
4
|
Summary: Credit risk modeling factory: WOE binning, scorecards, LightGBM, Excel reporting.
|
|
5
5
|
Home-page: https://github.com/Kyle-J-Sun/SuperModelingFactory
|
|
6
6
|
Author: Kyle Sun
|
|
@@ -265,7 +265,7 @@ em.close_workbook()
|
|
|
265
265
|
|
|
266
266
|
## 版本
|
|
267
267
|
|
|
268
|
-
- **Version**: 0.8.
|
|
268
|
+
- **Version**: 0.8.1
|
|
269
269
|
- **Author**: Jingkai Sun
|
|
270
270
|
|
|
271
271
|
## 许可证
|
|
@@ -12,7 +12,7 @@ build-backend = "setuptools.build_meta"
|
|
|
12
12
|
|
|
13
13
|
[project]
|
|
14
14
|
name = "SuperModelingFactory"
|
|
15
|
-
version = "0.8.
|
|
15
|
+
version = "0.8.1"
|
|
16
16
|
description = "Credit risk modeling factory: WOE binning, scorecards, LightGBM, Excel reporting."
|
|
17
17
|
readme = "README.md"
|
|
18
18
|
requires-python = ">=3.10"
|
|
@@ -15,7 +15,7 @@ def _read(path: str) -> str:
|
|
|
15
15
|
|
|
16
16
|
setup(
|
|
17
17
|
name="SuperModelingFactory",
|
|
18
|
-
version=os.environ.get("SMF_VERSION", "0.8.
|
|
18
|
+
version=os.environ.get("SMF_VERSION", "0.8.1"),
|
|
19
19
|
description="Credit risk modeling factory: WOE binning, scorecards, LightGBM, Excel reporting.",
|
|
20
20
|
long_description=_read("README.md"),
|
|
21
21
|
long_description_content_type="text/markdown",
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Binning_Tool.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Binning_Tool.pyi
RENAMED
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Json_Data_Converter.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Model_Registry_Tool.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/Proc_Compare.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/XOR_Encryptor.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/XOR_Encryptor.pyi
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Core/sample_weight_utils.py
RENAMED
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/Evaluation_Tool.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/Evaluation_Tool.pyi
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/Model_Eval_Tool.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/Model_Eval_Tool.pyi
RENAMED
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/evaluate_model.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/evaluate_model.pyi
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Eval/weighted_eval_utils.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Explainability/__init__.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Distribution_Tool.py
RENAMED
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Feature_Insights.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Feature_Insights.pyi
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Feature_Screen.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/PSI_Tool.pyi
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Screen_Gates.py
RENAMED
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Feature/Weighted_Screen.py
RENAMED
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/Backward_Tool.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/Backward_Tool.pyi
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Model/GBM_Search_Tool.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/credit_model.py
RENAMED
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/field_meta.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/mock_sample.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/orchestrator.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/reject_inference.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/sample_analysis.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Pipeline/score_comparison.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/Reject_Infer.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/Reject_Infer.pyi
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/Sample_Split.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/Sample/Sample_Split.pyi
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Plot_Tool.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Plot_Tool.pyi
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Report_Builder.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/WOE_Report_Builder.pyi
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/plot_woe_tool.py
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/WOE/plot_woe_tool.pyi
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/ref_font/WeiRuanYaHei.ttf
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/Modeling_Tool/ref_font/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/SuperModelingFactory.egg-info/SOURCES.txt
RENAMED
|
File without changes
|
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/SuperModelingFactory.egg-info/not-zip-safe
RENAMED
|
File without changes
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.1}/SuperModelingFactory.egg-info/requires.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|