SuperModelingFactory 0.8.0__tar.gz → 0.8.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/ExcelMaster/ExcelMaster.py +6 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/ExcelMaster/Template.py +0 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Binning_Tool.py +22 -17
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/ODPS_Tool.py +5 -7
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Parallel_Engine.py +6 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Slope_Tool.py +0 -6
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/kDataFrame.py +0 -6
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/utils.py +4 -2
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Eval/Model_Eval_Tool.py +9 -3
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Eval/evaluate_model.py +60 -18
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Explainability/Model_Explainer.py +10 -2
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/Feature_Screen.py +1 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Model/Backward_Tool.py +0 -4
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Model/LRM_Tool.py +28 -12
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/credit_model.py +27 -8
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/feature_validation.py +29 -2
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/reject_inference.py +3 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Sample/Sample_Split.py +1 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Monotone_Binner.py +490 -35
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Monotone_Binner.pyi +8 -2
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Plot_Tool.py +4 -2
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/plot_woe_tool.py +2 -2
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/__init__.py +1 -1
- supermodelingfactory-0.8.2/Modeling_Tool/_utils/frames.py +54 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/PKG-INFO +2 -2
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/README.md +1 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Report/Report_Tool.py +0 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/SuperModelingFactory.egg-info/PKG-INFO +2 -2
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/SuperModelingFactory.egg-info/SOURCES.txt +1 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/pyproject.toml +1 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/setup.py +1 -1
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/ExcelMaster/ExcelFormatTool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/ExcelMaster/Utility.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/ExcelMaster/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/LICENSE +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/MANIFEST.in +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Binning_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Check_DuckDB_Compatibility.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Json_Data_Converter.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Model_Registry_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Parallel_ODPS_Manager.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Proc_Compare.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Slope_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/XOR_Encryptor.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/XOR_Encryptor.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/kDataFrame.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/sample_weight_utils.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Eval/Evaluation_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Eval/Evaluation_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Eval/Model_Eval_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Eval/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Eval/evaluate_model.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Eval/weighted_eval_utils.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Explainability/Coalition_Structure.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Explainability/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/Distribution_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/Distribution_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/Feature_Insights.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/Feature_Insights.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/ODPS_Distribution_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/ODPS_Distribution_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/PSI_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/PSI_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/Screen_Gates.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/WOE_Engine_Feature_Patch.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/Weighted_Screen.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Feature/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Model/Backward_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Model/GBM_Search_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Model/GBM_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Model/GBM_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Model/LRM_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Model/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/_common.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/field_meta.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/mock_sample.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/orchestrator.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/sample_analysis.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/score_comparison.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/score_consistency_uat.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/screening_artifact.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Sample/Distribution_Adaptation.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Sample/Distribution_Adaptation.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Sample/Reject_Infer.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Sample/Reject_Infer.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Sample/Sample_Split.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Sample/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/UAT/UAT_Consistency_Checker.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/UAT/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Adapter.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Adapter.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Master.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Master.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Plot_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Report_Builder.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Report_Builder.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Tool.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/WOE_Tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/WOE/plot_woe_tool.pyi +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/_utils/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/_utils/nan_guard.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/_utils/robust.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/_utils/sentinels.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/ref_font/KaiTi.ttf +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/ref_font/WeiRuanYaHei.ttf +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/ref_font/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/ref_font/simsun.ttc +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Report/__init__.py +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/SuperModelingFactory.egg-info/dependency_links.txt +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/SuperModelingFactory.egg-info/not-zip-safe +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/SuperModelingFactory.egg-info/requires.txt +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/SuperModelingFactory.egg-info/top_level.txt +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/requirements.txt +0 -0
- {supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/setup.cfg +0 -0
|
@@ -241,7 +241,12 @@ class ExcelMaster(ExcelWorkbook):
|
|
|
241
241
|
start_col = loc[1] if loc else self.curr_col
|
|
242
242
|
written_range = [start_row, start_col, start_row + nrows - 1, start_col + ncols - 1]
|
|
243
243
|
|
|
244
|
-
|
|
244
|
+
if nrows == 1 and ncols == 1:
|
|
245
|
+
# xlsxwriter refuses to merge a single cell and writes nothing, so a
|
|
246
|
+
# one-column table used to lose its title; write the cell directly.
|
|
247
|
+
worksheet.write(start_row, start_col, text, self.dict_cell_format[cformat])
|
|
248
|
+
else:
|
|
249
|
+
worksheet.merge_range(*written_range, text, self.dict_cell_format[cformat])
|
|
245
250
|
|
|
246
251
|
if self.verbose:
|
|
247
252
|
logging.info(f"Merged Cells: {self.to_cell_range_text(*written_range)}")
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Binning_Tool.py
RENAMED
|
@@ -4,13 +4,8 @@ logger = logging.getLogger(__name__)
|
|
|
4
4
|
import pandas as pd
|
|
5
5
|
import numpy as np
|
|
6
6
|
from scipy.stats import chi2_contingency, chi2
|
|
7
|
+
from Modeling_Tool._utils.frames import as_binning_numeric
|
|
7
8
|
|
|
8
|
-
# Available only in newer pandas versions. Older Airflow images should skip it.
|
|
9
|
-
try:
|
|
10
|
-
pd.set_option('future.no_silent_downcasting', True)
|
|
11
|
-
except (KeyError, ValueError):
|
|
12
|
-
pass
|
|
13
|
-
pd.options.mode.chained_assignment = None # default='warn'
|
|
14
9
|
|
|
15
10
|
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
|
16
11
|
|
|
@@ -540,7 +535,7 @@ def cre_pvt(df, var_name, tgt_name):
|
|
|
540
535
|
>>> df_pvt = cre_pvt(df, var_name='income', tgt_name='default')
|
|
541
536
|
"""
|
|
542
537
|
|
|
543
|
-
df_pvt = df.groupby(var_name)[tgt_name].value_counts().unstack().fillna(0)
|
|
538
|
+
df_pvt = df.groupby(var_name, observed=False)[tgt_name].value_counts().unstack().fillna(0)
|
|
544
539
|
df_pvt["n"] = df_pvt[0] + df_pvt[1]
|
|
545
540
|
df_pvt["tr"] = df_pvt[1] / df_pvt["n"]
|
|
546
541
|
|
|
@@ -694,7 +689,9 @@ def get_bin_range(edges, precision = 5, ascending = False, left_sign = '(', righ
|
|
|
694
689
|
|
|
695
690
|
i = 0
|
|
696
691
|
reverse = not ascending
|
|
697
|
-
|
|
692
|
+
# 极大边界(如 float 最大值)按精度取整会溢出成 inf,历来如此,不为此告警
|
|
693
|
+
with np.errstate(over="ignore"):
|
|
694
|
+
edges = sorted([round(x, precision) for x in edges], reverse = reverse)
|
|
698
695
|
res = []
|
|
699
696
|
while i < len(edges) - 1:
|
|
700
697
|
left = edges[i]
|
|
@@ -708,6 +705,9 @@ def get_bin_range(edges, precision = 5, ascending = False, left_sign = '(', righ
|
|
|
708
705
|
|
|
709
706
|
def _materialize_bin_columns(data, binned, bin_range_list, bin_num_col, bin_range_col):
|
|
710
707
|
"""Attach categorical bin numbers and labels without Python row callbacks."""
|
|
708
|
+
# The two bin columns are ours to add: work on our own frame so callers that
|
|
709
|
+
# pass a slice or their own DataFrame never get these columns written back.
|
|
710
|
+
data = data.copy(deep = False)
|
|
711
711
|
codes = binned.cat.codes.to_numpy(dtype=np.intp, copy=False)
|
|
712
712
|
range_values = np.empty(len(codes), dtype=object)
|
|
713
713
|
range_values[:] = np.nan
|
|
@@ -935,7 +935,9 @@ def quick_binning(data, column, labels = None, nbins = 10, precision = 5, equal_
|
|
|
935
935
|
>>> binned, edges = quick_binning(data, 'income', nbins=10, equal_freq=True)
|
|
936
936
|
"""
|
|
937
937
|
|
|
938
|
-
|
|
938
|
+
# values near the float limits overflow when rounded, as they always did
|
|
939
|
+
with np.errstate(over="ignore"):
|
|
940
|
+
binning_series = as_binning_numeric(data[column]).round(precision)
|
|
939
941
|
|
|
940
942
|
if include_missing:
|
|
941
943
|
binning_series = binning_series.fillna(fillna)
|
|
@@ -986,14 +988,17 @@ def quick_binning(data, column, labels = None, nbins = 10, precision = 5, equal_
|
|
|
986
988
|
if len(spec_values) > 0:
|
|
987
989
|
fnl_breakpoints = sorted(list(set(list(fnl_breakpoints) + spec_values)))
|
|
988
990
|
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
991
|
+
# pandas rounds the edges to format interval labels; float-max edges overflow
|
|
992
|
+
# to inf there as they always did, so the numpy notice is not useful
|
|
993
|
+
with np.errstate(over="ignore"):
|
|
994
|
+
binned, bin_edges = pd.cut(
|
|
995
|
+
binning_series,
|
|
996
|
+
bins = fnl_breakpoints,
|
|
997
|
+
labels = labels,
|
|
998
|
+
right = right,
|
|
999
|
+
include_lowest = include_lowest,
|
|
1000
|
+
retbins = True
|
|
1001
|
+
)
|
|
997
1002
|
|
|
998
1003
|
orig_cat = [x for x in binned.cat.categories.tolist()]
|
|
999
1004
|
if ascending:
|
|
@@ -6,14 +6,12 @@ import threading
|
|
|
6
6
|
logger = logging.getLogger(__name__)
|
|
7
7
|
import pandas as pd
|
|
8
8
|
from odps import ODPS, options
|
|
9
|
-
from odps.models import
|
|
10
|
-
|
|
11
|
-
# Available only in newer pandas versions. Older Airflow images should skip it.
|
|
9
|
+
from odps.models import Column, Partition
|
|
12
10
|
try:
|
|
13
|
-
|
|
14
|
-
except
|
|
15
|
-
|
|
16
|
-
|
|
11
|
+
from odps.models import TableSchema as Schema
|
|
12
|
+
except ImportError: # older pyodps only ships the Schema name
|
|
13
|
+
from odps.models import Schema
|
|
14
|
+
|
|
17
15
|
|
|
18
16
|
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
|
19
17
|
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Core/Parallel_Engine.py
RENAMED
|
@@ -189,9 +189,15 @@ class ParallelApplyEngine:
|
|
|
189
189
|
func_args: tuple[Any, ...],
|
|
190
190
|
func_kwargs: dict[str, Any],
|
|
191
191
|
) -> None:
|
|
192
|
+
# joblib < 1.6 vendors cloudpickle; joblib >= 1.6 dropped the copy and depends on
|
|
193
|
+
# the cloudpickle package. Resolve the serializer outside the check below so an
|
|
194
|
+
# import problem is never reported as a non-serializable callable.
|
|
192
195
|
try:
|
|
193
196
|
from joblib.externals import cloudpickle
|
|
197
|
+
except ImportError:
|
|
198
|
+
import cloudpickle
|
|
194
199
|
|
|
200
|
+
try:
|
|
195
201
|
cloudpickle.dumps((func, func_args, func_kwargs))
|
|
196
202
|
except Exception as exc:
|
|
197
203
|
raise TypeError(
|
|
@@ -1,12 +1,6 @@
|
|
|
1
1
|
import logging
|
|
2
2
|
import numpy as np
|
|
3
3
|
import pandas as pd
|
|
4
|
-
# Available only in newer pandas versions. Older Airflow images should skip it.
|
|
5
|
-
try:
|
|
6
|
-
pd.set_option('future.no_silent_downcasting', True)
|
|
7
|
-
except (KeyError, ValueError):
|
|
8
|
-
pass
|
|
9
|
-
pd.options.mode.chained_assignment = None # default='warn'
|
|
10
4
|
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
|
11
5
|
|
|
12
6
|
|
|
@@ -5,12 +5,6 @@ from pandas import DataFrame
|
|
|
5
5
|
from pandas import Series
|
|
6
6
|
import numpy as np
|
|
7
7
|
|
|
8
|
-
# Available only in newer pandas versions. Older Airflow images should skip it.
|
|
9
|
-
try:
|
|
10
|
-
pd.set_option('future.no_silent_downcasting', True)
|
|
11
|
-
except (KeyError, ValueError):
|
|
12
|
-
pass
|
|
13
|
-
pd.options.mode.chained_assignment = None # default='warn'
|
|
14
8
|
|
|
15
9
|
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
|
16
10
|
|
|
@@ -1696,7 +1696,9 @@ def _calc_woe_iv_values(data, bad_pct, good_pct, fillwoe=True, filliv=True):
|
|
|
1696
1696
|
if len(data[bad_pct]) > 0 and len(data[good_pct]) > 0:
|
|
1697
1697
|
bad_values = data[bad_pct]
|
|
1698
1698
|
good_values = data[good_pct]
|
|
1699
|
-
|
|
1699
|
+
# 某一类占比为 0 的箱 WOE 为 ±inf(历来如此,调用方自行处理),不为此告警
|
|
1700
|
+
with np.errstate(divide="ignore"):
|
|
1701
|
+
woe = np.log(bad_values / good_values)
|
|
1700
1702
|
iv = (bad_values - good_values) * woe
|
|
1701
1703
|
else:
|
|
1702
1704
|
woe = 0 if fillwoe else np.nan
|
|
@@ -1854,7 +1856,7 @@ def scoring(data, model, varlist, scr_name, keeplist = None, all_missing_spec_va
|
|
|
1854
1856
|
nohit_condition = (pd.isnull(fnl_data[varlist]).sum(axis = 1) == len(varlist))
|
|
1855
1857
|
if fnl_data[nohit_condition].shape[0] > 0:
|
|
1856
1858
|
|
|
1857
|
-
all_missing_data = fnl_data[nohit_condition]
|
|
1859
|
+
all_missing_data = fnl_data[nohit_condition].copy()
|
|
1858
1860
|
other_data = fnl_data[~nohit_condition]
|
|
1859
1861
|
|
|
1860
1862
|
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Eval/Model_Eval_Tool.py
RENAMED
|
@@ -5,6 +5,7 @@ import numpy as np
|
|
|
5
5
|
import pandas as pd
|
|
6
6
|
from Modeling_Tool.Core.Binning_Tool import get_bin_range_list, super_binning
|
|
7
7
|
from Modeling_Tool.Core.utils import load_model, calc_iv, calc_woe
|
|
8
|
+
from Modeling_Tool._utils.frames import concat_non_empty
|
|
8
9
|
from .evaluate_model import evaluate_performance
|
|
9
10
|
from . import weighted_eval_utils as _weighted_eval
|
|
10
11
|
|
|
@@ -133,7 +134,9 @@ def _get_gains_table_scr(data, score, dep, nbins = 10, precision = 5,
|
|
|
133
134
|
|
|
134
135
|
|
|
135
136
|
if add_func is not None:
|
|
136
|
-
|
|
137
|
+
# 显式选中全部列(含分组列):add_func 照旧能看到 _bin_num / _bin_range,
|
|
138
|
+
# 也不触发 pandas 对 apply 默认包含分组列的弃用告警
|
|
139
|
+
gains_table_add = res.groupby(["_bin_num", "_bin_range"], dropna=False)[res.columns.unique().tolist()].apply(add_func)
|
|
137
140
|
gains_table = gains_table.merge(gains_table_add, right_index = True, left_index = True, how = 'left')
|
|
138
141
|
|
|
139
142
|
if retSummary:
|
|
@@ -257,6 +260,7 @@ def _get_gains_table_single(data, dep, nbins = 10, precision = 5, min_bin_prop =
|
|
|
257
260
|
return -3
|
|
258
261
|
|
|
259
262
|
if score is None:
|
|
263
|
+
data = data.copy() # the scored column is ours; do not leave it on the caller's frame
|
|
260
264
|
data['_mdl_scr'] = model.predict_proba(data.loc[:, varlist])[:, 1]
|
|
261
265
|
score = '_mdl_scr'
|
|
262
266
|
|
|
@@ -510,7 +514,8 @@ def _get_perf_summary_single(train,
|
|
|
510
514
|
oos_gains['index'] = 'oos'
|
|
511
515
|
oot_gains['index'] = 'oot'
|
|
512
516
|
|
|
513
|
-
|
|
517
|
+
# 缺失样本集的占位空表不参与拼接,否则整数列(N_BUMP / N_BINS)会被它变成 object
|
|
518
|
+
gains_summ = concat_non_empty([ins_gains, oos_gains, oot_gains])
|
|
514
519
|
|
|
515
520
|
model_eval_result_df = model_eval_result_df.merge(gains_summ, on = ['index'], how = 'left')
|
|
516
521
|
|
|
@@ -705,6 +710,7 @@ def _get_cust_gains_table_single(data, dep, nbins = 10, precision = 5, min_bin_p
|
|
|
705
710
|
return -3
|
|
706
711
|
|
|
707
712
|
if score is None:
|
|
713
|
+
data = data.copy() # the scored column is ours; do not leave it on the caller's frame
|
|
708
714
|
data['_mdl_scr'] = model.predict_proba(data.loc[:, varlist])[:, 1]
|
|
709
715
|
score = '_mdl_scr'
|
|
710
716
|
|
|
@@ -2609,7 +2615,7 @@ class PerformanceEvaluator:
|
|
|
2609
2615
|
gains_summ_list.append(gains_res)
|
|
2610
2616
|
|
|
2611
2617
|
if gains_summ_list:
|
|
2612
|
-
gains_summ =
|
|
2618
|
+
gains_summ = concat_non_empty(gains_summ_list)
|
|
2613
2619
|
model_eval_result_df = model_eval_result_df.merge(gains_summ, on = ['index'], how = 'left')
|
|
2614
2620
|
|
|
2615
2621
|
return model_eval_result_df
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Eval/evaluate_model.py
RENAMED
|
@@ -146,7 +146,11 @@ def summarize_pr(pr_df):
|
|
|
146
146
|
pr_info: dict
|
|
147
147
|
P-R曲线统计信息字典
|
|
148
148
|
"""
|
|
149
|
-
|
|
149
|
+
gap = abs(pr_df['precision'] - pr_df['recall'])
|
|
150
|
+
if not gap.notna().any():
|
|
151
|
+
# 精确率或召回率全为 NaN(如样本只有一个类别):平衡点无定义
|
|
152
|
+
return {'bep_index': np.nan, 'bep_threshold': np.nan, 'bep_precision': np.nan, 'bep_recall': np.nan}
|
|
153
|
+
equalind = np.argmin(gap)
|
|
150
154
|
pr_info = {
|
|
151
155
|
'bep_index': equalind,
|
|
152
156
|
'bep_threshold': pr_df['thresholds'][equalind],
|
|
@@ -331,6 +335,10 @@ def summarize_roc(roc_df):
|
|
|
331
335
|
return {'auc': np.nan, 'ks_index': np.nan, 'ks_threshold': np.nan, 'ks': np.nan}
|
|
332
336
|
|
|
333
337
|
f = roc_df['tpr'] - roc_df['fpr']
|
|
338
|
+
if not f.notna().any():
|
|
339
|
+
# 只有一个类别时 TPR 或 FPR 全为 NaN:AUC、KS 及其阈值都无定义。以前 argmax
|
|
340
|
+
# 返回 -1 后按标签取 thresholds[-1] 抛 KeyError,整个评估失败
|
|
341
|
+
return {'auc': np.nan, 'ks_index': np.nan, 'ks_threshold': np.nan, 'ks': np.nan}
|
|
334
342
|
roc_info = {
|
|
335
343
|
'auc': auc(roc_df['fpr'], roc_df['tpr']),
|
|
336
344
|
'ks_index': np.argmax(f),
|
|
@@ -388,11 +396,12 @@ def __plot_ks_axes(roc_df, ax):
|
|
|
388
396
|
ax.plot(X, roc_df['tpr'], color=palette['ClassicBlueRedGrey'][1], label='True Positive Rate', linewidth=2)
|
|
389
397
|
|
|
390
398
|
roc_info = summarize_roc(roc_df)
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
399
|
+
if not pd.isna(roc_info['ks_index']):
|
|
400
|
+
ks_vector = [
|
|
401
|
+
[X[roc_info['ks_index']], X[roc_info['ks_index']]],
|
|
402
|
+
[roc_df['fpr'][roc_info['ks_index']], roc_df['tpr'][roc_info['ks_index']]],
|
|
403
|
+
]
|
|
404
|
+
ax.plot(ks_vector[0], ks_vector[1], linewidth=4, color=palette['ClassicBlueRedGrey'][2], label='KS')
|
|
396
405
|
ax.set_title('Threshold={0:.3f} KS={1:.3f}'.format(roc_info['ks_threshold'], roc_info['ks']), fontsize=15)
|
|
397
406
|
ax.legend(loc=1, fontsize=12)
|
|
398
407
|
|
|
@@ -596,6 +605,36 @@ def _plot_weighted_kde_line(values, weights, ax, color, label):
|
|
|
596
605
|
ax.plot(centers, density, color=color, linewidth=1.5, label=label)
|
|
597
606
|
|
|
598
607
|
|
|
608
|
+
def _non_nan_1d(values):
|
|
609
|
+
"""distplot 的输入处理:转 1 维 float 数组并去掉 NaN(inf 保留)。"""
|
|
610
|
+
values = np.asarray(values, dtype=float)
|
|
611
|
+
if values.ndim > 1:
|
|
612
|
+
values = values.squeeze()
|
|
613
|
+
return values[~np.isnan(values)]
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
def _anchor_like_distplot(values, ax):
|
|
617
|
+
"""distplot 未指定 color 时先画再删一个 (均值, 0) 点取默认颜色;这个点仍计入坐标轴
|
|
618
|
+
数据范围,使纵轴包含 0。照做一遍,保证只有 KDE 曲线的图(多模型、近似常数分数)
|
|
619
|
+
坐标范围不变。"""
|
|
620
|
+
anchor, = ax.plot(values.mean() if values.size else np.nan, 0)
|
|
621
|
+
anchor.remove()
|
|
622
|
+
|
|
623
|
+
|
|
624
|
+
def _plot_score_hist(values, bins, ax, **hist_kws):
|
|
625
|
+
"""等价于已弃用的 ``sns.distplot(values, bins=bins, hist=True, kde=False, hist_kws=...)``。"""
|
|
626
|
+
values = _non_nan_1d(values)
|
|
627
|
+
_anchor_like_distplot(values, ax)
|
|
628
|
+
ax.hist(values, bins, orientation="vertical", **hist_kws)
|
|
629
|
+
|
|
630
|
+
|
|
631
|
+
def _plot_score_kde(values, bw_method, ax, color, label):
|
|
632
|
+
"""等价于已弃用的 ``sns.distplot(values, hist=False, kde=True, kde_kws={'bw': ...})``。"""
|
|
633
|
+
values = _non_nan_1d(values)
|
|
634
|
+
_anchor_like_distplot(values, ax)
|
|
635
|
+
sns.kdeplot(x=values, ax=ax, color=color, label=label, bw_method=bw_method)
|
|
636
|
+
|
|
637
|
+
|
|
599
638
|
def __plot_single_kde_axes(y_true, y_score, bins, ax, fontdicts, sample_weight=None):
|
|
600
639
|
"""在axes上绘制单个kde图.
|
|
601
640
|
|
|
@@ -615,12 +654,12 @@ def __plot_single_kde_axes(y_true, y_score, bins, ax, fontdicts, sample_weight=N
|
|
|
615
654
|
__plot_kde_axes_base(ax, fontdicts)
|
|
616
655
|
|
|
617
656
|
if sample_weight is None:
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
657
|
+
_plot_score_hist(y_score, bins, ax, density=True, rwidth=0.95,
|
|
658
|
+
color=palette['ClassicBlueRedGrey'][2], alpha=1, label='Total')
|
|
659
|
+
_plot_score_kde(y_score[np.where(y_true==0)], 1/bins/2, ax,
|
|
660
|
+
color=palette['ClassicBlueRedGrey'][0], label='Neg KDE')
|
|
661
|
+
_plot_score_kde(y_score[np.where(y_true==1)], 1/bins/2, ax,
|
|
662
|
+
color=palette['ClassicBlueRedGrey'][1], label='Pos KDE')
|
|
624
663
|
true_mean = np.mean(y_true)
|
|
625
664
|
score_mean = np.mean(y_score)
|
|
626
665
|
title = 'N={0:,} True={1:.2%} Score={2:.2%}'.format(
|
|
@@ -687,8 +726,8 @@ def __plot_multi_kde_axes(y_true, y_score_dict, bins, ax, fontdicts):
|
|
|
687
726
|
for i in range(len(models)):
|
|
688
727
|
md = models[i]
|
|
689
728
|
y_score = np.array(y_score_dict[md])
|
|
690
|
-
|
|
691
|
-
|
|
729
|
+
_plot_score_kde(y_score, 1/bins/2, ax, color=palette['MorandiDark'][i],
|
|
730
|
+
label='{0} (Score={1:.2%})'.format(md, np.mean(y_score)))
|
|
692
731
|
ax.axvline(x=np.mean(y_score), linestyle='--', linewidth=1, color=palette['MorandiDark'][i])
|
|
693
732
|
ax.axvline(x=np.mean(y_true), linestyle='-', linewidth=1, color=palette['ClassicGreyRed'][0], label='True')
|
|
694
733
|
|
|
@@ -720,7 +759,7 @@ def __agg(df):
|
|
|
720
759
|
group_cols = ['y_group', 'thresholds']
|
|
721
760
|
else:
|
|
722
761
|
group_cols = ['thresholds']
|
|
723
|
-
df_agg = df.groupby(group_cols).agg(
|
|
762
|
+
df_agg = df.groupby(group_cols, observed=False).agg(
|
|
724
763
|
min_score=pd.NamedAgg(column='y_score', aggfunc='min'),
|
|
725
764
|
max_score=pd.NamedAgg(column='y_score', aggfunc='max'),
|
|
726
765
|
n=pd.NamedAgg(column='y_true', aggfunc='count'),
|
|
@@ -987,7 +1026,9 @@ def calc_fixed_pct(y_true, y_score, y_group=None, bin_edges=None, ascending=True
|
|
|
987
1026
|
pct_df = __agg(df)
|
|
988
1027
|
|
|
989
1028
|
avg_true = np.mean(y_true)
|
|
990
|
-
|
|
1029
|
+
# 没有坏样本时 avg_true 为 0,lift 为 NaN / inf(结果照旧),不为此告警
|
|
1030
|
+
with np.errstate(divide="ignore", invalid="ignore"):
|
|
1031
|
+
pct_df['lift'] = [x / avg_true for x in pct_df['cumavg_true']]
|
|
991
1032
|
pct_df['gain'] = np.cumsum(pct_df['capture_rate'])
|
|
992
1033
|
|
|
993
1034
|
return pct_df
|
|
@@ -1902,8 +1943,9 @@ def __evaluate_performance(y_true, y_score, nrow, ncol, i, dist_bins, pct_bins,
|
|
|
1902
1943
|
|
|
1903
1944
|
pct_info = summarize_pct(pct_df, ascending=pct_ascending)
|
|
1904
1945
|
|
|
1905
|
-
if len(y_true) < 2
|
|
1906
|
-
#
|
|
1946
|
+
if len(y_true) < 2:
|
|
1947
|
+
# 有效样本不足两行:返回默认性能指标(全部为NaN)。只有一个类别时照常汇总与出图,
|
|
1948
|
+
# N / avgTrue / avgScore / 分位目标率都有定义,只有 KS / AUC 为 NaN
|
|
1907
1949
|
return {
|
|
1908
1950
|
'N': np.nan,
|
|
1909
1951
|
'avgTrue': np.nan,
|
|
@@ -21,6 +21,7 @@ from __future__ import annotations
|
|
|
21
21
|
|
|
22
22
|
import ast
|
|
23
23
|
import importlib.metadata
|
|
24
|
+
import inspect
|
|
24
25
|
import sys
|
|
25
26
|
import warnings
|
|
26
27
|
|
|
@@ -517,13 +518,20 @@ class ModelExplainer:
|
|
|
517
518
|
table["importance_pct"] = table["mean_abs_shap"] / total if total else 0.0
|
|
518
519
|
return table
|
|
519
520
|
|
|
520
|
-
def summary_plot(self, X=None, max_display=20, plot_type="dot", show=True, save_path=None):
|
|
521
|
-
"""SHAP summary (beeswarm / bar) plot.
|
|
521
|
+
def summary_plot(self, X=None, max_display=20, plot_type="dot", show=True, save_path=None, random_state=0):
|
|
522
|
+
"""SHAP summary (beeswarm / bar) plot.
|
|
523
|
+
|
|
524
|
+
``random_state`` seeds the point jitter through shap's ``rng`` argument
|
|
525
|
+
(shap >= 0.47), so the plot no longer depends on NumPy's global RNG;
|
|
526
|
+
older shap versions ignore it.
|
|
527
|
+
"""
|
|
522
528
|
shap = _lazy_shap()
|
|
523
529
|
import matplotlib.pyplot as plt
|
|
524
530
|
|
|
525
531
|
values, X_used = self._ensure_values(X)
|
|
526
532
|
kwargs = {"max_display": max_display, "plot_type": plot_type, "show": False}
|
|
533
|
+
if "rng" in inspect.signature(shap.summary_plot).parameters:
|
|
534
|
+
kwargs["rng"] = np.random.default_rng(random_state)
|
|
527
535
|
if not isinstance(X_used, pd.DataFrame) and self.feature_names is not None:
|
|
528
536
|
kwargs["feature_names"] = self.feature_names
|
|
529
537
|
shap.summary_plot(values, X_used, **kwargs)
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Model/Backward_Tool.py
RENAMED
|
@@ -24,7 +24,6 @@ import os
|
|
|
24
24
|
import sys
|
|
25
25
|
import copy
|
|
26
26
|
import logging
|
|
27
|
-
import warnings
|
|
28
27
|
import functools
|
|
29
28
|
from collections import OrderedDict
|
|
30
29
|
from typing import Optional, List, Dict, Union, Any, Tuple
|
|
@@ -37,9 +36,6 @@ import matplotlib.ticker as mticker
|
|
|
37
36
|
|
|
38
37
|
from Modeling_Tool.Core.sample_weight_utils import resolve_sample_weight
|
|
39
38
|
|
|
40
|
-
# Suppress warnings for cleaner output
|
|
41
|
-
warnings.filterwarnings('ignore')
|
|
42
|
-
|
|
43
39
|
|
|
44
40
|
def _resolve_backward_perf_weight_col(
|
|
45
41
|
split_name: str,
|
|
@@ -112,6 +112,19 @@ def lr_varimp(model):
|
|
|
112
112
|
return varimp_df.sort_values('importance', ascending=False).reset_index(drop=True)
|
|
113
113
|
|
|
114
114
|
|
|
115
|
+
def _predict_positive_proba(model, x_arr):
|
|
116
|
+
"""P(y=1) for a design matrix given as a numpy array.
|
|
117
|
+
|
|
118
|
+
A model fitted on a DataFrame gets the same column names back (values are
|
|
119
|
+
taken by position, so predictions are unchanged); this avoids sklearn's
|
|
120
|
+
"X does not have valid feature names" warning.
|
|
121
|
+
"""
|
|
122
|
+
names_in = getattr(model, 'feature_names_in_', None)
|
|
123
|
+
if names_in is not None and x_arr.ndim == 2 and x_arr.shape[1] == len(names_in):
|
|
124
|
+
x_arr = pd.DataFrame(x_arr, columns=names_in)
|
|
125
|
+
return model.predict_proba(x_arr)[:, 1]
|
|
126
|
+
|
|
127
|
+
|
|
115
128
|
def fast_lr_pvalues(model, x, feature_names):
|
|
116
129
|
"""Coefficient p-values for a fitted sklearn LogisticRegression via the
|
|
117
130
|
observed Fisher information — same formula as get_lr_statsmodel_summary
|
|
@@ -121,7 +134,7 @@ def fast_lr_pvalues(model, x, feature_names):
|
|
|
121
134
|
from scipy import stats
|
|
122
135
|
|
|
123
136
|
x_arr = x.values if hasattr(x, 'values') else np.array(x)
|
|
124
|
-
prob = model
|
|
137
|
+
prob = _predict_positive_proba(model, x_arr)
|
|
125
138
|
w = prob * (1 - prob)
|
|
126
139
|
X_design = np.hstack([np.ones((x_arr.shape[0], 1)), x_arr])
|
|
127
140
|
fisher = X_design.T @ (X_design * w[:, None])
|
|
@@ -174,7 +187,7 @@ def get_lr_statsmodel_summary(model, x, y, feature_names=None):
|
|
|
174
187
|
x_arr = x.values if hasattr(x, 'values') else np.array(x)
|
|
175
188
|
y_arr = y.values if hasattr(y, 'values') else np.array(y)
|
|
176
189
|
|
|
177
|
-
prob = model
|
|
190
|
+
prob = _predict_positive_proba(model, x_arr)
|
|
178
191
|
w = prob * (1 - prob)
|
|
179
192
|
W = np.diag(w)
|
|
180
193
|
X_design = np.hstack([np.ones((x_arr.shape[0], 1)), x_arr])
|
|
@@ -214,7 +227,7 @@ def _compute_log_likelihood(model, x, y, sample_weight=None):
|
|
|
214
227
|
y_arr = y.values if hasattr(y, 'values') else np.array(y)
|
|
215
228
|
weight = None if sample_weight is None else np.asarray(sample_weight, dtype=float)
|
|
216
229
|
|
|
217
|
-
prob = model
|
|
230
|
+
prob = _predict_positive_proba(model, x_arr)
|
|
218
231
|
prob = np.clip(prob, 1e-15, 1 - 1e-15)
|
|
219
232
|
point_ll = y_arr * np.log(prob) + (1 - y_arr) * np.log(1 - prob)
|
|
220
233
|
if weight is None:
|
|
@@ -434,16 +447,19 @@ class FeatureSelectionAnalyzer:
|
|
|
434
447
|
|
|
435
448
|
# Preserve the exact legacy call path for no weight and every
|
|
436
449
|
# constant-weight vector.
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
450
|
+
# A perfectly collinear feature has R² = 1 and VIF = inf; that is the
|
|
451
|
+
# expected result, not a floating-point problem worth a warning.
|
|
452
|
+
with np.errstate(divide="ignore"):
|
|
453
|
+
if weight is None or bool(np.all(weight == weight[0])):
|
|
454
|
+
vif_values = [variance_inflation_factor(x, i) for i in range(x.shape[1])]
|
|
455
|
+
else:
|
|
456
|
+
from statsmodels.regression.linear_model import WLS
|
|
441
457
|
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
458
|
+
vif_values = []
|
|
459
|
+
for i in range(x.shape[1]):
|
|
460
|
+
others = np.arange(x.shape[1]) != i
|
|
461
|
+
r_squared = WLS(x[:, i], x[:, others], weights=weight).fit().rsquared
|
|
462
|
+
vif_values.append(1.0 / (1.0 - r_squared))
|
|
447
463
|
vif_data = pd.DataFrame({
|
|
448
464
|
'feature': work.columns,
|
|
449
465
|
'VIF': vif_values,
|
{supermodelingfactory-0.8.0 → supermodelingfactory-0.8.2}/Modeling_Tool/Pipeline/credit_model.py
RENAMED
|
@@ -28,6 +28,19 @@ from ._common import (
|
|
|
28
28
|
)
|
|
29
29
|
|
|
30
30
|
|
|
31
|
+
def _any_column_has_value(frame: pd.DataFrame, columns: list[str], value: Any) -> bool:
|
|
32
|
+
"""Whether any of ``columns`` in ``frame`` holds ``value`` (dtype-incompatible columns never do)."""
|
|
33
|
+
for column in columns:
|
|
34
|
+
if column not in frame.columns:
|
|
35
|
+
continue
|
|
36
|
+
try:
|
|
37
|
+
if bool(frame[column].eq(value).any()):
|
|
38
|
+
return True
|
|
39
|
+
except TypeError:
|
|
40
|
+
continue
|
|
41
|
+
return False
|
|
42
|
+
|
|
43
|
+
|
|
31
44
|
@dataclass
|
|
32
45
|
class CreditModelPipelineConfig:
|
|
33
46
|
output_dir: str = "output"
|
|
@@ -76,6 +89,10 @@ class CreditModelPipelineConfig:
|
|
|
76
89
|
"sv_smoothing_alpha": 0.0,
|
|
77
90
|
}
|
|
78
91
|
)
|
|
92
|
+
# unseen_special_policy (monotone only): declared special values absent from
|
|
93
|
+
# the fit sample — "normal_bin" (legacy) or "neutral" placeholder bins.
|
|
94
|
+
# Without an explicit "special_values" key the monotone self-fit declares the
|
|
95
|
+
# legacy -999999 sentinel only when the WOE fit sample contains it.
|
|
79
96
|
monotone_woe_params: dict[str, Any] = field(
|
|
80
97
|
default_factory=lambda: {
|
|
81
98
|
"n_init_bins": 20,
|
|
@@ -85,6 +102,7 @@ class CreditModelPipelineConfig:
|
|
|
85
102
|
"sv_small_policy": "keep",
|
|
86
103
|
"sv_woe_smoothing": "none",
|
|
87
104
|
"sv_smoothing_alpha": 0.0,
|
|
105
|
+
"unseen_special_policy": "normal_bin",
|
|
88
106
|
}
|
|
89
107
|
)
|
|
90
108
|
|
|
@@ -901,14 +919,15 @@ class CreditModelPipeline:
|
|
|
901
919
|
)
|
|
902
920
|
|
|
903
921
|
if cfg.woe_engine.lower() == "monotone":
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
922
|
+
defaults = {"feature_cols": feature_cols, "target_col": cfg.target_col}
|
|
923
|
+
if "special_values" not in cfg.monotone_woe_params:
|
|
924
|
+
# 默认哨兵 -999999 只在拟合样本里真的出现时才声明:没出现时声明与否
|
|
925
|
+
# 分箱、打分完全一致,声明只会触发"声明了但没出现"告警(unseen_special_policy=
|
|
926
|
+
# 'neutral' 下还会给每个特征加占位箱)。显式传入的 special_values 不受影响。
|
|
927
|
+
defaults["special_values"] = (
|
|
928
|
+
[-999999] if _any_column_has_value(fit_ins, feature_cols, -999999) else []
|
|
929
|
+
)
|
|
930
|
+
params = merge_dict(defaults, cfg.monotone_woe_params)
|
|
912
931
|
# fit()-only kwargs must not reach MonotoneWOEBinner.__init__ —
|
|
913
932
|
# n_jobs / chi2_p / chi2_init_size in monotone_woe_params used to
|
|
914
933
|
# raise TypeError on this self-fit path (the screening-side
|