databubble-scoring 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- databubble_scoring/__init__.py +9 -0
- databubble_scoring/classifier.py +196 -0
- databubble_scoring/drift_monitoring.py +231 -0
- databubble_scoring/mlflow_pyfunc.py +198 -0
- databubble_scoring/model_card.py +440 -0
- databubble_scoring/model_comparison.py +201 -0
- databubble_scoring/predict.py +231 -0
- databubble_scoring/scorecard.py +133 -0
- databubble_scoring/segment_scorer.py +201 -0
- databubble_scoring-0.1.0.dist-info/METADATA +89 -0
- databubble_scoring-0.1.0.dist-info/RECORD +14 -0
- databubble_scoring-0.1.0.dist-info/WHEEL +5 -0
- databubble_scoring-0.1.0.dist-info/licenses/LICENSE +21 -0
- databubble_scoring-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""
|
|
2
|
+
databubble-scoring — pure, offline scoring/replay logic for DataBubble
|
|
3
|
+
model cards, scorecards, and segment scorers. numpy/pandas/scipy/pydantic
|
|
4
|
+
only; mlflow_pyfunc.py (mlflow support) is not imported here since it is an
|
|
5
|
+
optional extra — import it explicitly if you have `databubble-scoring[mlflow]`
|
|
6
|
+
installed.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Pure scoring core shared by classification and segmentation. This is a
|
|
3
|
+
SUBSET of the platform's skills/models/classifier.py — only the score-time
|
|
4
|
+
path lives here (zero sklearn dependency, confirmed by direct read of
|
|
5
|
+
score_observations, which reconstructs the linear function from stored
|
|
6
|
+
coef_/intercept_/scaler_mean/scaler_scale lists using nothing but numpy).
|
|
7
|
+
|
|
8
|
+
fit_classifier / fit_classifier_for_graph / check_segment_labels (sklearn-
|
|
9
|
+
heavy, fit-time only) stay in the platform's own classifier.py, which
|
|
10
|
+
re-imports FittedPipeline/score_observations/_sigmoid/_softmax from here so
|
|
11
|
+
exactly one definition of FittedPipeline exists repo-wide.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from collections import Counter
|
|
16
|
+
from typing import Optional
|
|
17
|
+
|
|
18
|
+
import numpy as np
|
|
19
|
+
import pandas as pd
|
|
20
|
+
from pydantic import BaseModel
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class FittedPipeline(BaseModel):
|
|
24
|
+
"""
|
|
25
|
+
Serialisable representation of a fitted classifier.
|
|
26
|
+
No sklearn objects — only lists and primitives.
|
|
27
|
+
score_observations reconstructs the linear function from these fields.
|
|
28
|
+
"""
|
|
29
|
+
model_type: str # "binary" | "multinomial"
|
|
30
|
+
feature_names: list[str] # column order expected at score time
|
|
31
|
+
scaler_mean: list[float]
|
|
32
|
+
scaler_scale: list[float]
|
|
33
|
+
classes_: list # class labels in order (as strings)
|
|
34
|
+
coef_: list[list[float]] # shape (n_classes_or_1, n_features)
|
|
35
|
+
intercept_: list[float] # shape (n_classes_or_1,)
|
|
36
|
+
threshold: float # decision threshold (binary: cost-adjusted; multinomial: argmax)
|
|
37
|
+
dummy_mapping: Optional[dict] = None # for plain-English labels at score time
|
|
38
|
+
imputer_medians: Optional[list[float]] = None # median per feature for NaN imputation at score time
|
|
39
|
+
|
|
40
|
+
class Config:
|
|
41
|
+
arbitrary_types_allowed = True
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class ScoringOutput(BaseModel):
|
|
45
|
+
"""
|
|
46
|
+
Structurally identical to the platform's skills.formatter.SkillOutput
|
|
47
|
+
(skill_name/summary/findings/warnings/recommendations/chapter_ref) —
|
|
48
|
+
this package has no platform dependency, so score_observations returns
|
|
49
|
+
this local model instead. .findings["labels"] etc. access is unchanged
|
|
50
|
+
for every caller.
|
|
51
|
+
"""
|
|
52
|
+
skill_name: str
|
|
53
|
+
summary: str
|
|
54
|
+
findings: dict = {}
|
|
55
|
+
warnings: list[str] = []
|
|
56
|
+
recommendations: list[str] = []
|
|
57
|
+
chapter_ref: Optional[str] = None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _sigmoid(z: np.ndarray) -> np.ndarray:
|
|
61
|
+
return 1.0 / (1.0 + np.exp(-np.clip(z, -500, 500)))
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _softmax(z: np.ndarray) -> np.ndarray:
|
|
65
|
+
# z shape: (n_obs, n_classes)
|
|
66
|
+
z_shifted = z - z.max(axis=1, keepdims=True)
|
|
67
|
+
exp_z = np.exp(z_shifted)
|
|
68
|
+
return exp_z / exp_z.sum(axis=1, keepdims=True)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def score_observations(
|
|
72
|
+
new_X: pd.DataFrame,
|
|
73
|
+
pipeline: FittedPipeline,
|
|
74
|
+
low_confidence_threshold: float = 0.60,
|
|
75
|
+
) -> ScoringOutput:
|
|
76
|
+
"""
|
|
77
|
+
Score new observations using a FittedPipeline.
|
|
78
|
+
Reconstructs the linear function from stored coefficients — no sklearn needed.
|
|
79
|
+
|
|
80
|
+
Returns findings:
|
|
81
|
+
labels : list of predicted class labels
|
|
82
|
+
probabilities : list of dicts {class: prob} per observation
|
|
83
|
+
low_confidence : list of bool
|
|
84
|
+
low_confidence_count: int
|
|
85
|
+
class_distribution : dict {label: count}
|
|
86
|
+
"""
|
|
87
|
+
missing = [f for f in pipeline.feature_names if f not in new_X.columns]
|
|
88
|
+
if missing:
|
|
89
|
+
return ScoringOutput(
|
|
90
|
+
skill_name="score_observations",
|
|
91
|
+
summary="Scoring failed — input columns do not match training features.",
|
|
92
|
+
findings={
|
|
93
|
+
"error": f"Missing columns: {missing}",
|
|
94
|
+
"expected": pipeline.feature_names,
|
|
95
|
+
"received": list(new_X.columns),
|
|
96
|
+
},
|
|
97
|
+
warnings=[f"Input is missing {len(missing)} column(s) from the training feature set."],
|
|
98
|
+
recommendations=["Ensure new data passes through the same feature preparation step as training data."],
|
|
99
|
+
chapter_ref="Chapter 19",
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
X_mat = new_X[pipeline.feature_names].values.astype(float)
|
|
103
|
+
|
|
104
|
+
if pipeline.imputer_medians is not None and np.isnan(X_mat).any():
|
|
105
|
+
_medians = np.array(pipeline.imputer_medians)
|
|
106
|
+
_nan_mask = np.isnan(X_mat)
|
|
107
|
+
X_mat[_nan_mask] = np.take(_medians, np.where(_nan_mask)[1])
|
|
108
|
+
|
|
109
|
+
mean_ = np.array(pipeline.scaler_mean)
|
|
110
|
+
scale_ = np.array(pipeline.scaler_scale)
|
|
111
|
+
X_scaled = (X_mat - mean_) / scale_
|
|
112
|
+
|
|
113
|
+
coef_ = np.array(pipeline.coef_) # (1, n_feat) binary | (n_cls, n_feat) multinomial
|
|
114
|
+
intercept_ = np.array(pipeline.intercept_)
|
|
115
|
+
|
|
116
|
+
if pipeline.model_type == "binary":
|
|
117
|
+
logits = X_scaled @ coef_[0] + intercept_[0] # (n_obs,)
|
|
118
|
+
probs_pos = _sigmoid(logits) # (n_obs,)
|
|
119
|
+
prob_matrix = np.column_stack([1 - probs_pos, probs_pos]) # (n_obs, 2)
|
|
120
|
+
else:
|
|
121
|
+
logits = X_scaled @ coef_.T + intercept_ # (n_obs, n_classes)
|
|
122
|
+
prob_matrix = _softmax(logits) # (n_obs, n_classes)
|
|
123
|
+
|
|
124
|
+
classes = pipeline.classes_
|
|
125
|
+
max_probs = prob_matrix.max(axis=1)
|
|
126
|
+
predicted_indices = prob_matrix.argmax(axis=1)
|
|
127
|
+
labels = [classes[i] for i in predicted_indices]
|
|
128
|
+
|
|
129
|
+
if pipeline.model_type == "binary" and pipeline.threshold != 0.50:
|
|
130
|
+
labels = [
|
|
131
|
+
classes[1] if prob_matrix[i, 1] >= pipeline.threshold else classes[0]
|
|
132
|
+
for i in range(len(labels))
|
|
133
|
+
]
|
|
134
|
+
|
|
135
|
+
low_confidence = [bool(p < low_confidence_threshold) for p in max_probs]
|
|
136
|
+
low_confidence_count = sum(low_confidence)
|
|
137
|
+
|
|
138
|
+
probabilities = [
|
|
139
|
+
{cls: round(float(prob_matrix[i, j]), 4) for j, cls in enumerate(classes)}
|
|
140
|
+
for i in range(len(labels))
|
|
141
|
+
]
|
|
142
|
+
|
|
143
|
+
class_dist = {str(k): v for k, v in Counter(labels).items()}
|
|
144
|
+
for _k in (str(c) for c in pipeline.classes_):
|
|
145
|
+
if _k not in class_dist:
|
|
146
|
+
class_dist[_k] = 0
|
|
147
|
+
|
|
148
|
+
n_obs = len(labels)
|
|
149
|
+
lc_pct = low_confidence_count / n_obs if n_obs > 0 else 0
|
|
150
|
+
|
|
151
|
+
warnings_out = []
|
|
152
|
+
if lc_pct > 0.20:
|
|
153
|
+
warnings_out.append(
|
|
154
|
+
f"{low_confidence_count} of {n_obs} observations ({lc_pct:.1%}) scored below "
|
|
155
|
+
f"the {low_confidence_threshold:.0%} confidence threshold. "
|
|
156
|
+
"These assignments should be treated as uncertain."
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
confidence_implausibly_low = False
|
|
160
|
+
if n_obs >= 100 and lc_pct < 0.005:
|
|
161
|
+
confidence_implausibly_low = True
|
|
162
|
+
warnings_out.append(
|
|
163
|
+
f"Only {lc_pct:.1%} of observations fell below the 0.60 confidence threshold. "
|
|
164
|
+
f"With real-world data this is statistically implausible and may indicate overfitting or "
|
|
165
|
+
f"data leakage — validate on genuinely unseen data."
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
summary = (
|
|
169
|
+
f"{n_obs} observations scored. "
|
|
170
|
+
f"Class distribution: {class_dist}. "
|
|
171
|
+
f"{low_confidence_count} flagged as low confidence (<{low_confidence_threshold:.0%})."
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
return ScoringOutput(
|
|
175
|
+
skill_name="score_observations",
|
|
176
|
+
summary=summary,
|
|
177
|
+
findings={
|
|
178
|
+
"labels": labels,
|
|
179
|
+
"probabilities": probabilities,
|
|
180
|
+
"low_confidence": low_confidence,
|
|
181
|
+
"low_confidence_count": low_confidence_count,
|
|
182
|
+
"low_confidence_pct": round(lc_pct, 4),
|
|
183
|
+
"confidence_implausibly_low": confidence_implausibly_low,
|
|
184
|
+
"class_distribution": class_dist,
|
|
185
|
+
"n_observations": n_obs,
|
|
186
|
+
"threshold_used": pipeline.threshold,
|
|
187
|
+
"low_confidence_threshold": low_confidence_threshold,
|
|
188
|
+
},
|
|
189
|
+
warnings=warnings_out,
|
|
190
|
+
recommendations=(
|
|
191
|
+
["Review low-confidence observations manually before acting on segment assignments."]
|
|
192
|
+
if lc_pct > 0.20 else
|
|
193
|
+
["Segment assignments are ready to use."]
|
|
194
|
+
),
|
|
195
|
+
chapter_ref="Chapter 19",
|
|
196
|
+
)
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""
|
|
2
|
+
DataBubble — Drift Monitoring for Exported ModelCards
|
|
3
|
+
|
|
4
|
+
Compares a new batch of rows against a ModelCard's fit-time
|
|
5
|
+
feature_snapshot (per-column mean/std/skewness or proportion, added by the
|
|
6
|
+
platform's linear_regression.py::run_regression at fit time — see
|
|
7
|
+
model_card.py::ColumnSnapshot). Point-summary only, never raw reference
|
|
8
|
+
rows — a one-sample t-test (mean) and two-sample z-tests (skewness,
|
|
9
|
+
proportion) are the right tests here, not a full two-sample PSI/KS, since
|
|
10
|
+
no second full distribution exists to compare against.
|
|
11
|
+
|
|
12
|
+
apply_recipe (model_card.py) rebuilds new rows into the exact fit-time
|
|
13
|
+
model-matrix shape — reused here for column alignment and for its existing
|
|
14
|
+
missing-column/unseen-level validation, rather than duplicating either.
|
|
15
|
+
|
|
16
|
+
The proportion two-sample z-test is hand-rolled with scipy.stats.norm
|
|
17
|
+
(standard pooled-variance two-proportion z-test) rather than taking
|
|
18
|
+
statsmodels as a runtime dependency, keeping this package to
|
|
19
|
+
numpy/pandas/scipy/pydantic only. Cross-checked against
|
|
20
|
+
statsmodels.stats.proportion.proportions_ztest in this package's own test
|
|
21
|
+
suite (a dev-only dependency, never a runtime one).
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import math
|
|
26
|
+
import warnings
|
|
27
|
+
from typing import Any, Literal, Optional
|
|
28
|
+
|
|
29
|
+
import numpy as np
|
|
30
|
+
import pandas as pd
|
|
31
|
+
from scipy import stats
|
|
32
|
+
from pydantic import BaseModel
|
|
33
|
+
|
|
34
|
+
try:
|
|
35
|
+
from .model_card import ModelCard, apply_recipe
|
|
36
|
+
except ImportError:
|
|
37
|
+
from model_card import ModelCard, apply_recipe
|
|
38
|
+
|
|
39
|
+
DRIFT_ALPHA = 0.05 # two-tailed significance threshold
|
|
40
|
+
SKEW_DRIFT_Z = 1.96 # ~95% two-tailed z-critical
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class ColumnDriftResult(BaseModel):
|
|
44
|
+
column: str
|
|
45
|
+
kind: Literal["continuous", "proportion"]
|
|
46
|
+
n_observed: int
|
|
47
|
+
reference: dict[str, Any]
|
|
48
|
+
observed: dict[str, Any]
|
|
49
|
+
drifted: bool
|
|
50
|
+
detail: str
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class DriftResult(BaseModel):
|
|
54
|
+
applicable: bool
|
|
55
|
+
reason: Optional[str] = None
|
|
56
|
+
n_features_checked: int = 0
|
|
57
|
+
n_features_drifted: int = 0
|
|
58
|
+
drift_detected: bool = False
|
|
59
|
+
per_feature: list[ColumnDriftResult] = []
|
|
60
|
+
interpretation: str = ""
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _skewness_se(n: int) -> float:
|
|
64
|
+
"""
|
|
65
|
+
Fisher's standard error of the sample skewness coefficient. Undefined
|
|
66
|
+
for n < 3 — callers must guard that themselves.
|
|
67
|
+
"""
|
|
68
|
+
return math.sqrt((6 * n * (n - 1)) / ((n - 2) * (n + 1) * (n + 3)))
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _two_proportion_ztest(count: list[int], nobs: list[int]) -> tuple[float, float]:
|
|
72
|
+
"""
|
|
73
|
+
Standard pooled-variance two-proportion z-test, two-sided. Hand-rolled
|
|
74
|
+
replacement for statsmodels.stats.proportion.proportions_ztest(count,
|
|
75
|
+
nobs) with its default prop_var (pooled) behavior — cross-checked
|
|
76
|
+
against the real thing in test_drift_monitoring.py.
|
|
77
|
+
Returns (z_statistic, p_value). p_value is nan when the pooled variance
|
|
78
|
+
is exactly 0 (both samples identical, e.g. all-0 or all-1) — caller
|
|
79
|
+
already has its own "std=0" fallback for a nan p-value.
|
|
80
|
+
"""
|
|
81
|
+
count1, count2 = count
|
|
82
|
+
n1, n2 = nobs
|
|
83
|
+
p1, p2 = count1 / n1, count2 / n2
|
|
84
|
+
p_pool = (count1 + count2) / (n1 + n2)
|
|
85
|
+
var = p_pool * (1 - p_pool) * (1.0 / n1 + 1.0 / n2)
|
|
86
|
+
if var <= 0:
|
|
87
|
+
return float("nan"), float("nan")
|
|
88
|
+
z = (p1 - p2) / math.sqrt(var)
|
|
89
|
+
p_value = 2 * float(stats.norm.sf(abs(z)))
|
|
90
|
+
return float(z), p_value
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _check_continuous_drift(column: str, snapshot: dict, observed: pd.Series, alpha: float) -> ColumnDriftResult:
|
|
94
|
+
values = observed.dropna()
|
|
95
|
+
n_obs = len(values)
|
|
96
|
+
notes: list[str] = []
|
|
97
|
+
drifted = False
|
|
98
|
+
|
|
99
|
+
ref_mean = snapshot.get("mean")
|
|
100
|
+
if ref_mean is not None and n_obs >= 2:
|
|
101
|
+
try:
|
|
102
|
+
with warnings.catch_warnings():
|
|
103
|
+
warnings.simplefilter("ignore")
|
|
104
|
+
_, p_value = stats.ttest_1samp(values, popmean=ref_mean)
|
|
105
|
+
if math.isnan(p_value):
|
|
106
|
+
shifted = float(values.mean()) != ref_mean
|
|
107
|
+
p_note = "std=0"
|
|
108
|
+
else:
|
|
109
|
+
shifted = p_value < alpha
|
|
110
|
+
p_note = f"p={p_value:.4g}"
|
|
111
|
+
except Exception:
|
|
112
|
+
shifted, p_note = False, ""
|
|
113
|
+
if shifted:
|
|
114
|
+
drifted = True
|
|
115
|
+
new_mean = float(values.mean())
|
|
116
|
+
direction = "up" if new_mean > ref_mean else "down"
|
|
117
|
+
notes.append(f"mean shifted {direction} from {ref_mean:.4g} to {new_mean:.4g} ({p_note})")
|
|
118
|
+
|
|
119
|
+
ref_skew = snapshot.get("skewness")
|
|
120
|
+
ref_n = snapshot.get("n")
|
|
121
|
+
if ref_skew is not None and ref_n is not None and n_obs >= 3 and ref_n >= 3:
|
|
122
|
+
try:
|
|
123
|
+
new_skew = float(values.skew())
|
|
124
|
+
se_diff = math.sqrt(_skewness_se(n_obs) ** 2 + _skewness_se(ref_n) ** 2)
|
|
125
|
+
z = (new_skew - ref_skew) / se_diff if se_diff > 0 else 0.0
|
|
126
|
+
if abs(z) >= SKEW_DRIFT_Z:
|
|
127
|
+
drifted = True
|
|
128
|
+
notes.append(f"skewness shifted from {ref_skew:.4g} to {new_skew:.4g} (z={z:.2f})")
|
|
129
|
+
except Exception:
|
|
130
|
+
pass
|
|
131
|
+
|
|
132
|
+
return ColumnDriftResult(
|
|
133
|
+
column=column, kind="continuous", n_observed=n_obs,
|
|
134
|
+
reference=snapshot,
|
|
135
|
+
observed={"mean": float(values.mean()) if n_obs else None,
|
|
136
|
+
"std": float(values.std(ddof=1)) if n_obs >= 2 else None,
|
|
137
|
+
"skewness": float(values.skew()) if n_obs >= 3 else None},
|
|
138
|
+
drifted=drifted,
|
|
139
|
+
detail="; ".join(notes) if notes else "No significant shift detected.",
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _check_proportion_drift(column: str, snapshot: dict, observed: pd.Series, alpha: float) -> ColumnDriftResult:
|
|
144
|
+
values = observed.dropna()
|
|
145
|
+
n_obs = len(values)
|
|
146
|
+
ref_proportion = snapshot.get("proportion")
|
|
147
|
+
ref_n = snapshot.get("n")
|
|
148
|
+
new_proportion = float(values.mean()) if n_obs else None
|
|
149
|
+
drifted = False
|
|
150
|
+
detail = "No significant shift detected."
|
|
151
|
+
|
|
152
|
+
if ref_proportion is not None and ref_n and n_obs >= 1:
|
|
153
|
+
new_count = int(values.sum())
|
|
154
|
+
ref_count = round(ref_proportion * ref_n)
|
|
155
|
+
try:
|
|
156
|
+
_stat, p_value = _two_proportion_ztest(
|
|
157
|
+
count=[new_count, ref_count], nobs=[n_obs, ref_n],
|
|
158
|
+
)
|
|
159
|
+
if math.isnan(p_value):
|
|
160
|
+
drifted = new_proportion != ref_proportion
|
|
161
|
+
detail = "std=0"
|
|
162
|
+
elif p_value < alpha:
|
|
163
|
+
drifted = True
|
|
164
|
+
direction = "up" if new_proportion > ref_proportion else "down"
|
|
165
|
+
detail = (
|
|
166
|
+
f"proportion shifted {direction} from {ref_proportion:.4g} to "
|
|
167
|
+
f"{new_proportion:.4g} (p={p_value:.4g})"
|
|
168
|
+
)
|
|
169
|
+
except Exception:
|
|
170
|
+
pass
|
|
171
|
+
|
|
172
|
+
return ColumnDriftResult(
|
|
173
|
+
column=column, kind="proportion", n_observed=n_obs,
|
|
174
|
+
reference=snapshot, observed={"proportion": new_proportion},
|
|
175
|
+
drifted=drifted, detail=detail,
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def diagnose_feature_drift(card: ModelCard, new_rows: pd.DataFrame, alpha: float = DRIFT_ALPHA) -> DriftResult:
|
|
180
|
+
"""
|
|
181
|
+
Compares new_rows against card.feature_snapshot (the training-time
|
|
182
|
+
per-column reference captured at fit time). Raises RecipeError (same
|
|
183
|
+
as predict_from_card) when new_rows can't be rebuilt into the card's
|
|
184
|
+
model matrix — missing columns or unseen categorical levels.
|
|
185
|
+
"""
|
|
186
|
+
if not card.feature_snapshot:
|
|
187
|
+
return DriftResult(
|
|
188
|
+
applicable=False,
|
|
189
|
+
reason=(
|
|
190
|
+
"This card predates drift monitoring (schema < 1.2) — "
|
|
191
|
+
"re-export the model to enable it."
|
|
192
|
+
),
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
matrix = apply_recipe(new_rows, card.recipe)
|
|
196
|
+
|
|
197
|
+
per_feature: list[ColumnDriftResult] = []
|
|
198
|
+
for column, snap in card.feature_snapshot.items():
|
|
199
|
+
snapshot = snap if isinstance(snap, dict) else snap.model_dump()
|
|
200
|
+
if column not in matrix.columns:
|
|
201
|
+
continue
|
|
202
|
+
if snapshot.get("kind") == "proportion":
|
|
203
|
+
per_feature.append(_check_proportion_drift(column, snapshot, matrix[column], alpha))
|
|
204
|
+
else:
|
|
205
|
+
per_feature.append(_check_continuous_drift(column, snapshot, matrix[column], alpha))
|
|
206
|
+
|
|
207
|
+
n_drifted = sum(1 for pf in per_feature if pf.drifted)
|
|
208
|
+
drift_detected = n_drifted > 0
|
|
209
|
+
|
|
210
|
+
if drift_detected:
|
|
211
|
+
drifted_cols = [pf.column for pf in per_feature if pf.drifted]
|
|
212
|
+
interpretation = (
|
|
213
|
+
f"{n_drifted} of {len(per_feature)} feature(s) shifted significantly "
|
|
214
|
+
f"from their training-time distribution: {', '.join(drifted_cols)}. "
|
|
215
|
+
f"Predictions from this model may no longer be reliable for this "
|
|
216
|
+
f"population — review before continuing to use it."
|
|
217
|
+
)
|
|
218
|
+
else:
|
|
219
|
+
interpretation = (
|
|
220
|
+
f"No significant shift detected across {len(per_feature)} feature(s) "
|
|
221
|
+
f"checked against their training-time distribution."
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
return DriftResult(
|
|
225
|
+
applicable=True,
|
|
226
|
+
n_features_checked=len(per_feature),
|
|
227
|
+
n_features_drifted=n_drifted,
|
|
228
|
+
drift_detected=drift_detected,
|
|
229
|
+
per_feature=per_feature,
|
|
230
|
+
interpretation=interpretation,
|
|
231
|
+
)
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
"""
|
|
2
|
+
DataBubble — MLflow pyfunc artifact for offline scoring.
|
|
3
|
+
|
|
4
|
+
Wraps a ModelCard / Scorecard / SegmentScorer as a standard MLflow pyfunc
|
|
5
|
+
model: `predict()` runs the exact same predict_from_card / score_from_scorecard
|
|
6
|
+
/ score_from_segment_scorer path the DataBubble platform itself uses at
|
|
7
|
+
export time, so a locally-loaded artifact and the hosted API can never
|
|
8
|
+
silently diverge. No network call at inference.
|
|
9
|
+
|
|
10
|
+
Optional module: only import this if you have `databubble-scoring[mlflow]`
|
|
11
|
+
installed. databubble_scoring/__init__.py never imports it.
|
|
12
|
+
|
|
13
|
+
ModelCardBundle (multi-group cards) is not supported in this version —
|
|
14
|
+
export a single group's ModelCard instead.
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import json
|
|
19
|
+
import os
|
|
20
|
+
import tempfile
|
|
21
|
+
from typing import Optional, Union
|
|
22
|
+
|
|
23
|
+
import pandas as pd
|
|
24
|
+
from pydantic import BaseModel
|
|
25
|
+
|
|
26
|
+
try:
|
|
27
|
+
import mlflow.pyfunc
|
|
28
|
+
except ImportError as e:
|
|
29
|
+
raise ImportError(
|
|
30
|
+
"mlflow is required to use databubble_scoring.mlflow_pyfunc — install "
|
|
31
|
+
"with `pip install databubble-scoring[mlflow]`."
|
|
32
|
+
) from e
|
|
33
|
+
|
|
34
|
+
try:
|
|
35
|
+
from . import __version__
|
|
36
|
+
from .model_card import ModelCard, RecipeError
|
|
37
|
+
from .predict import predict_from_card, PredictionResult
|
|
38
|
+
from .scorecard import Scorecard, score_from_scorecard
|
|
39
|
+
from .segment_scorer import SegmentScorer, score_from_segment_scorer
|
|
40
|
+
except ImportError:
|
|
41
|
+
from databubble_scoring import __version__
|
|
42
|
+
from model_card import ModelCard, RecipeError
|
|
43
|
+
from predict import predict_from_card, PredictionResult
|
|
44
|
+
from scorecard import Scorecard, score_from_scorecard
|
|
45
|
+
from segment_scorer import SegmentScorer, score_from_segment_scorer
|
|
46
|
+
|
|
47
|
+
Artifact = Union[ModelCard, Scorecard, SegmentScorer, dict]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _detect_kind(d: dict) -> str:
|
|
51
|
+
kind = d.get("kind")
|
|
52
|
+
if kind == "classification_scorecard":
|
|
53
|
+
return "classification_scorecard"
|
|
54
|
+
if kind == "segment_scorer":
|
|
55
|
+
return "segment_scorer"
|
|
56
|
+
if kind == "bundle":
|
|
57
|
+
raise ValueError(
|
|
58
|
+
"ModelCardBundle (multi-group cards) is not supported by the MLflow "
|
|
59
|
+
"pyfunc wrapper — export and save a single group's ModelCard instead."
|
|
60
|
+
)
|
|
61
|
+
if kind is not None:
|
|
62
|
+
raise ValueError(f"Unrecognised card 'kind': {kind!r}")
|
|
63
|
+
# ModelCard carries no `kind` field at all — its required fields are what
|
|
64
|
+
# distinguish it from a malformed payload.
|
|
65
|
+
if {"outcome", "intercept", "terms", "recipe", "provenance"} <= d.keys():
|
|
66
|
+
return "model_card"
|
|
67
|
+
raise ValueError(
|
|
68
|
+
"Could not determine artifact kind from card.json — expected a ModelCard "
|
|
69
|
+
"(no 'kind' field), a Scorecard ('kind': 'classification_scorecard'), or "
|
|
70
|
+
"a SegmentScorer ('kind': 'segment_scorer')."
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _artifact_dict(artifact: Artifact) -> dict:
|
|
75
|
+
if isinstance(artifact, BaseModel):
|
|
76
|
+
return artifact.model_dump()
|
|
77
|
+
if isinstance(artifact, dict):
|
|
78
|
+
return artifact
|
|
79
|
+
raise TypeError(f"artifact must be a ModelCard/Scorecard/SegmentScorer or dict, got {type(artifact)}")
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _prediction_result_to_frame(result: PredictionResult) -> pd.DataFrame:
|
|
83
|
+
data = {"prediction": result.predictions}
|
|
84
|
+
for col in ("ci_lower", "ci_upper", "pi_lower", "pi_upper"):
|
|
85
|
+
val = getattr(result, col)
|
|
86
|
+
if val is not None:
|
|
87
|
+
data[col] = val
|
|
88
|
+
return pd.DataFrame(data)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _scoring_output_to_frame(findings: dict, kind: str) -> pd.DataFrame:
|
|
92
|
+
if findings.get("labels") is None:
|
|
93
|
+
raise ValueError(
|
|
94
|
+
f"Scoring failed: {findings.get('error', 'input columns did not match the fitted pipeline')} "
|
|
95
|
+
f"(expected columns: {findings.get('expected')}, received: {findings.get('received')})"
|
|
96
|
+
)
|
|
97
|
+
data = {"label": findings["labels"]}
|
|
98
|
+
df = pd.DataFrame(data)
|
|
99
|
+
probs = findings.get("probabilities") or []
|
|
100
|
+
if probs:
|
|
101
|
+
df = pd.concat([df, pd.DataFrame(probs).add_prefix("probability_")], axis=1)
|
|
102
|
+
if "low_confidence" in findings:
|
|
103
|
+
df["low_confidence"] = findings["low_confidence"]
|
|
104
|
+
if kind == "segment_scorer":
|
|
105
|
+
assignments = findings.get("assignments") or []
|
|
106
|
+
if assignments:
|
|
107
|
+
df["segment"] = [a["segment"] for a in assignments]
|
|
108
|
+
return df
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class DataBubbleScoringModel(mlflow.pyfunc.PythonModel):
|
|
112
|
+
"""
|
|
113
|
+
Generic pyfunc wrapper for any DataBubble scoring artifact. The concrete
|
|
114
|
+
kind (ModelCard / Scorecard / SegmentScorer) is auto-detected from
|
|
115
|
+
card.json at load time — one model class covers all three, since the
|
|
116
|
+
scoring call each dispatches to is already the single source of truth
|
|
117
|
+
(predict_from_card / score_from_scorecard / score_from_segment_scorer).
|
|
118
|
+
"""
|
|
119
|
+
|
|
120
|
+
def load_context(self, context):
|
|
121
|
+
with open(context.artifacts["card_json"]) as f:
|
|
122
|
+
d = json.load(f)
|
|
123
|
+
self._kind = _detect_kind(d)
|
|
124
|
+
if self._kind == "model_card":
|
|
125
|
+
self._artifact = ModelCard(**d)
|
|
126
|
+
elif self._kind == "classification_scorecard":
|
|
127
|
+
self._artifact = Scorecard(**d)
|
|
128
|
+
else:
|
|
129
|
+
self._artifact = SegmentScorer(**d)
|
|
130
|
+
|
|
131
|
+
def predict(self, context, model_input, params=None):
|
|
132
|
+
if not isinstance(model_input, pd.DataFrame):
|
|
133
|
+
model_input = pd.DataFrame(model_input)
|
|
134
|
+
|
|
135
|
+
if self._kind == "model_card":
|
|
136
|
+
result = predict_from_card(self._artifact, model_input)
|
|
137
|
+
return _prediction_result_to_frame(result)
|
|
138
|
+
|
|
139
|
+
try:
|
|
140
|
+
if self._kind == "classification_scorecard":
|
|
141
|
+
scoring_result = score_from_scorecard(self._artifact, model_input)
|
|
142
|
+
else:
|
|
143
|
+
scoring_result = score_from_segment_scorer(self._artifact, model_input)
|
|
144
|
+
except RecipeError as e:
|
|
145
|
+
raise ValueError(f"Recipe replay failed: {e}") from e
|
|
146
|
+
return _scoring_output_to_frame(scoring_result.findings, self._kind)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _write_card_json(artifact: Artifact) -> str:
|
|
150
|
+
d = _artifact_dict(artifact)
|
|
151
|
+
_detect_kind(d) # validate (raises early on a bundle/malformed payload) before any file/mlflow I/O
|
|
152
|
+
fd, path = tempfile.mkstemp(suffix=".json", prefix="databubble_card_")
|
|
153
|
+
with os.fdopen(fd, "w") as f:
|
|
154
|
+
json.dump(d, f)
|
|
155
|
+
return path
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def save_databubble_model(artifact: Artifact, path: str, **save_model_kwargs) -> None:
|
|
159
|
+
"""
|
|
160
|
+
Save a ModelCard/Scorecard/SegmentScorer as a local MLflow pyfunc model
|
|
161
|
+
directory. `artifact` may be the pydantic instance or its .model_dump()
|
|
162
|
+
dict. Reload with `mlflow.pyfunc.load_model(path)`.
|
|
163
|
+
"""
|
|
164
|
+
card_path = _write_card_json(artifact)
|
|
165
|
+
try:
|
|
166
|
+
mlflow.pyfunc.save_model(
|
|
167
|
+
path=path,
|
|
168
|
+
python_model=DataBubbleScoringModel(),
|
|
169
|
+
artifacts={"card_json": card_path},
|
|
170
|
+
pip_requirements=[f"databubble-scoring=={__version__}"],
|
|
171
|
+
**save_model_kwargs,
|
|
172
|
+
)
|
|
173
|
+
finally:
|
|
174
|
+
os.remove(card_path)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def log_databubble_model(
|
|
178
|
+
artifact: Artifact,
|
|
179
|
+
artifact_path: str,
|
|
180
|
+
registered_model_name: Optional[str] = None,
|
|
181
|
+
**log_model_kwargs,
|
|
182
|
+
):
|
|
183
|
+
"""
|
|
184
|
+
Log a ModelCard/Scorecard/SegmentScorer as an MLflow pyfunc model to the
|
|
185
|
+
active run. Returns the mlflow.models.model.ModelInfo from log_model.
|
|
186
|
+
"""
|
|
187
|
+
card_path = _write_card_json(artifact)
|
|
188
|
+
try:
|
|
189
|
+
return mlflow.pyfunc.log_model(
|
|
190
|
+
artifact_path=artifact_path,
|
|
191
|
+
python_model=DataBubbleScoringModel(),
|
|
192
|
+
artifacts={"card_json": card_path},
|
|
193
|
+
pip_requirements=[f"databubble-scoring=={__version__}"],
|
|
194
|
+
registered_model_name=registered_model_name,
|
|
195
|
+
**log_model_kwargs,
|
|
196
|
+
)
|
|
197
|
+
finally:
|
|
198
|
+
os.remove(card_path)
|