databubble-scoring 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,9 @@
1
+ """
2
+ databubble-scoring — pure, offline scoring/replay logic for DataBubble
3
+ model cards, scorecards, and segment scorers. numpy/pandas/scipy/pydantic
4
+ only; mlflow_pyfunc.py (mlflow support) is not imported here since it is an
5
+ optional extra — import it explicitly if you have `databubble-scoring[mlflow]`
6
+ installed.
7
+ """
8
+
9
+ __version__ = "0.1.0"
@@ -0,0 +1,196 @@
1
+ """
2
+ Pure scoring core shared by classification and segmentation. This is a
3
+ SUBSET of the platform's skills/models/classifier.py — only the score-time
4
+ path lives here (zero sklearn dependency, confirmed by direct read of
5
+ score_observations, which reconstructs the linear function from stored
6
+ coef_/intercept_/scaler_mean/scaler_scale lists using nothing but numpy).
7
+
8
+ fit_classifier / fit_classifier_for_graph / check_segment_labels (sklearn-
9
+ heavy, fit-time only) stay in the platform's own classifier.py, which
10
+ re-imports FittedPipeline/score_observations/_sigmoid/_softmax from here so
11
+ exactly one definition of FittedPipeline exists repo-wide.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ from collections import Counter
16
+ from typing import Optional
17
+
18
+ import numpy as np
19
+ import pandas as pd
20
+ from pydantic import BaseModel
21
+
22
+
23
+ class FittedPipeline(BaseModel):
24
+ """
25
+ Serialisable representation of a fitted classifier.
26
+ No sklearn objects — only lists and primitives.
27
+ score_observations reconstructs the linear function from these fields.
28
+ """
29
+ model_type: str # "binary" | "multinomial"
30
+ feature_names: list[str] # column order expected at score time
31
+ scaler_mean: list[float]
32
+ scaler_scale: list[float]
33
+ classes_: list # class labels in order (as strings)
34
+ coef_: list[list[float]] # shape (n_classes_or_1, n_features)
35
+ intercept_: list[float] # shape (n_classes_or_1,)
36
+ threshold: float # decision threshold (binary: cost-adjusted; multinomial: argmax)
37
+ dummy_mapping: Optional[dict] = None # for plain-English labels at score time
38
+ imputer_medians: Optional[list[float]] = None # median per feature for NaN imputation at score time
39
+
40
+ class Config:
41
+ arbitrary_types_allowed = True
42
+
43
+
44
+ class ScoringOutput(BaseModel):
45
+ """
46
+ Structurally identical to the platform's skills.formatter.SkillOutput
47
+ (skill_name/summary/findings/warnings/recommendations/chapter_ref) —
48
+ this package has no platform dependency, so score_observations returns
49
+ this local model instead. .findings["labels"] etc. access is unchanged
50
+ for every caller.
51
+ """
52
+ skill_name: str
53
+ summary: str
54
+ findings: dict = {}
55
+ warnings: list[str] = []
56
+ recommendations: list[str] = []
57
+ chapter_ref: Optional[str] = None
58
+
59
+
60
+ def _sigmoid(z: np.ndarray) -> np.ndarray:
61
+ return 1.0 / (1.0 + np.exp(-np.clip(z, -500, 500)))
62
+
63
+
64
+ def _softmax(z: np.ndarray) -> np.ndarray:
65
+ # z shape: (n_obs, n_classes)
66
+ z_shifted = z - z.max(axis=1, keepdims=True)
67
+ exp_z = np.exp(z_shifted)
68
+ return exp_z / exp_z.sum(axis=1, keepdims=True)
69
+
70
+
71
+ def score_observations(
72
+ new_X: pd.DataFrame,
73
+ pipeline: FittedPipeline,
74
+ low_confidence_threshold: float = 0.60,
75
+ ) -> ScoringOutput:
76
+ """
77
+ Score new observations using a FittedPipeline.
78
+ Reconstructs the linear function from stored coefficients — no sklearn needed.
79
+
80
+ Returns findings:
81
+ labels : list of predicted class labels
82
+ probabilities : list of dicts {class: prob} per observation
83
+ low_confidence : list of bool
84
+ low_confidence_count: int
85
+ class_distribution : dict {label: count}
86
+ """
87
+ missing = [f for f in pipeline.feature_names if f not in new_X.columns]
88
+ if missing:
89
+ return ScoringOutput(
90
+ skill_name="score_observations",
91
+ summary="Scoring failed — input columns do not match training features.",
92
+ findings={
93
+ "error": f"Missing columns: {missing}",
94
+ "expected": pipeline.feature_names,
95
+ "received": list(new_X.columns),
96
+ },
97
+ warnings=[f"Input is missing {len(missing)} column(s) from the training feature set."],
98
+ recommendations=["Ensure new data passes through the same feature preparation step as training data."],
99
+ chapter_ref="Chapter 19",
100
+ )
101
+
102
+ X_mat = new_X[pipeline.feature_names].values.astype(float)
103
+
104
+ if pipeline.imputer_medians is not None and np.isnan(X_mat).any():
105
+ _medians = np.array(pipeline.imputer_medians)
106
+ _nan_mask = np.isnan(X_mat)
107
+ X_mat[_nan_mask] = np.take(_medians, np.where(_nan_mask)[1])
108
+
109
+ mean_ = np.array(pipeline.scaler_mean)
110
+ scale_ = np.array(pipeline.scaler_scale)
111
+ X_scaled = (X_mat - mean_) / scale_
112
+
113
+ coef_ = np.array(pipeline.coef_) # (1, n_feat) binary | (n_cls, n_feat) multinomial
114
+ intercept_ = np.array(pipeline.intercept_)
115
+
116
+ if pipeline.model_type == "binary":
117
+ logits = X_scaled @ coef_[0] + intercept_[0] # (n_obs,)
118
+ probs_pos = _sigmoid(logits) # (n_obs,)
119
+ prob_matrix = np.column_stack([1 - probs_pos, probs_pos]) # (n_obs, 2)
120
+ else:
121
+ logits = X_scaled @ coef_.T + intercept_ # (n_obs, n_classes)
122
+ prob_matrix = _softmax(logits) # (n_obs, n_classes)
123
+
124
+ classes = pipeline.classes_
125
+ max_probs = prob_matrix.max(axis=1)
126
+ predicted_indices = prob_matrix.argmax(axis=1)
127
+ labels = [classes[i] for i in predicted_indices]
128
+
129
+ if pipeline.model_type == "binary" and pipeline.threshold != 0.50:
130
+ labels = [
131
+ classes[1] if prob_matrix[i, 1] >= pipeline.threshold else classes[0]
132
+ for i in range(len(labels))
133
+ ]
134
+
135
+ low_confidence = [bool(p < low_confidence_threshold) for p in max_probs]
136
+ low_confidence_count = sum(low_confidence)
137
+
138
+ probabilities = [
139
+ {cls: round(float(prob_matrix[i, j]), 4) for j, cls in enumerate(classes)}
140
+ for i in range(len(labels))
141
+ ]
142
+
143
+ class_dist = {str(k): v for k, v in Counter(labels).items()}
144
+ for _k in (str(c) for c in pipeline.classes_):
145
+ if _k not in class_dist:
146
+ class_dist[_k] = 0
147
+
148
+ n_obs = len(labels)
149
+ lc_pct = low_confidence_count / n_obs if n_obs > 0 else 0
150
+
151
+ warnings_out = []
152
+ if lc_pct > 0.20:
153
+ warnings_out.append(
154
+ f"{low_confidence_count} of {n_obs} observations ({lc_pct:.1%}) scored below "
155
+ f"the {low_confidence_threshold:.0%} confidence threshold. "
156
+ "These assignments should be treated as uncertain."
157
+ )
158
+
159
+ confidence_implausibly_low = False
160
+ if n_obs >= 100 and lc_pct < 0.005:
161
+ confidence_implausibly_low = True
162
+ warnings_out.append(
163
+ f"Only {lc_pct:.1%} of observations fell below the 0.60 confidence threshold. "
164
+ f"With real-world data this is statistically implausible and may indicate overfitting or "
165
+ f"data leakage — validate on genuinely unseen data."
166
+ )
167
+
168
+ summary = (
169
+ f"{n_obs} observations scored. "
170
+ f"Class distribution: {class_dist}. "
171
+ f"{low_confidence_count} flagged as low confidence (<{low_confidence_threshold:.0%})."
172
+ )
173
+
174
+ return ScoringOutput(
175
+ skill_name="score_observations",
176
+ summary=summary,
177
+ findings={
178
+ "labels": labels,
179
+ "probabilities": probabilities,
180
+ "low_confidence": low_confidence,
181
+ "low_confidence_count": low_confidence_count,
182
+ "low_confidence_pct": round(lc_pct, 4),
183
+ "confidence_implausibly_low": confidence_implausibly_low,
184
+ "class_distribution": class_dist,
185
+ "n_observations": n_obs,
186
+ "threshold_used": pipeline.threshold,
187
+ "low_confidence_threshold": low_confidence_threshold,
188
+ },
189
+ warnings=warnings_out,
190
+ recommendations=(
191
+ ["Review low-confidence observations manually before acting on segment assignments."]
192
+ if lc_pct > 0.20 else
193
+ ["Segment assignments are ready to use."]
194
+ ),
195
+ chapter_ref="Chapter 19",
196
+ )
@@ -0,0 +1,231 @@
1
+ """
2
+ DataBubble — Drift Monitoring for Exported ModelCards
3
+
4
+ Compares a new batch of rows against a ModelCard's fit-time
5
+ feature_snapshot (per-column mean/std/skewness or proportion, added by the
6
+ platform's linear_regression.py::run_regression at fit time — see
7
+ model_card.py::ColumnSnapshot). Point-summary only, never raw reference
8
+ rows — a one-sample t-test (mean) and two-sample z-tests (skewness,
9
+ proportion) are the right tests here, not a full two-sample PSI/KS, since
10
+ no second full distribution exists to compare against.
11
+
12
+ apply_recipe (model_card.py) rebuilds new rows into the exact fit-time
13
+ model-matrix shape — reused here for column alignment and for its existing
14
+ missing-column/unseen-level validation, rather than duplicating either.
15
+
16
+ The proportion two-sample z-test is hand-rolled with scipy.stats.norm
17
+ (standard pooled-variance two-proportion z-test) rather than taking
18
+ statsmodels as a runtime dependency, keeping this package to
19
+ numpy/pandas/scipy/pydantic only. Cross-checked against
20
+ statsmodels.stats.proportion.proportions_ztest in this package's own test
21
+ suite (a dev-only dependency, never a runtime one).
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import math
26
+ import warnings
27
+ from typing import Any, Literal, Optional
28
+
29
+ import numpy as np
30
+ import pandas as pd
31
+ from scipy import stats
32
+ from pydantic import BaseModel
33
+
34
+ try:
35
+ from .model_card import ModelCard, apply_recipe
36
+ except ImportError:
37
+ from model_card import ModelCard, apply_recipe
38
+
39
+ DRIFT_ALPHA = 0.05 # two-tailed significance threshold
40
+ SKEW_DRIFT_Z = 1.96 # ~95% two-tailed z-critical
41
+
42
+
43
+ class ColumnDriftResult(BaseModel):
44
+ column: str
45
+ kind: Literal["continuous", "proportion"]
46
+ n_observed: int
47
+ reference: dict[str, Any]
48
+ observed: dict[str, Any]
49
+ drifted: bool
50
+ detail: str
51
+
52
+
53
+ class DriftResult(BaseModel):
54
+ applicable: bool
55
+ reason: Optional[str] = None
56
+ n_features_checked: int = 0
57
+ n_features_drifted: int = 0
58
+ drift_detected: bool = False
59
+ per_feature: list[ColumnDriftResult] = []
60
+ interpretation: str = ""
61
+
62
+
63
+ def _skewness_se(n: int) -> float:
64
+ """
65
+ Fisher's standard error of the sample skewness coefficient. Undefined
66
+ for n < 3 — callers must guard that themselves.
67
+ """
68
+ return math.sqrt((6 * n * (n - 1)) / ((n - 2) * (n + 1) * (n + 3)))
69
+
70
+
71
+ def _two_proportion_ztest(count: list[int], nobs: list[int]) -> tuple[float, float]:
72
+ """
73
+ Standard pooled-variance two-proportion z-test, two-sided. Hand-rolled
74
+ replacement for statsmodels.stats.proportion.proportions_ztest(count,
75
+ nobs) with its default prop_var (pooled) behavior — cross-checked
76
+ against the real thing in test_drift_monitoring.py.
77
+ Returns (z_statistic, p_value). p_value is nan when the pooled variance
78
+ is exactly 0 (both samples identical, e.g. all-0 or all-1) — caller
79
+ already has its own "std=0" fallback for a nan p-value.
80
+ """
81
+ count1, count2 = count
82
+ n1, n2 = nobs
83
+ p1, p2 = count1 / n1, count2 / n2
84
+ p_pool = (count1 + count2) / (n1 + n2)
85
+ var = p_pool * (1 - p_pool) * (1.0 / n1 + 1.0 / n2)
86
+ if var <= 0:
87
+ return float("nan"), float("nan")
88
+ z = (p1 - p2) / math.sqrt(var)
89
+ p_value = 2 * float(stats.norm.sf(abs(z)))
90
+ return float(z), p_value
91
+
92
+
93
+ def _check_continuous_drift(column: str, snapshot: dict, observed: pd.Series, alpha: float) -> ColumnDriftResult:
94
+ values = observed.dropna()
95
+ n_obs = len(values)
96
+ notes: list[str] = []
97
+ drifted = False
98
+
99
+ ref_mean = snapshot.get("mean")
100
+ if ref_mean is not None and n_obs >= 2:
101
+ try:
102
+ with warnings.catch_warnings():
103
+ warnings.simplefilter("ignore")
104
+ _, p_value = stats.ttest_1samp(values, popmean=ref_mean)
105
+ if math.isnan(p_value):
106
+ shifted = float(values.mean()) != ref_mean
107
+ p_note = "std=0"
108
+ else:
109
+ shifted = p_value < alpha
110
+ p_note = f"p={p_value:.4g}"
111
+ except Exception:
112
+ shifted, p_note = False, ""
113
+ if shifted:
114
+ drifted = True
115
+ new_mean = float(values.mean())
116
+ direction = "up" if new_mean > ref_mean else "down"
117
+ notes.append(f"mean shifted {direction} from {ref_mean:.4g} to {new_mean:.4g} ({p_note})")
118
+
119
+ ref_skew = snapshot.get("skewness")
120
+ ref_n = snapshot.get("n")
121
+ if ref_skew is not None and ref_n is not None and n_obs >= 3 and ref_n >= 3:
122
+ try:
123
+ new_skew = float(values.skew())
124
+ se_diff = math.sqrt(_skewness_se(n_obs) ** 2 + _skewness_se(ref_n) ** 2)
125
+ z = (new_skew - ref_skew) / se_diff if se_diff > 0 else 0.0
126
+ if abs(z) >= SKEW_DRIFT_Z:
127
+ drifted = True
128
+ notes.append(f"skewness shifted from {ref_skew:.4g} to {new_skew:.4g} (z={z:.2f})")
129
+ except Exception:
130
+ pass
131
+
132
+ return ColumnDriftResult(
133
+ column=column, kind="continuous", n_observed=n_obs,
134
+ reference=snapshot,
135
+ observed={"mean": float(values.mean()) if n_obs else None,
136
+ "std": float(values.std(ddof=1)) if n_obs >= 2 else None,
137
+ "skewness": float(values.skew()) if n_obs >= 3 else None},
138
+ drifted=drifted,
139
+ detail="; ".join(notes) if notes else "No significant shift detected.",
140
+ )
141
+
142
+
143
+ def _check_proportion_drift(column: str, snapshot: dict, observed: pd.Series, alpha: float) -> ColumnDriftResult:
144
+ values = observed.dropna()
145
+ n_obs = len(values)
146
+ ref_proportion = snapshot.get("proportion")
147
+ ref_n = snapshot.get("n")
148
+ new_proportion = float(values.mean()) if n_obs else None
149
+ drifted = False
150
+ detail = "No significant shift detected."
151
+
152
+ if ref_proportion is not None and ref_n and n_obs >= 1:
153
+ new_count = int(values.sum())
154
+ ref_count = round(ref_proportion * ref_n)
155
+ try:
156
+ _stat, p_value = _two_proportion_ztest(
157
+ count=[new_count, ref_count], nobs=[n_obs, ref_n],
158
+ )
159
+ if math.isnan(p_value):
160
+ drifted = new_proportion != ref_proportion
161
+ detail = "std=0"
162
+ elif p_value < alpha:
163
+ drifted = True
164
+ direction = "up" if new_proportion > ref_proportion else "down"
165
+ detail = (
166
+ f"proportion shifted {direction} from {ref_proportion:.4g} to "
167
+ f"{new_proportion:.4g} (p={p_value:.4g})"
168
+ )
169
+ except Exception:
170
+ pass
171
+
172
+ return ColumnDriftResult(
173
+ column=column, kind="proportion", n_observed=n_obs,
174
+ reference=snapshot, observed={"proportion": new_proportion},
175
+ drifted=drifted, detail=detail,
176
+ )
177
+
178
+
179
+ def diagnose_feature_drift(card: ModelCard, new_rows: pd.DataFrame, alpha: float = DRIFT_ALPHA) -> DriftResult:
180
+ """
181
+ Compares new_rows against card.feature_snapshot (the training-time
182
+ per-column reference captured at fit time). Raises RecipeError (same
183
+ as predict_from_card) when new_rows can't be rebuilt into the card's
184
+ model matrix — missing columns or unseen categorical levels.
185
+ """
186
+ if not card.feature_snapshot:
187
+ return DriftResult(
188
+ applicable=False,
189
+ reason=(
190
+ "This card predates drift monitoring (schema < 1.2) — "
191
+ "re-export the model to enable it."
192
+ ),
193
+ )
194
+
195
+ matrix = apply_recipe(new_rows, card.recipe)
196
+
197
+ per_feature: list[ColumnDriftResult] = []
198
+ for column, snap in card.feature_snapshot.items():
199
+ snapshot = snap if isinstance(snap, dict) else snap.model_dump()
200
+ if column not in matrix.columns:
201
+ continue
202
+ if snapshot.get("kind") == "proportion":
203
+ per_feature.append(_check_proportion_drift(column, snapshot, matrix[column], alpha))
204
+ else:
205
+ per_feature.append(_check_continuous_drift(column, snapshot, matrix[column], alpha))
206
+
207
+ n_drifted = sum(1 for pf in per_feature if pf.drifted)
208
+ drift_detected = n_drifted > 0
209
+
210
+ if drift_detected:
211
+ drifted_cols = [pf.column for pf in per_feature if pf.drifted]
212
+ interpretation = (
213
+ f"{n_drifted} of {len(per_feature)} feature(s) shifted significantly "
214
+ f"from their training-time distribution: {', '.join(drifted_cols)}. "
215
+ f"Predictions from this model may no longer be reliable for this "
216
+ f"population — review before continuing to use it."
217
+ )
218
+ else:
219
+ interpretation = (
220
+ f"No significant shift detected across {len(per_feature)} feature(s) "
221
+ f"checked against their training-time distribution."
222
+ )
223
+
224
+ return DriftResult(
225
+ applicable=True,
226
+ n_features_checked=len(per_feature),
227
+ n_features_drifted=n_drifted,
228
+ drift_detected=drift_detected,
229
+ per_feature=per_feature,
230
+ interpretation=interpretation,
231
+ )
@@ -0,0 +1,198 @@
1
+ """
2
+ DataBubble — MLflow pyfunc artifact for offline scoring.
3
+
4
+ Wraps a ModelCard / Scorecard / SegmentScorer as a standard MLflow pyfunc
5
+ model: `predict()` runs the exact same predict_from_card / score_from_scorecard
6
+ / score_from_segment_scorer path the DataBubble platform itself uses at
7
+ export time, so a locally-loaded artifact and the hosted API can never
8
+ silently diverge. No network call at inference.
9
+
10
+ Optional module: only import this if you have `databubble-scoring[mlflow]`
11
+ installed. databubble_scoring/__init__.py never imports it.
12
+
13
+ ModelCardBundle (multi-group cards) is not supported in this version —
14
+ export a single group's ModelCard instead.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import json
19
+ import os
20
+ import tempfile
21
+ from typing import Optional, Union
22
+
23
+ import pandas as pd
24
+ from pydantic import BaseModel
25
+
26
+ try:
27
+ import mlflow.pyfunc
28
+ except ImportError as e:
29
+ raise ImportError(
30
+ "mlflow is required to use databubble_scoring.mlflow_pyfunc — install "
31
+ "with `pip install databubble-scoring[mlflow]`."
32
+ ) from e
33
+
34
+ try:
35
+ from . import __version__
36
+ from .model_card import ModelCard, RecipeError
37
+ from .predict import predict_from_card, PredictionResult
38
+ from .scorecard import Scorecard, score_from_scorecard
39
+ from .segment_scorer import SegmentScorer, score_from_segment_scorer
40
+ except ImportError:
41
+ from databubble_scoring import __version__
42
+ from model_card import ModelCard, RecipeError
43
+ from predict import predict_from_card, PredictionResult
44
+ from scorecard import Scorecard, score_from_scorecard
45
+ from segment_scorer import SegmentScorer, score_from_segment_scorer
46
+
47
+ Artifact = Union[ModelCard, Scorecard, SegmentScorer, dict]
48
+
49
+
50
+ def _detect_kind(d: dict) -> str:
51
+ kind = d.get("kind")
52
+ if kind == "classification_scorecard":
53
+ return "classification_scorecard"
54
+ if kind == "segment_scorer":
55
+ return "segment_scorer"
56
+ if kind == "bundle":
57
+ raise ValueError(
58
+ "ModelCardBundle (multi-group cards) is not supported by the MLflow "
59
+ "pyfunc wrapper — export and save a single group's ModelCard instead."
60
+ )
61
+ if kind is not None:
62
+ raise ValueError(f"Unrecognised card 'kind': {kind!r}")
63
+ # ModelCard carries no `kind` field at all — its required fields are what
64
+ # distinguish it from a malformed payload.
65
+ if {"outcome", "intercept", "terms", "recipe", "provenance"} <= d.keys():
66
+ return "model_card"
67
+ raise ValueError(
68
+ "Could not determine artifact kind from card.json — expected a ModelCard "
69
+ "(no 'kind' field), a Scorecard ('kind': 'classification_scorecard'), or "
70
+ "a SegmentScorer ('kind': 'segment_scorer')."
71
+ )
72
+
73
+
74
+ def _artifact_dict(artifact: Artifact) -> dict:
75
+ if isinstance(artifact, BaseModel):
76
+ return artifact.model_dump()
77
+ if isinstance(artifact, dict):
78
+ return artifact
79
+ raise TypeError(f"artifact must be a ModelCard/Scorecard/SegmentScorer or dict, got {type(artifact)}")
80
+
81
+
82
+ def _prediction_result_to_frame(result: PredictionResult) -> pd.DataFrame:
83
+ data = {"prediction": result.predictions}
84
+ for col in ("ci_lower", "ci_upper", "pi_lower", "pi_upper"):
85
+ val = getattr(result, col)
86
+ if val is not None:
87
+ data[col] = val
88
+ return pd.DataFrame(data)
89
+
90
+
91
+ def _scoring_output_to_frame(findings: dict, kind: str) -> pd.DataFrame:
92
+ if findings.get("labels") is None:
93
+ raise ValueError(
94
+ f"Scoring failed: {findings.get('error', 'input columns did not match the fitted pipeline')} "
95
+ f"(expected columns: {findings.get('expected')}, received: {findings.get('received')})"
96
+ )
97
+ data = {"label": findings["labels"]}
98
+ df = pd.DataFrame(data)
99
+ probs = findings.get("probabilities") or []
100
+ if probs:
101
+ df = pd.concat([df, pd.DataFrame(probs).add_prefix("probability_")], axis=1)
102
+ if "low_confidence" in findings:
103
+ df["low_confidence"] = findings["low_confidence"]
104
+ if kind == "segment_scorer":
105
+ assignments = findings.get("assignments") or []
106
+ if assignments:
107
+ df["segment"] = [a["segment"] for a in assignments]
108
+ return df
109
+
110
+
111
+ class DataBubbleScoringModel(mlflow.pyfunc.PythonModel):
112
+ """
113
+ Generic pyfunc wrapper for any DataBubble scoring artifact. The concrete
114
+ kind (ModelCard / Scorecard / SegmentScorer) is auto-detected from
115
+ card.json at load time — one model class covers all three, since the
116
+ scoring call each dispatches to is already the single source of truth
117
+ (predict_from_card / score_from_scorecard / score_from_segment_scorer).
118
+ """
119
+
120
+ def load_context(self, context):
121
+ with open(context.artifacts["card_json"]) as f:
122
+ d = json.load(f)
123
+ self._kind = _detect_kind(d)
124
+ if self._kind == "model_card":
125
+ self._artifact = ModelCard(**d)
126
+ elif self._kind == "classification_scorecard":
127
+ self._artifact = Scorecard(**d)
128
+ else:
129
+ self._artifact = SegmentScorer(**d)
130
+
131
+ def predict(self, context, model_input, params=None):
132
+ if not isinstance(model_input, pd.DataFrame):
133
+ model_input = pd.DataFrame(model_input)
134
+
135
+ if self._kind == "model_card":
136
+ result = predict_from_card(self._artifact, model_input)
137
+ return _prediction_result_to_frame(result)
138
+
139
+ try:
140
+ if self._kind == "classification_scorecard":
141
+ scoring_result = score_from_scorecard(self._artifact, model_input)
142
+ else:
143
+ scoring_result = score_from_segment_scorer(self._artifact, model_input)
144
+ except RecipeError as e:
145
+ raise ValueError(f"Recipe replay failed: {e}") from e
146
+ return _scoring_output_to_frame(scoring_result.findings, self._kind)
147
+
148
+
149
+ def _write_card_json(artifact: Artifact) -> str:
150
+ d = _artifact_dict(artifact)
151
+ _detect_kind(d) # validate (raises early on a bundle/malformed payload) before any file/mlflow I/O
152
+ fd, path = tempfile.mkstemp(suffix=".json", prefix="databubble_card_")
153
+ with os.fdopen(fd, "w") as f:
154
+ json.dump(d, f)
155
+ return path
156
+
157
+
158
+ def save_databubble_model(artifact: Artifact, path: str, **save_model_kwargs) -> None:
159
+ """
160
+ Save a ModelCard/Scorecard/SegmentScorer as a local MLflow pyfunc model
161
+ directory. `artifact` may be the pydantic instance or its .model_dump()
162
+ dict. Reload with `mlflow.pyfunc.load_model(path)`.
163
+ """
164
+ card_path = _write_card_json(artifact)
165
+ try:
166
+ mlflow.pyfunc.save_model(
167
+ path=path,
168
+ python_model=DataBubbleScoringModel(),
169
+ artifacts={"card_json": card_path},
170
+ pip_requirements=[f"databubble-scoring=={__version__}"],
171
+ **save_model_kwargs,
172
+ )
173
+ finally:
174
+ os.remove(card_path)
175
+
176
+
177
+ def log_databubble_model(
178
+ artifact: Artifact,
179
+ artifact_path: str,
180
+ registered_model_name: Optional[str] = None,
181
+ **log_model_kwargs,
182
+ ):
183
+ """
184
+ Log a ModelCard/Scorecard/SegmentScorer as an MLflow pyfunc model to the
185
+ active run. Returns the mlflow.models.model.ModelInfo from log_model.
186
+ """
187
+ card_path = _write_card_json(artifact)
188
+ try:
189
+ return mlflow.pyfunc.log_model(
190
+ artifact_path=artifact_path,
191
+ python_model=DataBubbleScoringModel(),
192
+ artifacts={"card_json": card_path},
193
+ pip_requirements=[f"databubble-scoring=={__version__}"],
194
+ registered_model_name=registered_model_name,
195
+ **log_model_kwargs,
196
+ )
197
+ finally:
198
+ os.remove(card_path)