proxyml 0.5.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {proxyml-0.5.0 → proxyml-0.7.0}/PKG-INFO +1 -1
- {proxyml-0.5.0 → proxyml-0.7.0}/pyproject.toml +1 -1
- {proxyml-0.5.0 → proxyml-0.7.0}/src/proxyml/local/challenger.py +137 -14
- {proxyml-0.5.0 → proxyml-0.7.0}/src/proxyml.egg-info/PKG-INFO +1 -1
- {proxyml-0.5.0 → proxyml-0.7.0}/tests/test_local_challenger.py +180 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/LICENSE +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/README.md +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/setup.cfg +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/src/proxyml/__init__.py +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/src/proxyml/client.py +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/src/proxyml/local/__init__.py +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/src/proxyml/schema_builder.py +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/src/proxyml.egg-info/SOURCES.txt +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/src/proxyml.egg-info/dependency_links.txt +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/src/proxyml.egg-info/requires.txt +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/src/proxyml.egg-info/top_level.txt +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/tests/test_client.py +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/tests/test_dependency_boundaries.py +0 -0
- {proxyml-0.5.0 → proxyml-0.7.0}/tests/test_schema_builder.py +0 -0
|
@@ -10,6 +10,8 @@ training target was real ground truth or a black box's predictions.
|
|
|
10
10
|
|
|
11
11
|
from __future__ import annotations
|
|
12
12
|
|
|
13
|
+
import hashlib
|
|
14
|
+
import json
|
|
13
15
|
from dataclasses import dataclass, replace
|
|
14
16
|
from enum import Enum
|
|
15
17
|
from importlib.metadata import version as _pkg_version
|
|
@@ -118,6 +120,36 @@ class TrainedChallenger:
|
|
|
118
120
|
metrics: dict[str, float]
|
|
119
121
|
hyperparameters: dict[str, Any]
|
|
120
122
|
export: SurrogateExport
|
|
123
|
+
n_samples_total: int
|
|
124
|
+
n_samples_dropped_unlabeled: int
|
|
125
|
+
population_note: str
|
|
126
|
+
target_fingerprint: str
|
|
127
|
+
champion_metrics: dict[str, float] | None = None
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _fingerprint_values(values: np.ndarray | list) -> str:
|
|
131
|
+
"""Hash an array of labels, deterministically, without the data ever leaving this process.
|
|
132
|
+
|
|
133
|
+
``.tolist()`` converts numpy scalars to native Python types before
|
|
134
|
+
serializing, so the hash doesn't drift across numpy versions with
|
|
135
|
+
different scalar repr behavior. Order is preserved (not sorted) since
|
|
136
|
+
it encodes row alignment — that's exactly what a "same data?" check
|
|
137
|
+
needs to be sensitive to.
|
|
138
|
+
"""
|
|
139
|
+
canonical = json.dumps(np.asarray(values).tolist())
|
|
140
|
+
return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _population_note(target_name: str, n_total: int, n_labeled: int, n_dropped: int) -> str:
|
|
144
|
+
if n_dropped == 0:
|
|
145
|
+
return f"Evaluated on all {n_total} row(s) — '{target_name}' had no missing values."
|
|
146
|
+
return (
|
|
147
|
+
f"Evaluated on {n_labeled} of {n_total} row(s) with a non-null '{target_name}' value "
|
|
148
|
+
f"({n_dropped} unlabeled row(s) dropped before training/scoring). "
|
|
149
|
+
"Labeled-vs-unlabeled selection may not be random; treat this as a declared "
|
|
150
|
+
"scope limitation on the evaluation population, not a claim about performance "
|
|
151
|
+
"on the full dataset."
|
|
152
|
+
)
|
|
121
153
|
|
|
122
154
|
|
|
123
155
|
def train_challenger(
|
|
@@ -129,6 +161,8 @@ def train_challenger(
|
|
|
129
161
|
feature_names: list[str] | None = None,
|
|
130
162
|
task: Literal["classification", "regression", "auto"] = "auto",
|
|
131
163
|
test_size: float = 0.2,
|
|
164
|
+
target_name: str = "target",
|
|
165
|
+
champion_predictions: np.ndarray | list | None = None,
|
|
132
166
|
) -> TrainedChallenger:
|
|
133
167
|
"""Train a linear challenger model on ``df`` against ``target``, locally.
|
|
134
168
|
|
|
@@ -142,6 +176,15 @@ def train_challenger(
|
|
|
142
176
|
can be compared with the same ``proxyml_core.export.predict_from_export``
|
|
143
177
|
arithmetic.
|
|
144
178
|
|
|
179
|
+
Rows where ``target`` is missing (NaN/None) are dropped before training,
|
|
180
|
+
the CV split, and champion scoring — never silently included. The drop
|
|
181
|
+
count and a human-readable scope-limitation note are recorded on the
|
|
182
|
+
result (``n_samples_total``, ``n_samples_dropped_unlabeled``,
|
|
183
|
+
``population_note``). If ``champion_predictions`` is given, it must have
|
|
184
|
+
one entry per row of ``df``/``target`` (same order) so the identical rows
|
|
185
|
+
are dropped from both sides — champion and challenger are always
|
|
186
|
+
evaluated on the same labeled population, never on different ones.
|
|
187
|
+
|
|
145
188
|
Args:
|
|
146
189
|
df: samples to train on, one column per schema feature.
|
|
147
190
|
target: the value to predict for each row of ``df`` — ground-truth
|
|
@@ -151,7 +194,36 @@ def train_challenger(
|
|
|
151
194
|
feature_names: subset of ``schema.features`` to train on; omit for all.
|
|
152
195
|
task: "classification", "regression", or "auto" to infer from ``target``.
|
|
153
196
|
test_size: fraction of data held out to compute fidelity metrics.
|
|
197
|
+
target_name: human-readable name for ``target``, used in
|
|
198
|
+
``population_note`` (e.g. the column name, if known).
|
|
199
|
+
champion_predictions: a champion model's predictions, one per row of
|
|
200
|
+
``df``/``target`` (same order). If given, scored via
|
|
201
|
+
``score_champion()`` against the same (row-dropped) ``target``,
|
|
202
|
+
and the result is attached as ``TrainedChallenger.champion_metrics``.
|
|
154
203
|
"""
|
|
204
|
+
target_arr = np.asarray(target)
|
|
205
|
+
target_fingerprint = _fingerprint_values(target_arr)
|
|
206
|
+
if champion_predictions is not None and len(champion_predictions) != len(target_arr):
|
|
207
|
+
raise ValueError(
|
|
208
|
+
f"champion_predictions must have one entry per row of target "
|
|
209
|
+
f"({len(target_arr)} rows, got {len(champion_predictions)}) — same order — so "
|
|
210
|
+
f"rows with a missing {target_name!r} value can be dropped from both the "
|
|
211
|
+
f"challenger and the champion, keeping them evaluated on the same population."
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
labeled_mask = ~pd.isna(target_arr)
|
|
215
|
+
n_total = len(target_arr)
|
|
216
|
+
n_labeled = int(labeled_mask.sum())
|
|
217
|
+
n_dropped = n_total - n_labeled
|
|
218
|
+
if n_labeled == 0:
|
|
219
|
+
raise ValueError(f"All {n_total} row(s) have a missing {target_name!r} value; nothing to train on")
|
|
220
|
+
|
|
221
|
+
df = df.iloc[labeled_mask].reset_index(drop=True)
|
|
222
|
+
target_arr = target_arr[labeled_mask]
|
|
223
|
+
champion_predictions_labeled = (
|
|
224
|
+
np.asarray(champion_predictions)[labeled_mask] if champion_predictions is not None else None
|
|
225
|
+
)
|
|
226
|
+
|
|
155
227
|
rung = LADDERS[complexity]
|
|
156
228
|
|
|
157
229
|
features: list[Feature] = schema.features
|
|
@@ -161,7 +233,7 @@ def train_challenger(
|
|
|
161
233
|
col_order = [f.name for f in features]
|
|
162
234
|
|
|
163
235
|
X = df[col_order].to_numpy(dtype=object)
|
|
164
|
-
y =
|
|
236
|
+
y = target_arr
|
|
165
237
|
|
|
166
238
|
if task == "auto":
|
|
167
239
|
classification = is_classification(y)
|
|
@@ -174,6 +246,10 @@ def train_challenger(
|
|
|
174
246
|
if classification:
|
|
175
247
|
y = binarize_if_probabilities(y)
|
|
176
248
|
|
|
249
|
+
champion_metrics = None
|
|
250
|
+
if champion_predictions_labeled is not None:
|
|
251
|
+
champion_metrics = score_champion(y, champion_predictions_labeled, task=resolved_task)
|
|
252
|
+
|
|
177
253
|
preprocessor = build_preprocessor(features)
|
|
178
254
|
estimator = rung.build_classifier() if classification else rung.build_regressor()
|
|
179
255
|
pipeline = Pipeline(steps=[("preprocessor", preprocessor), ("estimator", estimator)])
|
|
@@ -202,6 +278,11 @@ def train_challenger(
|
|
|
202
278
|
metrics=metrics,
|
|
203
279
|
hyperparameters=hyperparameters,
|
|
204
280
|
export=export,
|
|
281
|
+
n_samples_total=n_total,
|
|
282
|
+
n_samples_dropped_unlabeled=n_dropped,
|
|
283
|
+
population_note=_population_note(target_name, n_total, n_labeled, n_dropped),
|
|
284
|
+
target_fingerprint=target_fingerprint,
|
|
285
|
+
champion_metrics=champion_metrics,
|
|
205
286
|
)
|
|
206
287
|
|
|
207
288
|
|
|
@@ -223,6 +304,13 @@ def score_champion(
|
|
|
223
304
|
|
|
224
305
|
Returns ``{"f1":..., "accuracy":...}`` for classification or ``{"r2":...}``
|
|
225
306
|
for regression — the same shape as ``TrainedChallenger.metrics``.
|
|
307
|
+
|
|
308
|
+
If you're calling this decoupled from ``train_challenger()`` (i.e. not
|
|
309
|
+
via its ``champion_predictions=`` param), pass the same ``labels`` you
|
|
310
|
+
used here as ``champion_labels=`` to ``to_challenger_upload()`` — that
|
|
311
|
+
lets the upload endpoint confirm the challenger and champion were
|
|
312
|
+
actually scored on the same data, catching an accidental mismatched
|
|
313
|
+
file before it silently produces a misleading comparison.
|
|
226
314
|
"""
|
|
227
315
|
return score_predictions(np.asarray(labels), np.asarray(predictions), task=task)
|
|
228
316
|
|
|
@@ -230,8 +318,9 @@ def score_champion(
|
|
|
230
318
|
def to_challenger_upload(
|
|
231
319
|
result: TrainedChallenger,
|
|
232
320
|
*,
|
|
233
|
-
n_samples: int,
|
|
321
|
+
n_samples: int | None = None,
|
|
234
322
|
champion_metrics: dict[str, float] | None = None,
|
|
323
|
+
champion_labels: np.ndarray | list | None = None,
|
|
235
324
|
sdk_version: str | None = None,
|
|
236
325
|
proxyml_core_version: str | None = None,
|
|
237
326
|
) -> dict[str, Any]:
|
|
@@ -245,27 +334,42 @@ def to_challenger_upload(
|
|
|
245
334
|
``json.dump`` — upload it either by POSTing it directly, or by saving it
|
|
246
335
|
to a file and using the dashboard's "Upload challenger" button.
|
|
247
336
|
|
|
248
|
-
``champion_metrics`` is optional: pass ``None`` (the default) to
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
``
|
|
252
|
-
|
|
253
|
-
|
|
337
|
+
``champion_metrics`` is optional: pass ``None`` (the default) to fall
|
|
338
|
+
back to ``result.champion_metrics`` (populated automatically if you
|
|
339
|
+
passed ``champion_predictions`` to ``train_challenger()``/
|
|
340
|
+
``train_auto_challenger()``) — or, if that's also ``None``, to get a
|
|
341
|
+
self-contained export of the challenger alone, e.g. to save/share it
|
|
342
|
+
before you have a champion to compare against. The upload endpoint
|
|
343
|
+
itself still requires ``champion_metrics`` at upload time; this function
|
|
344
|
+
just doesn't force you to have it up front.
|
|
254
345
|
|
|
255
346
|
Args:
|
|
256
347
|
result: output of ``train_challenger()``/``train_auto_challenger()``.
|
|
257
348
|
n_samples: size of the evaluation set both ``result.metrics`` and
|
|
258
|
-
``champion_metrics`` were scored on.
|
|
259
|
-
``
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
the only one who actually knows this number.
|
|
349
|
+
``champion_metrics`` were scored on. Defaults to
|
|
350
|
+
``result.n_samples_total - result.n_samples_dropped_unlabeled``
|
|
351
|
+
(the labeled-row count) — override only if you scored on some
|
|
352
|
+
other population.
|
|
263
353
|
champion_metrics: the champion's real-world performance, from
|
|
264
354
|
``score_champion()`` — same metric keys as ``result.metrics``.
|
|
265
|
-
|
|
355
|
+
Defaults to ``result.champion_metrics``.
|
|
356
|
+
champion_labels: the ``labels`` array you passed to a standalone
|
|
357
|
+
``score_champion()`` call, if ``champion_metrics`` didn't come
|
|
358
|
+
from ``train_challenger()``'s internal ``champion_predictions=``
|
|
359
|
+
path. Used only to compute ``champion_data_fingerprint`` — the
|
|
360
|
+
labels themselves are never included in the payload. If
|
|
361
|
+
``champion_metrics`` resolves from ``result.champion_metrics``
|
|
362
|
+
instead, the fingerprint defaults to ``result.target_fingerprint``
|
|
363
|
+
(guaranteed identical, since that internal path scores against
|
|
364
|
+
the exact same data).
|
|
266
365
|
sdk_version: defaults to the installed ``proxyml`` version.
|
|
267
366
|
proxyml_core_version: defaults to the installed ``proxyml-core`` version.
|
|
268
367
|
"""
|
|
368
|
+
if n_samples is None:
|
|
369
|
+
n_samples = result.n_samples_total - result.n_samples_dropped_unlabeled
|
|
370
|
+
used_internal_champion_metrics = champion_metrics is None
|
|
371
|
+
if champion_metrics is None:
|
|
372
|
+
champion_metrics = result.champion_metrics
|
|
269
373
|
if sdk_version is None:
|
|
270
374
|
sdk_version = _pkg_version("proxyml")
|
|
271
375
|
if proxyml_core_version is None:
|
|
@@ -275,12 +379,20 @@ def to_challenger_upload(
|
|
|
275
379
|
"export": result.export.to_dict(),
|
|
276
380
|
"challenger_metrics": result.metrics,
|
|
277
381
|
"n_samples": n_samples,
|
|
382
|
+
"n_samples_total": result.n_samples_total,
|
|
383
|
+
"n_samples_dropped_unlabeled": result.n_samples_dropped_unlabeled,
|
|
384
|
+
"population_note": result.population_note,
|
|
278
385
|
"complexity": result.complexity.value,
|
|
279
386
|
"sdk_version": sdk_version,
|
|
280
387
|
"proxyml_core_version": proxyml_core_version,
|
|
281
388
|
}
|
|
282
389
|
if champion_metrics is not None:
|
|
283
390
|
payload["champion_metrics"] = champion_metrics
|
|
391
|
+
payload["challenger_data_fingerprint"] = result.target_fingerprint
|
|
392
|
+
if champion_labels is not None:
|
|
393
|
+
payload["champion_data_fingerprint"] = _fingerprint_values(champion_labels)
|
|
394
|
+
elif used_internal_champion_metrics:
|
|
395
|
+
payload["champion_data_fingerprint"] = result.target_fingerprint
|
|
284
396
|
return payload
|
|
285
397
|
|
|
286
398
|
|
|
@@ -293,6 +405,7 @@ def train_auto_challenger(
|
|
|
293
405
|
feature_names: list[str] | None = None,
|
|
294
406
|
task: Literal["classification", "regression", "auto"] = "auto",
|
|
295
407
|
test_size: float = 0.2,
|
|
408
|
+
champion_predictions: np.ndarray | list | None = None,
|
|
296
409
|
) -> TrainedChallenger:
|
|
297
410
|
"""Load data, infer a schema, and train a linear challenger in one call.
|
|
298
411
|
|
|
@@ -302,6 +415,12 @@ def train_auto_challenger(
|
|
|
302
415
|
remains overridable; this does not search across ``LADDERS`` to find the
|
|
303
416
|
best-fitting rung.
|
|
304
417
|
|
|
418
|
+
Rows with a missing ``target_col`` value are dropped before training and
|
|
419
|
+
champion scoring — see ``train_challenger()`` for details. Schema
|
|
420
|
+
inference (feature means/stds/categories) still runs over every row,
|
|
421
|
+
including ones later dropped for a missing target — only training and
|
|
422
|
+
evaluation are restricted to the labeled subset.
|
|
423
|
+
|
|
305
424
|
Args:
|
|
306
425
|
data: a CSV path, or an already-loaded DataFrame containing both the
|
|
307
426
|
feature columns and ``target_col``.
|
|
@@ -312,6 +431,8 @@ def train_auto_challenger(
|
|
|
312
431
|
feature_names: subset of feature columns to train on; omit for all.
|
|
313
432
|
task: "classification", "regression", or "auto" to infer from ``target_col``.
|
|
314
433
|
test_size: fraction of data held out to compute fidelity metrics.
|
|
434
|
+
champion_predictions: a champion model's predictions, one per row of
|
|
435
|
+
``data`` (same order) — see ``train_challenger()``.
|
|
315
436
|
"""
|
|
316
437
|
df = data if isinstance(data, pd.DataFrame) else pd.read_csv(data)
|
|
317
438
|
target = df[target_col]
|
|
@@ -326,4 +447,6 @@ def train_auto_challenger(
|
|
|
326
447
|
feature_names=feature_names,
|
|
327
448
|
task=task,
|
|
328
449
|
test_size=test_size,
|
|
450
|
+
target_name=target_col,
|
|
451
|
+
champion_predictions=champion_predictions,
|
|
329
452
|
)
|
|
@@ -286,6 +286,108 @@ def test_to_challenger_upload_payload_is_json_serializable():
|
|
|
286
286
|
json.dumps(payload) # must not raise
|
|
287
287
|
|
|
288
288
|
|
|
289
|
+
def _labeled_df_with_nan_target(n=200, n_nan=20, seed=20):
|
|
290
|
+
df = _labeled_df(n=n, seed=seed)
|
|
291
|
+
df["approved"] = df["approved"].astype(float)
|
|
292
|
+
df.loc[df.index[:n_nan], "approved"] = np.nan
|
|
293
|
+
return df
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def test_nan_target_rows_are_dropped_and_counted():
|
|
297
|
+
df = _labeled_df_with_nan_target(n=200, n_nan=20)
|
|
298
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
299
|
+
|
|
300
|
+
assert result.n_samples_total == 200
|
|
301
|
+
assert result.n_samples_dropped_unlabeled == 20
|
|
302
|
+
assert "20" in result.population_note
|
|
303
|
+
assert "180" in result.population_note
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def test_no_nan_targets_reports_zero_dropped():
|
|
307
|
+
df = _labeled_df(seed=21)
|
|
308
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
309
|
+
|
|
310
|
+
assert result.n_samples_total == len(df)
|
|
311
|
+
assert result.n_samples_dropped_unlabeled == 0
|
|
312
|
+
assert "no missing values" in result.population_note
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def test_all_nan_target_raises():
|
|
316
|
+
df = _labeled_df(n=20, seed=22)
|
|
317
|
+
df["approved"] = np.nan
|
|
318
|
+
with pytest.raises(ValueError, match="missing"):
|
|
319
|
+
train_auto_challenger(df, "approved", task="classification")
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def test_champion_predictions_wrong_length_raises():
|
|
323
|
+
df = _labeled_df(seed=23)
|
|
324
|
+
with pytest.raises(ValueError, match="one entry per row"):
|
|
325
|
+
train_auto_challenger(
|
|
326
|
+
df, "approved", task="classification", champion_predictions=[True, False]
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def test_champion_predictions_scored_only_on_labeled_rows():
|
|
331
|
+
# Champion predictions mirror the (possibly-NaN) target itself, except on
|
|
332
|
+
# rows that get dropped as unlabeled, where they're deliberately wrong.
|
|
333
|
+
# If those rows leaked into scoring, champion accuracy would come in
|
|
334
|
+
# under 1.0 instead of exactly 1.0 — proving the shared-drop guarantee,
|
|
335
|
+
# not just that nothing crashes.
|
|
336
|
+
df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=24)
|
|
337
|
+
champion_predictions = [False if pd.isna(v) else v for v in df["approved"]]
|
|
338
|
+
|
|
339
|
+
result = train_auto_challenger(
|
|
340
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
assert result.champion_metrics is not None
|
|
344
|
+
assert result.champion_metrics["accuracy"] == 1.0
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def test_champion_predictions_not_given_leaves_champion_metrics_none():
|
|
348
|
+
df = _labeled_df(seed=25)
|
|
349
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
350
|
+
assert result.champion_metrics is None
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def test_to_challenger_upload_defaults_n_samples_from_result():
|
|
354
|
+
df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=26)
|
|
355
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
356
|
+
|
|
357
|
+
payload = to_challenger_upload(result)
|
|
358
|
+
|
|
359
|
+
assert payload["n_samples"] == 180
|
|
360
|
+
assert payload["n_samples_total"] == 200
|
|
361
|
+
assert payload["n_samples_dropped_unlabeled"] == 20
|
|
362
|
+
assert payload["population_note"] == result.population_note
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def test_to_challenger_upload_defaults_champion_metrics_from_result():
|
|
366
|
+
df = _labeled_df(seed=27)
|
|
367
|
+
champion_predictions = df["approved"].tolist()
|
|
368
|
+
result = train_auto_challenger(
|
|
369
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
payload = to_challenger_upload(result)
|
|
373
|
+
|
|
374
|
+
assert payload["champion_metrics"] == result.champion_metrics
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def test_to_challenger_upload_explicit_args_override_result_defaults():
|
|
378
|
+
df = _labeled_df(seed=28)
|
|
379
|
+
champion_predictions = df["approved"].tolist()
|
|
380
|
+
result = train_auto_challenger(
|
|
381
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
382
|
+
)
|
|
383
|
+
|
|
384
|
+
override_metrics = {"f1": 0.1, "accuracy": 0.1}
|
|
385
|
+
payload = to_challenger_upload(result, n_samples=999, champion_metrics=override_metrics)
|
|
386
|
+
|
|
387
|
+
assert payload["n_samples"] == 999
|
|
388
|
+
assert payload["champion_metrics"] == override_metrics
|
|
389
|
+
|
|
390
|
+
|
|
289
391
|
def test_train_auto_challenger_passes_immutable_cols_to_get_schema():
|
|
290
392
|
from unittest.mock import patch
|
|
291
393
|
|
|
@@ -300,3 +402,81 @@ def test_train_auto_challenger_passes_immutable_cols_to_get_schema():
|
|
|
300
402
|
called_kwargs = mock_get_schema.call_args.kwargs
|
|
301
403
|
assert list(called_df.columns) == list(features_df.columns)
|
|
302
404
|
assert called_kwargs["immutable_cols"] == ["age"]
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def test_target_fingerprint_is_deterministic_for_identical_data():
|
|
408
|
+
df = _labeled_df(seed=29)
|
|
409
|
+
result_a = train_auto_challenger(df, "approved", task="classification")
|
|
410
|
+
result_b = train_auto_challenger(df, "approved", task="classification")
|
|
411
|
+
|
|
412
|
+
assert result_a.target_fingerprint == result_b.target_fingerprint
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def test_target_fingerprint_differs_for_different_data():
|
|
416
|
+
df_a = _labeled_df(seed=29)
|
|
417
|
+
df_b = _labeled_df(seed=30)
|
|
418
|
+
result_a = train_auto_challenger(df_a, "approved", task="classification")
|
|
419
|
+
result_b = train_auto_challenger(df_b, "approved", task="classification")
|
|
420
|
+
|
|
421
|
+
assert result_a.target_fingerprint != result_b.target_fingerprint
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def test_to_challenger_upload_includes_matching_fingerprints_on_internal_champion_path():
|
|
425
|
+
df = _labeled_df(seed=31)
|
|
426
|
+
champion_predictions = df["approved"].tolist()
|
|
427
|
+
result = train_auto_challenger(
|
|
428
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
429
|
+
)
|
|
430
|
+
|
|
431
|
+
payload = to_challenger_upload(result)
|
|
432
|
+
|
|
433
|
+
assert payload["challenger_data_fingerprint"] == result.target_fingerprint
|
|
434
|
+
assert payload["champion_data_fingerprint"] == result.target_fingerprint
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def test_to_challenger_upload_champion_labels_fingerprint_for_decoupled_path():
|
|
438
|
+
df = _labeled_df(seed=32)
|
|
439
|
+
target = df["approved"]
|
|
440
|
+
result = train_challenger(df, target, _schema(), task="classification")
|
|
441
|
+
champion_metrics = score_champion(target, target, task="classification")
|
|
442
|
+
|
|
443
|
+
payload = to_challenger_upload(result, champion_metrics=champion_metrics, champion_labels=target)
|
|
444
|
+
|
|
445
|
+
assert payload["challenger_data_fingerprint"] == result.target_fingerprint
|
|
446
|
+
assert payload["champion_data_fingerprint"] == result.target_fingerprint
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def test_to_challenger_upload_champion_labels_fingerprint_differs_for_different_data():
|
|
450
|
+
df = _labeled_df(seed=33)
|
|
451
|
+
target = df["approved"]
|
|
452
|
+
result = train_challenger(df, target, _schema(), task="classification")
|
|
453
|
+
champion_metrics = score_champion(target, target, task="classification")
|
|
454
|
+
other_labels = ~target
|
|
455
|
+
|
|
456
|
+
payload = to_challenger_upload(
|
|
457
|
+
result, champion_metrics=champion_metrics, champion_labels=other_labels
|
|
458
|
+
)
|
|
459
|
+
|
|
460
|
+
assert payload["challenger_data_fingerprint"] != payload["champion_data_fingerprint"]
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def test_to_challenger_upload_omits_champion_fingerprint_without_labels_or_internal_path():
|
|
464
|
+
df = _labeled_df(seed=34)
|
|
465
|
+
target = df["approved"]
|
|
466
|
+
result = train_challenger(df, target, _schema(), task="classification")
|
|
467
|
+
champion_metrics = score_champion(target, target, task="classification")
|
|
468
|
+
|
|
469
|
+
payload = to_challenger_upload(result, champion_metrics=champion_metrics)
|
|
470
|
+
|
|
471
|
+
assert payload["challenger_data_fingerprint"] == result.target_fingerprint
|
|
472
|
+
assert "champion_data_fingerprint" not in payload
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def test_to_challenger_upload_without_champion_metrics_omits_fingerprints():
|
|
476
|
+
df = _labeled_df(seed=35)
|
|
477
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
478
|
+
|
|
479
|
+
payload = to_challenger_upload(result)
|
|
480
|
+
|
|
481
|
+
assert "challenger_data_fingerprint" not in payload
|
|
482
|
+
assert "champion_data_fingerprint" not in payload
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|