proxyml 0.4.1__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {proxyml-0.4.1 → proxyml-0.6.0}/PKG-INFO +1 -1
- {proxyml-0.4.1 → proxyml-0.6.0}/pyproject.toml +1 -1
- {proxyml-0.4.1 → proxyml-0.6.0}/src/proxyml/local/__init__.py +2 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/src/proxyml/local/challenger.py +143 -1
- {proxyml-0.4.1 → proxyml-0.6.0}/src/proxyml.egg-info/PKG-INFO +1 -1
- {proxyml-0.4.1 → proxyml-0.6.0}/tests/test_local_challenger.py +191 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/LICENSE +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/README.md +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/setup.cfg +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/src/proxyml/__init__.py +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/src/proxyml/client.py +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/src/proxyml/schema_builder.py +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/src/proxyml.egg-info/SOURCES.txt +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/src/proxyml.egg-info/dependency_links.txt +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/src/proxyml.egg-info/requires.txt +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/src/proxyml.egg-info/top_level.txt +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/tests/test_client.py +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/tests/test_dependency_boundaries.py +0 -0
- {proxyml-0.4.1 → proxyml-0.6.0}/tests/test_schema_builder.py +0 -0
|
@@ -11,6 +11,7 @@ from proxyml.local.challenger import (
|
|
|
11
11
|
Rung,
|
|
12
12
|
TrainedChallenger,
|
|
13
13
|
score_champion,
|
|
14
|
+
to_challenger_upload,
|
|
14
15
|
train_auto_challenger,
|
|
15
16
|
train_challenger,
|
|
16
17
|
)
|
|
@@ -19,6 +20,7 @@ __all__ = [
|
|
|
19
20
|
"train_challenger",
|
|
20
21
|
"train_auto_challenger",
|
|
21
22
|
"score_champion",
|
|
23
|
+
"to_challenger_upload",
|
|
22
24
|
"Complexity",
|
|
23
25
|
"Rung",
|
|
24
26
|
"TrainedChallenger",
|
|
@@ -12,6 +12,7 @@ from __future__ import annotations
|
|
|
12
12
|
|
|
13
13
|
from dataclasses import dataclass, replace
|
|
14
14
|
from enum import Enum
|
|
15
|
+
from importlib.metadata import version as _pkg_version
|
|
15
16
|
from pathlib import Path
|
|
16
17
|
from typing import Any, Callable, Literal
|
|
17
18
|
|
|
@@ -113,9 +114,26 @@ LADDERS: dict[Complexity, Rung] = {
|
|
|
113
114
|
class TrainedChallenger:
|
|
114
115
|
pipeline: Pipeline
|
|
115
116
|
task: Literal["classification", "regression"]
|
|
117
|
+
complexity: Complexity
|
|
116
118
|
metrics: dict[str, float]
|
|
117
119
|
hyperparameters: dict[str, Any]
|
|
118
120
|
export: SurrogateExport
|
|
121
|
+
n_samples_total: int
|
|
122
|
+
n_samples_dropped_unlabeled: int
|
|
123
|
+
population_note: str
|
|
124
|
+
champion_metrics: dict[str, float] | None = None
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _population_note(target_name: str, n_total: int, n_labeled: int, n_dropped: int) -> str:
|
|
128
|
+
if n_dropped == 0:
|
|
129
|
+
return f"Evaluated on all {n_total} row(s) — '{target_name}' had no missing values."
|
|
130
|
+
return (
|
|
131
|
+
f"Evaluated on {n_labeled} of {n_total} row(s) with a non-null '{target_name}' value "
|
|
132
|
+
f"({n_dropped} unlabeled row(s) dropped before training/scoring). "
|
|
133
|
+
"Labeled-vs-unlabeled selection may not be random; treat this as a declared "
|
|
134
|
+
"scope limitation on the evaluation population, not a claim about performance "
|
|
135
|
+
"on the full dataset."
|
|
136
|
+
)
|
|
119
137
|
|
|
120
138
|
|
|
121
139
|
def train_challenger(
|
|
@@ -127,6 +145,8 @@ def train_challenger(
|
|
|
127
145
|
feature_names: list[str] | None = None,
|
|
128
146
|
task: Literal["classification", "regression", "auto"] = "auto",
|
|
129
147
|
test_size: float = 0.2,
|
|
148
|
+
target_name: str = "target",
|
|
149
|
+
champion_predictions: np.ndarray | list | None = None,
|
|
130
150
|
) -> TrainedChallenger:
|
|
131
151
|
"""Train a linear challenger model on ``df`` against ``target``, locally.
|
|
132
152
|
|
|
@@ -140,6 +160,15 @@ def train_challenger(
|
|
|
140
160
|
can be compared with the same ``proxyml_core.export.predict_from_export``
|
|
141
161
|
arithmetic.
|
|
142
162
|
|
|
163
|
+
Rows where ``target`` is missing (NaN/None) are dropped before training,
|
|
164
|
+
the CV split, and champion scoring — never silently included. The drop
|
|
165
|
+
count and a human-readable scope-limitation note are recorded on the
|
|
166
|
+
result (``n_samples_total``, ``n_samples_dropped_unlabeled``,
|
|
167
|
+
``population_note``). If ``champion_predictions`` is given, it must have
|
|
168
|
+
one entry per row of ``df``/``target`` (same order) so the identical rows
|
|
169
|
+
are dropped from both sides — champion and challenger are always
|
|
170
|
+
evaluated on the same labeled population, never on different ones.
|
|
171
|
+
|
|
143
172
|
Args:
|
|
144
173
|
df: samples to train on, one column per schema feature.
|
|
145
174
|
target: the value to predict for each row of ``df`` — ground-truth
|
|
@@ -149,7 +178,35 @@ def train_challenger(
|
|
|
149
178
|
feature_names: subset of ``schema.features`` to train on; omit for all.
|
|
150
179
|
task: "classification", "regression", or "auto" to infer from ``target``.
|
|
151
180
|
test_size: fraction of data held out to compute fidelity metrics.
|
|
181
|
+
target_name: human-readable name for ``target``, used in
|
|
182
|
+
``population_note`` (e.g. the column name, if known).
|
|
183
|
+
champion_predictions: a champion model's predictions, one per row of
|
|
184
|
+
``df``/``target`` (same order). If given, scored via
|
|
185
|
+
``score_champion()`` against the same (row-dropped) ``target``,
|
|
186
|
+
and the result is attached as ``TrainedChallenger.champion_metrics``.
|
|
152
187
|
"""
|
|
188
|
+
target_arr = np.asarray(target)
|
|
189
|
+
if champion_predictions is not None and len(champion_predictions) != len(target_arr):
|
|
190
|
+
raise ValueError(
|
|
191
|
+
f"champion_predictions must have one entry per row of target "
|
|
192
|
+
f"({len(target_arr)} rows, got {len(champion_predictions)}) — same order — so "
|
|
193
|
+
f"rows with a missing {target_name!r} value can be dropped from both the "
|
|
194
|
+
f"challenger and the champion, keeping them evaluated on the same population."
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
labeled_mask = ~pd.isna(target_arr)
|
|
198
|
+
n_total = len(target_arr)
|
|
199
|
+
n_labeled = int(labeled_mask.sum())
|
|
200
|
+
n_dropped = n_total - n_labeled
|
|
201
|
+
if n_labeled == 0:
|
|
202
|
+
raise ValueError(f"All {n_total} row(s) have a missing {target_name!r} value; nothing to train on")
|
|
203
|
+
|
|
204
|
+
df = df.iloc[labeled_mask].reset_index(drop=True)
|
|
205
|
+
target_arr = target_arr[labeled_mask]
|
|
206
|
+
champion_predictions_labeled = (
|
|
207
|
+
np.asarray(champion_predictions)[labeled_mask] if champion_predictions is not None else None
|
|
208
|
+
)
|
|
209
|
+
|
|
153
210
|
rung = LADDERS[complexity]
|
|
154
211
|
|
|
155
212
|
features: list[Feature] = schema.features
|
|
@@ -159,7 +216,7 @@ def train_challenger(
|
|
|
159
216
|
col_order = [f.name for f in features]
|
|
160
217
|
|
|
161
218
|
X = df[col_order].to_numpy(dtype=object)
|
|
162
|
-
y =
|
|
219
|
+
y = target_arr
|
|
163
220
|
|
|
164
221
|
if task == "auto":
|
|
165
222
|
classification = is_classification(y)
|
|
@@ -172,6 +229,10 @@ def train_challenger(
|
|
|
172
229
|
if classification:
|
|
173
230
|
y = binarize_if_probabilities(y)
|
|
174
231
|
|
|
232
|
+
champion_metrics = None
|
|
233
|
+
if champion_predictions_labeled is not None:
|
|
234
|
+
champion_metrics = score_champion(y, champion_predictions_labeled, task=resolved_task)
|
|
235
|
+
|
|
175
236
|
preprocessor = build_preprocessor(features)
|
|
176
237
|
estimator = rung.build_classifier() if classification else rung.build_regressor()
|
|
177
238
|
pipeline = Pipeline(steps=[("preprocessor", preprocessor), ("estimator", estimator)])
|
|
@@ -196,9 +257,14 @@ def train_challenger(
|
|
|
196
257
|
return TrainedChallenger(
|
|
197
258
|
pipeline=pipeline,
|
|
198
259
|
task=resolved_task,
|
|
260
|
+
complexity=complexity,
|
|
199
261
|
metrics=metrics,
|
|
200
262
|
hyperparameters=hyperparameters,
|
|
201
263
|
export=export,
|
|
264
|
+
n_samples_total=n_total,
|
|
265
|
+
n_samples_dropped_unlabeled=n_dropped,
|
|
266
|
+
population_note=_population_note(target_name, n_total, n_labeled, n_dropped),
|
|
267
|
+
champion_metrics=champion_metrics,
|
|
202
268
|
)
|
|
203
269
|
|
|
204
270
|
|
|
@@ -224,6 +290,71 @@ def score_champion(
|
|
|
224
290
|
return score_predictions(np.asarray(labels), np.asarray(predictions), task=task)
|
|
225
291
|
|
|
226
292
|
|
|
293
|
+
def to_challenger_upload(
|
|
294
|
+
result: TrainedChallenger,
|
|
295
|
+
*,
|
|
296
|
+
n_samples: int | None = None,
|
|
297
|
+
champion_metrics: dict[str, float] | None = None,
|
|
298
|
+
sdk_version: str | None = None,
|
|
299
|
+
proxyml_core_version: str | None = None,
|
|
300
|
+
) -> dict[str, Any]:
|
|
301
|
+
"""Assemble the JSON-serializable payload for a challenger upload.
|
|
302
|
+
|
|
303
|
+
Matches the shape ProxyML's dashboard/API expects at
|
|
304
|
+
``POST /app/projects/{id}/challenger`` — handles the mechanical assembly
|
|
305
|
+
(serializing the export, stamping SDK/core versions, converting
|
|
306
|
+
``complexity`` to a plain string) so callers don't have to hand-roll it.
|
|
307
|
+
The result is plain ``dict``/``str``/``float`` data, ready for
|
|
308
|
+
``json.dump`` — upload it either by POSTing it directly, or by saving it
|
|
309
|
+
to a file and using the dashboard's "Upload challenger" button.
|
|
310
|
+
|
|
311
|
+
``champion_metrics`` is optional: pass ``None`` (the default) to fall
|
|
312
|
+
back to ``result.champion_metrics`` (populated automatically if you
|
|
313
|
+
passed ``champion_predictions`` to ``train_challenger()``/
|
|
314
|
+
``train_auto_challenger()``) — or, if that's also ``None``, to get a
|
|
315
|
+
self-contained export of the challenger alone, e.g. to save/share it
|
|
316
|
+
before you have a champion to compare against. The upload endpoint
|
|
317
|
+
itself still requires ``champion_metrics`` at upload time; this function
|
|
318
|
+
just doesn't force you to have it up front.
|
|
319
|
+
|
|
320
|
+
Args:
|
|
321
|
+
result: output of ``train_challenger()``/``train_auto_challenger()``.
|
|
322
|
+
n_samples: size of the evaluation set both ``result.metrics`` and
|
|
323
|
+
``champion_metrics`` were scored on. Defaults to
|
|
324
|
+
``result.n_samples_total - result.n_samples_dropped_unlabeled``
|
|
325
|
+
(the labeled-row count) — override only if you scored on some
|
|
326
|
+
other population.
|
|
327
|
+
champion_metrics: the champion's real-world performance, from
|
|
328
|
+
``score_champion()`` — same metric keys as ``result.metrics``.
|
|
329
|
+
Defaults to ``result.champion_metrics``.
|
|
330
|
+
sdk_version: defaults to the installed ``proxyml`` version.
|
|
331
|
+
proxyml_core_version: defaults to the installed ``proxyml-core`` version.
|
|
332
|
+
"""
|
|
333
|
+
if n_samples is None:
|
|
334
|
+
n_samples = result.n_samples_total - result.n_samples_dropped_unlabeled
|
|
335
|
+
if champion_metrics is None:
|
|
336
|
+
champion_metrics = result.champion_metrics
|
|
337
|
+
if sdk_version is None:
|
|
338
|
+
sdk_version = _pkg_version("proxyml")
|
|
339
|
+
if proxyml_core_version is None:
|
|
340
|
+
proxyml_core_version = _pkg_version("proxyml-core")
|
|
341
|
+
|
|
342
|
+
payload: dict[str, Any] = {
|
|
343
|
+
"export": result.export.to_dict(),
|
|
344
|
+
"challenger_metrics": result.metrics,
|
|
345
|
+
"n_samples": n_samples,
|
|
346
|
+
"n_samples_total": result.n_samples_total,
|
|
347
|
+
"n_samples_dropped_unlabeled": result.n_samples_dropped_unlabeled,
|
|
348
|
+
"population_note": result.population_note,
|
|
349
|
+
"complexity": result.complexity.value,
|
|
350
|
+
"sdk_version": sdk_version,
|
|
351
|
+
"proxyml_core_version": proxyml_core_version,
|
|
352
|
+
}
|
|
353
|
+
if champion_metrics is not None:
|
|
354
|
+
payload["champion_metrics"] = champion_metrics
|
|
355
|
+
return payload
|
|
356
|
+
|
|
357
|
+
|
|
227
358
|
def train_auto_challenger(
|
|
228
359
|
data: str | Path | pd.DataFrame,
|
|
229
360
|
target_col: str,
|
|
@@ -233,6 +364,7 @@ def train_auto_challenger(
|
|
|
233
364
|
feature_names: list[str] | None = None,
|
|
234
365
|
task: Literal["classification", "regression", "auto"] = "auto",
|
|
235
366
|
test_size: float = 0.2,
|
|
367
|
+
champion_predictions: np.ndarray | list | None = None,
|
|
236
368
|
) -> TrainedChallenger:
|
|
237
369
|
"""Load data, infer a schema, and train a linear challenger in one call.
|
|
238
370
|
|
|
@@ -242,6 +374,12 @@ def train_auto_challenger(
|
|
|
242
374
|
remains overridable; this does not search across ``LADDERS`` to find the
|
|
243
375
|
best-fitting rung.
|
|
244
376
|
|
|
377
|
+
Rows with a missing ``target_col`` value are dropped before training and
|
|
378
|
+
champion scoring — see ``train_challenger()`` for details. Schema
|
|
379
|
+
inference (feature means/stds/categories) still runs over every row,
|
|
380
|
+
including ones later dropped for a missing target — only training and
|
|
381
|
+
evaluation are restricted to the labeled subset.
|
|
382
|
+
|
|
245
383
|
Args:
|
|
246
384
|
data: a CSV path, or an already-loaded DataFrame containing both the
|
|
247
385
|
feature columns and ``target_col``.
|
|
@@ -252,6 +390,8 @@ def train_auto_challenger(
|
|
|
252
390
|
feature_names: subset of feature columns to train on; omit for all.
|
|
253
391
|
task: "classification", "regression", or "auto" to infer from ``target_col``.
|
|
254
392
|
test_size: fraction of data held out to compute fidelity metrics.
|
|
393
|
+
champion_predictions: a champion model's predictions, one per row of
|
|
394
|
+
``data`` (same order) — see ``train_challenger()``.
|
|
255
395
|
"""
|
|
256
396
|
df = data if isinstance(data, pd.DataFrame) else pd.read_csv(data)
|
|
257
397
|
target = df[target_col]
|
|
@@ -266,4 +406,6 @@ def train_auto_challenger(
|
|
|
266
406
|
feature_names=feature_names,
|
|
267
407
|
task=task,
|
|
268
408
|
test_size=test_size,
|
|
409
|
+
target_name=target_col,
|
|
410
|
+
champion_predictions=champion_predictions,
|
|
269
411
|
)
|
|
@@ -9,6 +9,7 @@ from proxyml.local import (
|
|
|
9
9
|
LADDERS,
|
|
10
10
|
TrainedChallenger,
|
|
11
11
|
score_champion,
|
|
12
|
+
to_challenger_upload,
|
|
12
13
|
train_auto_challenger,
|
|
13
14
|
train_challenger,
|
|
14
15
|
)
|
|
@@ -197,6 +198,196 @@ def test_score_champion_uses_same_scoring_as_train_challenger():
|
|
|
197
198
|
assert reproduced_metrics["r2"] == pytest.approx(r2_score(target, y_pred))
|
|
198
199
|
|
|
199
200
|
|
|
201
|
+
def test_train_challenger_result_carries_complexity():
|
|
202
|
+
schema = _schema()
|
|
203
|
+
df = _df(seed=13)
|
|
204
|
+
target = df["age"] * 0.5 + df["income"] * 0.0001
|
|
205
|
+
|
|
206
|
+
result = train_challenger(df, target, schema, complexity=Complexity.FLEXIBLE, task="regression")
|
|
207
|
+
|
|
208
|
+
assert result.complexity is Complexity.FLEXIBLE
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def test_train_auto_challenger_result_carries_complexity():
|
|
212
|
+
df = _labeled_df(seed=14)
|
|
213
|
+
result = train_auto_challenger(df, "approved", task="classification", complexity=Complexity.SIMPLE)
|
|
214
|
+
|
|
215
|
+
assert result.complexity is Complexity.SIMPLE
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def test_to_challenger_upload_shape_without_champion_metrics():
|
|
219
|
+
schema = _schema()
|
|
220
|
+
df = _df(seed=15)
|
|
221
|
+
target = df["age"] * 0.5 + df["income"] * 0.0001
|
|
222
|
+
result = train_challenger(df, target, schema, complexity=Complexity.MODERATE, task="regression")
|
|
223
|
+
|
|
224
|
+
payload = to_challenger_upload(result, n_samples=40)
|
|
225
|
+
|
|
226
|
+
assert payload["export"] == result.export.to_dict()
|
|
227
|
+
assert payload["challenger_metrics"] == result.metrics
|
|
228
|
+
assert payload["n_samples"] == 40
|
|
229
|
+
assert payload["complexity"] == "moderate"
|
|
230
|
+
assert "champion_metrics" not in payload
|
|
231
|
+
assert isinstance(payload["sdk_version"], str) and payload["sdk_version"]
|
|
232
|
+
assert isinstance(payload["proxyml_core_version"], str) and payload["proxyml_core_version"]
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def test_to_challenger_upload_includes_champion_metrics_when_given():
|
|
236
|
+
schema = _schema()
|
|
237
|
+
df = _df(seed=16)
|
|
238
|
+
target = df["age"] * 0.5 + df["income"] * 0.0001
|
|
239
|
+
result = train_challenger(df, target, schema, complexity=Complexity.MODERATE, task="regression")
|
|
240
|
+
champion_metrics = score_champion(target, target, task="regression")
|
|
241
|
+
|
|
242
|
+
payload = to_challenger_upload(result, n_samples=40, champion_metrics=champion_metrics)
|
|
243
|
+
|
|
244
|
+
assert payload["champion_metrics"] == champion_metrics
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def test_to_challenger_upload_defaults_versions_to_installed_packages():
|
|
248
|
+
from importlib.metadata import version as pkg_version
|
|
249
|
+
|
|
250
|
+
schema = _schema()
|
|
251
|
+
df = _df(seed=17)
|
|
252
|
+
target = df["age"] * 0.5 + df["income"] * 0.0001
|
|
253
|
+
result = train_challenger(df, target, schema, task="regression")
|
|
254
|
+
|
|
255
|
+
payload = to_challenger_upload(result, n_samples=10)
|
|
256
|
+
|
|
257
|
+
assert payload["sdk_version"] == pkg_version("proxyml")
|
|
258
|
+
assert payload["proxyml_core_version"] == pkg_version("proxyml-core")
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def test_to_challenger_upload_allows_version_override():
|
|
262
|
+
schema = _schema()
|
|
263
|
+
df = _df(seed=18)
|
|
264
|
+
target = df["age"] * 0.5 + df["income"] * 0.0001
|
|
265
|
+
result = train_challenger(df, target, schema, task="regression")
|
|
266
|
+
|
|
267
|
+
payload = to_challenger_upload(
|
|
268
|
+
result, n_samples=10, sdk_version="9.9.9", proxyml_core_version="8.8.8"
|
|
269
|
+
)
|
|
270
|
+
|
|
271
|
+
assert payload["sdk_version"] == "9.9.9"
|
|
272
|
+
assert payload["proxyml_core_version"] == "8.8.8"
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def test_to_challenger_upload_payload_is_json_serializable():
|
|
276
|
+
import json
|
|
277
|
+
|
|
278
|
+
schema = _schema()
|
|
279
|
+
df = _df(seed=19, n=300)
|
|
280
|
+
target = np.where(df["age"] > 50, "senior", "junior")
|
|
281
|
+
result = train_challenger(df, target, schema, task="classification")
|
|
282
|
+
champion_metrics = score_champion(target, target, task="classification")
|
|
283
|
+
|
|
284
|
+
payload = to_challenger_upload(result, n_samples=300, champion_metrics=champion_metrics)
|
|
285
|
+
|
|
286
|
+
json.dumps(payload) # must not raise
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _labeled_df_with_nan_target(n=200, n_nan=20, seed=20):
|
|
290
|
+
df = _labeled_df(n=n, seed=seed)
|
|
291
|
+
df["approved"] = df["approved"].astype(float)
|
|
292
|
+
df.loc[df.index[:n_nan], "approved"] = np.nan
|
|
293
|
+
return df
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def test_nan_target_rows_are_dropped_and_counted():
|
|
297
|
+
df = _labeled_df_with_nan_target(n=200, n_nan=20)
|
|
298
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
299
|
+
|
|
300
|
+
assert result.n_samples_total == 200
|
|
301
|
+
assert result.n_samples_dropped_unlabeled == 20
|
|
302
|
+
assert "20" in result.population_note
|
|
303
|
+
assert "180" in result.population_note
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def test_no_nan_targets_reports_zero_dropped():
|
|
307
|
+
df = _labeled_df(seed=21)
|
|
308
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
309
|
+
|
|
310
|
+
assert result.n_samples_total == len(df)
|
|
311
|
+
assert result.n_samples_dropped_unlabeled == 0
|
|
312
|
+
assert "no missing values" in result.population_note
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def test_all_nan_target_raises():
|
|
316
|
+
df = _labeled_df(n=20, seed=22)
|
|
317
|
+
df["approved"] = np.nan
|
|
318
|
+
with pytest.raises(ValueError, match="missing"):
|
|
319
|
+
train_auto_challenger(df, "approved", task="classification")
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def test_champion_predictions_wrong_length_raises():
|
|
323
|
+
df = _labeled_df(seed=23)
|
|
324
|
+
with pytest.raises(ValueError, match="one entry per row"):
|
|
325
|
+
train_auto_challenger(
|
|
326
|
+
df, "approved", task="classification", champion_predictions=[True, False]
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def test_champion_predictions_scored_only_on_labeled_rows():
|
|
331
|
+
# Champion predictions mirror the (possibly-NaN) target itself, except on
|
|
332
|
+
# rows that get dropped as unlabeled, where they're deliberately wrong.
|
|
333
|
+
# If those rows leaked into scoring, champion accuracy would come in
|
|
334
|
+
# under 1.0 instead of exactly 1.0 — proving the shared-drop guarantee,
|
|
335
|
+
# not just that nothing crashes.
|
|
336
|
+
df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=24)
|
|
337
|
+
champion_predictions = [False if pd.isna(v) else v for v in df["approved"]]
|
|
338
|
+
|
|
339
|
+
result = train_auto_challenger(
|
|
340
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
assert result.champion_metrics is not None
|
|
344
|
+
assert result.champion_metrics["accuracy"] == 1.0
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def test_champion_predictions_not_given_leaves_champion_metrics_none():
|
|
348
|
+
df = _labeled_df(seed=25)
|
|
349
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
350
|
+
assert result.champion_metrics is None
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def test_to_challenger_upload_defaults_n_samples_from_result():
|
|
354
|
+
df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=26)
|
|
355
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
356
|
+
|
|
357
|
+
payload = to_challenger_upload(result)
|
|
358
|
+
|
|
359
|
+
assert payload["n_samples"] == 180
|
|
360
|
+
assert payload["n_samples_total"] == 200
|
|
361
|
+
assert payload["n_samples_dropped_unlabeled"] == 20
|
|
362
|
+
assert payload["population_note"] == result.population_note
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def test_to_challenger_upload_defaults_champion_metrics_from_result():
|
|
366
|
+
df = _labeled_df(seed=27)
|
|
367
|
+
champion_predictions = df["approved"].tolist()
|
|
368
|
+
result = train_auto_challenger(
|
|
369
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
payload = to_challenger_upload(result)
|
|
373
|
+
|
|
374
|
+
assert payload["champion_metrics"] == result.champion_metrics
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def test_to_challenger_upload_explicit_args_override_result_defaults():
|
|
378
|
+
df = _labeled_df(seed=28)
|
|
379
|
+
champion_predictions = df["approved"].tolist()
|
|
380
|
+
result = train_auto_challenger(
|
|
381
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
382
|
+
)
|
|
383
|
+
|
|
384
|
+
override_metrics = {"f1": 0.1, "accuracy": 0.1}
|
|
385
|
+
payload = to_challenger_upload(result, n_samples=999, champion_metrics=override_metrics)
|
|
386
|
+
|
|
387
|
+
assert payload["n_samples"] == 999
|
|
388
|
+
assert payload["champion_metrics"] == override_metrics
|
|
389
|
+
|
|
390
|
+
|
|
200
391
|
def test_train_auto_challenger_passes_immutable_cols_to_get_schema():
|
|
201
392
|
from unittest.mock import patch
|
|
202
393
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|