proxyml 0.5.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {proxyml-0.5.0 → proxyml-0.6.0}/PKG-INFO +1 -1
- {proxyml-0.5.0 → proxyml-0.6.0}/pyproject.toml +1 -1
- {proxyml-0.5.0 → proxyml-0.6.0}/src/proxyml/local/challenger.py +96 -14
- {proxyml-0.5.0 → proxyml-0.6.0}/src/proxyml.egg-info/PKG-INFO +1 -1
- {proxyml-0.5.0 → proxyml-0.6.0}/tests/test_local_challenger.py +102 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/LICENSE +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/README.md +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/setup.cfg +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/src/proxyml/__init__.py +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/src/proxyml/client.py +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/src/proxyml/local/__init__.py +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/src/proxyml/schema_builder.py +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/src/proxyml.egg-info/SOURCES.txt +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/src/proxyml.egg-info/dependency_links.txt +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/src/proxyml.egg-info/requires.txt +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/src/proxyml.egg-info/top_level.txt +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/tests/test_client.py +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/tests/test_dependency_boundaries.py +0 -0
- {proxyml-0.5.0 → proxyml-0.6.0}/tests/test_schema_builder.py +0 -0
|
@@ -118,6 +118,22 @@ class TrainedChallenger:
|
|
|
118
118
|
metrics: dict[str, float]
|
|
119
119
|
hyperparameters: dict[str, Any]
|
|
120
120
|
export: SurrogateExport
|
|
121
|
+
n_samples_total: int
|
|
122
|
+
n_samples_dropped_unlabeled: int
|
|
123
|
+
population_note: str
|
|
124
|
+
champion_metrics: dict[str, float] | None = None
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _population_note(target_name: str, n_total: int, n_labeled: int, n_dropped: int) -> str:
|
|
128
|
+
if n_dropped == 0:
|
|
129
|
+
return f"Evaluated on all {n_total} row(s) — '{target_name}' had no missing values."
|
|
130
|
+
return (
|
|
131
|
+
f"Evaluated on {n_labeled} of {n_total} row(s) with a non-null '{target_name}' value "
|
|
132
|
+
f"({n_dropped} unlabeled row(s) dropped before training/scoring). "
|
|
133
|
+
"Labeled-vs-unlabeled selection may not be random; treat this as a declared "
|
|
134
|
+
"scope limitation on the evaluation population, not a claim about performance "
|
|
135
|
+
"on the full dataset."
|
|
136
|
+
)
|
|
121
137
|
|
|
122
138
|
|
|
123
139
|
def train_challenger(
|
|
@@ -129,6 +145,8 @@ def train_challenger(
|
|
|
129
145
|
feature_names: list[str] | None = None,
|
|
130
146
|
task: Literal["classification", "regression", "auto"] = "auto",
|
|
131
147
|
test_size: float = 0.2,
|
|
148
|
+
target_name: str = "target",
|
|
149
|
+
champion_predictions: np.ndarray | list | None = None,
|
|
132
150
|
) -> TrainedChallenger:
|
|
133
151
|
"""Train a linear challenger model on ``df`` against ``target``, locally.
|
|
134
152
|
|
|
@@ -142,6 +160,15 @@ def train_challenger(
|
|
|
142
160
|
can be compared with the same ``proxyml_core.export.predict_from_export``
|
|
143
161
|
arithmetic.
|
|
144
162
|
|
|
163
|
+
Rows where ``target`` is missing (NaN/None) are dropped before training,
|
|
164
|
+
the CV split, and champion scoring — never silently included. The drop
|
|
165
|
+
count and a human-readable scope-limitation note are recorded on the
|
|
166
|
+
result (``n_samples_total``, ``n_samples_dropped_unlabeled``,
|
|
167
|
+
``population_note``). If ``champion_predictions`` is given, it must have
|
|
168
|
+
one entry per row of ``df``/``target`` (same order) so the identical rows
|
|
169
|
+
are dropped from both sides — champion and challenger are always
|
|
170
|
+
evaluated on the same labeled population, never on different ones.
|
|
171
|
+
|
|
145
172
|
Args:
|
|
146
173
|
df: samples to train on, one column per schema feature.
|
|
147
174
|
target: the value to predict for each row of ``df`` — ground-truth
|
|
@@ -151,7 +178,35 @@ def train_challenger(
|
|
|
151
178
|
feature_names: subset of ``schema.features`` to train on; omit for all.
|
|
152
179
|
task: "classification", "regression", or "auto" to infer from ``target``.
|
|
153
180
|
test_size: fraction of data held out to compute fidelity metrics.
|
|
181
|
+
target_name: human-readable name for ``target``, used in
|
|
182
|
+
``population_note`` (e.g. the column name, if known).
|
|
183
|
+
champion_predictions: a champion model's predictions, one per row of
|
|
184
|
+
``df``/``target`` (same order). If given, scored via
|
|
185
|
+
``score_champion()`` against the same (row-dropped) ``target``,
|
|
186
|
+
and the result is attached as ``TrainedChallenger.champion_metrics``.
|
|
154
187
|
"""
|
|
188
|
+
target_arr = np.asarray(target)
|
|
189
|
+
if champion_predictions is not None and len(champion_predictions) != len(target_arr):
|
|
190
|
+
raise ValueError(
|
|
191
|
+
f"champion_predictions must have one entry per row of target "
|
|
192
|
+
f"({len(target_arr)} rows, got {len(champion_predictions)}) — same order — so "
|
|
193
|
+
f"rows with a missing {target_name!r} value can be dropped from both the "
|
|
194
|
+
f"challenger and the champion, keeping them evaluated on the same population."
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
labeled_mask = ~pd.isna(target_arr)
|
|
198
|
+
n_total = len(target_arr)
|
|
199
|
+
n_labeled = int(labeled_mask.sum())
|
|
200
|
+
n_dropped = n_total - n_labeled
|
|
201
|
+
if n_labeled == 0:
|
|
202
|
+
raise ValueError(f"All {n_total} row(s) have a missing {target_name!r} value; nothing to train on")
|
|
203
|
+
|
|
204
|
+
df = df.iloc[labeled_mask].reset_index(drop=True)
|
|
205
|
+
target_arr = target_arr[labeled_mask]
|
|
206
|
+
champion_predictions_labeled = (
|
|
207
|
+
np.asarray(champion_predictions)[labeled_mask] if champion_predictions is not None else None
|
|
208
|
+
)
|
|
209
|
+
|
|
155
210
|
rung = LADDERS[complexity]
|
|
156
211
|
|
|
157
212
|
features: list[Feature] = schema.features
|
|
@@ -161,7 +216,7 @@ def train_challenger(
|
|
|
161
216
|
col_order = [f.name for f in features]
|
|
162
217
|
|
|
163
218
|
X = df[col_order].to_numpy(dtype=object)
|
|
164
|
-
y =
|
|
219
|
+
y = target_arr
|
|
165
220
|
|
|
166
221
|
if task == "auto":
|
|
167
222
|
classification = is_classification(y)
|
|
@@ -174,6 +229,10 @@ def train_challenger(
|
|
|
174
229
|
if classification:
|
|
175
230
|
y = binarize_if_probabilities(y)
|
|
176
231
|
|
|
232
|
+
champion_metrics = None
|
|
233
|
+
if champion_predictions_labeled is not None:
|
|
234
|
+
champion_metrics = score_champion(y, champion_predictions_labeled, task=resolved_task)
|
|
235
|
+
|
|
177
236
|
preprocessor = build_preprocessor(features)
|
|
178
237
|
estimator = rung.build_classifier() if classification else rung.build_regressor()
|
|
179
238
|
pipeline = Pipeline(steps=[("preprocessor", preprocessor), ("estimator", estimator)])
|
|
@@ -202,6 +261,10 @@ def train_challenger(
|
|
|
202
261
|
metrics=metrics,
|
|
203
262
|
hyperparameters=hyperparameters,
|
|
204
263
|
export=export,
|
|
264
|
+
n_samples_total=n_total,
|
|
265
|
+
n_samples_dropped_unlabeled=n_dropped,
|
|
266
|
+
population_note=_population_note(target_name, n_total, n_labeled, n_dropped),
|
|
267
|
+
champion_metrics=champion_metrics,
|
|
205
268
|
)
|
|
206
269
|
|
|
207
270
|
|
|
@@ -230,7 +293,7 @@ def score_champion(
|
|
|
230
293
|
def to_challenger_upload(
|
|
231
294
|
result: TrainedChallenger,
|
|
232
295
|
*,
|
|
233
|
-
n_samples: int,
|
|
296
|
+
n_samples: int | None = None,
|
|
234
297
|
champion_metrics: dict[str, float] | None = None,
|
|
235
298
|
sdk_version: str | None = None,
|
|
236
299
|
proxyml_core_version: str | None = None,
|
|
@@ -245,27 +308,32 @@ def to_challenger_upload(
|
|
|
245
308
|
``json.dump`` — upload it either by POSTing it directly, or by saving it
|
|
246
309
|
to a file and using the dashboard's "Upload challenger" button.
|
|
247
310
|
|
|
248
|
-
``champion_metrics`` is optional: pass ``None`` (the default) to
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
``
|
|
252
|
-
|
|
253
|
-
|
|
311
|
+
``champion_metrics`` is optional: pass ``None`` (the default) to fall
|
|
312
|
+
back to ``result.champion_metrics`` (populated automatically if you
|
|
313
|
+
passed ``champion_predictions`` to ``train_challenger()``/
|
|
314
|
+
``train_auto_challenger()``) — or, if that's also ``None``, to get a
|
|
315
|
+
self-contained export of the challenger alone, e.g. to save/share it
|
|
316
|
+
before you have a champion to compare against. The upload endpoint
|
|
317
|
+
itself still requires ``champion_metrics`` at upload time; this function
|
|
318
|
+
just doesn't force you to have it up front.
|
|
254
319
|
|
|
255
320
|
Args:
|
|
256
321
|
result: output of ``train_challenger()``/``train_auto_challenger()``.
|
|
257
322
|
n_samples: size of the evaluation set both ``result.metrics`` and
|
|
258
|
-
``champion_metrics`` were scored on.
|
|
259
|
-
``
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
the only one who actually knows this number.
|
|
323
|
+
``champion_metrics`` were scored on. Defaults to
|
|
324
|
+
``result.n_samples_total - result.n_samples_dropped_unlabeled``
|
|
325
|
+
(the labeled-row count) — override only if you scored on some
|
|
326
|
+
other population.
|
|
263
327
|
champion_metrics: the champion's real-world performance, from
|
|
264
328
|
``score_champion()`` — same metric keys as ``result.metrics``.
|
|
265
|
-
|
|
329
|
+
Defaults to ``result.champion_metrics``.
|
|
266
330
|
sdk_version: defaults to the installed ``proxyml`` version.
|
|
267
331
|
proxyml_core_version: defaults to the installed ``proxyml-core`` version.
|
|
268
332
|
"""
|
|
333
|
+
if n_samples is None:
|
|
334
|
+
n_samples = result.n_samples_total - result.n_samples_dropped_unlabeled
|
|
335
|
+
if champion_metrics is None:
|
|
336
|
+
champion_metrics = result.champion_metrics
|
|
269
337
|
if sdk_version is None:
|
|
270
338
|
sdk_version = _pkg_version("proxyml")
|
|
271
339
|
if proxyml_core_version is None:
|
|
@@ -275,6 +343,9 @@ def to_challenger_upload(
|
|
|
275
343
|
"export": result.export.to_dict(),
|
|
276
344
|
"challenger_metrics": result.metrics,
|
|
277
345
|
"n_samples": n_samples,
|
|
346
|
+
"n_samples_total": result.n_samples_total,
|
|
347
|
+
"n_samples_dropped_unlabeled": result.n_samples_dropped_unlabeled,
|
|
348
|
+
"population_note": result.population_note,
|
|
278
349
|
"complexity": result.complexity.value,
|
|
279
350
|
"sdk_version": sdk_version,
|
|
280
351
|
"proxyml_core_version": proxyml_core_version,
|
|
@@ -293,6 +364,7 @@ def train_auto_challenger(
|
|
|
293
364
|
feature_names: list[str] | None = None,
|
|
294
365
|
task: Literal["classification", "regression", "auto"] = "auto",
|
|
295
366
|
test_size: float = 0.2,
|
|
367
|
+
champion_predictions: np.ndarray | list | None = None,
|
|
296
368
|
) -> TrainedChallenger:
|
|
297
369
|
"""Load data, infer a schema, and train a linear challenger in one call.
|
|
298
370
|
|
|
@@ -302,6 +374,12 @@ def train_auto_challenger(
|
|
|
302
374
|
remains overridable; this does not search across ``LADDERS`` to find the
|
|
303
375
|
best-fitting rung.
|
|
304
376
|
|
|
377
|
+
Rows with a missing ``target_col`` value are dropped before training and
|
|
378
|
+
champion scoring — see ``train_challenger()`` for details. Schema
|
|
379
|
+
inference (feature means/stds/categories) still runs over every row,
|
|
380
|
+
including ones later dropped for a missing target — only training and
|
|
381
|
+
evaluation are restricted to the labeled subset.
|
|
382
|
+
|
|
305
383
|
Args:
|
|
306
384
|
data: a CSV path, or an already-loaded DataFrame containing both the
|
|
307
385
|
feature columns and ``target_col``.
|
|
@@ -312,6 +390,8 @@ def train_auto_challenger(
|
|
|
312
390
|
feature_names: subset of feature columns to train on; omit for all.
|
|
313
391
|
task: "classification", "regression", or "auto" to infer from ``target_col``.
|
|
314
392
|
test_size: fraction of data held out to compute fidelity metrics.
|
|
393
|
+
champion_predictions: a champion model's predictions, one per row of
|
|
394
|
+
``data`` (same order) — see ``train_challenger()``.
|
|
315
395
|
"""
|
|
316
396
|
df = data if isinstance(data, pd.DataFrame) else pd.read_csv(data)
|
|
317
397
|
target = df[target_col]
|
|
@@ -326,4 +406,6 @@ def train_auto_challenger(
|
|
|
326
406
|
feature_names=feature_names,
|
|
327
407
|
task=task,
|
|
328
408
|
test_size=test_size,
|
|
409
|
+
target_name=target_col,
|
|
410
|
+
champion_predictions=champion_predictions,
|
|
329
411
|
)
|
|
@@ -286,6 +286,108 @@ def test_to_challenger_upload_payload_is_json_serializable():
|
|
|
286
286
|
json.dumps(payload) # must not raise
|
|
287
287
|
|
|
288
288
|
|
|
289
|
+
def _labeled_df_with_nan_target(n=200, n_nan=20, seed=20):
|
|
290
|
+
df = _labeled_df(n=n, seed=seed)
|
|
291
|
+
df["approved"] = df["approved"].astype(float)
|
|
292
|
+
df.loc[df.index[:n_nan], "approved"] = np.nan
|
|
293
|
+
return df
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def test_nan_target_rows_are_dropped_and_counted():
|
|
297
|
+
df = _labeled_df_with_nan_target(n=200, n_nan=20)
|
|
298
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
299
|
+
|
|
300
|
+
assert result.n_samples_total == 200
|
|
301
|
+
assert result.n_samples_dropped_unlabeled == 20
|
|
302
|
+
assert "20" in result.population_note
|
|
303
|
+
assert "180" in result.population_note
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def test_no_nan_targets_reports_zero_dropped():
|
|
307
|
+
df = _labeled_df(seed=21)
|
|
308
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
309
|
+
|
|
310
|
+
assert result.n_samples_total == len(df)
|
|
311
|
+
assert result.n_samples_dropped_unlabeled == 0
|
|
312
|
+
assert "no missing values" in result.population_note
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def test_all_nan_target_raises():
|
|
316
|
+
df = _labeled_df(n=20, seed=22)
|
|
317
|
+
df["approved"] = np.nan
|
|
318
|
+
with pytest.raises(ValueError, match="missing"):
|
|
319
|
+
train_auto_challenger(df, "approved", task="classification")
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def test_champion_predictions_wrong_length_raises():
|
|
323
|
+
df = _labeled_df(seed=23)
|
|
324
|
+
with pytest.raises(ValueError, match="one entry per row"):
|
|
325
|
+
train_auto_challenger(
|
|
326
|
+
df, "approved", task="classification", champion_predictions=[True, False]
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def test_champion_predictions_scored_only_on_labeled_rows():
|
|
331
|
+
# Champion predictions mirror the (possibly-NaN) target itself, except on
|
|
332
|
+
# rows that get dropped as unlabeled, where they're deliberately wrong.
|
|
333
|
+
# If those rows leaked into scoring, champion accuracy would come in
|
|
334
|
+
# under 1.0 instead of exactly 1.0 — proving the shared-drop guarantee,
|
|
335
|
+
# not just that nothing crashes.
|
|
336
|
+
df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=24)
|
|
337
|
+
champion_predictions = [False if pd.isna(v) else v for v in df["approved"]]
|
|
338
|
+
|
|
339
|
+
result = train_auto_challenger(
|
|
340
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
assert result.champion_metrics is not None
|
|
344
|
+
assert result.champion_metrics["accuracy"] == 1.0
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def test_champion_predictions_not_given_leaves_champion_metrics_none():
|
|
348
|
+
df = _labeled_df(seed=25)
|
|
349
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
350
|
+
assert result.champion_metrics is None
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def test_to_challenger_upload_defaults_n_samples_from_result():
|
|
354
|
+
df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=26)
|
|
355
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
356
|
+
|
|
357
|
+
payload = to_challenger_upload(result)
|
|
358
|
+
|
|
359
|
+
assert payload["n_samples"] == 180
|
|
360
|
+
assert payload["n_samples_total"] == 200
|
|
361
|
+
assert payload["n_samples_dropped_unlabeled"] == 20
|
|
362
|
+
assert payload["population_note"] == result.population_note
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def test_to_challenger_upload_defaults_champion_metrics_from_result():
|
|
366
|
+
df = _labeled_df(seed=27)
|
|
367
|
+
champion_predictions = df["approved"].tolist()
|
|
368
|
+
result = train_auto_challenger(
|
|
369
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
payload = to_challenger_upload(result)
|
|
373
|
+
|
|
374
|
+
assert payload["champion_metrics"] == result.champion_metrics
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def test_to_challenger_upload_explicit_args_override_result_defaults():
|
|
378
|
+
df = _labeled_df(seed=28)
|
|
379
|
+
champion_predictions = df["approved"].tolist()
|
|
380
|
+
result = train_auto_challenger(
|
|
381
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
382
|
+
)
|
|
383
|
+
|
|
384
|
+
override_metrics = {"f1": 0.1, "accuracy": 0.1}
|
|
385
|
+
payload = to_challenger_upload(result, n_samples=999, champion_metrics=override_metrics)
|
|
386
|
+
|
|
387
|
+
assert payload["n_samples"] == 999
|
|
388
|
+
assert payload["champion_metrics"] == override_metrics
|
|
389
|
+
|
|
390
|
+
|
|
289
391
|
def test_train_auto_challenger_passes_immutable_cols_to_get_schema():
|
|
290
392
|
from unittest.mock import patch
|
|
291
393
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|