proxyml 0.6.0__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {proxyml-0.6.0 → proxyml-0.8.0}/PKG-INFO +1 -1
- {proxyml-0.6.0 → proxyml-0.8.0}/pyproject.toml +1 -1
- {proxyml-0.6.0 → proxyml-0.8.0}/src/proxyml/local/__init__.py +2 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/src/proxyml/local/challenger.py +48 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/src/proxyml.egg-info/PKG-INFO +1 -1
- {proxyml-0.6.0 → proxyml-0.8.0}/tests/test_local_challenger.py +90 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/LICENSE +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/README.md +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/setup.cfg +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/src/proxyml/__init__.py +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/src/proxyml/client.py +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/src/proxyml/schema_builder.py +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/src/proxyml.egg-info/SOURCES.txt +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/src/proxyml.egg-info/dependency_links.txt +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/src/proxyml.egg-info/requires.txt +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/src/proxyml.egg-info/top_level.txt +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/tests/test_client.py +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/tests/test_dependency_boundaries.py +0 -0
- {proxyml-0.6.0 → proxyml-0.8.0}/tests/test_schema_builder.py +0 -0
|
@@ -10,6 +10,7 @@ from proxyml.local.challenger import (
|
|
|
10
10
|
Complexity,
|
|
11
11
|
Rung,
|
|
12
12
|
TrainedChallenger,
|
|
13
|
+
fingerprint_labels,
|
|
13
14
|
score_champion,
|
|
14
15
|
to_challenger_upload,
|
|
15
16
|
train_auto_challenger,
|
|
@@ -21,6 +22,7 @@ __all__ = [
|
|
|
21
22
|
"train_auto_challenger",
|
|
22
23
|
"score_champion",
|
|
23
24
|
"to_challenger_upload",
|
|
25
|
+
"fingerprint_labels",
|
|
24
26
|
"Complexity",
|
|
25
27
|
"Rung",
|
|
26
28
|
"TrainedChallenger",
|
|
@@ -10,6 +10,8 @@ training target was real ground truth or a black box's predictions.
|
|
|
10
10
|
|
|
11
11
|
from __future__ import annotations
|
|
12
12
|
|
|
13
|
+
import hashlib
|
|
14
|
+
import json
|
|
13
15
|
from dataclasses import dataclass, replace
|
|
14
16
|
from enum import Enum
|
|
15
17
|
from importlib.metadata import version as _pkg_version
|
|
@@ -121,9 +123,30 @@ class TrainedChallenger:
|
|
|
121
123
|
n_samples_total: int
|
|
122
124
|
n_samples_dropped_unlabeled: int
|
|
123
125
|
population_note: str
|
|
126
|
+
target_fingerprint: str
|
|
124
127
|
champion_metrics: dict[str, float] | None = None
|
|
125
128
|
|
|
126
129
|
|
|
130
|
+
def fingerprint_labels(values: np.ndarray | list) -> str:
|
|
131
|
+
"""Hash an array of labels, deterministically, without the data ever leaving this process.
|
|
132
|
+
|
|
133
|
+
``.tolist()`` converts numpy scalars to native Python types before
|
|
134
|
+
serializing, so the hash doesn't drift across numpy versions with
|
|
135
|
+
different scalar repr behavior. Order is preserved (not sorted) since
|
|
136
|
+
it encodes row alignment — that's exactly what a "same data?" check
|
|
137
|
+
needs to be sensitive to.
|
|
138
|
+
|
|
139
|
+
Public so callers that score a champion decoupled from
|
|
140
|
+
``train_challenger()`` — e.g. two separate tool calls in an MCP
|
|
141
|
+
server, with no shared Python state between them — can compute
|
|
142
|
+
``champion_data_fingerprint`` themselves and carry it into a payload
|
|
143
|
+
assembled by hand, the same way ``to_challenger_upload(champion_labels=...)``
|
|
144
|
+
does internally.
|
|
145
|
+
"""
|
|
146
|
+
canonical = json.dumps(np.asarray(values).tolist())
|
|
147
|
+
return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
|
|
148
|
+
|
|
149
|
+
|
|
127
150
|
def _population_note(target_name: str, n_total: int, n_labeled: int, n_dropped: int) -> str:
|
|
128
151
|
if n_dropped == 0:
|
|
129
152
|
return f"Evaluated on all {n_total} row(s) — '{target_name}' had no missing values."
|
|
@@ -186,6 +209,7 @@ def train_challenger(
|
|
|
186
209
|
and the result is attached as ``TrainedChallenger.champion_metrics``.
|
|
187
210
|
"""
|
|
188
211
|
target_arr = np.asarray(target)
|
|
212
|
+
target_fingerprint = fingerprint_labels(target_arr)
|
|
189
213
|
if champion_predictions is not None and len(champion_predictions) != len(target_arr):
|
|
190
214
|
raise ValueError(
|
|
191
215
|
f"champion_predictions must have one entry per row of target "
|
|
@@ -264,6 +288,7 @@ def train_challenger(
|
|
|
264
288
|
n_samples_total=n_total,
|
|
265
289
|
n_samples_dropped_unlabeled=n_dropped,
|
|
266
290
|
population_note=_population_note(target_name, n_total, n_labeled, n_dropped),
|
|
291
|
+
target_fingerprint=target_fingerprint,
|
|
267
292
|
champion_metrics=champion_metrics,
|
|
268
293
|
)
|
|
269
294
|
|
|
@@ -286,6 +311,13 @@ def score_champion(
|
|
|
286
311
|
|
|
287
312
|
Returns ``{"f1":..., "accuracy":...}`` for classification or ``{"r2":...}``
|
|
288
313
|
for regression — the same shape as ``TrainedChallenger.metrics``.
|
|
314
|
+
|
|
315
|
+
If you're calling this decoupled from ``train_challenger()`` (i.e. not
|
|
316
|
+
via its ``champion_predictions=`` param), pass the same ``labels`` you
|
|
317
|
+
used here as ``champion_labels=`` to ``to_challenger_upload()`` — that
|
|
318
|
+
lets the upload endpoint confirm the challenger and champion were
|
|
319
|
+
actually scored on the same data, catching an accidental mismatched
|
|
320
|
+
file before it silently produces a misleading comparison.
|
|
289
321
|
"""
|
|
290
322
|
return score_predictions(np.asarray(labels), np.asarray(predictions), task=task)
|
|
291
323
|
|
|
@@ -295,6 +327,7 @@ def to_challenger_upload(
|
|
|
295
327
|
*,
|
|
296
328
|
n_samples: int | None = None,
|
|
297
329
|
champion_metrics: dict[str, float] | None = None,
|
|
330
|
+
champion_labels: np.ndarray | list | None = None,
|
|
298
331
|
sdk_version: str | None = None,
|
|
299
332
|
proxyml_core_version: str | None = None,
|
|
300
333
|
) -> dict[str, Any]:
|
|
@@ -327,11 +360,21 @@ def to_challenger_upload(
|
|
|
327
360
|
champion_metrics: the champion's real-world performance, from
|
|
328
361
|
``score_champion()`` — same metric keys as ``result.metrics``.
|
|
329
362
|
Defaults to ``result.champion_metrics``.
|
|
363
|
+
champion_labels: the ``labels`` array you passed to a standalone
|
|
364
|
+
``score_champion()`` call, if ``champion_metrics`` didn't come
|
|
365
|
+
from ``train_challenger()``'s internal ``champion_predictions=``
|
|
366
|
+
path. Used only to compute ``champion_data_fingerprint`` — the
|
|
367
|
+
labels themselves are never included in the payload. If
|
|
368
|
+
``champion_metrics`` resolves from ``result.champion_metrics``
|
|
369
|
+
instead, the fingerprint defaults to ``result.target_fingerprint``
|
|
370
|
+
(guaranteed identical, since that internal path scores against
|
|
371
|
+
the exact same data).
|
|
330
372
|
sdk_version: defaults to the installed ``proxyml`` version.
|
|
331
373
|
proxyml_core_version: defaults to the installed ``proxyml-core`` version.
|
|
332
374
|
"""
|
|
333
375
|
if n_samples is None:
|
|
334
376
|
n_samples = result.n_samples_total - result.n_samples_dropped_unlabeled
|
|
377
|
+
used_internal_champion_metrics = champion_metrics is None
|
|
335
378
|
if champion_metrics is None:
|
|
336
379
|
champion_metrics = result.champion_metrics
|
|
337
380
|
if sdk_version is None:
|
|
@@ -352,6 +395,11 @@ def to_challenger_upload(
|
|
|
352
395
|
}
|
|
353
396
|
if champion_metrics is not None:
|
|
354
397
|
payload["champion_metrics"] = champion_metrics
|
|
398
|
+
payload["challenger_data_fingerprint"] = result.target_fingerprint
|
|
399
|
+
if champion_labels is not None:
|
|
400
|
+
payload["champion_data_fingerprint"] = fingerprint_labels(champion_labels)
|
|
401
|
+
elif used_internal_champion_metrics:
|
|
402
|
+
payload["champion_data_fingerprint"] = result.target_fingerprint
|
|
355
403
|
return payload
|
|
356
404
|
|
|
357
405
|
|
|
@@ -8,6 +8,7 @@ from proxyml.local import (
|
|
|
8
8
|
Complexity,
|
|
9
9
|
LADDERS,
|
|
10
10
|
TrainedChallenger,
|
|
11
|
+
fingerprint_labels,
|
|
11
12
|
score_champion,
|
|
12
13
|
to_challenger_upload,
|
|
13
14
|
train_auto_challenger,
|
|
@@ -402,3 +403,92 @@ def test_train_auto_challenger_passes_immutable_cols_to_get_schema():
|
|
|
402
403
|
called_kwargs = mock_get_schema.call_args.kwargs
|
|
403
404
|
assert list(called_df.columns) == list(features_df.columns)
|
|
404
405
|
assert called_kwargs["immutable_cols"] == ["age"]
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def test_target_fingerprint_is_deterministic_for_identical_data():
|
|
409
|
+
df = _labeled_df(seed=29)
|
|
410
|
+
result_a = train_auto_challenger(df, "approved", task="classification")
|
|
411
|
+
result_b = train_auto_challenger(df, "approved", task="classification")
|
|
412
|
+
|
|
413
|
+
assert result_a.target_fingerprint == result_b.target_fingerprint
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def test_target_fingerprint_differs_for_different_data():
|
|
417
|
+
df_a = _labeled_df(seed=29)
|
|
418
|
+
df_b = _labeled_df(seed=30)
|
|
419
|
+
result_a = train_auto_challenger(df_a, "approved", task="classification")
|
|
420
|
+
result_b = train_auto_challenger(df_b, "approved", task="classification")
|
|
421
|
+
|
|
422
|
+
assert result_a.target_fingerprint != result_b.target_fingerprint
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def test_to_challenger_upload_includes_matching_fingerprints_on_internal_champion_path():
|
|
426
|
+
df = _labeled_df(seed=31)
|
|
427
|
+
champion_predictions = df["approved"].tolist()
|
|
428
|
+
result = train_auto_challenger(
|
|
429
|
+
df, "approved", task="classification", champion_predictions=champion_predictions
|
|
430
|
+
)
|
|
431
|
+
|
|
432
|
+
payload = to_challenger_upload(result)
|
|
433
|
+
|
|
434
|
+
assert payload["challenger_data_fingerprint"] == result.target_fingerprint
|
|
435
|
+
assert payload["champion_data_fingerprint"] == result.target_fingerprint
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def test_to_challenger_upload_champion_labels_fingerprint_for_decoupled_path():
|
|
439
|
+
df = _labeled_df(seed=32)
|
|
440
|
+
target = df["approved"]
|
|
441
|
+
result = train_challenger(df, target, _schema(), task="classification")
|
|
442
|
+
champion_metrics = score_champion(target, target, task="classification")
|
|
443
|
+
|
|
444
|
+
payload = to_challenger_upload(result, champion_metrics=champion_metrics, champion_labels=target)
|
|
445
|
+
|
|
446
|
+
assert payload["challenger_data_fingerprint"] == result.target_fingerprint
|
|
447
|
+
assert payload["champion_data_fingerprint"] == result.target_fingerprint
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
def test_to_challenger_upload_champion_labels_fingerprint_differs_for_different_data():
|
|
451
|
+
df = _labeled_df(seed=33)
|
|
452
|
+
target = df["approved"]
|
|
453
|
+
result = train_challenger(df, target, _schema(), task="classification")
|
|
454
|
+
champion_metrics = score_champion(target, target, task="classification")
|
|
455
|
+
other_labels = ~target
|
|
456
|
+
|
|
457
|
+
payload = to_challenger_upload(
|
|
458
|
+
result, champion_metrics=champion_metrics, champion_labels=other_labels
|
|
459
|
+
)
|
|
460
|
+
|
|
461
|
+
assert payload["challenger_data_fingerprint"] != payload["champion_data_fingerprint"]
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
def test_to_challenger_upload_omits_champion_fingerprint_without_labels_or_internal_path():
|
|
465
|
+
df = _labeled_df(seed=34)
|
|
466
|
+
target = df["approved"]
|
|
467
|
+
result = train_challenger(df, target, _schema(), task="classification")
|
|
468
|
+
champion_metrics = score_champion(target, target, task="classification")
|
|
469
|
+
|
|
470
|
+
payload = to_challenger_upload(result, champion_metrics=champion_metrics)
|
|
471
|
+
|
|
472
|
+
assert payload["challenger_data_fingerprint"] == result.target_fingerprint
|
|
473
|
+
assert "champion_data_fingerprint" not in payload
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def test_to_challenger_upload_without_champion_metrics_omits_fingerprints():
|
|
477
|
+
df = _labeled_df(seed=35)
|
|
478
|
+
result = train_auto_challenger(df, "approved", task="classification")
|
|
479
|
+
|
|
480
|
+
payload = to_challenger_upload(result)
|
|
481
|
+
|
|
482
|
+
assert "challenger_data_fingerprint" not in payload
|
|
483
|
+
assert "champion_data_fingerprint" not in payload
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def test_fingerprint_labels_is_public_and_matches_target_fingerprint():
|
|
487
|
+
"""Exercises the exact use case this was made public for: a caller with
|
|
488
|
+
no TrainedChallenger in hand (e.g. a decoupled score_champion() call)
|
|
489
|
+
computing the same fingerprint train_challenger() would have."""
|
|
490
|
+
df = _labeled_df(seed=36)
|
|
491
|
+
target = df["approved"]
|
|
492
|
+
result = train_challenger(df, target, _schema(), task="classification")
|
|
493
|
+
|
|
494
|
+
assert fingerprint_labels(target) == result.target_fingerprint
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|