proxyml 0.6.0__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proxyml
3
- Version: 0.6.0
3
+ Version: 0.8.0
4
4
  Summary: Python SDK for calling the ProxyML API
5
5
  Author-email: ProxyML <contact@proxyml.ai>
6
6
  License: Apache License
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "proxyml"
7
- version = "0.6.0"
7
+ version = "0.8.0"
8
8
  description = "Python SDK for calling the ProxyML API"
9
9
  readme = "README.md"
10
10
  license = {file = "LICENSE"}
@@ -10,6 +10,7 @@ from proxyml.local.challenger import (
10
10
  Complexity,
11
11
  Rung,
12
12
  TrainedChallenger,
13
+ fingerprint_labels,
13
14
  score_champion,
14
15
  to_challenger_upload,
15
16
  train_auto_challenger,
@@ -21,6 +22,7 @@ __all__ = [
21
22
  "train_auto_challenger",
22
23
  "score_champion",
23
24
  "to_challenger_upload",
25
+ "fingerprint_labels",
24
26
  "Complexity",
25
27
  "Rung",
26
28
  "TrainedChallenger",
@@ -10,6 +10,8 @@ training target was real ground truth or a black box's predictions.
10
10
 
11
11
  from __future__ import annotations
12
12
 
13
+ import hashlib
14
+ import json
13
15
  from dataclasses import dataclass, replace
14
16
  from enum import Enum
15
17
  from importlib.metadata import version as _pkg_version
@@ -121,9 +123,30 @@ class TrainedChallenger:
121
123
  n_samples_total: int
122
124
  n_samples_dropped_unlabeled: int
123
125
  population_note: str
126
+ target_fingerprint: str
124
127
  champion_metrics: dict[str, float] | None = None
125
128
 
126
129
 
130
+ def fingerprint_labels(values: np.ndarray | list) -> str:
131
+ """Hash an array of labels, deterministically, without the data ever leaving this process.
132
+
133
+ ``.tolist()`` converts numpy scalars to native Python types before
134
+ serializing, so the hash doesn't drift across numpy versions with
135
+ different scalar repr behavior. Order is preserved (not sorted) since
136
+ it encodes row alignment — that's exactly what a "same data?" check
137
+ needs to be sensitive to.
138
+
139
+ Public so callers that score a champion decoupled from
140
+ ``train_challenger()`` — e.g. two separate tool calls in an MCP
141
+ server, with no shared Python state between them — can compute
142
+ ``champion_data_fingerprint`` themselves and carry it into a payload
143
+ assembled by hand, the same way ``to_challenger_upload(champion_labels=...)``
144
+ does internally.
145
+ """
146
+ canonical = json.dumps(np.asarray(values).tolist())
147
+ return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
148
+
149
+
127
150
  def _population_note(target_name: str, n_total: int, n_labeled: int, n_dropped: int) -> str:
128
151
  if n_dropped == 0:
129
152
  return f"Evaluated on all {n_total} row(s) — '{target_name}' had no missing values."
@@ -186,6 +209,7 @@ def train_challenger(
186
209
  and the result is attached as ``TrainedChallenger.champion_metrics``.
187
210
  """
188
211
  target_arr = np.asarray(target)
212
+ target_fingerprint = fingerprint_labels(target_arr)
189
213
  if champion_predictions is not None and len(champion_predictions) != len(target_arr):
190
214
  raise ValueError(
191
215
  f"champion_predictions must have one entry per row of target "
@@ -264,6 +288,7 @@ def train_challenger(
264
288
  n_samples_total=n_total,
265
289
  n_samples_dropped_unlabeled=n_dropped,
266
290
  population_note=_population_note(target_name, n_total, n_labeled, n_dropped),
291
+ target_fingerprint=target_fingerprint,
267
292
  champion_metrics=champion_metrics,
268
293
  )
269
294
 
@@ -286,6 +311,13 @@ def score_champion(
286
311
 
287
312
  Returns ``{"f1":..., "accuracy":...}`` for classification or ``{"r2":...}``
288
313
  for regression — the same shape as ``TrainedChallenger.metrics``.
314
+
315
+ If you're calling this decoupled from ``train_challenger()`` (i.e. not
316
+ via its ``champion_predictions=`` param), pass the same ``labels`` you
317
+ used here as ``champion_labels=`` to ``to_challenger_upload()`` — that
318
+ lets the upload endpoint confirm the challenger and champion were
319
+ actually scored on the same data, catching an accidental mismatched
320
+ file before it silently produces a misleading comparison.
289
321
  """
290
322
  return score_predictions(np.asarray(labels), np.asarray(predictions), task=task)
291
323
 
@@ -295,6 +327,7 @@ def to_challenger_upload(
295
327
  *,
296
328
  n_samples: int | None = None,
297
329
  champion_metrics: dict[str, float] | None = None,
330
+ champion_labels: np.ndarray | list | None = None,
298
331
  sdk_version: str | None = None,
299
332
  proxyml_core_version: str | None = None,
300
333
  ) -> dict[str, Any]:
@@ -327,11 +360,21 @@ def to_challenger_upload(
327
360
  champion_metrics: the champion's real-world performance, from
328
361
  ``score_champion()`` — same metric keys as ``result.metrics``.
329
362
  Defaults to ``result.champion_metrics``.
363
+ champion_labels: the ``labels`` array you passed to a standalone
364
+ ``score_champion()`` call, if ``champion_metrics`` didn't come
365
+ from ``train_challenger()``'s internal ``champion_predictions=``
366
+ path. Used only to compute ``champion_data_fingerprint`` — the
367
+ labels themselves are never included in the payload. If
368
+ ``champion_metrics`` resolves from ``result.champion_metrics``
369
+ instead, the fingerprint defaults to ``result.target_fingerprint``
370
+ (guaranteed identical, since that internal path scores against
371
+ the exact same data).
330
372
  sdk_version: defaults to the installed ``proxyml`` version.
331
373
  proxyml_core_version: defaults to the installed ``proxyml-core`` version.
332
374
  """
333
375
  if n_samples is None:
334
376
  n_samples = result.n_samples_total - result.n_samples_dropped_unlabeled
377
+ used_internal_champion_metrics = champion_metrics is None
335
378
  if champion_metrics is None:
336
379
  champion_metrics = result.champion_metrics
337
380
  if sdk_version is None:
@@ -352,6 +395,11 @@ def to_challenger_upload(
352
395
  }
353
396
  if champion_metrics is not None:
354
397
  payload["champion_metrics"] = champion_metrics
398
+ payload["challenger_data_fingerprint"] = result.target_fingerprint
399
+ if champion_labels is not None:
400
+ payload["champion_data_fingerprint"] = fingerprint_labels(champion_labels)
401
+ elif used_internal_champion_metrics:
402
+ payload["champion_data_fingerprint"] = result.target_fingerprint
355
403
  return payload
356
404
 
357
405
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proxyml
3
- Version: 0.6.0
3
+ Version: 0.8.0
4
4
  Summary: Python SDK for calling the ProxyML API
5
5
  Author-email: ProxyML <contact@proxyml.ai>
6
6
  License: Apache License
@@ -8,6 +8,7 @@ from proxyml.local import (
8
8
  Complexity,
9
9
  LADDERS,
10
10
  TrainedChallenger,
11
+ fingerprint_labels,
11
12
  score_champion,
12
13
  to_challenger_upload,
13
14
  train_auto_challenger,
@@ -402,3 +403,92 @@ def test_train_auto_challenger_passes_immutable_cols_to_get_schema():
402
403
  called_kwargs = mock_get_schema.call_args.kwargs
403
404
  assert list(called_df.columns) == list(features_df.columns)
404
405
  assert called_kwargs["immutable_cols"] == ["age"]
406
+
407
+
408
+ def test_target_fingerprint_is_deterministic_for_identical_data():
409
+ df = _labeled_df(seed=29)
410
+ result_a = train_auto_challenger(df, "approved", task="classification")
411
+ result_b = train_auto_challenger(df, "approved", task="classification")
412
+
413
+ assert result_a.target_fingerprint == result_b.target_fingerprint
414
+
415
+
416
+ def test_target_fingerprint_differs_for_different_data():
417
+ df_a = _labeled_df(seed=29)
418
+ df_b = _labeled_df(seed=30)
419
+ result_a = train_auto_challenger(df_a, "approved", task="classification")
420
+ result_b = train_auto_challenger(df_b, "approved", task="classification")
421
+
422
+ assert result_a.target_fingerprint != result_b.target_fingerprint
423
+
424
+
425
+ def test_to_challenger_upload_includes_matching_fingerprints_on_internal_champion_path():
426
+ df = _labeled_df(seed=31)
427
+ champion_predictions = df["approved"].tolist()
428
+ result = train_auto_challenger(
429
+ df, "approved", task="classification", champion_predictions=champion_predictions
430
+ )
431
+
432
+ payload = to_challenger_upload(result)
433
+
434
+ assert payload["challenger_data_fingerprint"] == result.target_fingerprint
435
+ assert payload["champion_data_fingerprint"] == result.target_fingerprint
436
+
437
+
438
+ def test_to_challenger_upload_champion_labels_fingerprint_for_decoupled_path():
439
+ df = _labeled_df(seed=32)
440
+ target = df["approved"]
441
+ result = train_challenger(df, target, _schema(), task="classification")
442
+ champion_metrics = score_champion(target, target, task="classification")
443
+
444
+ payload = to_challenger_upload(result, champion_metrics=champion_metrics, champion_labels=target)
445
+
446
+ assert payload["challenger_data_fingerprint"] == result.target_fingerprint
447
+ assert payload["champion_data_fingerprint"] == result.target_fingerprint
448
+
449
+
450
+ def test_to_challenger_upload_champion_labels_fingerprint_differs_for_different_data():
451
+ df = _labeled_df(seed=33)
452
+ target = df["approved"]
453
+ result = train_challenger(df, target, _schema(), task="classification")
454
+ champion_metrics = score_champion(target, target, task="classification")
455
+ other_labels = ~target
456
+
457
+ payload = to_challenger_upload(
458
+ result, champion_metrics=champion_metrics, champion_labels=other_labels
459
+ )
460
+
461
+ assert payload["challenger_data_fingerprint"] != payload["champion_data_fingerprint"]
462
+
463
+
464
+ def test_to_challenger_upload_omits_champion_fingerprint_without_labels_or_internal_path():
465
+ df = _labeled_df(seed=34)
466
+ target = df["approved"]
467
+ result = train_challenger(df, target, _schema(), task="classification")
468
+ champion_metrics = score_champion(target, target, task="classification")
469
+
470
+ payload = to_challenger_upload(result, champion_metrics=champion_metrics)
471
+
472
+ assert payload["challenger_data_fingerprint"] == result.target_fingerprint
473
+ assert "champion_data_fingerprint" not in payload
474
+
475
+
476
+ def test_to_challenger_upload_without_champion_metrics_omits_fingerprints():
477
+ df = _labeled_df(seed=35)
478
+ result = train_auto_challenger(df, "approved", task="classification")
479
+
480
+ payload = to_challenger_upload(result)
481
+
482
+ assert "challenger_data_fingerprint" not in payload
483
+ assert "champion_data_fingerprint" not in payload
484
+
485
+
486
+ def test_fingerprint_labels_is_public_and_matches_target_fingerprint():
487
+ """Exercises the exact use case this was made public for: a caller with
488
+ no TrainedChallenger in hand (e.g. a decoupled score_champion() call)
489
+ computing the same fingerprint train_challenger() would have."""
490
+ df = _labeled_df(seed=36)
491
+ target = df["approved"]
492
+ result = train_challenger(df, target, _schema(), task="classification")
493
+
494
+ assert fingerprint_labels(target) == result.target_fingerprint
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes