proxyml 0.5.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proxyml
3
- Version: 0.5.0
3
+ Version: 0.7.0
4
4
  Summary: Python SDK for calling the ProxyML API
5
5
  Author-email: ProxyML <contact@proxyml.ai>
6
6
  License: Apache License
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "proxyml"
7
- version = "0.5.0"
7
+ version = "0.7.0"
8
8
  description = "Python SDK for calling the ProxyML API"
9
9
  readme = "README.md"
10
10
  license = {file = "LICENSE"}
@@ -10,6 +10,8 @@ training target was real ground truth or a black box's predictions.
10
10
 
11
11
  from __future__ import annotations
12
12
 
13
+ import hashlib
14
+ import json
13
15
  from dataclasses import dataclass, replace
14
16
  from enum import Enum
15
17
  from importlib.metadata import version as _pkg_version
@@ -118,6 +120,36 @@ class TrainedChallenger:
118
120
  metrics: dict[str, float]
119
121
  hyperparameters: dict[str, Any]
120
122
  export: SurrogateExport
123
+ n_samples_total: int
124
+ n_samples_dropped_unlabeled: int
125
+ population_note: str
126
+ target_fingerprint: str
127
+ champion_metrics: dict[str, float] | None = None
128
+
129
+
130
+ def _fingerprint_values(values: np.ndarray | list) -> str:
131
+ """Hash an array of labels, deterministically, without the data ever leaving this process.
132
+
133
+ ``.tolist()`` converts numpy scalars to native Python types before
134
+ serializing, so the hash doesn't drift across numpy versions with
135
+ different scalar repr behavior. Order is preserved (not sorted) since
136
+ it encodes row alignment — that's exactly what a "same data?" check
137
+ needs to be sensitive to.
138
+ """
139
+ canonical = json.dumps(np.asarray(values).tolist())
140
+ return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
141
+
142
+
143
+ def _population_note(target_name: str, n_total: int, n_labeled: int, n_dropped: int) -> str:
144
+ if n_dropped == 0:
145
+ return f"Evaluated on all {n_total} row(s) — '{target_name}' had no missing values."
146
+ return (
147
+ f"Evaluated on {n_labeled} of {n_total} row(s) with a non-null '{target_name}' value "
148
+ f"({n_dropped} unlabeled row(s) dropped before training/scoring). "
149
+ "Labeled-vs-unlabeled selection may not be random; treat this as a declared "
150
+ "scope limitation on the evaluation population, not a claim about performance "
151
+ "on the full dataset."
152
+ )
121
153
 
122
154
 
123
155
  def train_challenger(
@@ -129,6 +161,8 @@ def train_challenger(
129
161
  feature_names: list[str] | None = None,
130
162
  task: Literal["classification", "regression", "auto"] = "auto",
131
163
  test_size: float = 0.2,
164
+ target_name: str = "target",
165
+ champion_predictions: np.ndarray | list | None = None,
132
166
  ) -> TrainedChallenger:
133
167
  """Train a linear challenger model on ``df`` against ``target``, locally.
134
168
 
@@ -142,6 +176,15 @@ def train_challenger(
142
176
  can be compared with the same ``proxyml_core.export.predict_from_export``
143
177
  arithmetic.
144
178
 
179
+ Rows where ``target`` is missing (NaN/None) are dropped before training,
180
+ the CV split, and champion scoring — never silently included. The drop
181
+ count and a human-readable scope-limitation note are recorded on the
182
+ result (``n_samples_total``, ``n_samples_dropped_unlabeled``,
183
+ ``population_note``). If ``champion_predictions`` is given, it must have
184
+ one entry per row of ``df``/``target`` (same order) so the identical rows
185
+ are dropped from both sides — champion and challenger are always
186
+ evaluated on the same labeled population, never on different ones.
187
+
145
188
  Args:
146
189
  df: samples to train on, one column per schema feature.
147
190
  target: the value to predict for each row of ``df`` — ground-truth
@@ -151,7 +194,36 @@ def train_challenger(
151
194
  feature_names: subset of ``schema.features`` to train on; omit for all.
152
195
  task: "classification", "regression", or "auto" to infer from ``target``.
153
196
  test_size: fraction of data held out to compute fidelity metrics.
197
+ target_name: human-readable name for ``target``, used in
198
+ ``population_note`` (e.g. the column name, if known).
199
+ champion_predictions: a champion model's predictions, one per row of
200
+ ``df``/``target`` (same order). If given, scored via
201
+ ``score_champion()`` against the same (row-dropped) ``target``,
202
+ and the result is attached as ``TrainedChallenger.champion_metrics``.
154
203
  """
204
+ target_arr = np.asarray(target)
205
+ target_fingerprint = _fingerprint_values(target_arr)
206
+ if champion_predictions is not None and len(champion_predictions) != len(target_arr):
207
+ raise ValueError(
208
+ f"champion_predictions must have one entry per row of target "
209
+ f"({len(target_arr)} rows, got {len(champion_predictions)}) — same order — so "
210
+ f"rows with a missing {target_name!r} value can be dropped from both the "
211
+ f"challenger and the champion, keeping them evaluated on the same population."
212
+ )
213
+
214
+ labeled_mask = ~pd.isna(target_arr)
215
+ n_total = len(target_arr)
216
+ n_labeled = int(labeled_mask.sum())
217
+ n_dropped = n_total - n_labeled
218
+ if n_labeled == 0:
219
+ raise ValueError(f"All {n_total} row(s) have a missing {target_name!r} value; nothing to train on")
220
+
221
+ df = df.iloc[labeled_mask].reset_index(drop=True)
222
+ target_arr = target_arr[labeled_mask]
223
+ champion_predictions_labeled = (
224
+ np.asarray(champion_predictions)[labeled_mask] if champion_predictions is not None else None
225
+ )
226
+
155
227
  rung = LADDERS[complexity]
156
228
 
157
229
  features: list[Feature] = schema.features
@@ -161,7 +233,7 @@ def train_challenger(
161
233
  col_order = [f.name for f in features]
162
234
 
163
235
  X = df[col_order].to_numpy(dtype=object)
164
- y = np.asarray(target)
236
+ y = target_arr
165
237
 
166
238
  if task == "auto":
167
239
  classification = is_classification(y)
@@ -174,6 +246,10 @@ def train_challenger(
174
246
  if classification:
175
247
  y = binarize_if_probabilities(y)
176
248
 
249
+ champion_metrics = None
250
+ if champion_predictions_labeled is not None:
251
+ champion_metrics = score_champion(y, champion_predictions_labeled, task=resolved_task)
252
+
177
253
  preprocessor = build_preprocessor(features)
178
254
  estimator = rung.build_classifier() if classification else rung.build_regressor()
179
255
  pipeline = Pipeline(steps=[("preprocessor", preprocessor), ("estimator", estimator)])
@@ -202,6 +278,11 @@ def train_challenger(
202
278
  metrics=metrics,
203
279
  hyperparameters=hyperparameters,
204
280
  export=export,
281
+ n_samples_total=n_total,
282
+ n_samples_dropped_unlabeled=n_dropped,
283
+ population_note=_population_note(target_name, n_total, n_labeled, n_dropped),
284
+ target_fingerprint=target_fingerprint,
285
+ champion_metrics=champion_metrics,
205
286
  )
206
287
 
207
288
 
@@ -223,6 +304,13 @@ def score_champion(
223
304
 
224
305
  Returns ``{"f1":..., "accuracy":...}`` for classification or ``{"r2":...}``
225
306
  for regression — the same shape as ``TrainedChallenger.metrics``.
307
+
308
+ If you're calling this decoupled from ``train_challenger()`` (i.e. not
309
+ via its ``champion_predictions=`` param), pass the same ``labels`` you
310
+ used here as ``champion_labels=`` to ``to_challenger_upload()`` — that
311
+ lets the upload endpoint confirm the challenger and champion were
312
+ actually scored on the same data, catching an accidental mismatched
313
+ file before it silently produces a misleading comparison.
226
314
  """
227
315
  return score_predictions(np.asarray(labels), np.asarray(predictions), task=task)
228
316
 
@@ -230,8 +318,9 @@ def score_champion(
230
318
  def to_challenger_upload(
231
319
  result: TrainedChallenger,
232
320
  *,
233
- n_samples: int,
321
+ n_samples: int | None = None,
234
322
  champion_metrics: dict[str, float] | None = None,
323
+ champion_labels: np.ndarray | list | None = None,
235
324
  sdk_version: str | None = None,
236
325
  proxyml_core_version: str | None = None,
237
326
  ) -> dict[str, Any]:
@@ -245,27 +334,42 @@ def to_challenger_upload(
245
334
  ``json.dump`` — upload it either by POSTing it directly, or by saving it
246
335
  to a file and using the dashboard's "Upload challenger" button.
247
336
 
248
- ``champion_metrics`` is optional: pass ``None`` (the default) to get a
249
- self-contained export of the challenger alone — e.g. to save/share it
250
- before you have a champion to compare against — and fill in
251
- ``champion_metrics`` later. The upload endpoint itself still requires
252
- ``champion_metrics`` at upload time; this function just doesn't force you
253
- to have it up front.
337
+ ``champion_metrics`` is optional: pass ``None`` (the default) to fall
338
+ back to ``result.champion_metrics`` (populated automatically if you
339
+ passed ``champion_predictions`` to ``train_challenger()``/
340
+ ``train_auto_challenger()``) — or, if that's also ``None``, to get a
341
+ self-contained export of the challenger alone, e.g. to save/share it
342
+ before you have a champion to compare against. The upload endpoint
343
+ itself still requires ``champion_metrics`` at upload time; this function
344
+ just doesn't force you to have it up front.
254
345
 
255
346
  Args:
256
347
  result: output of ``train_challenger()``/``train_auto_challenger()``.
257
348
  n_samples: size of the evaluation set both ``result.metrics`` and
258
- ``champion_metrics`` were scored on. Not derived automatically —
259
- ``TrainedChallenger`` doesn't retain its internal held-out split,
260
- and ``champion_metrics`` typically comes from a separate
261
- ``score_champion()`` call the two need to share, so the caller is
262
- the only one who actually knows this number.
349
+ ``champion_metrics`` were scored on. Defaults to
350
+ ``result.n_samples_total - result.n_samples_dropped_unlabeled``
351
+ (the labeled-row count) — override only if you scored on some
352
+ other population.
263
353
  champion_metrics: the champion's real-world performance, from
264
354
  ``score_champion()`` — same metric keys as ``result.metrics``.
265
- Omit if you don't have it yet.
355
+ Defaults to ``result.champion_metrics``.
356
+ champion_labels: the ``labels`` array you passed to a standalone
357
+ ``score_champion()`` call, if ``champion_metrics`` didn't come
358
+ from ``train_challenger()``'s internal ``champion_predictions=``
359
+ path. Used only to compute ``champion_data_fingerprint`` — the
360
+ labels themselves are never included in the payload. If
361
+ ``champion_metrics`` resolves from ``result.champion_metrics``
362
+ instead, the fingerprint defaults to ``result.target_fingerprint``
363
+ (guaranteed identical, since that internal path scores against
364
+ the exact same data).
266
365
  sdk_version: defaults to the installed ``proxyml`` version.
267
366
  proxyml_core_version: defaults to the installed ``proxyml-core`` version.
268
367
  """
368
+ if n_samples is None:
369
+ n_samples = result.n_samples_total - result.n_samples_dropped_unlabeled
370
+ used_internal_champion_metrics = champion_metrics is None
371
+ if champion_metrics is None:
372
+ champion_metrics = result.champion_metrics
269
373
  if sdk_version is None:
270
374
  sdk_version = _pkg_version("proxyml")
271
375
  if proxyml_core_version is None:
@@ -275,12 +379,20 @@ def to_challenger_upload(
275
379
  "export": result.export.to_dict(),
276
380
  "challenger_metrics": result.metrics,
277
381
  "n_samples": n_samples,
382
+ "n_samples_total": result.n_samples_total,
383
+ "n_samples_dropped_unlabeled": result.n_samples_dropped_unlabeled,
384
+ "population_note": result.population_note,
278
385
  "complexity": result.complexity.value,
279
386
  "sdk_version": sdk_version,
280
387
  "proxyml_core_version": proxyml_core_version,
281
388
  }
282
389
  if champion_metrics is not None:
283
390
  payload["champion_metrics"] = champion_metrics
391
+ payload["challenger_data_fingerprint"] = result.target_fingerprint
392
+ if champion_labels is not None:
393
+ payload["champion_data_fingerprint"] = _fingerprint_values(champion_labels)
394
+ elif used_internal_champion_metrics:
395
+ payload["champion_data_fingerprint"] = result.target_fingerprint
284
396
  return payload
285
397
 
286
398
 
@@ -293,6 +405,7 @@ def train_auto_challenger(
293
405
  feature_names: list[str] | None = None,
294
406
  task: Literal["classification", "regression", "auto"] = "auto",
295
407
  test_size: float = 0.2,
408
+ champion_predictions: np.ndarray | list | None = None,
296
409
  ) -> TrainedChallenger:
297
410
  """Load data, infer a schema, and train a linear challenger in one call.
298
411
 
@@ -302,6 +415,12 @@ def train_auto_challenger(
302
415
  remains overridable; this does not search across ``LADDERS`` to find the
303
416
  best-fitting rung.
304
417
 
418
+ Rows with a missing ``target_col`` value are dropped before training and
419
+ champion scoring — see ``train_challenger()`` for details. Schema
420
+ inference (feature means/stds/categories) still runs over every row,
421
+ including ones later dropped for a missing target — only training and
422
+ evaluation are restricted to the labeled subset.
423
+
305
424
  Args:
306
425
  data: a CSV path, or an already-loaded DataFrame containing both the
307
426
  feature columns and ``target_col``.
@@ -312,6 +431,8 @@ def train_auto_challenger(
312
431
  feature_names: subset of feature columns to train on; omit for all.
313
432
  task: "classification", "regression", or "auto" to infer from ``target_col``.
314
433
  test_size: fraction of data held out to compute fidelity metrics.
434
+ champion_predictions: a champion model's predictions, one per row of
435
+ ``data`` (same order) — see ``train_challenger()``.
315
436
  """
316
437
  df = data if isinstance(data, pd.DataFrame) else pd.read_csv(data)
317
438
  target = df[target_col]
@@ -326,4 +447,6 @@ def train_auto_challenger(
326
447
  feature_names=feature_names,
327
448
  task=task,
328
449
  test_size=test_size,
450
+ target_name=target_col,
451
+ champion_predictions=champion_predictions,
329
452
  )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proxyml
3
- Version: 0.5.0
3
+ Version: 0.7.0
4
4
  Summary: Python SDK for calling the ProxyML API
5
5
  Author-email: ProxyML <contact@proxyml.ai>
6
6
  License: Apache License
@@ -286,6 +286,108 @@ def test_to_challenger_upload_payload_is_json_serializable():
286
286
  json.dumps(payload) # must not raise
287
287
 
288
288
 
289
+ def _labeled_df_with_nan_target(n=200, n_nan=20, seed=20):
290
+ df = _labeled_df(n=n, seed=seed)
291
+ df["approved"] = df["approved"].astype(float)
292
+ df.loc[df.index[:n_nan], "approved"] = np.nan
293
+ return df
294
+
295
+
296
+ def test_nan_target_rows_are_dropped_and_counted():
297
+ df = _labeled_df_with_nan_target(n=200, n_nan=20)
298
+ result = train_auto_challenger(df, "approved", task="classification")
299
+
300
+ assert result.n_samples_total == 200
301
+ assert result.n_samples_dropped_unlabeled == 20
302
+ assert "20" in result.population_note
303
+ assert "180" in result.population_note
304
+
305
+
306
+ def test_no_nan_targets_reports_zero_dropped():
307
+ df = _labeled_df(seed=21)
308
+ result = train_auto_challenger(df, "approved", task="classification")
309
+
310
+ assert result.n_samples_total == len(df)
311
+ assert result.n_samples_dropped_unlabeled == 0
312
+ assert "no missing values" in result.population_note
313
+
314
+
315
+ def test_all_nan_target_raises():
316
+ df = _labeled_df(n=20, seed=22)
317
+ df["approved"] = np.nan
318
+ with pytest.raises(ValueError, match="missing"):
319
+ train_auto_challenger(df, "approved", task="classification")
320
+
321
+
322
+ def test_champion_predictions_wrong_length_raises():
323
+ df = _labeled_df(seed=23)
324
+ with pytest.raises(ValueError, match="one entry per row"):
325
+ train_auto_challenger(
326
+ df, "approved", task="classification", champion_predictions=[True, False]
327
+ )
328
+
329
+
330
+ def test_champion_predictions_scored_only_on_labeled_rows():
331
+ # Champion predictions mirror the (possibly-NaN) target itself, except on
332
+ # rows that get dropped as unlabeled, where they're deliberately wrong.
333
+ # If those rows leaked into scoring, champion accuracy would come in
334
+ # under 1.0 instead of exactly 1.0 — proving the shared-drop guarantee,
335
+ # not just that nothing crashes.
336
+ df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=24)
337
+ champion_predictions = [False if pd.isna(v) else v for v in df["approved"]]
338
+
339
+ result = train_auto_challenger(
340
+ df, "approved", task="classification", champion_predictions=champion_predictions
341
+ )
342
+
343
+ assert result.champion_metrics is not None
344
+ assert result.champion_metrics["accuracy"] == 1.0
345
+
346
+
347
+ def test_champion_predictions_not_given_leaves_champion_metrics_none():
348
+ df = _labeled_df(seed=25)
349
+ result = train_auto_challenger(df, "approved", task="classification")
350
+ assert result.champion_metrics is None
351
+
352
+
353
+ def test_to_challenger_upload_defaults_n_samples_from_result():
354
+ df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=26)
355
+ result = train_auto_challenger(df, "approved", task="classification")
356
+
357
+ payload = to_challenger_upload(result)
358
+
359
+ assert payload["n_samples"] == 180
360
+ assert payload["n_samples_total"] == 200
361
+ assert payload["n_samples_dropped_unlabeled"] == 20
362
+ assert payload["population_note"] == result.population_note
363
+
364
+
365
+ def test_to_challenger_upload_defaults_champion_metrics_from_result():
366
+ df = _labeled_df(seed=27)
367
+ champion_predictions = df["approved"].tolist()
368
+ result = train_auto_challenger(
369
+ df, "approved", task="classification", champion_predictions=champion_predictions
370
+ )
371
+
372
+ payload = to_challenger_upload(result)
373
+
374
+ assert payload["champion_metrics"] == result.champion_metrics
375
+
376
+
377
+ def test_to_challenger_upload_explicit_args_override_result_defaults():
378
+ df = _labeled_df(seed=28)
379
+ champion_predictions = df["approved"].tolist()
380
+ result = train_auto_challenger(
381
+ df, "approved", task="classification", champion_predictions=champion_predictions
382
+ )
383
+
384
+ override_metrics = {"f1": 0.1, "accuracy": 0.1}
385
+ payload = to_challenger_upload(result, n_samples=999, champion_metrics=override_metrics)
386
+
387
+ assert payload["n_samples"] == 999
388
+ assert payload["champion_metrics"] == override_metrics
389
+
390
+
289
391
  def test_train_auto_challenger_passes_immutable_cols_to_get_schema():
290
392
  from unittest.mock import patch
291
393
 
@@ -300,3 +402,81 @@ def test_train_auto_challenger_passes_immutable_cols_to_get_schema():
300
402
  called_kwargs = mock_get_schema.call_args.kwargs
301
403
  assert list(called_df.columns) == list(features_df.columns)
302
404
  assert called_kwargs["immutable_cols"] == ["age"]
405
+
406
+
407
+ def test_target_fingerprint_is_deterministic_for_identical_data():
408
+ df = _labeled_df(seed=29)
409
+ result_a = train_auto_challenger(df, "approved", task="classification")
410
+ result_b = train_auto_challenger(df, "approved", task="classification")
411
+
412
+ assert result_a.target_fingerprint == result_b.target_fingerprint
413
+
414
+
415
+ def test_target_fingerprint_differs_for_different_data():
416
+ df_a = _labeled_df(seed=29)
417
+ df_b = _labeled_df(seed=30)
418
+ result_a = train_auto_challenger(df_a, "approved", task="classification")
419
+ result_b = train_auto_challenger(df_b, "approved", task="classification")
420
+
421
+ assert result_a.target_fingerprint != result_b.target_fingerprint
422
+
423
+
424
+ def test_to_challenger_upload_includes_matching_fingerprints_on_internal_champion_path():
425
+ df = _labeled_df(seed=31)
426
+ champion_predictions = df["approved"].tolist()
427
+ result = train_auto_challenger(
428
+ df, "approved", task="classification", champion_predictions=champion_predictions
429
+ )
430
+
431
+ payload = to_challenger_upload(result)
432
+
433
+ assert payload["challenger_data_fingerprint"] == result.target_fingerprint
434
+ assert payload["champion_data_fingerprint"] == result.target_fingerprint
435
+
436
+
437
+ def test_to_challenger_upload_champion_labels_fingerprint_for_decoupled_path():
438
+ df = _labeled_df(seed=32)
439
+ target = df["approved"]
440
+ result = train_challenger(df, target, _schema(), task="classification")
441
+ champion_metrics = score_champion(target, target, task="classification")
442
+
443
+ payload = to_challenger_upload(result, champion_metrics=champion_metrics, champion_labels=target)
444
+
445
+ assert payload["challenger_data_fingerprint"] == result.target_fingerprint
446
+ assert payload["champion_data_fingerprint"] == result.target_fingerprint
447
+
448
+
449
+ def test_to_challenger_upload_champion_labels_fingerprint_differs_for_different_data():
450
+ df = _labeled_df(seed=33)
451
+ target = df["approved"]
452
+ result = train_challenger(df, target, _schema(), task="classification")
453
+ champion_metrics = score_champion(target, target, task="classification")
454
+ other_labels = ~target
455
+
456
+ payload = to_challenger_upload(
457
+ result, champion_metrics=champion_metrics, champion_labels=other_labels
458
+ )
459
+
460
+ assert payload["challenger_data_fingerprint"] != payload["champion_data_fingerprint"]
461
+
462
+
463
+ def test_to_challenger_upload_omits_champion_fingerprint_without_labels_or_internal_path():
464
+ df = _labeled_df(seed=34)
465
+ target = df["approved"]
466
+ result = train_challenger(df, target, _schema(), task="classification")
467
+ champion_metrics = score_champion(target, target, task="classification")
468
+
469
+ payload = to_challenger_upload(result, champion_metrics=champion_metrics)
470
+
471
+ assert payload["challenger_data_fingerprint"] == result.target_fingerprint
472
+ assert "champion_data_fingerprint" not in payload
473
+
474
+
475
+ def test_to_challenger_upload_without_champion_metrics_omits_fingerprints():
476
+ df = _labeled_df(seed=35)
477
+ result = train_auto_challenger(df, "approved", task="classification")
478
+
479
+ payload = to_challenger_upload(result)
480
+
481
+ assert "challenger_data_fingerprint" not in payload
482
+ assert "champion_data_fingerprint" not in payload
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes