proxyml 0.4.1__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proxyml
3
- Version: 0.4.1
3
+ Version: 0.6.0
4
4
  Summary: Python SDK for calling the ProxyML API
5
5
  Author-email: ProxyML <contact@proxyml.ai>
6
6
  License: Apache License
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "proxyml"
7
- version = "0.4.1"
7
+ version = "0.6.0"
8
8
  description = "Python SDK for calling the ProxyML API"
9
9
  readme = "README.md"
10
10
  license = {file = "LICENSE"}
@@ -11,6 +11,7 @@ from proxyml.local.challenger import (
11
11
  Rung,
12
12
  TrainedChallenger,
13
13
  score_champion,
14
+ to_challenger_upload,
14
15
  train_auto_challenger,
15
16
  train_challenger,
16
17
  )
@@ -19,6 +20,7 @@ __all__ = [
19
20
  "train_challenger",
20
21
  "train_auto_challenger",
21
22
  "score_champion",
23
+ "to_challenger_upload",
22
24
  "Complexity",
23
25
  "Rung",
24
26
  "TrainedChallenger",
@@ -12,6 +12,7 @@ from __future__ import annotations
12
12
 
13
13
  from dataclasses import dataclass, replace
14
14
  from enum import Enum
15
+ from importlib.metadata import version as _pkg_version
15
16
  from pathlib import Path
16
17
  from typing import Any, Callable, Literal
17
18
 
@@ -113,9 +114,26 @@ LADDERS: dict[Complexity, Rung] = {
113
114
  class TrainedChallenger:
114
115
  pipeline: Pipeline
115
116
  task: Literal["classification", "regression"]
117
+ complexity: Complexity
116
118
  metrics: dict[str, float]
117
119
  hyperparameters: dict[str, Any]
118
120
  export: SurrogateExport
121
+ n_samples_total: int
122
+ n_samples_dropped_unlabeled: int
123
+ population_note: str
124
+ champion_metrics: dict[str, float] | None = None
125
+
126
+
127
+ def _population_note(target_name: str, n_total: int, n_labeled: int, n_dropped: int) -> str:
128
+ if n_dropped == 0:
129
+ return f"Evaluated on all {n_total} row(s) — '{target_name}' had no missing values."
130
+ return (
131
+ f"Evaluated on {n_labeled} of {n_total} row(s) with a non-null '{target_name}' value "
132
+ f"({n_dropped} unlabeled row(s) dropped before training/scoring). "
133
+ "Labeled-vs-unlabeled selection may not be random; treat this as a declared "
134
+ "scope limitation on the evaluation population, not a claim about performance "
135
+ "on the full dataset."
136
+ )
119
137
 
120
138
 
121
139
  def train_challenger(
@@ -127,6 +145,8 @@ def train_challenger(
127
145
  feature_names: list[str] | None = None,
128
146
  task: Literal["classification", "regression", "auto"] = "auto",
129
147
  test_size: float = 0.2,
148
+ target_name: str = "target",
149
+ champion_predictions: np.ndarray | list | None = None,
130
150
  ) -> TrainedChallenger:
131
151
  """Train a linear challenger model on ``df`` against ``target``, locally.
132
152
 
@@ -140,6 +160,15 @@ def train_challenger(
140
160
  can be compared with the same ``proxyml_core.export.predict_from_export``
141
161
  arithmetic.
142
162
 
163
+ Rows where ``target`` is missing (NaN/None) are dropped before training,
164
+ the CV split, and champion scoring — never silently included. The drop
165
+ count and a human-readable scope-limitation note are recorded on the
166
+ result (``n_samples_total``, ``n_samples_dropped_unlabeled``,
167
+ ``population_note``). If ``champion_predictions`` is given, it must have
168
+ one entry per row of ``df``/``target`` (same order) so the identical rows
169
+ are dropped from both sides — champion and challenger are always
170
+ evaluated on the same labeled population, never on different ones.
171
+
143
172
  Args:
144
173
  df: samples to train on, one column per schema feature.
145
174
  target: the value to predict for each row of ``df`` — ground-truth
@@ -149,7 +178,35 @@ def train_challenger(
149
178
  feature_names: subset of ``schema.features`` to train on; omit for all.
150
179
  task: "classification", "regression", or "auto" to infer from ``target``.
151
180
  test_size: fraction of data held out to compute fidelity metrics.
181
+ target_name: human-readable name for ``target``, used in
182
+ ``population_note`` (e.g. the column name, if known).
183
+ champion_predictions: a champion model's predictions, one per row of
184
+ ``df``/``target`` (same order). If given, scored via
185
+ ``score_champion()`` against the same (row-dropped) ``target``,
186
+ and the result is attached as ``TrainedChallenger.champion_metrics``.
152
187
  """
188
+ target_arr = np.asarray(target)
189
+ if champion_predictions is not None and len(champion_predictions) != len(target_arr):
190
+ raise ValueError(
191
+ f"champion_predictions must have one entry per row of target "
192
+ f"({len(target_arr)} rows, got {len(champion_predictions)}) — same order — so "
193
+ f"rows with a missing {target_name!r} value can be dropped from both the "
194
+ f"challenger and the champion, keeping them evaluated on the same population."
195
+ )
196
+
197
+ labeled_mask = ~pd.isna(target_arr)
198
+ n_total = len(target_arr)
199
+ n_labeled = int(labeled_mask.sum())
200
+ n_dropped = n_total - n_labeled
201
+ if n_labeled == 0:
202
+ raise ValueError(f"All {n_total} row(s) have a missing {target_name!r} value; nothing to train on")
203
+
204
+ df = df.iloc[labeled_mask].reset_index(drop=True)
205
+ target_arr = target_arr[labeled_mask]
206
+ champion_predictions_labeled = (
207
+ np.asarray(champion_predictions)[labeled_mask] if champion_predictions is not None else None
208
+ )
209
+
153
210
  rung = LADDERS[complexity]
154
211
 
155
212
  features: list[Feature] = schema.features
@@ -159,7 +216,7 @@ def train_challenger(
159
216
  col_order = [f.name for f in features]
160
217
 
161
218
  X = df[col_order].to_numpy(dtype=object)
162
- y = np.asarray(target)
219
+ y = target_arr
163
220
 
164
221
  if task == "auto":
165
222
  classification = is_classification(y)
@@ -172,6 +229,10 @@ def train_challenger(
172
229
  if classification:
173
230
  y = binarize_if_probabilities(y)
174
231
 
232
+ champion_metrics = None
233
+ if champion_predictions_labeled is not None:
234
+ champion_metrics = score_champion(y, champion_predictions_labeled, task=resolved_task)
235
+
175
236
  preprocessor = build_preprocessor(features)
176
237
  estimator = rung.build_classifier() if classification else rung.build_regressor()
177
238
  pipeline = Pipeline(steps=[("preprocessor", preprocessor), ("estimator", estimator)])
@@ -196,9 +257,14 @@ def train_challenger(
196
257
  return TrainedChallenger(
197
258
  pipeline=pipeline,
198
259
  task=resolved_task,
260
+ complexity=complexity,
199
261
  metrics=metrics,
200
262
  hyperparameters=hyperparameters,
201
263
  export=export,
264
+ n_samples_total=n_total,
265
+ n_samples_dropped_unlabeled=n_dropped,
266
+ population_note=_population_note(target_name, n_total, n_labeled, n_dropped),
267
+ champion_metrics=champion_metrics,
202
268
  )
203
269
 
204
270
 
@@ -224,6 +290,71 @@ def score_champion(
224
290
  return score_predictions(np.asarray(labels), np.asarray(predictions), task=task)
225
291
 
226
292
 
293
+ def to_challenger_upload(
294
+ result: TrainedChallenger,
295
+ *,
296
+ n_samples: int | None = None,
297
+ champion_metrics: dict[str, float] | None = None,
298
+ sdk_version: str | None = None,
299
+ proxyml_core_version: str | None = None,
300
+ ) -> dict[str, Any]:
301
+ """Assemble the JSON-serializable payload for a challenger upload.
302
+
303
+ Matches the shape ProxyML's dashboard/API expects at
304
+ ``POST /app/projects/{id}/challenger`` — handles the mechanical assembly
305
+ (serializing the export, stamping SDK/core versions, converting
306
+ ``complexity`` to a plain string) so callers don't have to hand-roll it.
307
+ The result is plain ``dict``/``str``/``float`` data, ready for
308
+ ``json.dump`` — upload it either by POSTing it directly, or by saving it
309
+ to a file and using the dashboard's "Upload challenger" button.
310
+
311
+ ``champion_metrics`` is optional: pass ``None`` (the default) to fall
312
+ back to ``result.champion_metrics`` (populated automatically if you
313
+ passed ``champion_predictions`` to ``train_challenger()``/
314
+ ``train_auto_challenger()``) — or, if that's also ``None``, to get a
315
+ self-contained export of the challenger alone, e.g. to save/share it
316
+ before you have a champion to compare against. The upload endpoint
317
+ itself still requires ``champion_metrics`` at upload time; this function
318
+ just doesn't force you to have it up front.
319
+
320
+ Args:
321
+ result: output of ``train_challenger()``/``train_auto_challenger()``.
322
+ n_samples: size of the evaluation set both ``result.metrics`` and
323
+ ``champion_metrics`` were scored on. Defaults to
324
+ ``result.n_samples_total - result.n_samples_dropped_unlabeled``
325
+ (the labeled-row count) — override only if you scored on some
326
+ other population.
327
+ champion_metrics: the champion's real-world performance, from
328
+ ``score_champion()`` — same metric keys as ``result.metrics``.
329
+ Defaults to ``result.champion_metrics``.
330
+ sdk_version: defaults to the installed ``proxyml`` version.
331
+ proxyml_core_version: defaults to the installed ``proxyml-core`` version.
332
+ """
333
+ if n_samples is None:
334
+ n_samples = result.n_samples_total - result.n_samples_dropped_unlabeled
335
+ if champion_metrics is None:
336
+ champion_metrics = result.champion_metrics
337
+ if sdk_version is None:
338
+ sdk_version = _pkg_version("proxyml")
339
+ if proxyml_core_version is None:
340
+ proxyml_core_version = _pkg_version("proxyml-core")
341
+
342
+ payload: dict[str, Any] = {
343
+ "export": result.export.to_dict(),
344
+ "challenger_metrics": result.metrics,
345
+ "n_samples": n_samples,
346
+ "n_samples_total": result.n_samples_total,
347
+ "n_samples_dropped_unlabeled": result.n_samples_dropped_unlabeled,
348
+ "population_note": result.population_note,
349
+ "complexity": result.complexity.value,
350
+ "sdk_version": sdk_version,
351
+ "proxyml_core_version": proxyml_core_version,
352
+ }
353
+ if champion_metrics is not None:
354
+ payload["champion_metrics"] = champion_metrics
355
+ return payload
356
+
357
+
227
358
  def train_auto_challenger(
228
359
  data: str | Path | pd.DataFrame,
229
360
  target_col: str,
@@ -233,6 +364,7 @@ def train_auto_challenger(
233
364
  feature_names: list[str] | None = None,
234
365
  task: Literal["classification", "regression", "auto"] = "auto",
235
366
  test_size: float = 0.2,
367
+ champion_predictions: np.ndarray | list | None = None,
236
368
  ) -> TrainedChallenger:
237
369
  """Load data, infer a schema, and train a linear challenger in one call.
238
370
 
@@ -242,6 +374,12 @@ def train_auto_challenger(
242
374
  remains overridable; this does not search across ``LADDERS`` to find the
243
375
  best-fitting rung.
244
376
 
377
+ Rows with a missing ``target_col`` value are dropped before training and
378
+ champion scoring — see ``train_challenger()`` for details. Schema
379
+ inference (feature means/stds/categories) still runs over every row,
380
+ including ones later dropped for a missing target — only training and
381
+ evaluation are restricted to the labeled subset.
382
+
245
383
  Args:
246
384
  data: a CSV path, or an already-loaded DataFrame containing both the
247
385
  feature columns and ``target_col``.
@@ -252,6 +390,8 @@ def train_auto_challenger(
252
390
  feature_names: subset of feature columns to train on; omit for all.
253
391
  task: "classification", "regression", or "auto" to infer from ``target_col``.
254
392
  test_size: fraction of data held out to compute fidelity metrics.
393
+ champion_predictions: a champion model's predictions, one per row of
394
+ ``data`` (same order) — see ``train_challenger()``.
255
395
  """
256
396
  df = data if isinstance(data, pd.DataFrame) else pd.read_csv(data)
257
397
  target = df[target_col]
@@ -266,4 +406,6 @@ def train_auto_challenger(
266
406
  feature_names=feature_names,
267
407
  task=task,
268
408
  test_size=test_size,
409
+ target_name=target_col,
410
+ champion_predictions=champion_predictions,
269
411
  )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proxyml
3
- Version: 0.4.1
3
+ Version: 0.6.0
4
4
  Summary: Python SDK for calling the ProxyML API
5
5
  Author-email: ProxyML <contact@proxyml.ai>
6
6
  License: Apache License
@@ -9,6 +9,7 @@ from proxyml.local import (
9
9
  LADDERS,
10
10
  TrainedChallenger,
11
11
  score_champion,
12
+ to_challenger_upload,
12
13
  train_auto_challenger,
13
14
  train_challenger,
14
15
  )
@@ -197,6 +198,196 @@ def test_score_champion_uses_same_scoring_as_train_challenger():
197
198
  assert reproduced_metrics["r2"] == pytest.approx(r2_score(target, y_pred))
198
199
 
199
200
 
201
+ def test_train_challenger_result_carries_complexity():
202
+ schema = _schema()
203
+ df = _df(seed=13)
204
+ target = df["age"] * 0.5 + df["income"] * 0.0001
205
+
206
+ result = train_challenger(df, target, schema, complexity=Complexity.FLEXIBLE, task="regression")
207
+
208
+ assert result.complexity is Complexity.FLEXIBLE
209
+
210
+
211
+ def test_train_auto_challenger_result_carries_complexity():
212
+ df = _labeled_df(seed=14)
213
+ result = train_auto_challenger(df, "approved", task="classification", complexity=Complexity.SIMPLE)
214
+
215
+ assert result.complexity is Complexity.SIMPLE
216
+
217
+
218
+ def test_to_challenger_upload_shape_without_champion_metrics():
219
+ schema = _schema()
220
+ df = _df(seed=15)
221
+ target = df["age"] * 0.5 + df["income"] * 0.0001
222
+ result = train_challenger(df, target, schema, complexity=Complexity.MODERATE, task="regression")
223
+
224
+ payload = to_challenger_upload(result, n_samples=40)
225
+
226
+ assert payload["export"] == result.export.to_dict()
227
+ assert payload["challenger_metrics"] == result.metrics
228
+ assert payload["n_samples"] == 40
229
+ assert payload["complexity"] == "moderate"
230
+ assert "champion_metrics" not in payload
231
+ assert isinstance(payload["sdk_version"], str) and payload["sdk_version"]
232
+ assert isinstance(payload["proxyml_core_version"], str) and payload["proxyml_core_version"]
233
+
234
+
235
+ def test_to_challenger_upload_includes_champion_metrics_when_given():
236
+ schema = _schema()
237
+ df = _df(seed=16)
238
+ target = df["age"] * 0.5 + df["income"] * 0.0001
239
+ result = train_challenger(df, target, schema, complexity=Complexity.MODERATE, task="regression")
240
+ champion_metrics = score_champion(target, target, task="regression")
241
+
242
+ payload = to_challenger_upload(result, n_samples=40, champion_metrics=champion_metrics)
243
+
244
+ assert payload["champion_metrics"] == champion_metrics
245
+
246
+
247
+ def test_to_challenger_upload_defaults_versions_to_installed_packages():
248
+ from importlib.metadata import version as pkg_version
249
+
250
+ schema = _schema()
251
+ df = _df(seed=17)
252
+ target = df["age"] * 0.5 + df["income"] * 0.0001
253
+ result = train_challenger(df, target, schema, task="regression")
254
+
255
+ payload = to_challenger_upload(result, n_samples=10)
256
+
257
+ assert payload["sdk_version"] == pkg_version("proxyml")
258
+ assert payload["proxyml_core_version"] == pkg_version("proxyml-core")
259
+
260
+
261
+ def test_to_challenger_upload_allows_version_override():
262
+ schema = _schema()
263
+ df = _df(seed=18)
264
+ target = df["age"] * 0.5 + df["income"] * 0.0001
265
+ result = train_challenger(df, target, schema, task="regression")
266
+
267
+ payload = to_challenger_upload(
268
+ result, n_samples=10, sdk_version="9.9.9", proxyml_core_version="8.8.8"
269
+ )
270
+
271
+ assert payload["sdk_version"] == "9.9.9"
272
+ assert payload["proxyml_core_version"] == "8.8.8"
273
+
274
+
275
+ def test_to_challenger_upload_payload_is_json_serializable():
276
+ import json
277
+
278
+ schema = _schema()
279
+ df = _df(seed=19, n=300)
280
+ target = np.where(df["age"] > 50, "senior", "junior")
281
+ result = train_challenger(df, target, schema, task="classification")
282
+ champion_metrics = score_champion(target, target, task="classification")
283
+
284
+ payload = to_challenger_upload(result, n_samples=300, champion_metrics=champion_metrics)
285
+
286
+ json.dumps(payload) # must not raise
287
+
288
+
289
+ def _labeled_df_with_nan_target(n=200, n_nan=20, seed=20):
290
+ df = _labeled_df(n=n, seed=seed)
291
+ df["approved"] = df["approved"].astype(float)
292
+ df.loc[df.index[:n_nan], "approved"] = np.nan
293
+ return df
294
+
295
+
296
+ def test_nan_target_rows_are_dropped_and_counted():
297
+ df = _labeled_df_with_nan_target(n=200, n_nan=20)
298
+ result = train_auto_challenger(df, "approved", task="classification")
299
+
300
+ assert result.n_samples_total == 200
301
+ assert result.n_samples_dropped_unlabeled == 20
302
+ assert "20" in result.population_note
303
+ assert "180" in result.population_note
304
+
305
+
306
+ def test_no_nan_targets_reports_zero_dropped():
307
+ df = _labeled_df(seed=21)
308
+ result = train_auto_challenger(df, "approved", task="classification")
309
+
310
+ assert result.n_samples_total == len(df)
311
+ assert result.n_samples_dropped_unlabeled == 0
312
+ assert "no missing values" in result.population_note
313
+
314
+
315
+ def test_all_nan_target_raises():
316
+ df = _labeled_df(n=20, seed=22)
317
+ df["approved"] = np.nan
318
+ with pytest.raises(ValueError, match="missing"):
319
+ train_auto_challenger(df, "approved", task="classification")
320
+
321
+
322
+ def test_champion_predictions_wrong_length_raises():
323
+ df = _labeled_df(seed=23)
324
+ with pytest.raises(ValueError, match="one entry per row"):
325
+ train_auto_challenger(
326
+ df, "approved", task="classification", champion_predictions=[True, False]
327
+ )
328
+
329
+
330
+ def test_champion_predictions_scored_only_on_labeled_rows():
331
+ # Champion predictions mirror the (possibly-NaN) target itself, except on
332
+ # rows that get dropped as unlabeled, where they're deliberately wrong.
333
+ # If those rows leaked into scoring, champion accuracy would come in
334
+ # under 1.0 instead of exactly 1.0 — proving the shared-drop guarantee,
335
+ # not just that nothing crashes.
336
+ df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=24)
337
+ champion_predictions = [False if pd.isna(v) else v for v in df["approved"]]
338
+
339
+ result = train_auto_challenger(
340
+ df, "approved", task="classification", champion_predictions=champion_predictions
341
+ )
342
+
343
+ assert result.champion_metrics is not None
344
+ assert result.champion_metrics["accuracy"] == 1.0
345
+
346
+
347
+ def test_champion_predictions_not_given_leaves_champion_metrics_none():
348
+ df = _labeled_df(seed=25)
349
+ result = train_auto_challenger(df, "approved", task="classification")
350
+ assert result.champion_metrics is None
351
+
352
+
353
+ def test_to_challenger_upload_defaults_n_samples_from_result():
354
+ df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=26)
355
+ result = train_auto_challenger(df, "approved", task="classification")
356
+
357
+ payload = to_challenger_upload(result)
358
+
359
+ assert payload["n_samples"] == 180
360
+ assert payload["n_samples_total"] == 200
361
+ assert payload["n_samples_dropped_unlabeled"] == 20
362
+ assert payload["population_note"] == result.population_note
363
+
364
+
365
+ def test_to_challenger_upload_defaults_champion_metrics_from_result():
366
+ df = _labeled_df(seed=27)
367
+ champion_predictions = df["approved"].tolist()
368
+ result = train_auto_challenger(
369
+ df, "approved", task="classification", champion_predictions=champion_predictions
370
+ )
371
+
372
+ payload = to_challenger_upload(result)
373
+
374
+ assert payload["champion_metrics"] == result.champion_metrics
375
+
376
+
377
+ def test_to_challenger_upload_explicit_args_override_result_defaults():
378
+ df = _labeled_df(seed=28)
379
+ champion_predictions = df["approved"].tolist()
380
+ result = train_auto_challenger(
381
+ df, "approved", task="classification", champion_predictions=champion_predictions
382
+ )
383
+
384
+ override_metrics = {"f1": 0.1, "accuracy": 0.1}
385
+ payload = to_challenger_upload(result, n_samples=999, champion_metrics=override_metrics)
386
+
387
+ assert payload["n_samples"] == 999
388
+ assert payload["champion_metrics"] == override_metrics
389
+
390
+
200
391
  def test_train_auto_challenger_passes_immutable_cols_to_get_schema():
201
392
  from unittest.mock import patch
202
393
 
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes