proxyml 0.5.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proxyml
3
- Version: 0.5.0
3
+ Version: 0.6.0
4
4
  Summary: Python SDK for calling the ProxyML API
5
5
  Author-email: ProxyML <contact@proxyml.ai>
6
6
  License: Apache License
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "proxyml"
7
- version = "0.5.0"
7
+ version = "0.6.0"
8
8
  description = "Python SDK for calling the ProxyML API"
9
9
  readme = "README.md"
10
10
  license = {file = "LICENSE"}
@@ -118,6 +118,22 @@ class TrainedChallenger:
118
118
  metrics: dict[str, float]
119
119
  hyperparameters: dict[str, Any]
120
120
  export: SurrogateExport
121
+ n_samples_total: int
122
+ n_samples_dropped_unlabeled: int
123
+ population_note: str
124
+ champion_metrics: dict[str, float] | None = None
125
+
126
+
127
+ def _population_note(target_name: str, n_total: int, n_labeled: int, n_dropped: int) -> str:
128
+ if n_dropped == 0:
129
+ return f"Evaluated on all {n_total} row(s) — '{target_name}' had no missing values."
130
+ return (
131
+ f"Evaluated on {n_labeled} of {n_total} row(s) with a non-null '{target_name}' value "
132
+ f"({n_dropped} unlabeled row(s) dropped before training/scoring). "
133
+ "Labeled-vs-unlabeled selection may not be random; treat this as a declared "
134
+ "scope limitation on the evaluation population, not a claim about performance "
135
+ "on the full dataset."
136
+ )
121
137
 
122
138
 
123
139
  def train_challenger(
@@ -129,6 +145,8 @@ def train_challenger(
129
145
  feature_names: list[str] | None = None,
130
146
  task: Literal["classification", "regression", "auto"] = "auto",
131
147
  test_size: float = 0.2,
148
+ target_name: str = "target",
149
+ champion_predictions: np.ndarray | list | None = None,
132
150
  ) -> TrainedChallenger:
133
151
  """Train a linear challenger model on ``df`` against ``target``, locally.
134
152
 
@@ -142,6 +160,15 @@ def train_challenger(
142
160
  can be compared with the same ``proxyml_core.export.predict_from_export``
143
161
  arithmetic.
144
162
 
163
+ Rows where ``target`` is missing (NaN/None) are dropped before training,
164
+ the CV split, and champion scoring — never silently included. The drop
165
+ count and a human-readable scope-limitation note are recorded on the
166
+ result (``n_samples_total``, ``n_samples_dropped_unlabeled``,
167
+ ``population_note``). If ``champion_predictions`` is given, it must have
168
+ one entry per row of ``df``/``target`` (same order) so the identical rows
169
+ are dropped from both sides — champion and challenger are always
170
+ evaluated on the same labeled population, never on different ones.
171
+
145
172
  Args:
146
173
  df: samples to train on, one column per schema feature.
147
174
  target: the value to predict for each row of ``df`` — ground-truth
@@ -151,7 +178,35 @@ def train_challenger(
151
178
  feature_names: subset of ``schema.features`` to train on; omit for all.
152
179
  task: "classification", "regression", or "auto" to infer from ``target``.
153
180
  test_size: fraction of data held out to compute fidelity metrics.
181
+ target_name: human-readable name for ``target``, used in
182
+ ``population_note`` (e.g. the column name, if known).
183
+ champion_predictions: a champion model's predictions, one per row of
184
+ ``df``/``target`` (same order). If given, scored via
185
+ ``score_champion()`` against the same (row-dropped) ``target``,
186
+ and the result is attached as ``TrainedChallenger.champion_metrics``.
154
187
  """
188
+ target_arr = np.asarray(target)
189
+ if champion_predictions is not None and len(champion_predictions) != len(target_arr):
190
+ raise ValueError(
191
+ f"champion_predictions must have one entry per row of target "
192
+ f"({len(target_arr)} rows, got {len(champion_predictions)}) — same order — so "
193
+ f"rows with a missing {target_name!r} value can be dropped from both the "
194
+ f"challenger and the champion, keeping them evaluated on the same population."
195
+ )
196
+
197
+ labeled_mask = ~pd.isna(target_arr)
198
+ n_total = len(target_arr)
199
+ n_labeled = int(labeled_mask.sum())
200
+ n_dropped = n_total - n_labeled
201
+ if n_labeled == 0:
202
+ raise ValueError(f"All {n_total} row(s) have a missing {target_name!r} value; nothing to train on")
203
+
204
+ df = df.iloc[labeled_mask].reset_index(drop=True)
205
+ target_arr = target_arr[labeled_mask]
206
+ champion_predictions_labeled = (
207
+ np.asarray(champion_predictions)[labeled_mask] if champion_predictions is not None else None
208
+ )
209
+
155
210
  rung = LADDERS[complexity]
156
211
 
157
212
  features: list[Feature] = schema.features
@@ -161,7 +216,7 @@ def train_challenger(
161
216
  col_order = [f.name for f in features]
162
217
 
163
218
  X = df[col_order].to_numpy(dtype=object)
164
- y = np.asarray(target)
219
+ y = target_arr
165
220
 
166
221
  if task == "auto":
167
222
  classification = is_classification(y)
@@ -174,6 +229,10 @@ def train_challenger(
174
229
  if classification:
175
230
  y = binarize_if_probabilities(y)
176
231
 
232
+ champion_metrics = None
233
+ if champion_predictions_labeled is not None:
234
+ champion_metrics = score_champion(y, champion_predictions_labeled, task=resolved_task)
235
+
177
236
  preprocessor = build_preprocessor(features)
178
237
  estimator = rung.build_classifier() if classification else rung.build_regressor()
179
238
  pipeline = Pipeline(steps=[("preprocessor", preprocessor), ("estimator", estimator)])
@@ -202,6 +261,10 @@ def train_challenger(
202
261
  metrics=metrics,
203
262
  hyperparameters=hyperparameters,
204
263
  export=export,
264
+ n_samples_total=n_total,
265
+ n_samples_dropped_unlabeled=n_dropped,
266
+ population_note=_population_note(target_name, n_total, n_labeled, n_dropped),
267
+ champion_metrics=champion_metrics,
205
268
  )
206
269
 
207
270
 
@@ -230,7 +293,7 @@ def score_champion(
230
293
  def to_challenger_upload(
231
294
  result: TrainedChallenger,
232
295
  *,
233
- n_samples: int,
296
+ n_samples: int | None = None,
234
297
  champion_metrics: dict[str, float] | None = None,
235
298
  sdk_version: str | None = None,
236
299
  proxyml_core_version: str | None = None,
@@ -245,27 +308,32 @@ def to_challenger_upload(
245
308
  ``json.dump`` — upload it either by POSTing it directly, or by saving it
246
309
  to a file and using the dashboard's "Upload challenger" button.
247
310
 
248
- ``champion_metrics`` is optional: pass ``None`` (the default) to get a
249
- self-contained export of the challenger alone — e.g. to save/share it
250
- before you have a champion to compare against — and fill in
251
- ``champion_metrics`` later. The upload endpoint itself still requires
252
- ``champion_metrics`` at upload time; this function just doesn't force you
253
- to have it up front.
311
+ ``champion_metrics`` is optional: pass ``None`` (the default) to fall
312
+ back to ``result.champion_metrics`` (populated automatically if you
313
+ passed ``champion_predictions`` to ``train_challenger()``/
314
+ ``train_auto_challenger()``) — or, if that's also ``None``, to get a
315
+ self-contained export of the challenger alone, e.g. to save/share it
316
+ before you have a champion to compare against. The upload endpoint
317
+ itself still requires ``champion_metrics`` at upload time; this function
318
+ just doesn't force you to have it up front.
254
319
 
255
320
  Args:
256
321
  result: output of ``train_challenger()``/``train_auto_challenger()``.
257
322
  n_samples: size of the evaluation set both ``result.metrics`` and
258
- ``champion_metrics`` were scored on. Not derived automatically —
259
- ``TrainedChallenger`` doesn't retain its internal held-out split,
260
- and ``champion_metrics`` typically comes from a separate
261
- ``score_champion()`` call the two need to share, so the caller is
262
- the only one who actually knows this number.
323
+ ``champion_metrics`` were scored on. Defaults to
324
+ ``result.n_samples_total - result.n_samples_dropped_unlabeled``
325
+ (the labeled-row count) — override only if you scored on some
326
+ other population.
263
327
  champion_metrics: the champion's real-world performance, from
264
328
  ``score_champion()`` — same metric keys as ``result.metrics``.
265
- Omit if you don't have it yet.
329
+ Defaults to ``result.champion_metrics``.
266
330
  sdk_version: defaults to the installed ``proxyml`` version.
267
331
  proxyml_core_version: defaults to the installed ``proxyml-core`` version.
268
332
  """
333
+ if n_samples is None:
334
+ n_samples = result.n_samples_total - result.n_samples_dropped_unlabeled
335
+ if champion_metrics is None:
336
+ champion_metrics = result.champion_metrics
269
337
  if sdk_version is None:
270
338
  sdk_version = _pkg_version("proxyml")
271
339
  if proxyml_core_version is None:
@@ -275,6 +343,9 @@ def to_challenger_upload(
275
343
  "export": result.export.to_dict(),
276
344
  "challenger_metrics": result.metrics,
277
345
  "n_samples": n_samples,
346
+ "n_samples_total": result.n_samples_total,
347
+ "n_samples_dropped_unlabeled": result.n_samples_dropped_unlabeled,
348
+ "population_note": result.population_note,
278
349
  "complexity": result.complexity.value,
279
350
  "sdk_version": sdk_version,
280
351
  "proxyml_core_version": proxyml_core_version,
@@ -293,6 +364,7 @@ def train_auto_challenger(
293
364
  feature_names: list[str] | None = None,
294
365
  task: Literal["classification", "regression", "auto"] = "auto",
295
366
  test_size: float = 0.2,
367
+ champion_predictions: np.ndarray | list | None = None,
296
368
  ) -> TrainedChallenger:
297
369
  """Load data, infer a schema, and train a linear challenger in one call.
298
370
 
@@ -302,6 +374,12 @@ def train_auto_challenger(
302
374
  remains overridable; this does not search across ``LADDERS`` to find the
303
375
  best-fitting rung.
304
376
 
377
+ Rows with a missing ``target_col`` value are dropped before training and
378
+ champion scoring — see ``train_challenger()`` for details. Schema
379
+ inference (feature means/stds/categories) still runs over every row,
380
+ including ones later dropped for a missing target — only training and
381
+ evaluation are restricted to the labeled subset.
382
+
305
383
  Args:
306
384
  data: a CSV path, or an already-loaded DataFrame containing both the
307
385
  feature columns and ``target_col``.
@@ -312,6 +390,8 @@ def train_auto_challenger(
312
390
  feature_names: subset of feature columns to train on; omit for all.
313
391
  task: "classification", "regression", or "auto" to infer from ``target_col``.
314
392
  test_size: fraction of data held out to compute fidelity metrics.
393
+ champion_predictions: a champion model's predictions, one per row of
394
+ ``data`` (same order) — see ``train_challenger()``.
315
395
  """
316
396
  df = data if isinstance(data, pd.DataFrame) else pd.read_csv(data)
317
397
  target = df[target_col]
@@ -326,4 +406,6 @@ def train_auto_challenger(
326
406
  feature_names=feature_names,
327
407
  task=task,
328
408
  test_size=test_size,
409
+ target_name=target_col,
410
+ champion_predictions=champion_predictions,
329
411
  )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: proxyml
3
- Version: 0.5.0
3
+ Version: 0.6.0
4
4
  Summary: Python SDK for calling the ProxyML API
5
5
  Author-email: ProxyML <contact@proxyml.ai>
6
6
  License: Apache License
@@ -286,6 +286,108 @@ def test_to_challenger_upload_payload_is_json_serializable():
286
286
  json.dumps(payload) # must not raise
287
287
 
288
288
 
289
+ def _labeled_df_with_nan_target(n=200, n_nan=20, seed=20):
290
+ df = _labeled_df(n=n, seed=seed)
291
+ df["approved"] = df["approved"].astype(float)
292
+ df.loc[df.index[:n_nan], "approved"] = np.nan
293
+ return df
294
+
295
+
296
+ def test_nan_target_rows_are_dropped_and_counted():
297
+ df = _labeled_df_with_nan_target(n=200, n_nan=20)
298
+ result = train_auto_challenger(df, "approved", task="classification")
299
+
300
+ assert result.n_samples_total == 200
301
+ assert result.n_samples_dropped_unlabeled == 20
302
+ assert "20" in result.population_note
303
+ assert "180" in result.population_note
304
+
305
+
306
+ def test_no_nan_targets_reports_zero_dropped():
307
+ df = _labeled_df(seed=21)
308
+ result = train_auto_challenger(df, "approved", task="classification")
309
+
310
+ assert result.n_samples_total == len(df)
311
+ assert result.n_samples_dropped_unlabeled == 0
312
+ assert "no missing values" in result.population_note
313
+
314
+
315
+ def test_all_nan_target_raises():
316
+ df = _labeled_df(n=20, seed=22)
317
+ df["approved"] = np.nan
318
+ with pytest.raises(ValueError, match="missing"):
319
+ train_auto_challenger(df, "approved", task="classification")
320
+
321
+
322
+ def test_champion_predictions_wrong_length_raises():
323
+ df = _labeled_df(seed=23)
324
+ with pytest.raises(ValueError, match="one entry per row"):
325
+ train_auto_challenger(
326
+ df, "approved", task="classification", champion_predictions=[True, False]
327
+ )
328
+
329
+
330
+ def test_champion_predictions_scored_only_on_labeled_rows():
331
+ # Champion predictions mirror the (possibly-NaN) target itself, except on
332
+ # rows that get dropped as unlabeled, where they're deliberately wrong.
333
+ # If those rows leaked into scoring, champion accuracy would come in
334
+ # under 1.0 instead of exactly 1.0 — proving the shared-drop guarantee,
335
+ # not just that nothing crashes.
336
+ df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=24)
337
+ champion_predictions = [False if pd.isna(v) else v for v in df["approved"]]
338
+
339
+ result = train_auto_challenger(
340
+ df, "approved", task="classification", champion_predictions=champion_predictions
341
+ )
342
+
343
+ assert result.champion_metrics is not None
344
+ assert result.champion_metrics["accuracy"] == 1.0
345
+
346
+
347
+ def test_champion_predictions_not_given_leaves_champion_metrics_none():
348
+ df = _labeled_df(seed=25)
349
+ result = train_auto_challenger(df, "approved", task="classification")
350
+ assert result.champion_metrics is None
351
+
352
+
353
+ def test_to_challenger_upload_defaults_n_samples_from_result():
354
+ df = _labeled_df_with_nan_target(n=200, n_nan=20, seed=26)
355
+ result = train_auto_challenger(df, "approved", task="classification")
356
+
357
+ payload = to_challenger_upload(result)
358
+
359
+ assert payload["n_samples"] == 180
360
+ assert payload["n_samples_total"] == 200
361
+ assert payload["n_samples_dropped_unlabeled"] == 20
362
+ assert payload["population_note"] == result.population_note
363
+
364
+
365
+ def test_to_challenger_upload_defaults_champion_metrics_from_result():
366
+ df = _labeled_df(seed=27)
367
+ champion_predictions = df["approved"].tolist()
368
+ result = train_auto_challenger(
369
+ df, "approved", task="classification", champion_predictions=champion_predictions
370
+ )
371
+
372
+ payload = to_challenger_upload(result)
373
+
374
+ assert payload["champion_metrics"] == result.champion_metrics
375
+
376
+
377
+ def test_to_challenger_upload_explicit_args_override_result_defaults():
378
+ df = _labeled_df(seed=28)
379
+ champion_predictions = df["approved"].tolist()
380
+ result = train_auto_challenger(
381
+ df, "approved", task="classification", champion_predictions=champion_predictions
382
+ )
383
+
384
+ override_metrics = {"f1": 0.1, "accuracy": 0.1}
385
+ payload = to_challenger_upload(result, n_samples=999, champion_metrics=override_metrics)
386
+
387
+ assert payload["n_samples"] == 999
388
+ assert payload["champion_metrics"] == override_metrics
389
+
390
+
289
391
  def test_train_auto_challenger_passes_immutable_cols_to_get_schema():
290
392
  from unittest.mock import patch
291
393
 
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes