auditkit 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. auditkit/README.md +99 -0
  2. auditkit/__init__.py +177 -0
  3. auditkit/__main__.py +3 -0
  4. auditkit/_bootstrap.py +77 -0
  5. auditkit/_identity_guard.py +99 -0
  6. auditkit/adapter.py +264 -0
  7. auditkit/annotator.py +339 -0
  8. auditkit/api.py +502 -0
  9. auditkit/assets/auditkit_logo.png +0 -0
  10. auditkit/cache.py +47 -0
  11. auditkit/cli.py +417 -0
  12. auditkit/comparison.py +563 -0
  13. auditkit/diff.py +265 -0
  14. auditkit/errors.py +54 -0
  15. auditkit/evaluator.py +20 -0
  16. auditkit/experiment.py +145 -0
  17. auditkit/hf_publish.py +262 -0
  18. auditkit/lmeval_engine.py +550 -0
  19. auditkit/loaders.py +121 -0
  20. auditkit/logs.py +18 -0
  21. auditkit/metric.py +199 -0
  22. auditkit/metrics/README.md +15 -0
  23. auditkit/metrics/__init__.py +0 -0
  24. auditkit/metrics/code.py +222 -0
  25. auditkit/metrics/embedding.py +131 -0
  26. auditkit/metrics/encoder_judge.py +423 -0
  27. auditkit/metrics/generation.py +331 -0
  28. auditkit/metrics/guard.py +412 -0
  29. auditkit/metrics/hallucination.py +45 -0
  30. auditkit/metrics/judge.py +547 -0
  31. auditkit/metrics/pairwise.py +153 -0
  32. auditkit/metrics/perf.py +53 -0
  33. auditkit/metrics/rag.py +149 -0
  34. auditkit/metrics/security.py +64 -0
  35. auditkit/metrics/toxicity.py +238 -0
  36. auditkit/model/README.md +16 -0
  37. auditkit/model/__init__.py +485 -0
  38. auditkit/model/anthropic.py +94 -0
  39. auditkit/model/api_gen.py +133 -0
  40. auditkit/model/groq_gen.py +121 -0
  41. auditkit/model/hf_gen.py +385 -0
  42. auditkit/model/lexsi.py +155 -0
  43. auditkit/model/litellm_gen.py +65 -0
  44. auditkit/model/openai.py +90 -0
  45. auditkit/model/openrouter_gen.py +152 -0
  46. auditkit/model/vllm_gen.py +316 -0
  47. auditkit/model_compare.py +655 -0
  48. auditkit/redteam/README.md +9 -0
  49. auditkit/redteam/__init__.py +26 -0
  50. auditkit/redteam/detector.py +37 -0
  51. auditkit/redteam/detectors/README.md +5 -0
  52. auditkit/redteam/detectors/builtin.py +126 -0
  53. auditkit/redteam/probe.py +39 -0
  54. auditkit/redteam/probes/README.md +5 -0
  55. auditkit/redteam/probes/builtin.py +85 -0
  56. auditkit/redteam/runner.py +206 -0
  57. auditkit/registry.py +65 -0
  58. auditkit/report.py +278 -0
  59. auditkit/report_format.py +52 -0
  60. auditkit/router.py +54 -0
  61. auditkit/runner.py +575 -0
  62. auditkit/runspec.py +159 -0
  63. auditkit/sample.py +40 -0
  64. auditkit/scenario.py +88 -0
  65. auditkit/scenarios/README.md +10 -0
  66. auditkit/scenarios/__init__.py +4 -0
  67. auditkit/scenarios/arc.py +33 -0
  68. auditkit/scenarios/gsm8k.py +32 -0
  69. auditkit/scenarios/hellaswag.py +33 -0
  70. auditkit/scenarios/humaneval.py +32 -0
  71. auditkit/scenarios/mmlu.py +34 -0
  72. auditkit/scenarios/truthfulqa.py +33 -0
  73. auditkit/score.py +165 -0
  74. auditkit/scorers.py +117 -0
  75. auditkit/scoring.py +79 -0
  76. auditkit/types.py +69 -0
  77. auditkit-1.0.0.dist-info/METADATA +396 -0
  78. auditkit-1.0.0.dist-info/RECORD +81 -0
  79. auditkit-1.0.0.dist-info/WHEEL +4 -0
  80. auditkit-1.0.0.dist-info/entry_points.txt +2 -0
  81. auditkit-1.0.0.dist-info/licenses/LICENSE.md +92 -0
auditkit/comparison.py ADDED
@@ -0,0 +1,563 @@
1
+ """Baseline-anchored comparison of two runs — the "did pruning/quantizing hurt?" view.
2
+
3
+ `RunComparison(baseline, candidate)` is a read-side helper over two immutable
4
+ `RunResult`s produced with the same eval config (e.g. a base model vs a pruned or
5
+ quantized one). It gives:
6
+
7
+ - **per-task deltas** (engine-agnostic: benchmark runs key per-task metrics in
8
+ `stats` as ``"task:metric"``; native runs carry a `task` on each `Prediction`
9
+ and a per-score breakdown in `metadata["scores"]` — both are folded in),
10
+ - **percentage differences** per metric (the summary reports the absolute delta
11
+ and the relative % -- no pass/warn/fail verdict and no better/worse framing;
12
+ a fixed threshold isn't meaningful across every metric, so it's the reader's
13
+ job to interpret the numbers. An opt-in threshold gate is still available via
14
+ .grade() with caller-chosen thresholds),
15
+ - **retention %** (candidate/baseline for MAXIMIZE, inverted for MINIMIZE),
16
+ - **significance** via the shared by-sample-id paired bootstrap,
17
+ - a **size/latency tradeoff** view over caller-supplied numbers,
18
+ - and the **regressed / improved** sample browser (delegated to `RunDiff`).
19
+
20
+ Lineage-based *auto* baseline selection is deliberately a platform concern; here
21
+ you hand it the two runs.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import logging
27
+ import statistics
28
+ from collections import defaultdict
29
+ from dataclasses import dataclass
30
+ from typing import Any, Optional
31
+
32
+ from ._bootstrap import paired_bootstrap, score_pairs
33
+ from .diff import (
34
+ DeltaGrade, RunDiff, direction_for, grade_delta, metric_directions,
35
+ relative_pct,
36
+ )
37
+ from .report import Prediction, RunResult
38
+ from .types import Direction
39
+
40
+ logger = logging.getLogger(__name__)
41
+
42
+ # NOT_COMPARABLE deliberately ranks below every real grade: a single metric
43
+ # that was only scored on one side shouldn't be able to make grade()'s
44
+ # worst-case reduction report a FAIL-equivalent when every metric that
45
+ # actually WAS compared passed cleanly. If every metric is NOT_COMPARABLE
46
+ # (nothing at all could be compared), that's still what grade() reports --
47
+ # there's nothing better to fall back to.
48
+ _GRADE_ORDER = {
49
+ DeltaGrade.NOT_COMPARABLE: -1,
50
+ DeltaGrade.PASS: 0,
51
+ DeltaGrade.WARN: 1,
52
+ DeltaGrade.FAIL: 2,
53
+ }
54
+
55
+
56
+ def _delta_line(label: str, d) -> str:
57
+ """One human line for a MetricDelta/TaskDelta: baseline -> candidate, the
58
+ absolute delta AND the relative %, with the sample count/std. No verdict and
59
+ no better/worse framing -- the reader interprets the numbers for their own
60
+ metric. ASCII only (printed to consoles that can't encode arrows)."""
61
+ if d.delta is None:
62
+ return (f"{label}: N/A -- only scored on one side "
63
+ f"(n={d.baseline_count}->{d.candidate_count})")
64
+ pct_s = f"{d.pct_change:+.1f}% rel" if d.pct_change is not None else "rel n/a"
65
+ return (
66
+ f"{label}: {d.baseline:.4f} -> {d.candidate:.4f} "
67
+ f"delta={d.delta:+.4f} ({pct_s}) "
68
+ f"(n={d.baseline_count}->{d.candidate_count}, "
69
+ f"std={d.baseline_std:.4f}->{d.candidate_std:.4f})"
70
+ )
71
+
72
+
73
+ @dataclass
74
+ class MetricDelta:
75
+ metric: str
76
+ # None when the metric was scored on only one side (grade is then
77
+ # NOT_COMPARABLE) -- never a fabricated 0.0 for whichever side didn't
78
+ # score it.
79
+ baseline: Optional[float]
80
+ candidate: Optional[float]
81
+ delta: Optional[float]
82
+ direction: str
83
+ # Threshold-based pass/warn/fail. Kept for callers who explicitly want a
84
+ # gate, but NOT what the default summary reports -- a fixed % threshold
85
+ # isn't meaningful across every metric (see summary()). Read the percentage
86
+ # difference (`delta` / `pct_change`) and decide acceptability yourself.
87
+ grade: DeltaGrade
88
+ # How many real samples fed each mean, and how spread out they were --
89
+ # .mean alone can't tell you a regression rode on far fewer samples than
90
+ # the baseline, or that a flat mean hid much wider variance. See
91
+ # RunComparison.summary()'s coverage warning, which reads these.
92
+ baseline_count: int = 0
93
+ candidate_count: int = 0
94
+ baseline_std: float = 0.0
95
+ candidate_std: float = 0.0
96
+ # The change as a percentage of the baseline (candidate vs baseline). None
97
+ # when incomparable or the baseline is <= 0 (percent-of-zero is undefined).
98
+ pct_change: Optional[float] = None
99
+
100
+
101
+ @dataclass
102
+ class TaskDelta:
103
+ task: str
104
+ metric: str
105
+ baseline: Optional[float]
106
+ candidate: Optional[float]
107
+ delta: Optional[float]
108
+ direction: str
109
+ grade: DeltaGrade
110
+ baseline_count: int = 0
111
+ candidate_count: int = 0
112
+ baseline_std: float = 0.0
113
+ candidate_std: float = 0.0
114
+ pct_change: Optional[float] = None
115
+
116
+
117
+ class RunComparison:
118
+ """Compare a ``candidate`` run against a ``baseline`` run on the same config."""
119
+
120
+ def __init__(
121
+ self,
122
+ baseline: RunResult,
123
+ candidate: RunResult,
124
+ *,
125
+ pass_threshold: float = 0.02,
126
+ warn_threshold: float = 0.05,
127
+ ) -> None:
128
+ self.baseline = baseline
129
+ self.candidate = candidate
130
+ self.pass_threshold = pass_threshold
131
+ self.warn_threshold = warn_threshold
132
+ self._diff = RunDiff(baseline, candidate)
133
+
134
+ # -- metric directions (read from stored per-score dicts) --------------
135
+ def _directions(self) -> dict[str, Direction]:
136
+ return metric_directions((self.baseline, self.candidate))
137
+
138
+ def _direction(self, metric: str) -> Direction:
139
+ return direction_for(metric, self._directions())
140
+
141
+ def _grade(self, delta: Optional[float], direction: Direction) -> DeltaGrade:
142
+ # None means "this metric wasn't scored on both sides" -- nothing to
143
+ # grade, not a fabricated PASS/FAIL against a phantom zero delta.
144
+ if delta is None:
145
+ return DeltaGrade.NOT_COMPARABLE
146
+ return grade_delta(delta, direction, self.pass_threshold, self.warn_threshold)
147
+
148
+ # -- overall per-metric deltas ----------------------------------------
149
+ def metric_deltas(self) -> list[MetricDelta]:
150
+ out: list[MetricDelta] = []
151
+ for name, info in self._diff.metric_deltas().items():
152
+ d = self._direction(name)
153
+ b_stat = self.baseline.stats.get(name)
154
+ c_stat = self.candidate.stats.get(name)
155
+ out.append(MetricDelta(
156
+ metric=name, baseline=info["baseline"], candidate=info["contrast"],
157
+ delta=info["delta"], direction=d.value, grade=self._grade(info["delta"], d),
158
+ baseline_count=b_stat.count if b_stat else 0,
159
+ candidate_count=c_stat.count if c_stat else 0,
160
+ baseline_std=b_stat.std if b_stat else 0.0,
161
+ candidate_std=c_stat.std if c_stat else 0.0,
162
+ pct_change=relative_pct(info["baseline"], info["delta"]),
163
+ ))
164
+ return out
165
+
166
+ # -- per-task deltas (engine-agnostic) --------------------------------
167
+ def _per_task_stats(self, run: RunResult) -> dict[tuple[str, str], dict[str, float]]:
168
+ stats: dict[tuple[str, str], dict[str, float]] = {}
169
+ # Benchmark engine: stats keyed "task:metric" -- Stat already tracks count/std.
170
+ for key, stat in run.stats.items():
171
+ if ":" in key:
172
+ task, metric = key.split(":", 1)
173
+ stats[(task, metric)] = {"mean": stat.mean, "count": stat.count, "std": stat.std}
174
+ # Native engine: group predictions by task via their per-score dicts.
175
+ acc: dict[tuple[str, str], list[float]] = defaultdict(list)
176
+ for p in run.predictions:
177
+ for s in p.metadata.get("scores", []):
178
+ name, val = s.get("name"), s.get("value")
179
+ if name is not None and isinstance(val, (int, float)):
180
+ acc[(p.task or "", name)].append(float(val))
181
+ for k, vals in acc.items():
182
+ if vals and k not in stats: # don't override benchmark stats
183
+ stats[k] = {
184
+ "mean": sum(vals) / len(vals),
185
+ "count": len(vals),
186
+ "std": statistics.pstdev(vals) if len(vals) > 1 else 0.0,
187
+ }
188
+ return stats
189
+
190
+ def per_task_deltas(self) -> list[TaskDelta]:
191
+ b = self._per_task_stats(self.baseline)
192
+ c = self._per_task_stats(self.candidate)
193
+ out: list[TaskDelta] = []
194
+ for (task, metric) in sorted(set(b) | set(c)):
195
+ bs, cs = b.get((task, metric)), c.get((task, metric))
196
+ d = self._direction(metric)
197
+ # None (not a fabricated 0.0) when either side never scored this
198
+ # task/metric pair at all -- same fix as the blended metric_deltas()
199
+ # above, applied at the per-task granularity.
200
+ delta = (cs["mean"] - bs["mean"]) if (bs is not None and cs is not None) else None
201
+ out.append(TaskDelta(
202
+ task=task, metric=metric,
203
+ baseline=bs["mean"] if bs else None, candidate=cs["mean"] if cs else None,
204
+ delta=delta, direction=d.value, grade=self._grade(delta, d),
205
+ baseline_count=bs["count"] if bs else 0, candidate_count=cs["count"] if cs else 0,
206
+ baseline_std=bs["std"] if bs else 0.0, candidate_std=cs["std"] if cs else 0.0,
207
+ pct_change=relative_pct(bs["mean"] if bs else None, delta),
208
+ ))
209
+ return out
210
+
211
+ # -- logging escalation: fires wherever these are called, not just via
212
+ # .summary()/.coverage_warnings() -- separate from metric_deltas()/
213
+ # per_task_deltas() themselves (coverage_warnings() calls those, so
214
+ # warning *inside* them would recurse).
215
+ def _warn_if_issues(self) -> None:
216
+ if self.has_failures:
217
+ fs = self.failure_summary()
218
+ logger.warning(
219
+ "RunComparison: baseline had %d failed sample(s), candidate had %d "
220
+ "-- excluded from means, not counted as wrong. See .failure_summary().",
221
+ fs["baseline_failed"], fs["candidate_failed"],
222
+ )
223
+ cw = self.coverage_warnings()
224
+ if cw:
225
+ logger.warning("RunComparison: sample coverage differs -- %s", cw)
226
+
227
+ # -- grading rollups ---------------------------------------------------
228
+ def grades(self) -> dict[str, DeltaGrade]:
229
+ """Grade per overall metric."""
230
+ self._warn_if_issues()
231
+ return {md.metric: md.grade for md in self.metric_deltas()}
232
+
233
+ def grade(self) -> DeltaGrade:
234
+ """Worst grade across all tasks/metrics — the ship / no-ship headline."""
235
+ self._warn_if_issues()
236
+ grades = [td.grade for td in self.per_task_deltas()] or [md.grade for md in self.metric_deltas()]
237
+ return max(grades, key=lambda g: _GRADE_ORDER[g]) if grades else DeltaGrade.PASS
238
+
239
+ # -- retention ---------------------------------------------------------
240
+ def retention(self, metric: Optional[str] = None):
241
+ """Fraction of the baseline's quality the candidate keeps (1.0 = parity).
242
+
243
+ candidate/baseline for MAXIMIZE metrics, baseline/candidate for MINIMIZE
244
+ (so 'lower is better' metrics also read as 'higher retention = better').
245
+ Returns a dict over all metrics, or a single float for a named metric.
246
+
247
+ Retention is a ratio, and a ratio is only a meaningful "how much
248
+ quality did we keep" answer for a non-negative-range metric -- for a
249
+ metric that can go negative (some judge/reward scores), a ratio of
250
+ two negative numbers can look like an improvement or a loss with the
251
+ wrong sign entirely (e.g. baseline=-2.0, candidate=-1.0 is a real
252
+ improvement, closer to zero, but -1.0/-2.0 = 0.5 reads as "lost half
253
+ the quality"). Rather than inventing a formula for an ill-defined
254
+ case, this logs a warning and reports ``nan`` -- an honest "not
255
+ computable," not a silently misleading number.
256
+ """
257
+ self._warn_if_issues()
258
+ mds = {md.metric: md for md in self.metric_deltas()}
259
+ out: dict[str, float] = {}
260
+ for name in ([metric] if metric else list(mds)):
261
+ md = mds.get(name)
262
+ if md is None or md.baseline is None or md.candidate is None:
263
+ continue
264
+ if md.baseline < 0 or md.candidate < 0:
265
+ logger.warning(
266
+ "retention(%r): baseline=%.6g candidate=%.6g -- not meaningful as a "
267
+ "ratio for a metric with negative values, reporting nan instead of "
268
+ "a misleading number.", name, md.baseline, md.candidate,
269
+ )
270
+ out[name] = float("nan")
271
+ continue
272
+ if self._direction(name) == Direction.MAXIMIZE:
273
+ out[name] = md.candidate / md.baseline if md.baseline else float("nan")
274
+ else:
275
+ out[name] = md.baseline / md.candidate if md.candidate else float("nan")
276
+ return out.get(metric) if metric else out
277
+
278
+ # -- significance (shared, by sample_id) ------------------------------
279
+ def significance(self, metric: Optional[str] = None) -> dict[str, Any]:
280
+ return paired_bootstrap(score_pairs(self.baseline, self.candidate, metric))
281
+
282
+ # -- size / latency tradeoff (everything measured, nothing caller-supplied) --
283
+ def tradeoff(self, *, metric: Optional[str] = None) -> dict[str, Any]:
284
+ """Quality-vs-cost view. Every number here is measured, never passed
285
+ in by a caller: latency from ``RunResult.perf`` (timed by ``Runner``
286
+ while it ran), size from ``RunResult.model_size`` (introspected for
287
+ local backends, identity-only for hosted-API ones -- see
288
+ ``Model.model_info()``). When either side is API-based there is no
289
+ on-disk size to ratio, so ``size_ratio``/``quality_per_mb`` are
290
+ omitted and ``api_based`` names the two models instead."""
291
+ self._warn_if_issues()
292
+ mds = {md.metric: md for md in self.metric_deltas()}
293
+ if metric is None:
294
+ metric = next(iter(mds), None)
295
+ out: dict[str, Any] = {"metric": metric}
296
+ if metric in mds:
297
+ out["baseline_quality"] = mds[metric].baseline
298
+ out["candidate_quality"] = mds[metric].candidate
299
+ out["retention"] = self.retention(metric)
300
+
301
+ # or {}: model_size is None (not {}) when RunConfig.track_performance
302
+ # was never enabled on that run -- treated the same as "measured but
303
+ # nothing recorded" here, since tradeoff() already omits size_ratio
304
+ # etc. whenever a side has no size_mb, whatever the reason.
305
+ base_size, cand_size = self.baseline.model_size or {}, self.candidate.model_size or {}
306
+ baseline_size_mb, candidate_size_mb = base_size.get("size_mb"), cand_size.get("size_mb")
307
+ if baseline_size_mb and candidate_size_mb:
308
+ out["baseline_size_mb"] = baseline_size_mb
309
+ out["candidate_size_mb"] = candidate_size_mb
310
+ out["size_ratio"] = candidate_size_mb / baseline_size_mb # <1 = smaller
311
+ if metric in mds:
312
+ # retention(), not the raw candidate value: for a MINIMIZE
313
+ # metric (latency, cost) a bigger raw number is WORSE, so
314
+ # dividing it by size and calling the result "quality" would
315
+ # be backwards. retention() already normalizes for direction
316
+ # (1.0 = parity, regardless of whether the metric counts up
317
+ # or down), so "quality per mb" means the same thing for
318
+ # every metric.
319
+ out["quality_per_mb"] = self.retention(metric) / candidate_size_mb
320
+ elif base_size or cand_size:
321
+ # No real size_mb on at least one side -- either it's a hosted
322
+ # -API/callable model (never measurable at all) or a local
323
+ # backend that's marked is_local but doesn't actually introspect
324
+ # size yet (e.g. VLLMModel today -- see model/vllm_gen.py). Either
325
+ # way there's no size ratio to report; name the two models
326
+ # instead of silently reporting nothing at all (the previous
327
+ # behavior when both sides were "local" but neither had size_mb).
328
+ out["api_based"] = {
329
+ "baseline": base_size.get("model_name", self.baseline.model_spec),
330
+ "candidate": cand_size.get("model_name", self.candidate.model_spec),
331
+ }
332
+
333
+ baseline_latency_ms = ((self.baseline.perf or {}).get("latency_ms") or {}).get("mean") or None
334
+ candidate_latency_ms = ((self.candidate.perf or {}).get("latency_ms") or {}).get("mean") or None
335
+ if baseline_latency_ms and candidate_latency_ms:
336
+ out["baseline_latency_ms"] = baseline_latency_ms
337
+ out["candidate_latency_ms"] = candidate_latency_ms
338
+ out["latency_ratio"] = candidate_latency_ms / baseline_latency_ms
339
+ out["speedup"] = baseline_latency_ms / candidate_latency_ms # >1 = faster
340
+ return out
341
+
342
+ # -- performance (latency/throughput, measured automatically) ---------
343
+ def performance(self) -> dict[str, Any]:
344
+ """Baseline-vs-candidate latency/throughput, straight from each run's
345
+ measured ``RunResult.perf`` -- no caller-supplied numbers needed
346
+ (unlike ``tradeoff()``, which folds in quality + optional size).
347
+
348
+ A side's fields here are ``None`` (not ``{}``/``0.0``) when that run
349
+ never enabled ``RunConfig.track_performance`` at all -- distinct from
350
+ an empty/zeroed value, which means it *was* tracked but nothing was
351
+ timed (e.g. a ``reads_actual_output`` run). Mixing a tracked and an
352
+ untracked run is fine: the tracked side still reports its real
353
+ numbers, the untracked side reports ``None`` for its own fields, and
354
+ ``speedup`` (which needs both) is simply omitted.
355
+ """
356
+ base_tracked = self.baseline.perf is not None
357
+ cand_tracked = self.candidate.perf is not None
358
+ base_lat = (self.baseline.perf.get("latency_ms") or {}) if base_tracked else None
359
+ cand_lat = (self.candidate.perf.get("latency_ms") or {}) if cand_tracked else None
360
+ base_thr = (self.baseline.perf.get("throughput") or {}) if base_tracked else None
361
+ cand_thr = (self.candidate.perf.get("throughput") or {}) if cand_tracked else None
362
+ out: dict[str, Any] = {
363
+ "baseline_latency_ms": base_lat,
364
+ "candidate_latency_ms": cand_lat,
365
+ "baseline_throughput_rps": base_thr.get("rps", 0.0) if base_thr is not None else None,
366
+ "candidate_throughput_rps": cand_thr.get("rps", 0.0) if cand_thr is not None else None,
367
+ "baseline_output_tokens_per_sec": base_thr.get("output_tokens_per_sec", 0.0) if base_thr is not None else None,
368
+ "candidate_output_tokens_per_sec": cand_thr.get("output_tokens_per_sec", 0.0) if cand_thr is not None else None,
369
+ }
370
+ if base_lat is not None and cand_lat is not None and base_lat.get("count") and cand_lat.get("count"):
371
+ out["speedup"] = base_lat["mean"] / cand_lat["mean"] if cand_lat["mean"] else float("nan")
372
+ return out
373
+
374
+ # -- cost (real measured tokens; pricing is the one thing you must supply) --
375
+ def cost(self, baseline_pricing: dict[str, float], candidate_pricing: dict[str, float]) -> dict[str, Any]:
376
+ """Dollar cost of baseline vs. candidate. Token counts are real,
377
+ provider-reported numbers already captured on each ``RunResult``
378
+ (see ``RunResult.token_usage``/``.cost()``) -- never estimated here.
379
+ ``$``/token pricing has no library default (it isn't measurable and
380
+ goes stale) so it's the one thing you pass in, per side, since a
381
+ baseline and candidate are often different paid models with
382
+ different prices. Returns ``{}`` for either side that recorded no
383
+ token usage at all (nothing to price)."""
384
+ out: dict[str, Any] = {}
385
+ base_cost = self.baseline.cost(baseline_pricing)
386
+ cand_cost = self.candidate.cost(candidate_pricing)
387
+ if base_cost is not None:
388
+ out["baseline_cost"] = base_cost
389
+ if cand_cost is not None:
390
+ out["candidate_cost"] = cand_cost
391
+ if base_cost and cand_cost:
392
+ out["cost_ratio"] = cand_cost / base_cost
393
+ return out
394
+
395
+ # -- sample browser (delegated to RunDiff, aligned by sample_id) ------
396
+ def regressed(self, metric_name: str = "") -> list[Prediction]:
397
+ """Samples correct in ``baseline`` but wrong in ``candidate`` -- what broke.
398
+
399
+ Named for what it reports (a two-way flip between exactly these two
400
+ runs), not "recently" -- there's no history or tracking involved, just
401
+ this one baseline-vs-candidate comparison. Was ``newly_wrong()``.
402
+ """
403
+ return self._diff.regressed(metric_name)
404
+
405
+ def improved(self, metric_name: str = "") -> list[Prediction]:
406
+ """Samples wrong in ``baseline`` but fixed in ``candidate``. Was ``newly_correct()``."""
407
+ return self._diff.improved(metric_name)
408
+
409
+ def still_wrong(self, metric_name: str = "") -> list[Prediction]:
410
+ return self._diff.still_wrong(metric_name)
411
+
412
+ def sample_diff(self, metric_name: str = "") -> list[dict]:
413
+ return self._diff.sample_diff(metric_name)
414
+
415
+ def sample_summary(self, metric_name: str = "") -> dict[str, int]:
416
+ return self._diff.sample_summary(metric_name)
417
+
418
+ # -- failed-sample visibility -------------------------------------------
419
+ def failure_summary(self) -> dict[str, Any]:
420
+ """Per-run failed-sample counts and errors.
421
+
422
+ ``metric_deltas()``/``per_task_deltas()``/``retention()``/``tradeoff()``
423
+ all read per-sample scores from ``Prediction.metadata["scores"]``, which
424
+ is empty on a sample that errored during annotation/scoring (see
425
+ ``Runner.run()``'s per-sample fallback) -- such samples are silently
426
+ **excluded** from those means, not counted as a miss. A run with several
427
+ failures can therefore report a better mean than a clean run would,
428
+ purely because its denominator shrank. Call this (or check
429
+ ``has_failures``) before trusting a grade; ``regressed()`` is the one
430
+ view that *does* catch a failure correctly (via ``Prediction.correct``,
431
+ which is ``False`` on the fallback), everything else needs this check.
432
+ """
433
+ return {
434
+ "baseline_failed": self.baseline.failed_count,
435
+ "candidate_failed": self.candidate.failed_count,
436
+ "baseline_errors": self.baseline.errors,
437
+ "candidate_errors": self.candidate.errors,
438
+ }
439
+
440
+ @property
441
+ def has_failures(self) -> bool:
442
+ return bool(self.baseline.failed_count or self.candidate.failed_count)
443
+
444
+ def coverage_warnings(self) -> list[str]:
445
+ """Metrics (or task/metric pairs) whose baseline/candidate sample
446
+ counts diverge.
447
+
448
+ ``metric_deltas()``/``per_task_deltas()`` derive their metric set purely
449
+ from whichever names actually appear in each run's stats/scores -- there
450
+ is no check against an "expected" scorer list, and a metric that failed
451
+ on every sample (or was scored under a different config) simply never
452
+ shows up, silently. A count mismatch between baseline and candidate for
453
+ the *same* metric name is the visible symptom of that: it means the two
454
+ means were computed over different sample sets, not a like-for-like
455
+ comparison -- treat a large mismatch as a reason to distrust that
456
+ metric's grade, not just a curiosity.
457
+
458
+ Checks both the blended (overall) counts *and* the per-task counts --
459
+ two tasks can individually have mismatched N (5 vs 3, 5 vs 7) while the
460
+ *totals* coincidentally agree (10 vs 10), invisible to a blended-only
461
+ check.
462
+ """
463
+ warnings = []
464
+ for md in self.metric_deltas():
465
+ if md.baseline_count != md.candidate_count:
466
+ warnings.append(
467
+ f"{md.metric}: baseline n={md.baseline_count}, candidate n={md.candidate_count}"
468
+ )
469
+ for td in self.per_task_deltas():
470
+ if td.baseline_count != td.candidate_count:
471
+ warnings.append(
472
+ f"{td.task}/{td.metric}: baseline n={td.baseline_count}, "
473
+ f"candidate n={td.candidate_count}"
474
+ )
475
+ return warnings
476
+
477
+ def summary(self) -> str:
478
+ lines = []
479
+ if self.has_failures:
480
+ fs = self.failure_summary()
481
+ lines.append(
482
+ f" WARNING: baseline had {fs['baseline_failed']} failed sample(s), "
483
+ f"candidate had {fs['candidate_failed']} -- EXCLUDED from the means "
484
+ f"below, not counted as wrong. See .failure_summary()."
485
+ )
486
+ cw = self.coverage_warnings()
487
+ if cw:
488
+ lines.append(
489
+ " WARNING: sample coverage differs per metric (means computed over "
490
+ "different Ns, not a like-for-like comparison):"
491
+ )
492
+ lines += [f" {w}" for w in cw]
493
+ mds = self.metric_deltas()
494
+ # Just the percentage differences per metric -- no pass/warn/fail and no
495
+ # better/worse framing. A fixed threshold isn't meaningful across every
496
+ # metric, so it's the reader's job to interpret the numbers for their own
497
+ # metrics. (An opt-in gate is still available via .grade()/.grades() with
498
+ # your own thresholds.)
499
+ lines += [
500
+ f"Comparison: {self.baseline.run_id} (baseline) -> {self.candidate.run_id} (candidate)",
501
+ " per-metric (baseline -> candidate | absolute delta | relative %):",
502
+ ]
503
+ for md in mds:
504
+ lines.append(" " + _delta_line(md.metric, md))
505
+ tasks = self.per_task_deltas()
506
+ # Show per-task detail whenever there's more than one distinct task --
507
+ # comparing this to len(metric_deltas()) instead (the old check) could
508
+ # wrongly hide a genuinely different per-task breakdown whenever the
509
+ # row counts happened to coincide numerically (e.g. 2 metrics each
510
+ # scored on a different, non-overlapping single task).
511
+ if len({td.task for td in tasks}) > 1:
512
+ lines.append(" per-task:")
513
+ for td in tasks:
514
+ lines.append(" " + _delta_line(f"{td.task}/{td.metric}", td))
515
+ sm = self.sample_summary()
516
+ lines.append(
517
+ f" samples: {sm['newly_wrong']} went correct->wrong, "
518
+ f"{sm['newly_correct']} went wrong->correct"
519
+ )
520
+ perf = self.performance()
521
+ if (perf["baseline_latency_ms"] is not None and perf["candidate_latency_ms"] is not None
522
+ and perf["baseline_latency_ms"].get("count") and perf["candidate_latency_ms"].get("count")):
523
+ bl, cl = perf["baseline_latency_ms"], perf["candidate_latency_ms"]
524
+ line = (
525
+ f" performance: latency {bl['mean']:.0f}ms -> {cl['mean']:.0f}ms, "
526
+ f"throughput {perf['baseline_throughput_rps']:.2f} -> "
527
+ f"{perf['candidate_throughput_rps']:.2f} req/s "
528
+ f"(speedup={perf['speedup']:.2f}x)"
529
+ )
530
+ # Only when the backend actually reported token usage (hosted
531
+ # APIs) -- 0 tok/s for a local/callable backend would
532
+ # misleadingly read as "measured zero" rather than "unavailable".
533
+ if perf["baseline_output_tokens_per_sec"] or perf["candidate_output_tokens_per_sec"]:
534
+ line += (
535
+ f", {perf['baseline_output_tokens_per_sec']:.1f} -> "
536
+ f"{perf['candidate_output_tokens_per_sec']:.1f} out-tok/s"
537
+ )
538
+ lines.append(line)
539
+ size_line = self._model_size_line()
540
+ if size_line:
541
+ lines.append(f" {size_line}")
542
+ return "\n".join(lines)
543
+
544
+ def _model_size_line(self) -> str:
545
+ """Model size for both sides, or '' if neither run recorded any
546
+ (RunResult.model_size is None when RunConfig.track_performance was
547
+ never enabled on that run, {} for a tracked run with nothing to
548
+ report -- both treated the same way here)."""
549
+ base_size, cand_size = self.baseline.model_size or {}, self.candidate.model_size or {}
550
+ if not base_size and not cand_size:
551
+ return ""
552
+ baseline_size_mb, candidate_size_mb = base_size.get("size_mb"), cand_size.get("size_mb")
553
+ if baseline_size_mb and candidate_size_mb:
554
+ return (
555
+ f"model size: {baseline_size_mb:.1f} MB -> {candidate_size_mb:.1f} MB "
556
+ f"({base_size.get('model_name', '?')} -> {cand_size.get('model_name', '?')})"
557
+ )
558
+ # Either a hosted-API/callable model (nothing to measure) or a local
559
+ # backend marked is_local that doesn't actually introspect size yet
560
+ # (e.g. VLLMModel) -- either way, no fake "0.0 MB" placeholder.
561
+ base_name = base_size.get("model_name", self.baseline.model_spec)
562
+ cand_name = cand_size.get("model_name", self.candidate.model_spec)
563
+ return f"model: {base_name} -> {cand_name} (no measurable size)"