auditkit 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- auditkit/README.md +99 -0
- auditkit/__init__.py +177 -0
- auditkit/__main__.py +3 -0
- auditkit/_bootstrap.py +77 -0
- auditkit/_identity_guard.py +99 -0
- auditkit/adapter.py +264 -0
- auditkit/annotator.py +339 -0
- auditkit/api.py +502 -0
- auditkit/assets/auditkit_logo.png +0 -0
- auditkit/cache.py +47 -0
- auditkit/cli.py +417 -0
- auditkit/comparison.py +563 -0
- auditkit/diff.py +265 -0
- auditkit/errors.py +54 -0
- auditkit/evaluator.py +20 -0
- auditkit/experiment.py +145 -0
- auditkit/hf_publish.py +262 -0
- auditkit/lmeval_engine.py +550 -0
- auditkit/loaders.py +121 -0
- auditkit/logs.py +18 -0
- auditkit/metric.py +199 -0
- auditkit/metrics/README.md +15 -0
- auditkit/metrics/__init__.py +0 -0
- auditkit/metrics/code.py +222 -0
- auditkit/metrics/embedding.py +131 -0
- auditkit/metrics/encoder_judge.py +423 -0
- auditkit/metrics/generation.py +331 -0
- auditkit/metrics/guard.py +412 -0
- auditkit/metrics/hallucination.py +45 -0
- auditkit/metrics/judge.py +547 -0
- auditkit/metrics/pairwise.py +153 -0
- auditkit/metrics/perf.py +53 -0
- auditkit/metrics/rag.py +149 -0
- auditkit/metrics/security.py +64 -0
- auditkit/metrics/toxicity.py +238 -0
- auditkit/model/README.md +16 -0
- auditkit/model/__init__.py +485 -0
- auditkit/model/anthropic.py +94 -0
- auditkit/model/api_gen.py +133 -0
- auditkit/model/groq_gen.py +121 -0
- auditkit/model/hf_gen.py +385 -0
- auditkit/model/lexsi.py +155 -0
- auditkit/model/litellm_gen.py +65 -0
- auditkit/model/openai.py +90 -0
- auditkit/model/openrouter_gen.py +152 -0
- auditkit/model/vllm_gen.py +316 -0
- auditkit/model_compare.py +655 -0
- auditkit/redteam/README.md +9 -0
- auditkit/redteam/__init__.py +26 -0
- auditkit/redteam/detector.py +37 -0
- auditkit/redteam/detectors/README.md +5 -0
- auditkit/redteam/detectors/builtin.py +126 -0
- auditkit/redteam/probe.py +39 -0
- auditkit/redteam/probes/README.md +5 -0
- auditkit/redteam/probes/builtin.py +85 -0
- auditkit/redteam/runner.py +206 -0
- auditkit/registry.py +65 -0
- auditkit/report.py +278 -0
- auditkit/report_format.py +52 -0
- auditkit/router.py +54 -0
- auditkit/runner.py +575 -0
- auditkit/runspec.py +159 -0
- auditkit/sample.py +40 -0
- auditkit/scenario.py +88 -0
- auditkit/scenarios/README.md +10 -0
- auditkit/scenarios/__init__.py +4 -0
- auditkit/scenarios/arc.py +33 -0
- auditkit/scenarios/gsm8k.py +32 -0
- auditkit/scenarios/hellaswag.py +33 -0
- auditkit/scenarios/humaneval.py +32 -0
- auditkit/scenarios/mmlu.py +34 -0
- auditkit/scenarios/truthfulqa.py +33 -0
- auditkit/score.py +165 -0
- auditkit/scorers.py +117 -0
- auditkit/scoring.py +79 -0
- auditkit/types.py +69 -0
- auditkit-1.0.0.dist-info/METADATA +396 -0
- auditkit-1.0.0.dist-info/RECORD +81 -0
- auditkit-1.0.0.dist-info/WHEEL +4 -0
- auditkit-1.0.0.dist-info/entry_points.txt +2 -0
- auditkit-1.0.0.dist-info/licenses/LICENSE.md +92 -0
auditkit/comparison.py
ADDED
|
@@ -0,0 +1,563 @@
|
|
|
1
|
+
"""Baseline-anchored comparison of two runs — the "did pruning/quantizing hurt?" view.
|
|
2
|
+
|
|
3
|
+
`RunComparison(baseline, candidate)` is a read-side helper over two immutable
|
|
4
|
+
`RunResult`s produced with the same eval config (e.g. a base model vs a pruned or
|
|
5
|
+
quantized one). It gives:
|
|
6
|
+
|
|
7
|
+
- **per-task deltas** (engine-agnostic: benchmark runs key per-task metrics in
|
|
8
|
+
`stats` as ``"task:metric"``; native runs carry a `task` on each `Prediction`
|
|
9
|
+
and a per-score breakdown in `metadata["scores"]` — both are folded in),
|
|
10
|
+
- **percentage differences** per metric (the summary reports the absolute delta
|
|
11
|
+
and the relative % -- no pass/warn/fail verdict and no better/worse framing;
|
|
12
|
+
a fixed threshold isn't meaningful across every metric, so it's the reader's
|
|
13
|
+
job to interpret the numbers. An opt-in threshold gate is still available via
|
|
14
|
+
.grade() with caller-chosen thresholds),
|
|
15
|
+
- **retention %** (candidate/baseline for MAXIMIZE, inverted for MINIMIZE),
|
|
16
|
+
- **significance** via the shared by-sample-id paired bootstrap,
|
|
17
|
+
- a **size/latency tradeoff** view over caller-supplied numbers,
|
|
18
|
+
- and the **regressed / improved** sample browser (delegated to `RunDiff`).
|
|
19
|
+
|
|
20
|
+
Lineage-based *auto* baseline selection is deliberately a platform concern; here
|
|
21
|
+
you hand it the two runs.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import logging
|
|
27
|
+
import statistics
|
|
28
|
+
from collections import defaultdict
|
|
29
|
+
from dataclasses import dataclass
|
|
30
|
+
from typing import Any, Optional
|
|
31
|
+
|
|
32
|
+
from ._bootstrap import paired_bootstrap, score_pairs
|
|
33
|
+
from .diff import (
|
|
34
|
+
DeltaGrade, RunDiff, direction_for, grade_delta, metric_directions,
|
|
35
|
+
relative_pct,
|
|
36
|
+
)
|
|
37
|
+
from .report import Prediction, RunResult
|
|
38
|
+
from .types import Direction
|
|
39
|
+
|
|
40
|
+
logger = logging.getLogger(__name__)
|
|
41
|
+
|
|
42
|
+
# NOT_COMPARABLE deliberately ranks below every real grade: a single metric
|
|
43
|
+
# that was only scored on one side shouldn't be able to make grade()'s
|
|
44
|
+
# worst-case reduction report a FAIL-equivalent when every metric that
|
|
45
|
+
# actually WAS compared passed cleanly. If every metric is NOT_COMPARABLE
|
|
46
|
+
# (nothing at all could be compared), that's still what grade() reports --
|
|
47
|
+
# there's nothing better to fall back to.
|
|
48
|
+
_GRADE_ORDER = {
|
|
49
|
+
DeltaGrade.NOT_COMPARABLE: -1,
|
|
50
|
+
DeltaGrade.PASS: 0,
|
|
51
|
+
DeltaGrade.WARN: 1,
|
|
52
|
+
DeltaGrade.FAIL: 2,
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _delta_line(label: str, d) -> str:
|
|
57
|
+
"""One human line for a MetricDelta/TaskDelta: baseline -> candidate, the
|
|
58
|
+
absolute delta AND the relative %, with the sample count/std. No verdict and
|
|
59
|
+
no better/worse framing -- the reader interprets the numbers for their own
|
|
60
|
+
metric. ASCII only (printed to consoles that can't encode arrows)."""
|
|
61
|
+
if d.delta is None:
|
|
62
|
+
return (f"{label}: N/A -- only scored on one side "
|
|
63
|
+
f"(n={d.baseline_count}->{d.candidate_count})")
|
|
64
|
+
pct_s = f"{d.pct_change:+.1f}% rel" if d.pct_change is not None else "rel n/a"
|
|
65
|
+
return (
|
|
66
|
+
f"{label}: {d.baseline:.4f} -> {d.candidate:.4f} "
|
|
67
|
+
f"delta={d.delta:+.4f} ({pct_s}) "
|
|
68
|
+
f"(n={d.baseline_count}->{d.candidate_count}, "
|
|
69
|
+
f"std={d.baseline_std:.4f}->{d.candidate_std:.4f})"
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass
|
|
74
|
+
class MetricDelta:
|
|
75
|
+
metric: str
|
|
76
|
+
# None when the metric was scored on only one side (grade is then
|
|
77
|
+
# NOT_COMPARABLE) -- never a fabricated 0.0 for whichever side didn't
|
|
78
|
+
# score it.
|
|
79
|
+
baseline: Optional[float]
|
|
80
|
+
candidate: Optional[float]
|
|
81
|
+
delta: Optional[float]
|
|
82
|
+
direction: str
|
|
83
|
+
# Threshold-based pass/warn/fail. Kept for callers who explicitly want a
|
|
84
|
+
# gate, but NOT what the default summary reports -- a fixed % threshold
|
|
85
|
+
# isn't meaningful across every metric (see summary()). Read the percentage
|
|
86
|
+
# difference (`delta` / `pct_change`) and decide acceptability yourself.
|
|
87
|
+
grade: DeltaGrade
|
|
88
|
+
# How many real samples fed each mean, and how spread out they were --
|
|
89
|
+
# .mean alone can't tell you a regression rode on far fewer samples than
|
|
90
|
+
# the baseline, or that a flat mean hid much wider variance. See
|
|
91
|
+
# RunComparison.summary()'s coverage warning, which reads these.
|
|
92
|
+
baseline_count: int = 0
|
|
93
|
+
candidate_count: int = 0
|
|
94
|
+
baseline_std: float = 0.0
|
|
95
|
+
candidate_std: float = 0.0
|
|
96
|
+
# The change as a percentage of the baseline (candidate vs baseline). None
|
|
97
|
+
# when incomparable or the baseline is <= 0 (percent-of-zero is undefined).
|
|
98
|
+
pct_change: Optional[float] = None
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@dataclass
|
|
102
|
+
class TaskDelta:
|
|
103
|
+
task: str
|
|
104
|
+
metric: str
|
|
105
|
+
baseline: Optional[float]
|
|
106
|
+
candidate: Optional[float]
|
|
107
|
+
delta: Optional[float]
|
|
108
|
+
direction: str
|
|
109
|
+
grade: DeltaGrade
|
|
110
|
+
baseline_count: int = 0
|
|
111
|
+
candidate_count: int = 0
|
|
112
|
+
baseline_std: float = 0.0
|
|
113
|
+
candidate_std: float = 0.0
|
|
114
|
+
pct_change: Optional[float] = None
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class RunComparison:
|
|
118
|
+
"""Compare a ``candidate`` run against a ``baseline`` run on the same config."""
|
|
119
|
+
|
|
120
|
+
def __init__(
|
|
121
|
+
self,
|
|
122
|
+
baseline: RunResult,
|
|
123
|
+
candidate: RunResult,
|
|
124
|
+
*,
|
|
125
|
+
pass_threshold: float = 0.02,
|
|
126
|
+
warn_threshold: float = 0.05,
|
|
127
|
+
) -> None:
|
|
128
|
+
self.baseline = baseline
|
|
129
|
+
self.candidate = candidate
|
|
130
|
+
self.pass_threshold = pass_threshold
|
|
131
|
+
self.warn_threshold = warn_threshold
|
|
132
|
+
self._diff = RunDiff(baseline, candidate)
|
|
133
|
+
|
|
134
|
+
# -- metric directions (read from stored per-score dicts) --------------
|
|
135
|
+
def _directions(self) -> dict[str, Direction]:
|
|
136
|
+
return metric_directions((self.baseline, self.candidate))
|
|
137
|
+
|
|
138
|
+
def _direction(self, metric: str) -> Direction:
|
|
139
|
+
return direction_for(metric, self._directions())
|
|
140
|
+
|
|
141
|
+
def _grade(self, delta: Optional[float], direction: Direction) -> DeltaGrade:
|
|
142
|
+
# None means "this metric wasn't scored on both sides" -- nothing to
|
|
143
|
+
# grade, not a fabricated PASS/FAIL against a phantom zero delta.
|
|
144
|
+
if delta is None:
|
|
145
|
+
return DeltaGrade.NOT_COMPARABLE
|
|
146
|
+
return grade_delta(delta, direction, self.pass_threshold, self.warn_threshold)
|
|
147
|
+
|
|
148
|
+
# -- overall per-metric deltas ----------------------------------------
|
|
149
|
+
def metric_deltas(self) -> list[MetricDelta]:
|
|
150
|
+
out: list[MetricDelta] = []
|
|
151
|
+
for name, info in self._diff.metric_deltas().items():
|
|
152
|
+
d = self._direction(name)
|
|
153
|
+
b_stat = self.baseline.stats.get(name)
|
|
154
|
+
c_stat = self.candidate.stats.get(name)
|
|
155
|
+
out.append(MetricDelta(
|
|
156
|
+
metric=name, baseline=info["baseline"], candidate=info["contrast"],
|
|
157
|
+
delta=info["delta"], direction=d.value, grade=self._grade(info["delta"], d),
|
|
158
|
+
baseline_count=b_stat.count if b_stat else 0,
|
|
159
|
+
candidate_count=c_stat.count if c_stat else 0,
|
|
160
|
+
baseline_std=b_stat.std if b_stat else 0.0,
|
|
161
|
+
candidate_std=c_stat.std if c_stat else 0.0,
|
|
162
|
+
pct_change=relative_pct(info["baseline"], info["delta"]),
|
|
163
|
+
))
|
|
164
|
+
return out
|
|
165
|
+
|
|
166
|
+
# -- per-task deltas (engine-agnostic) --------------------------------
|
|
167
|
+
def _per_task_stats(self, run: RunResult) -> dict[tuple[str, str], dict[str, float]]:
|
|
168
|
+
stats: dict[tuple[str, str], dict[str, float]] = {}
|
|
169
|
+
# Benchmark engine: stats keyed "task:metric" -- Stat already tracks count/std.
|
|
170
|
+
for key, stat in run.stats.items():
|
|
171
|
+
if ":" in key:
|
|
172
|
+
task, metric = key.split(":", 1)
|
|
173
|
+
stats[(task, metric)] = {"mean": stat.mean, "count": stat.count, "std": stat.std}
|
|
174
|
+
# Native engine: group predictions by task via their per-score dicts.
|
|
175
|
+
acc: dict[tuple[str, str], list[float]] = defaultdict(list)
|
|
176
|
+
for p in run.predictions:
|
|
177
|
+
for s in p.metadata.get("scores", []):
|
|
178
|
+
name, val = s.get("name"), s.get("value")
|
|
179
|
+
if name is not None and isinstance(val, (int, float)):
|
|
180
|
+
acc[(p.task or "", name)].append(float(val))
|
|
181
|
+
for k, vals in acc.items():
|
|
182
|
+
if vals and k not in stats: # don't override benchmark stats
|
|
183
|
+
stats[k] = {
|
|
184
|
+
"mean": sum(vals) / len(vals),
|
|
185
|
+
"count": len(vals),
|
|
186
|
+
"std": statistics.pstdev(vals) if len(vals) > 1 else 0.0,
|
|
187
|
+
}
|
|
188
|
+
return stats
|
|
189
|
+
|
|
190
|
+
def per_task_deltas(self) -> list[TaskDelta]:
|
|
191
|
+
b = self._per_task_stats(self.baseline)
|
|
192
|
+
c = self._per_task_stats(self.candidate)
|
|
193
|
+
out: list[TaskDelta] = []
|
|
194
|
+
for (task, metric) in sorted(set(b) | set(c)):
|
|
195
|
+
bs, cs = b.get((task, metric)), c.get((task, metric))
|
|
196
|
+
d = self._direction(metric)
|
|
197
|
+
# None (not a fabricated 0.0) when either side never scored this
|
|
198
|
+
# task/metric pair at all -- same fix as the blended metric_deltas()
|
|
199
|
+
# above, applied at the per-task granularity.
|
|
200
|
+
delta = (cs["mean"] - bs["mean"]) if (bs is not None and cs is not None) else None
|
|
201
|
+
out.append(TaskDelta(
|
|
202
|
+
task=task, metric=metric,
|
|
203
|
+
baseline=bs["mean"] if bs else None, candidate=cs["mean"] if cs else None,
|
|
204
|
+
delta=delta, direction=d.value, grade=self._grade(delta, d),
|
|
205
|
+
baseline_count=bs["count"] if bs else 0, candidate_count=cs["count"] if cs else 0,
|
|
206
|
+
baseline_std=bs["std"] if bs else 0.0, candidate_std=cs["std"] if cs else 0.0,
|
|
207
|
+
pct_change=relative_pct(bs["mean"] if bs else None, delta),
|
|
208
|
+
))
|
|
209
|
+
return out
|
|
210
|
+
|
|
211
|
+
# -- logging escalation: fires wherever these are called, not just via
|
|
212
|
+
# .summary()/.coverage_warnings() -- separate from metric_deltas()/
|
|
213
|
+
# per_task_deltas() themselves (coverage_warnings() calls those, so
|
|
214
|
+
# warning *inside* them would recurse).
|
|
215
|
+
def _warn_if_issues(self) -> None:
|
|
216
|
+
if self.has_failures:
|
|
217
|
+
fs = self.failure_summary()
|
|
218
|
+
logger.warning(
|
|
219
|
+
"RunComparison: baseline had %d failed sample(s), candidate had %d "
|
|
220
|
+
"-- excluded from means, not counted as wrong. See .failure_summary().",
|
|
221
|
+
fs["baseline_failed"], fs["candidate_failed"],
|
|
222
|
+
)
|
|
223
|
+
cw = self.coverage_warnings()
|
|
224
|
+
if cw:
|
|
225
|
+
logger.warning("RunComparison: sample coverage differs -- %s", cw)
|
|
226
|
+
|
|
227
|
+
# -- grading rollups ---------------------------------------------------
|
|
228
|
+
def grades(self) -> dict[str, DeltaGrade]:
|
|
229
|
+
"""Grade per overall metric."""
|
|
230
|
+
self._warn_if_issues()
|
|
231
|
+
return {md.metric: md.grade for md in self.metric_deltas()}
|
|
232
|
+
|
|
233
|
+
def grade(self) -> DeltaGrade:
|
|
234
|
+
"""Worst grade across all tasks/metrics — the ship / no-ship headline."""
|
|
235
|
+
self._warn_if_issues()
|
|
236
|
+
grades = [td.grade for td in self.per_task_deltas()] or [md.grade for md in self.metric_deltas()]
|
|
237
|
+
return max(grades, key=lambda g: _GRADE_ORDER[g]) if grades else DeltaGrade.PASS
|
|
238
|
+
|
|
239
|
+
# -- retention ---------------------------------------------------------
|
|
240
|
+
def retention(self, metric: Optional[str] = None):
|
|
241
|
+
"""Fraction of the baseline's quality the candidate keeps (1.0 = parity).
|
|
242
|
+
|
|
243
|
+
candidate/baseline for MAXIMIZE metrics, baseline/candidate for MINIMIZE
|
|
244
|
+
(so 'lower is better' metrics also read as 'higher retention = better').
|
|
245
|
+
Returns a dict over all metrics, or a single float for a named metric.
|
|
246
|
+
|
|
247
|
+
Retention is a ratio, and a ratio is only a meaningful "how much
|
|
248
|
+
quality did we keep" answer for a non-negative-range metric -- for a
|
|
249
|
+
metric that can go negative (some judge/reward scores), a ratio of
|
|
250
|
+
two negative numbers can look like an improvement or a loss with the
|
|
251
|
+
wrong sign entirely (e.g. baseline=-2.0, candidate=-1.0 is a real
|
|
252
|
+
improvement, closer to zero, but -1.0/-2.0 = 0.5 reads as "lost half
|
|
253
|
+
the quality"). Rather than inventing a formula for an ill-defined
|
|
254
|
+
case, this logs a warning and reports ``nan`` -- an honest "not
|
|
255
|
+
computable," not a silently misleading number.
|
|
256
|
+
"""
|
|
257
|
+
self._warn_if_issues()
|
|
258
|
+
mds = {md.metric: md for md in self.metric_deltas()}
|
|
259
|
+
out: dict[str, float] = {}
|
|
260
|
+
for name in ([metric] if metric else list(mds)):
|
|
261
|
+
md = mds.get(name)
|
|
262
|
+
if md is None or md.baseline is None or md.candidate is None:
|
|
263
|
+
continue
|
|
264
|
+
if md.baseline < 0 or md.candidate < 0:
|
|
265
|
+
logger.warning(
|
|
266
|
+
"retention(%r): baseline=%.6g candidate=%.6g -- not meaningful as a "
|
|
267
|
+
"ratio for a metric with negative values, reporting nan instead of "
|
|
268
|
+
"a misleading number.", name, md.baseline, md.candidate,
|
|
269
|
+
)
|
|
270
|
+
out[name] = float("nan")
|
|
271
|
+
continue
|
|
272
|
+
if self._direction(name) == Direction.MAXIMIZE:
|
|
273
|
+
out[name] = md.candidate / md.baseline if md.baseline else float("nan")
|
|
274
|
+
else:
|
|
275
|
+
out[name] = md.baseline / md.candidate if md.candidate else float("nan")
|
|
276
|
+
return out.get(metric) if metric else out
|
|
277
|
+
|
|
278
|
+
# -- significance (shared, by sample_id) ------------------------------
|
|
279
|
+
def significance(self, metric: Optional[str] = None) -> dict[str, Any]:
|
|
280
|
+
return paired_bootstrap(score_pairs(self.baseline, self.candidate, metric))
|
|
281
|
+
|
|
282
|
+
# -- size / latency tradeoff (everything measured, nothing caller-supplied) --
|
|
283
|
+
def tradeoff(self, *, metric: Optional[str] = None) -> dict[str, Any]:
|
|
284
|
+
"""Quality-vs-cost view. Every number here is measured, never passed
|
|
285
|
+
in by a caller: latency from ``RunResult.perf`` (timed by ``Runner``
|
|
286
|
+
while it ran), size from ``RunResult.model_size`` (introspected for
|
|
287
|
+
local backends, identity-only for hosted-API ones -- see
|
|
288
|
+
``Model.model_info()``). When either side is API-based there is no
|
|
289
|
+
on-disk size to ratio, so ``size_ratio``/``quality_per_mb`` are
|
|
290
|
+
omitted and ``api_based`` names the two models instead."""
|
|
291
|
+
self._warn_if_issues()
|
|
292
|
+
mds = {md.metric: md for md in self.metric_deltas()}
|
|
293
|
+
if metric is None:
|
|
294
|
+
metric = next(iter(mds), None)
|
|
295
|
+
out: dict[str, Any] = {"metric": metric}
|
|
296
|
+
if metric in mds:
|
|
297
|
+
out["baseline_quality"] = mds[metric].baseline
|
|
298
|
+
out["candidate_quality"] = mds[metric].candidate
|
|
299
|
+
out["retention"] = self.retention(metric)
|
|
300
|
+
|
|
301
|
+
# or {}: model_size is None (not {}) when RunConfig.track_performance
|
|
302
|
+
# was never enabled on that run -- treated the same as "measured but
|
|
303
|
+
# nothing recorded" here, since tradeoff() already omits size_ratio
|
|
304
|
+
# etc. whenever a side has no size_mb, whatever the reason.
|
|
305
|
+
base_size, cand_size = self.baseline.model_size or {}, self.candidate.model_size or {}
|
|
306
|
+
baseline_size_mb, candidate_size_mb = base_size.get("size_mb"), cand_size.get("size_mb")
|
|
307
|
+
if baseline_size_mb and candidate_size_mb:
|
|
308
|
+
out["baseline_size_mb"] = baseline_size_mb
|
|
309
|
+
out["candidate_size_mb"] = candidate_size_mb
|
|
310
|
+
out["size_ratio"] = candidate_size_mb / baseline_size_mb # <1 = smaller
|
|
311
|
+
if metric in mds:
|
|
312
|
+
# retention(), not the raw candidate value: for a MINIMIZE
|
|
313
|
+
# metric (latency, cost) a bigger raw number is WORSE, so
|
|
314
|
+
# dividing it by size and calling the result "quality" would
|
|
315
|
+
# be backwards. retention() already normalizes for direction
|
|
316
|
+
# (1.0 = parity, regardless of whether the metric counts up
|
|
317
|
+
# or down), so "quality per mb" means the same thing for
|
|
318
|
+
# every metric.
|
|
319
|
+
out["quality_per_mb"] = self.retention(metric) / candidate_size_mb
|
|
320
|
+
elif base_size or cand_size:
|
|
321
|
+
# No real size_mb on at least one side -- either it's a hosted
|
|
322
|
+
# -API/callable model (never measurable at all) or a local
|
|
323
|
+
# backend that's marked is_local but doesn't actually introspect
|
|
324
|
+
# size yet (e.g. VLLMModel today -- see model/vllm_gen.py). Either
|
|
325
|
+
# way there's no size ratio to report; name the two models
|
|
326
|
+
# instead of silently reporting nothing at all (the previous
|
|
327
|
+
# behavior when both sides were "local" but neither had size_mb).
|
|
328
|
+
out["api_based"] = {
|
|
329
|
+
"baseline": base_size.get("model_name", self.baseline.model_spec),
|
|
330
|
+
"candidate": cand_size.get("model_name", self.candidate.model_spec),
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
baseline_latency_ms = ((self.baseline.perf or {}).get("latency_ms") or {}).get("mean") or None
|
|
334
|
+
candidate_latency_ms = ((self.candidate.perf or {}).get("latency_ms") or {}).get("mean") or None
|
|
335
|
+
if baseline_latency_ms and candidate_latency_ms:
|
|
336
|
+
out["baseline_latency_ms"] = baseline_latency_ms
|
|
337
|
+
out["candidate_latency_ms"] = candidate_latency_ms
|
|
338
|
+
out["latency_ratio"] = candidate_latency_ms / baseline_latency_ms
|
|
339
|
+
out["speedup"] = baseline_latency_ms / candidate_latency_ms # >1 = faster
|
|
340
|
+
return out
|
|
341
|
+
|
|
342
|
+
# -- performance (latency/throughput, measured automatically) ---------
|
|
343
|
+
def performance(self) -> dict[str, Any]:
|
|
344
|
+
"""Baseline-vs-candidate latency/throughput, straight from each run's
|
|
345
|
+
measured ``RunResult.perf`` -- no caller-supplied numbers needed
|
|
346
|
+
(unlike ``tradeoff()``, which folds in quality + optional size).
|
|
347
|
+
|
|
348
|
+
A side's fields here are ``None`` (not ``{}``/``0.0``) when that run
|
|
349
|
+
never enabled ``RunConfig.track_performance`` at all -- distinct from
|
|
350
|
+
an empty/zeroed value, which means it *was* tracked but nothing was
|
|
351
|
+
timed (e.g. a ``reads_actual_output`` run). Mixing a tracked and an
|
|
352
|
+
untracked run is fine: the tracked side still reports its real
|
|
353
|
+
numbers, the untracked side reports ``None`` for its own fields, and
|
|
354
|
+
``speedup`` (which needs both) is simply omitted.
|
|
355
|
+
"""
|
|
356
|
+
base_tracked = self.baseline.perf is not None
|
|
357
|
+
cand_tracked = self.candidate.perf is not None
|
|
358
|
+
base_lat = (self.baseline.perf.get("latency_ms") or {}) if base_tracked else None
|
|
359
|
+
cand_lat = (self.candidate.perf.get("latency_ms") or {}) if cand_tracked else None
|
|
360
|
+
base_thr = (self.baseline.perf.get("throughput") or {}) if base_tracked else None
|
|
361
|
+
cand_thr = (self.candidate.perf.get("throughput") or {}) if cand_tracked else None
|
|
362
|
+
out: dict[str, Any] = {
|
|
363
|
+
"baseline_latency_ms": base_lat,
|
|
364
|
+
"candidate_latency_ms": cand_lat,
|
|
365
|
+
"baseline_throughput_rps": base_thr.get("rps", 0.0) if base_thr is not None else None,
|
|
366
|
+
"candidate_throughput_rps": cand_thr.get("rps", 0.0) if cand_thr is not None else None,
|
|
367
|
+
"baseline_output_tokens_per_sec": base_thr.get("output_tokens_per_sec", 0.0) if base_thr is not None else None,
|
|
368
|
+
"candidate_output_tokens_per_sec": cand_thr.get("output_tokens_per_sec", 0.0) if cand_thr is not None else None,
|
|
369
|
+
}
|
|
370
|
+
if base_lat is not None and cand_lat is not None and base_lat.get("count") and cand_lat.get("count"):
|
|
371
|
+
out["speedup"] = base_lat["mean"] / cand_lat["mean"] if cand_lat["mean"] else float("nan")
|
|
372
|
+
return out
|
|
373
|
+
|
|
374
|
+
# -- cost (real measured tokens; pricing is the one thing you must supply) --
|
|
375
|
+
def cost(self, baseline_pricing: dict[str, float], candidate_pricing: dict[str, float]) -> dict[str, Any]:
|
|
376
|
+
"""Dollar cost of baseline vs. candidate. Token counts are real,
|
|
377
|
+
provider-reported numbers already captured on each ``RunResult``
|
|
378
|
+
(see ``RunResult.token_usage``/``.cost()``) -- never estimated here.
|
|
379
|
+
``$``/token pricing has no library default (it isn't measurable and
|
|
380
|
+
goes stale) so it's the one thing you pass in, per side, since a
|
|
381
|
+
baseline and candidate are often different paid models with
|
|
382
|
+
different prices. Returns ``{}`` for either side that recorded no
|
|
383
|
+
token usage at all (nothing to price)."""
|
|
384
|
+
out: dict[str, Any] = {}
|
|
385
|
+
base_cost = self.baseline.cost(baseline_pricing)
|
|
386
|
+
cand_cost = self.candidate.cost(candidate_pricing)
|
|
387
|
+
if base_cost is not None:
|
|
388
|
+
out["baseline_cost"] = base_cost
|
|
389
|
+
if cand_cost is not None:
|
|
390
|
+
out["candidate_cost"] = cand_cost
|
|
391
|
+
if base_cost and cand_cost:
|
|
392
|
+
out["cost_ratio"] = cand_cost / base_cost
|
|
393
|
+
return out
|
|
394
|
+
|
|
395
|
+
# -- sample browser (delegated to RunDiff, aligned by sample_id) ------
|
|
396
|
+
def regressed(self, metric_name: str = "") -> list[Prediction]:
|
|
397
|
+
"""Samples correct in ``baseline`` but wrong in ``candidate`` -- what broke.
|
|
398
|
+
|
|
399
|
+
Named for what it reports (a two-way flip between exactly these two
|
|
400
|
+
runs), not "recently" -- there's no history or tracking involved, just
|
|
401
|
+
this one baseline-vs-candidate comparison. Was ``newly_wrong()``.
|
|
402
|
+
"""
|
|
403
|
+
return self._diff.regressed(metric_name)
|
|
404
|
+
|
|
405
|
+
def improved(self, metric_name: str = "") -> list[Prediction]:
|
|
406
|
+
"""Samples wrong in ``baseline`` but fixed in ``candidate``. Was ``newly_correct()``."""
|
|
407
|
+
return self._diff.improved(metric_name)
|
|
408
|
+
|
|
409
|
+
def still_wrong(self, metric_name: str = "") -> list[Prediction]:
|
|
410
|
+
return self._diff.still_wrong(metric_name)
|
|
411
|
+
|
|
412
|
+
def sample_diff(self, metric_name: str = "") -> list[dict]:
|
|
413
|
+
return self._diff.sample_diff(metric_name)
|
|
414
|
+
|
|
415
|
+
def sample_summary(self, metric_name: str = "") -> dict[str, int]:
|
|
416
|
+
return self._diff.sample_summary(metric_name)
|
|
417
|
+
|
|
418
|
+
# -- failed-sample visibility -------------------------------------------
|
|
419
|
+
def failure_summary(self) -> dict[str, Any]:
|
|
420
|
+
"""Per-run failed-sample counts and errors.
|
|
421
|
+
|
|
422
|
+
``metric_deltas()``/``per_task_deltas()``/``retention()``/``tradeoff()``
|
|
423
|
+
all read per-sample scores from ``Prediction.metadata["scores"]``, which
|
|
424
|
+
is empty on a sample that errored during annotation/scoring (see
|
|
425
|
+
``Runner.run()``'s per-sample fallback) -- such samples are silently
|
|
426
|
+
**excluded** from those means, not counted as a miss. A run with several
|
|
427
|
+
failures can therefore report a better mean than a clean run would,
|
|
428
|
+
purely because its denominator shrank. Call this (or check
|
|
429
|
+
``has_failures``) before trusting a grade; ``regressed()`` is the one
|
|
430
|
+
view that *does* catch a failure correctly (via ``Prediction.correct``,
|
|
431
|
+
which is ``False`` on the fallback), everything else needs this check.
|
|
432
|
+
"""
|
|
433
|
+
return {
|
|
434
|
+
"baseline_failed": self.baseline.failed_count,
|
|
435
|
+
"candidate_failed": self.candidate.failed_count,
|
|
436
|
+
"baseline_errors": self.baseline.errors,
|
|
437
|
+
"candidate_errors": self.candidate.errors,
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
@property
|
|
441
|
+
def has_failures(self) -> bool:
|
|
442
|
+
return bool(self.baseline.failed_count or self.candidate.failed_count)
|
|
443
|
+
|
|
444
|
+
def coverage_warnings(self) -> list[str]:
|
|
445
|
+
"""Metrics (or task/metric pairs) whose baseline/candidate sample
|
|
446
|
+
counts diverge.
|
|
447
|
+
|
|
448
|
+
``metric_deltas()``/``per_task_deltas()`` derive their metric set purely
|
|
449
|
+
from whichever names actually appear in each run's stats/scores -- there
|
|
450
|
+
is no check against an "expected" scorer list, and a metric that failed
|
|
451
|
+
on every sample (or was scored under a different config) simply never
|
|
452
|
+
shows up, silently. A count mismatch between baseline and candidate for
|
|
453
|
+
the *same* metric name is the visible symptom of that: it means the two
|
|
454
|
+
means were computed over different sample sets, not a like-for-like
|
|
455
|
+
comparison -- treat a large mismatch as a reason to distrust that
|
|
456
|
+
metric's grade, not just a curiosity.
|
|
457
|
+
|
|
458
|
+
Checks both the blended (overall) counts *and* the per-task counts --
|
|
459
|
+
two tasks can individually have mismatched N (5 vs 3, 5 vs 7) while the
|
|
460
|
+
*totals* coincidentally agree (10 vs 10), invisible to a blended-only
|
|
461
|
+
check.
|
|
462
|
+
"""
|
|
463
|
+
warnings = []
|
|
464
|
+
for md in self.metric_deltas():
|
|
465
|
+
if md.baseline_count != md.candidate_count:
|
|
466
|
+
warnings.append(
|
|
467
|
+
f"{md.metric}: baseline n={md.baseline_count}, candidate n={md.candidate_count}"
|
|
468
|
+
)
|
|
469
|
+
for td in self.per_task_deltas():
|
|
470
|
+
if td.baseline_count != td.candidate_count:
|
|
471
|
+
warnings.append(
|
|
472
|
+
f"{td.task}/{td.metric}: baseline n={td.baseline_count}, "
|
|
473
|
+
f"candidate n={td.candidate_count}"
|
|
474
|
+
)
|
|
475
|
+
return warnings
|
|
476
|
+
|
|
477
|
+
def summary(self) -> str:
|
|
478
|
+
lines = []
|
|
479
|
+
if self.has_failures:
|
|
480
|
+
fs = self.failure_summary()
|
|
481
|
+
lines.append(
|
|
482
|
+
f" WARNING: baseline had {fs['baseline_failed']} failed sample(s), "
|
|
483
|
+
f"candidate had {fs['candidate_failed']} -- EXCLUDED from the means "
|
|
484
|
+
f"below, not counted as wrong. See .failure_summary()."
|
|
485
|
+
)
|
|
486
|
+
cw = self.coverage_warnings()
|
|
487
|
+
if cw:
|
|
488
|
+
lines.append(
|
|
489
|
+
" WARNING: sample coverage differs per metric (means computed over "
|
|
490
|
+
"different Ns, not a like-for-like comparison):"
|
|
491
|
+
)
|
|
492
|
+
lines += [f" {w}" for w in cw]
|
|
493
|
+
mds = self.metric_deltas()
|
|
494
|
+
# Just the percentage differences per metric -- no pass/warn/fail and no
|
|
495
|
+
# better/worse framing. A fixed threshold isn't meaningful across every
|
|
496
|
+
# metric, so it's the reader's job to interpret the numbers for their own
|
|
497
|
+
# metrics. (An opt-in gate is still available via .grade()/.grades() with
|
|
498
|
+
# your own thresholds.)
|
|
499
|
+
lines += [
|
|
500
|
+
f"Comparison: {self.baseline.run_id} (baseline) -> {self.candidate.run_id} (candidate)",
|
|
501
|
+
" per-metric (baseline -> candidate | absolute delta | relative %):",
|
|
502
|
+
]
|
|
503
|
+
for md in mds:
|
|
504
|
+
lines.append(" " + _delta_line(md.metric, md))
|
|
505
|
+
tasks = self.per_task_deltas()
|
|
506
|
+
# Show per-task detail whenever there's more than one distinct task --
|
|
507
|
+
# comparing this to len(metric_deltas()) instead (the old check) could
|
|
508
|
+
# wrongly hide a genuinely different per-task breakdown whenever the
|
|
509
|
+
# row counts happened to coincide numerically (e.g. 2 metrics each
|
|
510
|
+
# scored on a different, non-overlapping single task).
|
|
511
|
+
if len({td.task for td in tasks}) > 1:
|
|
512
|
+
lines.append(" per-task:")
|
|
513
|
+
for td in tasks:
|
|
514
|
+
lines.append(" " + _delta_line(f"{td.task}/{td.metric}", td))
|
|
515
|
+
sm = self.sample_summary()
|
|
516
|
+
lines.append(
|
|
517
|
+
f" samples: {sm['newly_wrong']} went correct->wrong, "
|
|
518
|
+
f"{sm['newly_correct']} went wrong->correct"
|
|
519
|
+
)
|
|
520
|
+
perf = self.performance()
|
|
521
|
+
if (perf["baseline_latency_ms"] is not None and perf["candidate_latency_ms"] is not None
|
|
522
|
+
and perf["baseline_latency_ms"].get("count") and perf["candidate_latency_ms"].get("count")):
|
|
523
|
+
bl, cl = perf["baseline_latency_ms"], perf["candidate_latency_ms"]
|
|
524
|
+
line = (
|
|
525
|
+
f" performance: latency {bl['mean']:.0f}ms -> {cl['mean']:.0f}ms, "
|
|
526
|
+
f"throughput {perf['baseline_throughput_rps']:.2f} -> "
|
|
527
|
+
f"{perf['candidate_throughput_rps']:.2f} req/s "
|
|
528
|
+
f"(speedup={perf['speedup']:.2f}x)"
|
|
529
|
+
)
|
|
530
|
+
# Only when the backend actually reported token usage (hosted
|
|
531
|
+
# APIs) -- 0 tok/s for a local/callable backend would
|
|
532
|
+
# misleadingly read as "measured zero" rather than "unavailable".
|
|
533
|
+
if perf["baseline_output_tokens_per_sec"] or perf["candidate_output_tokens_per_sec"]:
|
|
534
|
+
line += (
|
|
535
|
+
f", {perf['baseline_output_tokens_per_sec']:.1f} -> "
|
|
536
|
+
f"{perf['candidate_output_tokens_per_sec']:.1f} out-tok/s"
|
|
537
|
+
)
|
|
538
|
+
lines.append(line)
|
|
539
|
+
size_line = self._model_size_line()
|
|
540
|
+
if size_line:
|
|
541
|
+
lines.append(f" {size_line}")
|
|
542
|
+
return "\n".join(lines)
|
|
543
|
+
|
|
544
|
+
def _model_size_line(self) -> str:
|
|
545
|
+
"""Model size for both sides, or '' if neither run recorded any
|
|
546
|
+
(RunResult.model_size is None when RunConfig.track_performance was
|
|
547
|
+
never enabled on that run, {} for a tracked run with nothing to
|
|
548
|
+
report -- both treated the same way here)."""
|
|
549
|
+
base_size, cand_size = self.baseline.model_size or {}, self.candidate.model_size or {}
|
|
550
|
+
if not base_size and not cand_size:
|
|
551
|
+
return ""
|
|
552
|
+
baseline_size_mb, candidate_size_mb = base_size.get("size_mb"), cand_size.get("size_mb")
|
|
553
|
+
if baseline_size_mb and candidate_size_mb:
|
|
554
|
+
return (
|
|
555
|
+
f"model size: {baseline_size_mb:.1f} MB -> {candidate_size_mb:.1f} MB "
|
|
556
|
+
f"({base_size.get('model_name', '?')} -> {cand_size.get('model_name', '?')})"
|
|
557
|
+
)
|
|
558
|
+
# Either a hosted-API/callable model (nothing to measure) or a local
|
|
559
|
+
# backend marked is_local that doesn't actually introspect size yet
|
|
560
|
+
# (e.g. VLLMModel) -- either way, no fake "0.0 MB" placeholder.
|
|
561
|
+
base_name = base_size.get("model_name", self.baseline.model_spec)
|
|
562
|
+
cand_name = cand_size.get("model_name", self.candidate.model_spec)
|
|
563
|
+
return f"model: {base_name} -> {cand_name} (no measurable size)"
|