quantdiff 0.1.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. quantdiff/__init__.py +53 -0
  2. quantdiff/__main__.py +5 -0
  3. quantdiff/_http.py +151 -0
  4. quantdiff/_text.py +13 -0
  5. quantdiff/_version.py +1 -0
  6. quantdiff/api.py +340 -0
  7. quantdiff/backends/__init__.py +28 -0
  8. quantdiff/backends/_common.py +342 -0
  9. quantdiff/backends/base.py +91 -0
  10. quantdiff/backends/llamacpp.py +428 -0
  11. quantdiff/backends/ollama.py +359 -0
  12. quantdiff/backends/openai_compat.py +338 -0
  13. quantdiff/cache.py +240 -0
  14. quantdiff/card.py +1664 -0
  15. quantdiff/cli.py +377 -0
  16. quantdiff/discover.py +488 -0
  17. quantdiff/errors.py +45 -0
  18. quantdiff/metrics/__init__.py +36 -0
  19. quantdiff/metrics/codeexec.py +428 -0
  20. quantdiff/metrics/jsonschema.py +610 -0
  21. quantdiff/metrics/logit.py +214 -0
  22. quantdiff/metrics/tasks.py +114 -0
  23. quantdiff/metrics/textsim.py +66 -0
  24. quantdiff/metrics/toolcheck.py +99 -0
  25. quantdiff/png.py +360 -0
  26. quantdiff/preflight.py +365 -0
  27. quantdiff/progress.py +283 -0
  28. quantdiff/py.typed +0 -0
  29. quantdiff/report.py +780 -0
  30. quantdiff/runner.py +492 -0
  31. quantdiff/spec.py +154 -0
  32. quantdiff/stats.py +226 -0
  33. quantdiff/suites/__init__.py +462 -0
  34. quantdiff/suites/data/chat.jsonl +22 -0
  35. quantdiff/suites/data/code.jsonl +32 -0
  36. quantdiff/suites/data/json.jsonl +34 -0
  37. quantdiff/suites/data/scoring.jsonl +41 -0
  38. quantdiff/suites/data/tools.jsonl +32 -0
  39. quantdiff/types.py +322 -0
  40. quantdiff/verdict.py +1513 -0
  41. quantdiff-0.1.0rc1.dist-info/METADATA +514 -0
  42. quantdiff-0.1.0rc1.dist-info/RECORD +45 -0
  43. quantdiff-0.1.0rc1.dist-info/WHEEL +4 -0
  44. quantdiff-0.1.0rc1.dist-info/entry_points.txt +2 -0
  45. quantdiff-0.1.0rc1.dist-info/licenses/LICENSE +202 -0
quantdiff/verdict.py ADDED
@@ -0,0 +1,1513 @@
1
+ """Turn a Report into a recommendation: which download to run, which to avoid, and why.
2
+
3
+ This is the contract the scorecards render. The dataclasses are shared with card.py and
4
+ the CLI; `judge` applies the rules below, which docs/methodology.md explains for users.
5
+
6
+ Every comparison is paired, because every model answers the same cases and is scored on
7
+ the same prompts: task outcomes are joined by case id and logit results by prompt id. Only
8
+ items present on both sides count.
9
+
10
+ Each candidate is judged on its own against the reference, never against the other
11
+ candidates, so adding or removing a candidate never changes another one's status. A
12
+ candidate is recommended only on positive evidence that it is close to the reference; thin
13
+ evidence leads to inconclusive, not to a pick. The KLD interval below is the 95% bootstrap
14
+ interval of the mean per-prompt KLD. Rules, applied in order:
15
+
16
+ 1. failed: the candidate produced no metrics at all.
17
+ 2. avoid, when any of these holds:
18
+ - a reliable suite (the reference passes at least half of its cases) shows a significant
19
+ paired regression (exact McNemar p < 0.05, candidate below the reference);
20
+ - the reliable suites pooled together show a significant paired regression;
21
+ - with at least MIN_PROMPTS paired prompts, the mean KLD is at least LARGE_KLD and so is
22
+ the lower end of its interval.
23
+ 3. close (eligible to run), only on positive evidence:
24
+ - with logit metrics, closeness rests on them: at least MIN_PROMPTS paired prompts and
25
+ the upper end of the KLD interval below CLOSE_KLD. Task suites of a few dozen cases
26
+ cannot bound small differences, so with logit evidence they act as a breakage
27
+ detector: any significant loss (rule 2) is avoid, and a wide but not significant
28
+ interval is shown, not used against the candidate. A KLD that is shown small also
29
+ bounds how far the two output distributions can differ.
30
+ - without logit metrics, closeness rests on tasks: the reliable suites pooled, at least
31
+ MIN_CASES cases, and the lower end of the 95% interval of the pass-rate difference at
32
+ or above -TASK_MARGIN points.
33
+ The reasons say which evidence the call rests on.
34
+ 4. usable: with at least MIN_PROMPTS paired prompts, the lower end of the KLD interval is
35
+ above CLOSE_KLD and the mean is below LARGE_KLD. A measured, moderate loss: the best
36
+ choice when nothing close fits the size budget.
37
+ 5. inconclusive: everything else, including a large-looking mean on fewer than MIN_PROMPTS
38
+ prompts and a mean at or above LARGE_KLD whose interval reaches below it. The reasons
39
+ name what is missing and estimate how many prompts or cases per suite would decide.
40
+ 6. The pick. Among the close candidates that fit the --max-size budget (every close one
41
+ when there is no budget) the smallest download is recommended (by size on disk when
42
+ each of them reports one, otherwise the lowest KLD). The other close ones are ok. A
43
+ candidate that is not close is never recommended or ok, so when no close candidate
44
+ fits, the headline names the usable candidate with the lowest KLD that fits instead.
45
+
46
+ Caveats do not change a status but sit next to it: a KLD interval near a bar (a rerun
47
+ could move the candidate across it), and a reliable suite whose estimate is more than
48
+ TASK_MARGIN points below the reference without being significant.
49
+
50
+ A candidate is never described as better than the reference. A significant gain is
51
+ reported as a warning sign about the suite or the reference, not as a win.
52
+
53
+ Ranking: status (recommended, ok, usable, inconclusive, avoid, failed), then mean KLD
54
+ ascending, size ascending, and input order.
55
+ """
56
+
57
+ from __future__ import annotations
58
+
59
+ import itertools
60
+ import math
61
+ import os.path
62
+ import unicodedata
63
+ from collections.abc import Iterable, Sequence
64
+ from dataclasses import dataclass, field
65
+ from typing import Final, Literal
66
+
67
+ from quantdiff.stats import (
68
+ Interval,
69
+ bootstrap_mean,
70
+ cases_to_bound_loss,
71
+ items_to_bound_below,
72
+ mcnemar_exact,
73
+ paired_proportion_diff,
74
+ )
75
+ from quantdiff.suites import load_builtin, load_scoring_prompts
76
+ from quantdiff.types import (
77
+ CandidateResult,
78
+ LogitMetrics,
79
+ PreflightFinding,
80
+ Report,
81
+ RunSettings,
82
+ ScoringPrompt,
83
+ TaskKind,
84
+ )
85
+
86
+ __all__ = [
87
+ "CALIBRATED_TOP_K",
88
+ "CLOSE_KLD",
89
+ "FULL_VOCAB_CLOSE",
90
+ "FULL_VOCAB_LARGE",
91
+ "FULL_VOCAB_NEAR_LOSSLESS",
92
+ "LABEL_SEPARATORS",
93
+ "LARGE_KLD",
94
+ "MIN_CASES",
95
+ "MIN_PROMPTS",
96
+ "MIN_SHARED_PREFIX",
97
+ "NEAR_BAR",
98
+ "NEAR_LOSSLESS_KLD",
99
+ "SCORED_TASK_KINDS",
100
+ "TASK_MARGIN",
101
+ "TOP_K_FRACTION",
102
+ "CandidateVerdict",
103
+ "Interval",
104
+ "KldBand",
105
+ "KldThresholds",
106
+ "ServerFinding",
107
+ "Status",
108
+ "TaskDelta",
109
+ "Verdict",
110
+ "describe_delta",
111
+ "display_labels",
112
+ "format_size",
113
+ "judge",
114
+ "kld_band",
115
+ "kld_thresholds",
116
+ "mostly_non_latin",
117
+ "top_k_fraction",
118
+ ]
119
+
120
+ Status = Literal["recommended", "ok", "usable", "avoid", "inconclusive", "failed"]
121
+ """recommended: the one to run. ok: also close to the reference, but not the pick (usually
122
+ larger). usable: a measured but moderate loss; a sound choice when nothing closer fits.
123
+ avoid: a large loss or measured task breakage. inconclusive: not enough evidence either way.
124
+ failed: the candidate produced no usable metrics."""
125
+
126
+ KldBand = Literal["near-lossless", "small", "moderate", "large"]
127
+
128
+ # Thresholds. Every number the rules use is here, so a calibration can change them in one
129
+ # place; docs/methodology.md, docs/calibration.md and README.md quote them.
130
+ #
131
+ # The KLD bars are set on llama.cpp's full-vocabulary scale and converted to quantdiff's
132
+ # top-k lower bound, which reads a fixed fraction of the full value. docs/calibration.md
133
+ # measured that fraction on identical positions: 0.24 at k=1, 0.57 at k=5, 0.67 at k=10 and
134
+ # 0.76 at k=20 (llama-perplexity --kl-divergence as the full-vocabulary truth).
135
+ FULL_VOCAB_NEAR_LOSSLESS: Final = 0.015
136
+ """Full-vocabulary KLD below this is near-lossless (typical of Q8_0)."""
137
+ FULL_VOCAB_CLOSE: Final = 0.06
138
+ """Full-vocabulary closeness bar: about Q4_K_M on a 7 to 8B model."""
139
+ FULL_VOCAB_LARGE: Final = 0.15
140
+ """Full-vocabulary KLD at or above this is a large loss (Q2_K territory)."""
141
+ TOP_K_FRACTION: Final[tuple[tuple[int, float], ...]] = (
142
+ (1, 0.24),
143
+ (5, 0.57),
144
+ (10, 0.67),
145
+ (20, 0.76),
146
+ )
147
+ """Measured top-k lower bound as a fraction of full-vocabulary KLD, by k."""
148
+ CALIBRATED_TOP_K: Final = 10
149
+ """The default --top-k, at which the module-level KLD constants below apply."""
150
+ MIN_PROMPTS: Final = 8
151
+ """Fewer paired prompts than this never prove a candidate close or far on logits: a
152
+ bootstrap over a handful of prompts has almost no distinct resamples."""
153
+ MIN_CASES: Final = 20
154
+ """Fewer pooled paired task cases than this never prove a candidate close on tasks."""
155
+ TASK_MARGIN: Final = 10.0
156
+ """Points of pass rate the pooled task interval may reach below the reference while the
157
+ candidate still counts as close."""
158
+ NEAR_BAR: Final = 0.1
159
+ """A KLD interval that straddles a bar, or ends within this fraction of it, is near it."""
160
+
161
+ SCORED_TASK_KINDS: Final[tuple[TaskKind, ...]] = ("json", "tools", "code")
162
+ """Task kinds with a pass/fail check. Chat is scored by agreement instead."""
163
+ LABEL_SEPARATORS: Final = ":-_/."
164
+ MIN_SHARED_PREFIX: Final = 8
165
+
166
+ _ALPHA: Final = 0.05
167
+ _RELIABLE_REFERENCE_RATE: Final = 0.5
168
+ _STATUS_ORDER: Final[dict[Status, int]] = {
169
+ "recommended": 0,
170
+ "ok": 1,
171
+ "usable": 2,
172
+ "inconclusive": 3,
173
+ "avoid": 4,
174
+ "failed": 5,
175
+ }
176
+ _MAX_CASES_TO_RESOLVE: Final = 5000
177
+ """Larger estimates of the cases that would resolve a task loss are not worth quoting."""
178
+ _NON_LATIN_SHARE: Final = 0.5
179
+ """Prompts whose letters are more than this share non-Latin are mostly non-Latin script."""
180
+ _SIZE_DIGITS: Final = 3
181
+ """Significant digits of the sizes and budgets that sentences quote."""
182
+ _CONFIRM_EXACT: Final = "Confirm with llama-server (exact token ids) before ruling out a download."
183
+ _SAME_SIZE: Final = 0.005
184
+ """Size changes under half a percent are reported as the same size."""
185
+ _ROUND_NEEDED_TO: Final = 10
186
+ """Evidence estimates are rough, so they are rounded up to a multiple of this."""
187
+ _QUANTIZED_REFERENCE: Final = (
188
+ "usually a sign the suite is too small or the reference is quantized itself"
189
+ )
190
+
191
+
192
+ @dataclass(frozen=True, slots=True)
193
+ class TaskDelta:
194
+ """Candidate minus reference pass rate for one suite, paired case by case, in points."""
195
+
196
+ kind: TaskKind
197
+ cases: int
198
+ candidate_rate: float
199
+ reference_rate: float
200
+ delta: Interval
201
+ significant: bool
202
+ """True when the paired test (exact McNemar) rejects equality at the 5% level."""
203
+ reference_reliable: bool
204
+ """False when the reference itself passes under half the cases, so the suite says little
205
+ about this model; such suites are shown but never used to judge a candidate."""
206
+
207
+
208
+ @dataclass(frozen=True, slots=True)
209
+ class CandidateVerdict:
210
+ label: str
211
+ status: Status
212
+ rank: int | None
213
+ """1 is best; None for failed candidates."""
214
+ kld_band: KldBand | None
215
+ size_bytes: int | None
216
+ size_change: float | None
217
+ """Fractional size change against the reference, e.g. -0.25 for 25% smaller."""
218
+ task_deltas: tuple[TaskDelta, ...]
219
+ reasons: tuple[str, ...]
220
+ """Short plain-English sentences that justify the status, most important first."""
221
+ caveats: tuple[str, ...] = ()
222
+ """Unresolved concerns that do not change the status but a reader must see next to it,
223
+ e.g. "tools -17 unresolved (95% CI -42 to +6); rerun with --max-cases 60"."""
224
+ near_bar: bool = False
225
+ """True when the KLD interval straddles a band edge closely enough that a rerun could
226
+ change the status."""
227
+ fits_budget: bool | None = None
228
+ """Whether the download fits the --max-size budget; None when no budget was given or
229
+ the size is unknown."""
230
+
231
+
232
+ @dataclass(frozen=True, slots=True)
233
+ class ServerFinding:
234
+ """A pre-flight finding, with whether it could have changed the scores on this card."""
235
+
236
+ finding: PreflightFinding
237
+ labels: tuple[str, ...]
238
+ """Models it applies to; every model when it is a server-wide issue."""
239
+ affects_scores: bool
240
+ impact: str
241
+ """One sentence, e.g. "does not affect these scores: the longest prompt is ~900 tokens"."""
242
+
243
+
244
+ @dataclass(frozen=True, slots=True)
245
+ class Verdict:
246
+ headline: str
247
+ """One short sentence that answers "which should I run?", e.g. "Run q4_K_M: 25% smaller
248
+ than q8_0, close on logits (KLD 0.03, CI up to 0.036) on 41 prompts.", "Best that fits
249
+ 6 GB: q4_K_M, moderate loss (KLD 0.044)." or an honest "Keep q8_0 for now: no candidate
250
+ is shown to be close on 12 prompts and 24 cases."."""
251
+ details: tuple[str, ...]
252
+ """Supporting sentences: the numbers behind the headline, what to avoid and why. An
253
+ unresolved task loss, when there is one, comes first."""
254
+ candidates: tuple[CandidateVerdict, ...]
255
+ """Every candidate, in rank order, failed ones last."""
256
+ findings: tuple[ServerFinding, ...]
257
+ cases_needed: int | None = None
258
+ """When nothing is recommended and a rerun would likely decide: the value of
259
+ --max-cases (cases per suite and scoring prompts, rounded up to a multiple of 10)."""
260
+ remedy: str | None = None
261
+ """When nothing is recommended: one sentence with the next step, e.g. "Rerun with
262
+ --max-cases 40 to decide." Kept out of the headline and the details."""
263
+
264
+
265
+ # Public helpers ---------------------------------------------------------------------------
266
+
267
+
268
+ @dataclass(frozen=True, slots=True)
269
+ class KldThresholds:
270
+ """The KLD bars on quantdiff's top-k scale for one --top-k setting."""
271
+
272
+ near_lossless: float
273
+ close: float
274
+ """A candidate is close on logits when its KLD interval ends below this, and usable at
275
+ best when the interval starts above it. Also the top of the small band."""
276
+ large: float
277
+ """Mean KLD at or above this is the large band; avoided when the interval starts at or
278
+ above it too."""
279
+
280
+ @property
281
+ def bar(self) -> str:
282
+ """The closeness bar as sentences quote it."""
283
+ return f"{self.close:.2g}"
284
+
285
+ @property
286
+ def large_bar(self) -> str:
287
+ """The large-loss bar as sentences quote it."""
288
+ return f"{self.large:.2f}"
289
+
290
+
291
+ def top_k_fraction(top_k: int) -> float:
292
+ """The measured fraction of full-vocabulary KLD that a top-k lower bound reads,
293
+ interpolated linearly between calibrated k values and held flat outside them."""
294
+ points = TOP_K_FRACTION
295
+ if top_k <= points[0][0]:
296
+ return points[0][1]
297
+ for (k_low, f_low), (k_high, f_high) in itertools.pairwise(points):
298
+ if top_k <= k_high:
299
+ return f_low + (f_high - f_low) * (top_k - k_low) / (k_high - k_low)
300
+ return points[-1][1]
301
+
302
+
303
+ def kld_thresholds(top_k: int = CALIBRATED_TOP_K) -> KldThresholds:
304
+ """The KLD bars for a run with this --top-k, rounded to the precision cards print."""
305
+ fraction = top_k_fraction(top_k)
306
+ return KldThresholds(
307
+ near_lossless=round(FULL_VOCAB_NEAR_LOSSLESS * fraction, 3),
308
+ close=round(FULL_VOCAB_CLOSE * fraction, 3),
309
+ large=round(FULL_VOCAB_LARGE * fraction, 2),
310
+ )
311
+
312
+
313
+ _DEFAULT_THRESHOLDS: Final = kld_thresholds()
314
+ NEAR_LOSSLESS_KLD: Final = _DEFAULT_THRESHOLDS.near_lossless
315
+ """Near-lossless bar at the default --top-k (0.01)."""
316
+ CLOSE_KLD: Final = _DEFAULT_THRESHOLDS.close
317
+ """Closeness bar at the default --top-k (0.04)."""
318
+ LARGE_KLD: Final = _DEFAULT_THRESHOLDS.large
319
+ """Large-loss bar at the default --top-k (0.10)."""
320
+
321
+
322
+ def kld_band(kld_mean: float, thresholds: KldThresholds = _DEFAULT_THRESHOLDS) -> KldBand:
323
+ """quantdiff's band for a mean KLD; see docs/calibration.md for the anchors."""
324
+ if kld_mean < thresholds.near_lossless:
325
+ return "near-lossless"
326
+ if kld_mean < thresholds.close:
327
+ return "small"
328
+ if kld_mean < thresholds.large:
329
+ return "moderate"
330
+ return "large"
331
+
332
+
333
+ def display_labels(report: Report) -> dict[str, str]:
334
+ """Short names for sentences: each full label mapped to the part that tells models apart.
335
+
336
+ Quant labels of one model usually differ only after a long common stem, such as
337
+ qwen2.5:7b-instruct-q4_K_M and qwen2.5:7b-instruct-q8_0. When every label (reference
338
+ included) shares a prefix that ends at one of LABEL_SEPARATORS, is at least
339
+ MIN_SHARED_PREFIX characters, and leaves something on every label, the prefix is
340
+ dropped. Otherwise labels are used as they are.
341
+ """
342
+ labels = [result.spec.label for result in (report.reference, *report.candidates)]
343
+ prefix = ""
344
+ if len(labels) >= 2:
345
+ common = os.path.commonprefix(labels)
346
+ end = max((i + 1 for i, char in enumerate(common) if char in LABEL_SEPARATORS), default=0)
347
+ prefix = common[:end]
348
+ if len(prefix) < MIN_SHARED_PREFIX or any(len(label) == len(prefix) for label in labels):
349
+ prefix = ""
350
+ return {label: label[len(prefix) :] for label in labels}
351
+
352
+
353
+ def describe_delta(delta: TaskDelta) -> str:
354
+ """The delta as a reader should take it, e.g. "-30 points (95% CI -48 to -12)".
355
+
356
+ A gain that is not significant reads "= reference (within noise)", so a lucky case or
357
+ two is never shown as a candidate beating the reference.
358
+ """
359
+ estimate = round(delta.delta.estimate)
360
+ if not delta.significant:
361
+ if estimate >= 0:
362
+ return "= reference (within noise)"
363
+ return f"{estimate} points (within noise)"
364
+ if estimate > 0:
365
+ return f"+{estimate} points vs reference; {_QUANTIZED_REFERENCE}"
366
+ return f"{estimate} points ({_interval_points(delta.delta)})"
367
+
368
+
369
+ def format_size(size_bytes: int, *, round_up: bool = False) -> str:
370
+ """A download size or budget to three significant digits in decimal units, e.g.
371
+ "7.1 GB", "6.25 GB" or "398 MB". With `round_up`, never less than `size_bytes`, so the
372
+ text can be passed back as a --max-size that the download fits."""
373
+ for unit, scale in (("TB", 1e12), ("GB", 1e9), ("MB", 1e6), ("KB", 1e3)):
374
+ if size_bytes >= scale:
375
+ value = size_bytes / scale
376
+ decimals = max(0, _SIZE_DIGITS - len(str(int(value))))
377
+ step = 10**decimals
378
+ if round_up:
379
+ value = math.ceil(value * step) / step
380
+ text = f"{value:.{decimals}f}"
381
+ return (text.rstrip("0").rstrip(".") if decimals else text) + f" {unit}"
382
+ return f"{size_bytes} bytes"
383
+
384
+
385
+ def mostly_non_latin(texts: Iterable[str]) -> bool:
386
+ """True when more than half the letters in `texts` are outside the Latin script.
387
+
388
+ Only letters count, so digits, punctuation, code symbols and whitespace never tip the
389
+ balance. Text forcing re-tokenizes the reference's text, which is least faithful for
390
+ scripts that byte-level tokenizers split into many pieces.
391
+ """
392
+ letters = non_latin = 0
393
+ for text in texts:
394
+ for char in text:
395
+ if not char.isalpha():
396
+ continue
397
+ letters += 1
398
+ if not (char.isascii() or unicodedata.name(char, "").startswith("LATIN ")):
399
+ non_latin += 1
400
+ return letters > 0 and non_latin > letters * _NON_LATIN_SHARE
401
+
402
+
403
+ # Evidence ---------------------------------------------------------------------------------
404
+
405
+
406
+ @dataclass(frozen=True, slots=True)
407
+ class _LogitEvidence:
408
+ """Mean per-prompt KLD against the reference, with its bootstrap interval over prompts."""
409
+
410
+ prompts: int
411
+ """Prompts with a per-prompt KLD, the unit the interval resamples."""
412
+ mean: float
413
+ interval: Interval | None
414
+ """None when the report has no per-prompt results (reports before schema version 2)."""
415
+ thresholds: KldThresholds
416
+ exact: bool
417
+ """True when the backend scored by token id; False for text forcing."""
418
+
419
+ def _bounded(self) -> Interval | None:
420
+ """The interval, when there are enough prompts for it to decide anything."""
421
+ return self.interval if self.prompts >= MIN_PROMPTS else None
422
+
423
+ @property
424
+ def close(self) -> bool:
425
+ interval = self._bounded()
426
+ return interval is not None and interval.high < self.thresholds.close
427
+
428
+ @property
429
+ def moderate(self) -> bool:
430
+ interval = self._bounded()
431
+ return (
432
+ interval is not None
433
+ and interval.low > self.thresholds.close
434
+ and self.mean < self.thresholds.large
435
+ )
436
+
437
+ @property
438
+ def large(self) -> bool:
439
+ interval = self._bounded()
440
+ return (
441
+ interval is not None
442
+ and self.mean >= self.thresholds.large
443
+ and interval.low >= self.thresholds.large
444
+ )
445
+
446
+ @property
447
+ def near_bar(self) -> str | None:
448
+ """The name of the bar the interval is near ("closeness" or "large-loss"), if any."""
449
+ interval = self._bounded()
450
+ if interval is None:
451
+ return None
452
+ for name, bar in (
453
+ ("closeness", self.thresholds.close),
454
+ ("large-loss", self.thresholds.large),
455
+ ):
456
+ nearest = min(abs(interval.low - bar), abs(interval.high - bar))
457
+ if interval.low <= bar <= interval.high or nearest <= NEAR_BAR * bar:
458
+ return name
459
+ return None
460
+
461
+ @property
462
+ def prompts_needed(self) -> int | None:
463
+ """Prompts that would likely settle which side of the nearest bar the mean is on;
464
+ None if more will not.
465
+
466
+ Below the closeness bar that is the interval's upper end dropping under it. At or
467
+ above a bar it is the lower end clearing that bar, estimated on the mirrored
468
+ interval with the same square-root scaling.
469
+ """
470
+ interval = self.interval
471
+ if interval is None:
472
+ return None
473
+ if self.mean < self.thresholds.close:
474
+ needed = items_to_bound_below(interval, self.prompts, self.thresholds.close)
475
+ else:
476
+ bar = (
477
+ self.thresholds.large
478
+ if self.mean >= self.thresholds.large
479
+ else self.thresholds.close
480
+ )
481
+ mirrored = Interval(-interval.estimate, -interval.high, -interval.low)
482
+ needed = items_to_bound_below(mirrored, self.prompts, -bar)
483
+ return None if needed is None else max(MIN_PROMPTS, needed)
484
+
485
+
486
+ @dataclass(frozen=True, slots=True)
487
+ class _TaskEvidence:
488
+ """Every reliable suite's paired outcomes pooled into one pass-rate comparison."""
489
+
490
+ kinds: tuple[TaskKind, ...]
491
+ cases: int
492
+ delta: Interval
493
+ """Candidate minus reference pass rate in points, with a Newcombe 95% interval."""
494
+ pooled_needed: int | None
495
+ """Pooled cases at which the interval's lower end would likely clear -TASK_MARGIN (the
496
+ current count when it already does); None if more cases will not."""
497
+ p_value: float
498
+ """Exact McNemar p-value of the pooled comparison."""
499
+
500
+ @property
501
+ def close(self) -> bool:
502
+ return self.cases >= MIN_CASES and self.delta.low >= -TASK_MARGIN
503
+
504
+ @property
505
+ def significant_loss(self) -> bool:
506
+ return self.p_value < _ALPHA and self.delta.estimate < 0.0
507
+
508
+ @property
509
+ def cases_needed(self) -> int | None:
510
+ """The pooled requirement spread over the suites, as cases per suite."""
511
+ if self.pooled_needed is None:
512
+ return None
513
+ return math.ceil(max(self.pooled_needed, MIN_CASES) / len(self.kinds))
514
+
515
+
516
+ def _logit_evidence(logit: LogitMetrics | None, thresholds: KldThresholds) -> _LogitEvidence | None:
517
+ if logit is None or logit.prompts == 0:
518
+ return None
519
+ values = [p.kld_mean for p in logit.per_prompt if p.kld_mean is not None]
520
+ if not values:
521
+ return _LogitEvidence(0, logit.kld_mean, None, thresholds, logit.exact_token_ids)
522
+ interval = bootstrap_mean(values)
523
+ return _LogitEvidence(
524
+ len(values), interval.estimate, interval, thresholds, logit.exact_token_ids
525
+ )
526
+
527
+
528
+ def _task_evidence(
529
+ reference: dict[TaskKind, dict[str, bool]],
530
+ candidate: dict[TaskKind, dict[str, bool]],
531
+ reliable: frozenset[TaskKind],
532
+ ) -> _TaskEvidence | None:
533
+ kinds: list[TaskKind] = []
534
+ ref_pass: list[bool] = []
535
+ cand_pass: list[bool] = []
536
+ for kind in SCORED_TASK_KINDS:
537
+ if kind not in reliable:
538
+ continue
539
+ ours, theirs = _pair_cases(reference[kind], candidate[kind])
540
+ if ours:
541
+ kinds.append(kind)
542
+ ref_pass += ours
543
+ cand_pass += theirs
544
+ if not kinds:
545
+ return None
546
+ return _TaskEvidence(
547
+ kinds=tuple(kinds),
548
+ cases=len(ref_pass),
549
+ delta=paired_proportion_diff(ref_pass, cand_pass),
550
+ pooled_needed=cases_to_bound_loss(ref_pass, cand_pass, TASK_MARGIN),
551
+ p_value=mcnemar_exact(ref_pass, cand_pass),
552
+ )
553
+
554
+
555
+ def _outcomes(result: CandidateResult) -> dict[TaskKind, dict[str, bool]]:
556
+ joined: dict[TaskKind, dict[str, bool]] = {kind: {} for kind in SCORED_TASK_KINDS}
557
+ for outcome in result.outcomes:
558
+ if outcome.kind in joined and outcome.passed is not None:
559
+ joined[outcome.kind][outcome.case_id] = outcome.passed
560
+ return joined
561
+
562
+
563
+ def _pair_cases(first: dict[str, bool], second: dict[str, bool]) -> tuple[list[bool], list[bool]]:
564
+ shared = [case_id for case_id in first if case_id in second]
565
+ return [first[c] for c in shared], [second[c] for c in shared]
566
+
567
+
568
+ # Per-candidate measurements ---------------------------------------------------------------
569
+
570
+
571
+ @dataclass(frozen=True, slots=True)
572
+ class _Candidate:
573
+ result: CandidateResult
574
+ index: int
575
+ name: str
576
+ kld: float | None
577
+ """Mean KLD over all scored positions, as the card shows it."""
578
+ logit: _LogitEvidence | None
579
+ tasks: _TaskEvidence | None
580
+ size: int | None
581
+ size_change: float | None
582
+ deltas: tuple[TaskDelta, ...]
583
+ has_task_mean: bool
584
+ unresolved: tuple[tuple[TaskDelta, int | None], ...]
585
+ """Reliable suites more than TASK_MARGIN points below the reference whose interval still
586
+ reaches zero and whose loss is not significant, each with the paired cases that would
587
+ likely make the loss significant."""
588
+
589
+ @property
590
+ def label(self) -> str:
591
+ return self.result.spec.label
592
+
593
+ @property
594
+ def measured(self) -> bool:
595
+ agreement = self.result.agreement
596
+ return (
597
+ self.kld is not None
598
+ or self.has_task_mean
599
+ or (agreement is not None and agreement.cases > 0)
600
+ )
601
+
602
+ @property
603
+ def regressions(self) -> list[TaskDelta]:
604
+ return sorted(
605
+ (
606
+ d
607
+ for d in self.deltas
608
+ if d.reference_reliable and d.significant and d.delta.estimate < 0
609
+ ),
610
+ key=lambda d: d.delta.estimate,
611
+ )
612
+
613
+ @property
614
+ def gains(self) -> list[TaskDelta]:
615
+ return [d for d in self.deltas if d.significant and d.delta.estimate > 0]
616
+
617
+ def rank_key(self) -> tuple[bool, float, bool, int, int]:
618
+ return (
619
+ self.kld is None,
620
+ self.kld or 0.0,
621
+ self.size is None,
622
+ self.size or 0,
623
+ self.index,
624
+ )
625
+
626
+
627
+ def _measure(
628
+ result: CandidateResult,
629
+ index: int,
630
+ name: str,
631
+ *,
632
+ reference: CandidateResult,
633
+ reference_outcomes: dict[TaskKind, dict[str, bool]],
634
+ reliable: frozenset[TaskKind],
635
+ thresholds: KldThresholds,
636
+ ) -> _Candidate:
637
+ logit = result.logit if result.logit is not None and result.logit.prompts > 0 else None
638
+ size = None if result.info is None else result.info.size_bytes
639
+ reference_size = None if reference.info is None else reference.info.size_bytes
640
+ outcomes = _outcomes(result)
641
+ deltas = []
642
+ unresolved = []
643
+ for kind in SCORED_TASK_KINDS:
644
+ ref_pass, cand_pass = _pair_cases(reference_outcomes[kind], outcomes[kind])
645
+ if not ref_pass:
646
+ continue
647
+ delta = TaskDelta(
648
+ kind=kind,
649
+ cases=len(ref_pass),
650
+ candidate_rate=sum(cand_pass) / len(cand_pass),
651
+ reference_rate=sum(ref_pass) / len(ref_pass),
652
+ delta=paired_proportion_diff(ref_pass, cand_pass),
653
+ significant=mcnemar_exact(ref_pass, cand_pass) < _ALPHA,
654
+ reference_reliable=kind in reliable,
655
+ )
656
+ deltas.append(delta)
657
+ if (
658
+ delta.reference_reliable
659
+ and not delta.significant
660
+ and delta.delta.estimate < -TASK_MARGIN
661
+ and delta.delta.high >= 0.0
662
+ ):
663
+ unresolved.append((delta, _cases_to_resolve(ref_pass, cand_pass)))
664
+ return _Candidate(
665
+ result=result,
666
+ index=index,
667
+ name=name,
668
+ kld=None if logit is None else logit.kld_mean,
669
+ logit=_logit_evidence(logit, thresholds),
670
+ tasks=_task_evidence(reference_outcomes, outcomes, reliable),
671
+ size=size,
672
+ size_change=(None if size is None or not reference_size else size / reference_size - 1.0),
673
+ deltas=tuple(deltas),
674
+ has_task_mean=any(
675
+ task.kind in SCORED_TASK_KINDS and task.rate is not None for task in result.tasks
676
+ ),
677
+ unresolved=tuple(unresolved),
678
+ )
679
+
680
+
681
+ def _cases_to_resolve(ref: Sequence[bool], cand: Sequence[bool]) -> int | None:
682
+ """Roughly how many paired cases would make an observed loss significant.
683
+
684
+ The observed shares of the four paired outcomes are held fixed while the case count
685
+ grows, as for the task estimate in stats.cases_to_bound_loss, and the exact McNemar
686
+ test is rerun on the scaled counts. None past _MAX_CASES_TO_RESOLVE cases.
687
+ """
688
+ n = len(ref)
689
+ lost = sum(1 for r, c in zip(ref, cand, strict=True) if r and not c)
690
+ gained = sum(1 for r, c in zip(ref, cand, strict=True) if c and not r)
691
+
692
+ def significant(cases: int) -> bool:
693
+ scaled_lost = round(lost * cases / n)
694
+ scaled_gained = round(gained * cases / n)
695
+ reference = [True] * scaled_lost + [False] * scaled_gained
696
+ candidate = [False] * scaled_lost + [True] * scaled_gained
697
+ return mcnemar_exact(reference, candidate) < _ALPHA
698
+
699
+ low, high = n, 2 * n
700
+ while not significant(high):
701
+ if high >= _MAX_CASES_TO_RESOLVE:
702
+ return None
703
+ low, high = high, 2 * high
704
+ while high - low > 1:
705
+ middle = (low + high) // 2
706
+ if significant(middle):
707
+ high = middle
708
+ else:
709
+ low = middle
710
+ return high
711
+
712
+
713
+ # Judging ----------------------------------------------------------------------------------
714
+
715
+
716
+ @dataclass(slots=True)
717
+ class _Call:
718
+ """Status and reasons for one candidate while the verdict is being built."""
719
+
720
+ candidate: _Candidate
721
+ status: Status
722
+ reasons: list[str]
723
+ """Phrases, most important first; turned into sentences at the end."""
724
+ detail: str | None = None
725
+ """The sentence this candidate contributes to Verdict.details, if any."""
726
+ prompts_needed: int | None = None
727
+ cases_needed: int | None = None
728
+ """For an inconclusive candidate that more evidence could decide: scoring prompts and
729
+ cases per suite, rounded up. Both None when no rerun is likely to decide."""
730
+ fits: bool | None = None
731
+ """Whether the download fits the --max-size budget; None without a budget or a size."""
732
+ caveats: list[str] = field(default_factory=list)
733
+ task_caveat: str | None = None
734
+ """The first unresolved task loss, as a caveat phrase."""
735
+
736
+ @property
737
+ def name(self) -> str:
738
+ return self.candidate.name
739
+
740
+
741
+ def judge(report: Report, *, scoring: Sequence[ScoringPrompt] = ()) -> Verdict:
742
+ """Build the recommendation for `report`. Deterministic: same report, same verdict.
743
+
744
+ `scoring` is the text of the scoring prompts the run used, which reports do not store.
745
+ When it is mostly non-Latin script and a text-forced candidate reads above the
746
+ closeness bar, the details say that text forcing may read high on such prompts.
747
+ """
748
+ names = display_labels(report)
749
+ reference = report.reference
750
+ ref_name = names[reference.spec.label]
751
+ ref_outcomes = _outcomes(reference)
752
+ settings = report.settings
753
+ thresholds = kld_thresholds(settings.top_k)
754
+ reliable = frozenset(
755
+ kind
756
+ for kind in SCORED_TASK_KINDS
757
+ if ref_outcomes[kind]
758
+ and sum(ref_outcomes[kind].values()) / len(ref_outcomes[kind]) >= _RELIABLE_REFERENCE_RATE
759
+ )
760
+ candidates = [
761
+ _measure(
762
+ result,
763
+ index,
764
+ names[result.spec.label],
765
+ reference=reference,
766
+ reference_outcomes=ref_outcomes,
767
+ reliable=reliable,
768
+ thresholds=thresholds,
769
+ )
770
+ for index, result in enumerate(report.candidates)
771
+ ]
772
+ calls = [_assess(candidate, ref_name) for candidate in candidates]
773
+ for call in calls:
774
+ _annotate(call, settings)
775
+ _pick(calls, ref_name, settings.max_size_bytes)
776
+ for call in calls:
777
+ if call.status != "failed":
778
+ call.reasons += [
779
+ f"scored higher than the reference on {d.kind}; {_QUANTIZED_REFERENCE}"
780
+ for d in call.candidate.gains
781
+ ]
782
+
783
+ ordered = sorted(
784
+ calls, key=lambda call: (_STATUS_ORDER[call.status], call.candidate.rank_key())
785
+ )
786
+ verdicts = tuple(
787
+ CandidateVerdict(
788
+ label=call.candidate.label,
789
+ status=call.status,
790
+ rank=None if call.status == "failed" else rank,
791
+ kld_band=(
792
+ None if call.candidate.kld is None else kld_band(call.candidate.kld, thresholds)
793
+ ),
794
+ size_bytes=call.candidate.size,
795
+ size_change=call.candidate.size_change,
796
+ task_deltas=call.candidate.deltas,
797
+ reasons=tuple(_sentence(reason) for reason in call.reasons),
798
+ caveats=tuple(call.caveats),
799
+ near_bar=call.candidate.logit is not None and call.candidate.logit.near_bar is not None,
800
+ fits_budget=call.fits,
801
+ )
802
+ for rank, call in enumerate(ordered, start=1)
803
+ )
804
+ summary = _summarize(
805
+ ordered, ref_name, settings, non_latin=mostly_non_latin(p.text for p in scoring)
806
+ )
807
+ return Verdict(
808
+ headline=summary.headline,
809
+ details=summary.details,
810
+ candidates=verdicts,
811
+ findings=_findings(report, names),
812
+ cases_needed=summary.cases_needed,
813
+ remedy=summary.remedy,
814
+ )
815
+
816
+
817
+ def _assess(candidate: _Candidate, ref_name: str) -> _Call:
818
+ """The candidate's status on its own evidence; close candidates come back as ok."""
819
+ if not candidate.measured:
820
+ return _Call(candidate, "failed", [_failure_reason(candidate.result)])
821
+ avoid = _avoid_reasons(candidate, ref_name)
822
+ if avoid:
823
+ return _Call(
824
+ candidate, "avoid", avoid, f"Avoid {candidate.name}: {' and '.join(avoid[:2])}."
825
+ )
826
+ logit, tasks = candidate.logit, candidate.tasks
827
+ # Significant task losses were handled above, so with logit evidence only KLD decides.
828
+ close = logit.close if logit is not None else tasks is not None and tasks.close
829
+ if close:
830
+ return _Call(candidate, "ok", _close_reasons(candidate, ref_name))
831
+ if logit is not None and logit.moderate and logit.interval is not None:
832
+ return _usable(candidate, logit.interval, ref_name)
833
+ return _inconclusive(candidate, ref_name)
834
+
835
+
836
+ def _annotate(call: _Call, settings: RunSettings) -> None:
837
+ """Budget fit and caveats, which sit next to the status without changing it."""
838
+ candidate = call.candidate
839
+ budget = settings.max_size_bytes
840
+ if budget is not None and candidate.size is not None:
841
+ call.fits = candidate.size <= budget
842
+ if call.status == "failed":
843
+ return
844
+ near = None if candidate.logit is None else candidate.logit.near_bar
845
+ if near is not None:
846
+ call.caveats.append(f"near the {near} bar; a rerun could change this")
847
+ tasks = [_task_caveat(delta, needed, settings) for delta, needed in candidate.unresolved]
848
+ call.caveats += tasks
849
+ call.task_caveat = tasks[0] if tasks else None
850
+
851
+
852
+ def _task_caveat(delta: TaskDelta, needed: int | None, settings: RunSettings) -> str:
853
+ """E.g. "tools -17 unresolved (95% CI -42 to +6); rerun with --max-cases 60"."""
854
+ text = (
855
+ f"{delta.kind} {round(delta.delta.estimate)} unresolved ({_interval_points(delta.delta)})"
856
+ )
857
+ if needed is None:
858
+ return text
859
+ cases = _round_up(needed)
860
+ if settings.prompts_file is None and cases <= len(load_builtin(delta.kind)):
861
+ return f"{text}; rerun with --max-cases {cases}"
862
+ return f"{text}; about {cases} {delta.kind} cases would settle it"
863
+
864
+
865
+ def _failure_reason(result: CandidateResult) -> str:
866
+ if result.errors:
867
+ return "no metrics: " + " ".join(result.errors[0].split()).rstrip(".")
868
+ return "no metrics were produced"
869
+
870
+
871
+ def _avoid_reasons(candidate: _Candidate, ref_name: str) -> list[str]:
872
+ reasons = [
873
+ f"{d.kind} drops {-round(d.delta.estimate)} points vs {ref_name} "
874
+ f"({_interval_points(d.delta)})"
875
+ for d in candidate.regressions
876
+ ]
877
+ tasks = candidate.tasks
878
+ if not reasons and tasks is not None and tasks.significant_loss:
879
+ reasons.append(
880
+ f"task scores drop {_points(-tasks.delta.estimate)} points overall vs {ref_name} "
881
+ f"({_interval_points(tasks.delta)} on {_plural(tasks.cases, 'case')})"
882
+ )
883
+ logit = candidate.logit
884
+ if logit is not None and logit.large and logit.interval is not None:
885
+ reasons.append(f"KLD is large ({_kld(logit.mean)}, {_interval_kld(logit.interval)})")
886
+ return reasons
887
+
888
+
889
+ def _close_reasons(candidate: _Candidate, ref_name: str) -> list[str]:
890
+ reasons = []
891
+ logit, tasks = candidate.logit, candidate.tasks
892
+ if logit is not None and logit.interval is not None:
893
+ reasons.append(
894
+ f"KLD {_kld(logit.mean)} ({_interval_kld(logit.interval)}) "
895
+ f"on {_plural(logit.prompts, 'prompt')}"
896
+ )
897
+ if tasks is not None:
898
+ reasons.append(
899
+ _tasks_within(tasks, ref_name) if tasks.close else _no_task_loss(tasks, ref_name)
900
+ )
901
+ note = _evidence_note(candidate, ref_name)
902
+ if note:
903
+ reasons.append(f"rests on {note}")
904
+ return reasons
905
+
906
+
907
+ def _usable(candidate: _Candidate, interval: Interval, ref_name: str) -> _Call:
908
+ loss = f"moderate loss: KLD {_kld(interval.estimate)} ({_interval_kld(interval)})"
909
+ reasons = [loss]
910
+ tasks = candidate.tasks
911
+ if tasks is not None:
912
+ reasons.append(
913
+ _tasks_within(tasks, ref_name) if tasks.close else _no_task_loss(tasks, ref_name)
914
+ )
915
+ return _Call(candidate, "usable", reasons, f"{candidate.name} has a {loss}.")
916
+
917
+
918
+ def _evidence_note(candidate: _Candidate, ref_name: str) -> str | None:
919
+ """Which evidence a close call rests on, when it is only one kind."""
920
+ if candidate.tasks is None:
921
+ if candidate.deltas:
922
+ return (
923
+ f"logit evidence only; {ref_name} passes under half of every task suite, "
924
+ "so the suites cannot judge it"
925
+ )
926
+ return "logit evidence only; no task suites ran"
927
+ if candidate.logit is None:
928
+ return "task evidence only; no logit metrics"
929
+ return None
930
+
931
+
932
+ def _inconclusive(candidate: _Candidate, ref_name: str) -> _Call:
933
+ logit, tasks = candidate.logit, candidate.tasks
934
+ reasons: list[str] = []
935
+ gaps: list[str] = []
936
+ needs: list[int | None] = []
937
+ prompts_needed = cases_needed = None
938
+ if logit is not None:
939
+ if logit.close and logit.interval is not None:
940
+ reasons.append(
941
+ f"KLD {_kld(logit.mean)} is under the {logit.thresholds.bar} closeness bar "
942
+ f"({_interval_kld(logit.interval)})"
943
+ )
944
+ else:
945
+ gaps.append(_logit_gap(logit))
946
+ prompts_needed = logit.prompts_needed
947
+ needs.append(prompts_needed)
948
+ if tasks is not None:
949
+ if tasks.close:
950
+ reasons.append(_tasks_within(tasks, ref_name))
951
+ else:
952
+ gaps.append(_task_gap(tasks, ref_name))
953
+ cases_needed = tasks.cases_needed
954
+ needs.append(cases_needed)
955
+ if logit is None and tasks is None:
956
+ gaps.append("no logit metrics and no reliable task results to judge it by")
957
+ needs.append(None)
958
+ call = _Call(candidate, "inconclusive", gaps + reasons)
959
+ if all(need is not None for need in needs):
960
+ call.prompts_needed = None if prompts_needed is None else _round_up(prompts_needed)
961
+ call.cases_needed = None if cases_needed is None else _round_up(cases_needed)
962
+ call.reasons.append(f"{_wanted(call)} would likely decide")
963
+ above = logit is not None and logit.mean >= logit.thresholds.close
964
+ state = "is undecided" if above else "is not shown to be close"
965
+ call.detail = f"{candidate.name} {state}: {gaps[0]}."
966
+ return call
967
+
968
+
969
+ def _logit_gap(logit: _LogitEvidence) -> str:
970
+ kld = _kld(logit.mean)
971
+ interval = logit.interval
972
+ if interval is None:
973
+ return f"KLD {kld}, but the report has no per-prompt KLD to bound it"
974
+ if logit.prompts < MIN_PROMPTS:
975
+ return _thin_logit_gap(logit)
976
+ thresholds = logit.thresholds
977
+ bar = thresholds.bar
978
+ if logit.mean >= thresholds.large:
979
+ return (
980
+ f"KLD {kld} may be a large loss, but its 95% CI starts at {_kld(interval.low)}, "
981
+ f"under the {thresholds.large_bar} large-loss bar"
982
+ )
983
+ if logit.mean >= thresholds.close:
984
+ return (
985
+ f"KLD {kld} is above the {bar} closeness bar, but its 95% CI reaches down to "
986
+ f"{_kld(interval.low)}"
987
+ )
988
+ return f"KLD {kld}, but its 95% CI reaches {_kld(interval.high)}, above the {bar} closeness bar"
989
+
990
+
991
+ def _thin_logit_gap(logit: _LogitEvidence) -> str:
992
+ """Why fewer than MIN_PROMPTS prompts decide nothing about this KLD."""
993
+ kld = _kld(logit.mean)
994
+ bar = logit.thresholds.bar
995
+ if logit.mean >= logit.thresholds.large:
996
+ return f"KLD {kld} looks large on {_plural(logit.prompts, 'prompt')}, too few to call it"
997
+ if logit.mean >= logit.thresholds.close:
998
+ return f"KLD {kld} is above the {bar} closeness bar"
999
+ return f"KLD {kld}, but {_plural(logit.prompts, 'prompt')} cannot bound it below {bar}"
1000
+
1001
+
1002
+ def _task_gap(tasks: _TaskEvidence, ref_name: str) -> str:
1003
+ margin = _points(TASK_MARGIN)
1004
+ cases = _plural(tasks.cases, "case")
1005
+ if tasks.delta.estimate <= -TASK_MARGIN:
1006
+ return f"task scores are {_points(-tasks.delta.estimate)} points lower than {ref_name}"
1007
+ if tasks.cases < MIN_CASES:
1008
+ return f"{cases} cannot show task scores within {margin} points of {ref_name}"
1009
+ return (
1010
+ f"task scores could be up to {_points(-tasks.delta.low)} points lower than {ref_name} "
1011
+ f"({_interval_points(tasks.delta)} on {cases})"
1012
+ )
1013
+
1014
+
1015
+ def _no_task_loss(tasks: _TaskEvidence, ref_name: str) -> str:
1016
+ return (
1017
+ f"no significant task loss vs {ref_name} on {_plural(tasks.cases, 'case')} "
1018
+ f"({_interval_points(tasks.delta)})"
1019
+ )
1020
+
1021
+
1022
+ def _tasks_within(tasks: _TaskEvidence, ref_name: str) -> str:
1023
+ return (
1024
+ f"task scores within {_points(TASK_MARGIN)} points of {ref_name} "
1025
+ f"on {_plural(tasks.cases, 'case')}"
1026
+ )
1027
+
1028
+
1029
+ def _wanted(call: _Call) -> str:
1030
+ """The evidence an inconclusive candidate needs, e.g. "about 20 scoring prompts"."""
1031
+ parts = []
1032
+ if call.prompts_needed is not None:
1033
+ parts.append(f"{call.prompts_needed} scoring prompts")
1034
+ if call.cases_needed is not None:
1035
+ parts.append(f"{call.cases_needed} cases per suite")
1036
+ return "about " + " and ".join(parts)
1037
+
1038
+
1039
+ def _pick(calls: Sequence[_Call], ref_name: str, budget: int | None) -> None:
1040
+ """Recommend the smallest close candidate that fits the budget; the other close ones
1041
+ stay ok, and those over the budget say so."""
1042
+ close = [call for call in calls if call.status == "ok"]
1043
+ if budget is not None:
1044
+ for call in close:
1045
+ if not call.fits:
1046
+ call.reasons.insert(0, _over_budget(call, ref_name, budget))
1047
+ call.detail = f"{call.name} is close but {_needs(call)}."
1048
+ eligible = [call for call in close if budget is None or call.fits]
1049
+ if not eligible:
1050
+ return
1051
+ if all(call.candidate.size is not None for call in eligible):
1052
+ pick = min(eligible, key=lambda call: (call.candidate.size, call.candidate.rank_key()))
1053
+ else:
1054
+ pick = min(eligible, key=lambda call: call.candidate.rank_key())
1055
+ pick.status = "recommended"
1056
+ if len(eligible) > 1:
1057
+ pick.reasons.append(f"smallest download that is close to {ref_name}")
1058
+ for call in eligible:
1059
+ if call is pick:
1060
+ continue
1061
+ reason = _also_close(call.candidate, pick.candidate, ref_name)
1062
+ call.reasons.insert(0, reason)
1063
+ call.detail = f"{call.name} is {reason}."
1064
+
1065
+
1066
+ def _also_close(candidate: _Candidate, pick: _Candidate, ref_name: str) -> str:
1067
+ if candidate.size is not None and pick.size:
1068
+ larger = candidate.size / pick.size - 1.0
1069
+ if larger > _SAME_SIZE:
1070
+ return f"also close to {ref_name} but {_percent(larger)} larger"
1071
+ return f"also close to {ref_name}"
1072
+
1073
+
1074
+ def _over_budget(call: _Call, ref_name: str, budget: int) -> str:
1075
+ size = call.candidate.size
1076
+ if size is None:
1077
+ return (
1078
+ f"close to {ref_name}, but its size is unknown, so it cannot be checked against "
1079
+ f"the {format_size(budget)} budget"
1080
+ )
1081
+ return (
1082
+ f"close to {ref_name} but needs {format_size(size)}, over the {format_size(budget)} budget"
1083
+ )
1084
+
1085
+
1086
+ def _needs(call: _Call) -> str:
1087
+ size = call.candidate.size
1088
+ return "its size is unknown" if size is None else f"needs {format_size(size)}"
1089
+
1090
+
1091
+ # Headline and details ---------------------------------------------------------------------
1092
+
1093
+
1094
+ @dataclass(slots=True)
1095
+ class _Summary:
1096
+ headline: str
1097
+ details: tuple[str, ...] = ()
1098
+ remedy: str | None = None
1099
+ cases_needed: int | None = None
1100
+
1101
+
1102
+ def _summarize(
1103
+ ordered: Sequence[_Call], ref_name: str, settings: RunSettings, *, non_latin: bool
1104
+ ) -> _Summary:
1105
+ if not ordered:
1106
+ return _Summary("No candidates were evaluated.")
1107
+ live = [call for call in ordered if call.status != "failed"]
1108
+ failed = [call.name for call in ordered if call.status == "failed"]
1109
+ if not live:
1110
+ return _Summary("No candidate produced metrics; see the errors for each candidate.")
1111
+ lead = _lead(live, ref_name, settings.max_size_bytes)
1112
+ summary = _Summary(lead.headline)
1113
+ if live[0].status != "recommended":
1114
+ summary.remedy, summary.cases_needed = _next_step(
1115
+ live, ref_name, settings, non_latin=non_latin
1116
+ )
1117
+ details = list(lead.lines)
1118
+ caveated = [lead.about, *(call for call in live if call is not lead.about)]
1119
+ unresolved = next((call for call in caveated if call.task_caveat), None)
1120
+ if unresolved is not None:
1121
+ # The headline's own candidate first, right under the headline; another one's
1122
+ # unresolved loss after the numbers behind the headline.
1123
+ at = 0 if unresolved is lead.about else len(details)
1124
+ details.insert(at, f"{unresolved.name}: {unresolved.task_caveat}.")
1125
+ details += [
1126
+ call.detail
1127
+ for call in live
1128
+ if call.detail and not (call is lead.about and lead.replaces_detail)
1129
+ ]
1130
+ if non_latin and _text_forced_above_bar(live):
1131
+ note = "Most scoring prompts are not Latin script; text forcing may read higher there."
1132
+ details.append(note if summary.remedy == _CONFIRM_EXACT else f"{note} {_CONFIRM_EXACT}")
1133
+ details += [
1134
+ f"{call.name} scored higher than the reference on {d.kind}; {_QUANTIZED_REFERENCE}."
1135
+ for call in live
1136
+ for d in call.candidate.gains
1137
+ ]
1138
+ if failed:
1139
+ details.append(f"{_join(failed)} produced no metrics; see the errors.")
1140
+ summary.details = tuple(details)
1141
+ return summary
1142
+
1143
+
1144
+ @dataclass(frozen=True, slots=True)
1145
+ class _Lead:
1146
+ headline: str
1147
+ about: _Call
1148
+ """The candidate the headline is about."""
1149
+ lines: tuple[str, ...] = ()
1150
+ """The numbers behind the headline, first in the details."""
1151
+ replaces_detail: bool = False
1152
+ """True when the headline or the lines already say what the candidate's detail says."""
1153
+
1154
+
1155
+ def _lead(live: Sequence[_Call], ref_name: str, budget: int | None) -> _Lead:
1156
+ """The headline for candidates in rank order, at least one of them not failed."""
1157
+ top = live[0]
1158
+ if top.status == "recommended":
1159
+ return _Lead(
1160
+ _run_headline(top.candidate, ref_name),
1161
+ top,
1162
+ (_run_detail(top.candidate, ref_name),),
1163
+ replaces_detail=True,
1164
+ )
1165
+ close = [call for call in live if call.status == "ok"]
1166
+ usable = [
1167
+ (call, call.candidate.logit)
1168
+ for call in live
1169
+ if call.status == "usable" and call.candidate.logit is not None
1170
+ ]
1171
+ fitting = [(call, logit) for call, logit in usable if budget is None or call.fits]
1172
+ if fitting:
1173
+ best, logit = fitting[0]
1174
+ if budget is None:
1175
+ unsettled = any(call.status == "inconclusive" for call in live)
1176
+ headline = _smallest_loss_headline(best, logit, ref_name, unsettled=unsettled)
1177
+ else:
1178
+ headline = (
1179
+ f"Best that fits {format_size(budget)}: {best.name}, moderate loss "
1180
+ f"(KLD {_kld(logit.mean)})."
1181
+ )
1182
+ return _Lead(headline, best)
1183
+ if budget is not None and (close or usable):
1184
+ nearest = close[0] if close else usable[0][0]
1185
+ state = "is close" if close else "has a moderate loss"
1186
+ headline = (
1187
+ f"Nothing that fits {format_size(budget)} is close to {ref_name}; "
1188
+ f"{nearest.name} {state} but {_needs(nearest)}."
1189
+ )
1190
+ return _Lead(headline, nearest, replaces_detail=bool(close))
1191
+ if top.status == "inconclusive":
1192
+ closest = _closest_detail(top.candidate, ref_name)
1193
+ return _Lead(
1194
+ _keep_for_now_headline(top.candidate, ref_name),
1195
+ top,
1196
+ () if closest is None else (closest,),
1197
+ replaces_detail=True,
1198
+ )
1199
+ who = f"{top.name} shows" if len(live) == 1 else "every candidate shows"
1200
+ return _Lead(f"Keep {ref_name}: {who} a measured loss.", top)
1201
+
1202
+
1203
+ def _text_forced_above_bar(live: Sequence[_Call]) -> bool:
1204
+ """Whether a text-forced candidate reads above the closeness bar."""
1205
+ return any(
1206
+ call.candidate.logit is not None
1207
+ and not call.candidate.logit.exact
1208
+ and call.candidate.logit.mean > call.candidate.logit.thresholds.close
1209
+ for call in live
1210
+ )
1211
+
1212
+
1213
+ def _next_step(
1214
+ live: Sequence[_Call], ref_name: str, settings: RunSettings, *, non_latin: bool
1215
+ ) -> tuple[str, int | None]:
1216
+ """What to do when nothing is recommended, and the --max-cases value of a rerun."""
1217
+ for call in live:
1218
+ if call.status == "inconclusive":
1219
+ rerun = _remedy(call, settings)
1220
+ if rerun is not None:
1221
+ return rerun
1222
+ return _advice(live, ref_name, settings, non_latin=non_latin), None
1223
+
1224
+
1225
+ def _advice(live: Sequence[_Call], ref_name: str, settings: RunSettings, *, non_latin: bool) -> str:
1226
+ """The next step when no rerun of the same kind is likely to decide."""
1227
+ text_forced = _text_forced_above_bar(live)
1228
+ budget = settings.max_size_bytes
1229
+ over = sorted(
1230
+ (call.candidate.size, call.name)
1231
+ for call in live
1232
+ if call.status == "ok" and call.candidate.size is not None
1233
+ )
1234
+ if non_latin and text_forced:
1235
+ return _CONFIRM_EXACT
1236
+ if budget is None and any(call.status == "usable" for call in live):
1237
+ return (
1238
+ "Pass --max-size with the memory you can spare, for example --max-size 6GB, to "
1239
+ "pick the best download that fits."
1240
+ )
1241
+ if budget is not None and over:
1242
+ size, name = over[0]
1243
+ flag = format_size(size, round_up=True).replace(" ", "")
1244
+ return f"Rerun with --max-size {flag} to run {name}, which is close."
1245
+ if text_forced:
1246
+ return _CONFIRM_EXACT
1247
+ return _missing_evidence(live, ref_name)
1248
+
1249
+
1250
+ def _missing_evidence(live: Sequence[_Call], ref_name: str) -> str:
1251
+ undecided = [call.candidate for call in live if call.status == "inconclusive"]
1252
+ if any(c.logit is None and c.tasks is None for c in undecided):
1253
+ return (
1254
+ "Rerun on a server that returns logprobs, or with json or tools cases the "
1255
+ "reference passes, so there is evidence to judge by."
1256
+ )
1257
+ if any(c.logit is not None and c.logit.interval is None for c in undecided):
1258
+ return (
1259
+ "Rerun with this version of quantdiff, which records the per-prompt KLD needed "
1260
+ "to decide."
1261
+ )
1262
+ return f"Try a larger quant against {ref_name}; nothing tested here is close."
1263
+
1264
+
1265
+ def _run_headline(pick: _Candidate, ref_name: str) -> str:
1266
+ logit = pick.logit
1267
+ if logit is not None and logit.interval is not None:
1268
+ evidence = (
1269
+ f"close on logits (KLD {_kld(logit.mean)}, CI up to {_kld(logit.interval.high)}) "
1270
+ f"on {_plural(logit.prompts, 'prompt')}"
1271
+ )
1272
+ else:
1273
+ cases = 0 if pick.tasks is None else pick.tasks.cases
1274
+ evidence = f"close on task scores on {_plural(cases, 'case')} (no logit metrics)"
1275
+ change = pick.size_change
1276
+ if change is not None and change < -_SAME_SIZE:
1277
+ return f"Run {pick.name}: {_percent(-change)} smaller than {ref_name}, {evidence}."
1278
+ if change is not None and change > _SAME_SIZE:
1279
+ return f"Run {pick.name}: {_percent(change)} larger than {ref_name}, {evidence}."
1280
+ return f"Run {pick.name}: {evidence}."
1281
+
1282
+
1283
+ def _smallest_loss_headline(
1284
+ best: _Call, logit: _LogitEvidence, ref_name: str, *, unsettled: bool
1285
+ ) -> str:
1286
+ """The headline when nothing is close but a candidate has a moderate loss."""
1287
+ close = "shown to be close" if unsettled else "close"
1288
+ change = best.candidate.size_change
1289
+ among = " among the smaller downloads" if change is not None and change < -_SAME_SIZE else ""
1290
+ return (
1291
+ f"No download is {close} to {ref_name}; {best.name} has the smallest loss{among} "
1292
+ f"(moderate, KLD {_kld(logit.mean)})."
1293
+ )
1294
+
1295
+
1296
+ def _run_detail(pick: _Candidate, ref_name: str) -> str:
1297
+ """The numbers behind a Run headline, as one sentence."""
1298
+ logit, tasks = pick.logit, pick.tasks
1299
+ parts = []
1300
+ if logit is not None and logit.interval is not None:
1301
+ parts.append(f"KLD {_kld(logit.mean)} ({_interval_kld(logit.interval)})")
1302
+ if tasks is not None:
1303
+ parts.append(
1304
+ _tasks_within(tasks, ref_name) if tasks.close else _no_task_loss(tasks, ref_name)
1305
+ )
1306
+ sentence = f"{pick.name}: {' and '.join(parts)}"
1307
+ note = _evidence_note(pick, ref_name)
1308
+ return f"{sentence}; {note}." if note else f"{sentence}."
1309
+
1310
+
1311
+ def _keep_for_now_headline(closest: _Candidate, ref_name: str) -> str:
1312
+ logit, tasks = closest.logit, closest.tasks
1313
+ if logit is None and tasks is None:
1314
+ return f"Keep {ref_name} for now: no candidate has logit or reliable task results."
1315
+ scope = []
1316
+ if logit is not None and logit.prompts:
1317
+ scope.append(_plural(logit.prompts, "prompt"))
1318
+ if tasks is not None:
1319
+ scope.append(_plural(tasks.cases, "case"))
1320
+ on = f" on {' and '.join(scope)}" if scope else ""
1321
+ return f"Keep {ref_name} for now: no candidate is shown to be close{on}."
1322
+
1323
+
1324
+ def _closest_detail(closest: _Candidate, ref_name: str) -> str | None:
1325
+ """Why the candidate that looks closest is still not proven close."""
1326
+ logit, tasks = closest.logit, closest.tasks
1327
+ looks = []
1328
+ if logit is not None:
1329
+ if logit.interval is None:
1330
+ looks.append(f"KLD {_kld(logit.mean)}, with no per-prompt results to bound it")
1331
+ elif logit.prompts >= MIN_PROMPTS:
1332
+ looks.append(f"KLD {_kld(logit.mean)}, 95% CI up to {_kld(logit.interval.high)}")
1333
+ else:
1334
+ looks.append(f"KLD {_kld(logit.mean)} on {_plural(logit.prompts, 'prompt')}")
1335
+ if tasks is not None:
1336
+ lower = round(-tasks.delta.estimate)
1337
+ level = f"{lower} points lower" if lower > 0 else f"level with {ref_name}"
1338
+ looks.append(f"task scores {level}, 95% CI down to {_signed(tasks.delta.low)}")
1339
+ if not looks:
1340
+ return None
1341
+ return f"{closest.name} looks closest: {'; '.join(looks)}."
1342
+
1343
+
1344
+ def _remedy(call: _Call, settings: RunSettings) -> tuple[str, int] | None:
1345
+ """What to rerun with to decide an inconclusive candidate, and the --max-cases value."""
1346
+ prompts, cases = call.prompts_needed, call.cases_needed
1347
+ if prompts is None and cases is None:
1348
+ return None
1349
+ flag = max(prompts or 0, cases or 0)
1350
+ wanted = _sentence(_wanted(call)).removesuffix(".")
1351
+ if settings.prompts_file is not None:
1352
+ return f"{wanted} would decide; add them to {settings.prompts_file}.", flag
1353
+ kinds = () if cases is None or call.candidate.tasks is None else call.candidate.tasks.kinds
1354
+ room = [len(load_builtin(kind)) for kind in kinds]
1355
+ if prompts is not None:
1356
+ room.append(len(load_scoring_prompts()))
1357
+ if flag <= min(room):
1358
+ return f"Rerun with --max-cases {flag} to decide.", flag
1359
+ return (
1360
+ f"{wanted} would decide. That is more than the built-in suites hold, so add your "
1361
+ "own with --prompts.",
1362
+ flag,
1363
+ )
1364
+
1365
+
1366
+ # Server findings --------------------------------------------------------------------------
1367
+
1368
+
1369
+ def _findings(report: Report, names: dict[str, str]) -> tuple[ServerFinding, ...]:
1370
+ results = (report.reference, *report.candidates)
1371
+ grouped: dict[PreflightFinding, list[str]] = {}
1372
+ for result in results:
1373
+ failed = {f.check for f in result.preflight if f.severity == "fail"}
1374
+ for finding in result.preflight:
1375
+ if finding.severity not in ("warn", "fail"):
1376
+ continue
1377
+ if finding.severity == "warn" and finding.check in failed:
1378
+ continue
1379
+ labels = grouped.setdefault(finding, [])
1380
+ if result.spec.label not in labels:
1381
+ labels.append(result.spec.label)
1382
+ by_label = {result.spec.label: result for result in results}
1383
+ findings = [
1384
+ ServerFinding(finding, tuple(labels), *_impact(finding, labels, by_label, report))
1385
+ for finding, labels in grouped.items()
1386
+ ]
1387
+ findings += _same_weights(report, names)
1388
+ severity = {"fail": 0, "warn": 1}
1389
+ findings.sort(key=lambda f: (not f.affects_scores, severity.get(f.finding.severity, 2)))
1390
+ return tuple(findings)
1391
+
1392
+
1393
+ def _impact(
1394
+ finding: PreflightFinding,
1395
+ labels: Sequence[str],
1396
+ by_label: dict[str, CandidateResult],
1397
+ report: Report,
1398
+ ) -> tuple[bool, str]:
1399
+ if finding.check == "context":
1400
+ return _context_impact(labels, by_label, report.settings.longest_prompt_tokens)
1401
+ return _FIXED_IMPACT.get(
1402
+ finding.check, (True, "may affect scores: a serving problem can lower a model's results")
1403
+ )
1404
+
1405
+
1406
+ _FIXED_IMPACT: Final[dict[str, tuple[bool, str]]] = {
1407
+ "template": (True, "affects these scores: every task case runs through the chat template"),
1408
+ "tokenizer": (True, "affects these scores: logit positions do not line up with the reference"),
1409
+ "logprobs": (
1410
+ False,
1411
+ "does not affect these scores: logit metrics are missing, so only task results count",
1412
+ ),
1413
+ }
1414
+
1415
+
1416
+ def _context_impact(
1417
+ labels: Sequence[str], by_label: dict[str, CandidateResult], longest: int | None
1418
+ ) -> tuple[bool, str]:
1419
+ if longest is None:
1420
+ return True, "may affect scores: the length of the longest prompt is not known"
1421
+ infos = [by_label[label].info for label in labels]
1422
+ known = [info.context_length for info in infos if info and info.context_length]
1423
+ if len(known) < len(infos):
1424
+ return True, "may affect scores: the context window is not known"
1425
+ window = min(known)
1426
+ if window >= longest:
1427
+ return False, (
1428
+ f"does not affect these scores: the longest prompt is ~{longest} tokens and "
1429
+ f"the context window is {window}"
1430
+ )
1431
+ return True, (
1432
+ f"may affect scores: the longest prompt is ~{longest} tokens but the context "
1433
+ f"window is {window}"
1434
+ )
1435
+
1436
+
1437
+ def _same_weights(report: Report, names: dict[str, str]) -> list[ServerFinding]:
1438
+ info = report.reference.info
1439
+ reference_id = None if info is None else info.weights_id
1440
+ if reference_id is None:
1441
+ return []
1442
+ return [
1443
+ ServerFinding(
1444
+ finding=PreflightFinding(
1445
+ check="same-weights",
1446
+ severity="warn",
1447
+ message=f"{names[result.spec.label]} serves the same weights as the reference",
1448
+ fix="Point the reference at a different download, such as the BF16 or Q8_0 file.",
1449
+ ),
1450
+ labels=(result.spec.label,),
1451
+ affects_scores=True,
1452
+ impact="affects these scores: the candidate is being compared with itself",
1453
+ )
1454
+ for result in report.candidates
1455
+ if result.info is not None and result.info.weights_id == reference_id
1456
+ ]
1457
+
1458
+
1459
+ # Formatting -------------------------------------------------------------------------------
1460
+
1461
+
1462
+ def _sentence(phrase: str) -> str:
1463
+ # Suite names are identifiers ("json", "tools") and stay lowercase at a sentence start.
1464
+ starts_with_suite = phrase.split(" ", 1)[0] in SCORED_TASK_KINDS
1465
+ text = phrase if starts_with_suite else phrase[:1].upper() + phrase[1:]
1466
+ return text if text.endswith(".") else text + "."
1467
+
1468
+
1469
+ def _percent(fraction: float) -> str:
1470
+ return f"{fraction * 100:.0f}%"
1471
+
1472
+
1473
+ def _kld(value: float) -> str:
1474
+ """Up to three decimals from 0.01 up, which keeps values near the 0.05 bar readable;
1475
+ two significant digits below that."""
1476
+ if value >= 10:
1477
+ return f"{value:.0f}"
1478
+ if value >= 0.01:
1479
+ return f"{value:.3f}".rstrip("0").rstrip(".")
1480
+ if 0 < value < 0.0001:
1481
+ return "under 0.0001"
1482
+ return f"{value:.2g}"
1483
+
1484
+
1485
+ def _interval_kld(interval: Interval) -> str:
1486
+ return f"95% CI {_kld(interval.low)} to {_kld(interval.high)}"
1487
+
1488
+
1489
+ def _interval_points(interval: Interval) -> str:
1490
+ return f"95% CI {_signed(interval.low)} to {_signed(interval.high)}"
1491
+
1492
+
1493
+ def _signed(points: float) -> str:
1494
+ rounded = round(points)
1495
+ return f"+{rounded}" if rounded > 0 else str(rounded)
1496
+
1497
+
1498
+ def _points(points: float) -> str:
1499
+ return f"{points:.0f}"
1500
+
1501
+
1502
+ def _round_up(count: int) -> int:
1503
+ return math.ceil(count / _ROUND_NEEDED_TO) * _ROUND_NEEDED_TO
1504
+
1505
+
1506
+ def _plural(count: int, noun: str) -> str:
1507
+ return f"{count} {noun}" if count == 1 else f"{count} {noun}s"
1508
+
1509
+
1510
+ def _join(items: Sequence[str]) -> str:
1511
+ if len(items) <= 2:
1512
+ return " and ".join(items)
1513
+ return ", ".join(items[:-1]) + " and " + items[-1]