quantdiff 0.1.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantdiff/__init__.py +53 -0
- quantdiff/__main__.py +5 -0
- quantdiff/_http.py +151 -0
- quantdiff/_text.py +13 -0
- quantdiff/_version.py +1 -0
- quantdiff/api.py +340 -0
- quantdiff/backends/__init__.py +28 -0
- quantdiff/backends/_common.py +342 -0
- quantdiff/backends/base.py +91 -0
- quantdiff/backends/llamacpp.py +428 -0
- quantdiff/backends/ollama.py +359 -0
- quantdiff/backends/openai_compat.py +338 -0
- quantdiff/cache.py +240 -0
- quantdiff/card.py +1664 -0
- quantdiff/cli.py +377 -0
- quantdiff/discover.py +488 -0
- quantdiff/errors.py +45 -0
- quantdiff/metrics/__init__.py +36 -0
- quantdiff/metrics/codeexec.py +428 -0
- quantdiff/metrics/jsonschema.py +610 -0
- quantdiff/metrics/logit.py +214 -0
- quantdiff/metrics/tasks.py +114 -0
- quantdiff/metrics/textsim.py +66 -0
- quantdiff/metrics/toolcheck.py +99 -0
- quantdiff/png.py +360 -0
- quantdiff/preflight.py +365 -0
- quantdiff/progress.py +283 -0
- quantdiff/py.typed +0 -0
- quantdiff/report.py +780 -0
- quantdiff/runner.py +492 -0
- quantdiff/spec.py +154 -0
- quantdiff/stats.py +226 -0
- quantdiff/suites/__init__.py +462 -0
- quantdiff/suites/data/chat.jsonl +22 -0
- quantdiff/suites/data/code.jsonl +32 -0
- quantdiff/suites/data/json.jsonl +34 -0
- quantdiff/suites/data/scoring.jsonl +41 -0
- quantdiff/suites/data/tools.jsonl +32 -0
- quantdiff/types.py +322 -0
- quantdiff/verdict.py +1513 -0
- quantdiff-0.1.0rc1.dist-info/METADATA +514 -0
- quantdiff-0.1.0rc1.dist-info/RECORD +45 -0
- quantdiff-0.1.0rc1.dist-info/WHEEL +4 -0
- quantdiff-0.1.0rc1.dist-info/entry_points.txt +2 -0
- quantdiff-0.1.0rc1.dist-info/licenses/LICENSE +202 -0
quantdiff/verdict.py
ADDED
|
@@ -0,0 +1,1513 @@
|
|
|
1
|
+
"""Turn a Report into a recommendation: which download to run, which to avoid, and why.
|
|
2
|
+
|
|
3
|
+
This is the contract the scorecards render. The dataclasses are shared with card.py and
|
|
4
|
+
the CLI; `judge` applies the rules below, which docs/methodology.md explains for users.
|
|
5
|
+
|
|
6
|
+
Every comparison is paired, because every model answers the same cases and is scored on
|
|
7
|
+
the same prompts: task outcomes are joined by case id and logit results by prompt id. Only
|
|
8
|
+
items present on both sides count.
|
|
9
|
+
|
|
10
|
+
Each candidate is judged on its own against the reference, never against the other
|
|
11
|
+
candidates, so adding or removing a candidate never changes another one's status. A
|
|
12
|
+
candidate is recommended only on positive evidence that it is close to the reference; thin
|
|
13
|
+
evidence leads to inconclusive, not to a pick. The KLD interval below is the 95% bootstrap
|
|
14
|
+
interval of the mean per-prompt KLD. Rules, applied in order:
|
|
15
|
+
|
|
16
|
+
1. failed: the candidate produced no metrics at all.
|
|
17
|
+
2. avoid, when any of these holds:
|
|
18
|
+
- a reliable suite (the reference passes at least half of its cases) shows a significant
|
|
19
|
+
paired regression (exact McNemar p < 0.05, candidate below the reference);
|
|
20
|
+
- the reliable suites pooled together show a significant paired regression;
|
|
21
|
+
- with at least MIN_PROMPTS paired prompts, the mean KLD is at least LARGE_KLD and so is
|
|
22
|
+
the lower end of its interval.
|
|
23
|
+
3. close (eligible to run), only on positive evidence:
|
|
24
|
+
- with logit metrics, closeness rests on them: at least MIN_PROMPTS paired prompts and
|
|
25
|
+
the upper end of the KLD interval below CLOSE_KLD. Task suites of a few dozen cases
|
|
26
|
+
cannot bound small differences, so with logit evidence they act as a breakage
|
|
27
|
+
detector: any significant loss (rule 2) is avoid, and a wide but not significant
|
|
28
|
+
interval is shown, not used against the candidate. A KLD that is shown small also
|
|
29
|
+
bounds how far the two output distributions can differ.
|
|
30
|
+
- without logit metrics, closeness rests on tasks: the reliable suites pooled, at least
|
|
31
|
+
MIN_CASES cases, and the lower end of the 95% interval of the pass-rate difference at
|
|
32
|
+
or above -TASK_MARGIN points.
|
|
33
|
+
The reasons say which evidence the call rests on.
|
|
34
|
+
4. usable: with at least MIN_PROMPTS paired prompts, the lower end of the KLD interval is
|
|
35
|
+
above CLOSE_KLD and the mean is below LARGE_KLD. A measured, moderate loss: the best
|
|
36
|
+
choice when nothing close fits the size budget.
|
|
37
|
+
5. inconclusive: everything else, including a large-looking mean on fewer than MIN_PROMPTS
|
|
38
|
+
prompts and a mean at or above LARGE_KLD whose interval reaches below it. The reasons
|
|
39
|
+
name what is missing and estimate how many prompts or cases per suite would decide.
|
|
40
|
+
6. The pick. Among the close candidates that fit the --max-size budget (every close one
|
|
41
|
+
when there is no budget) the smallest download is recommended (by size on disk when
|
|
42
|
+
each of them reports one, otherwise the lowest KLD). The other close ones are ok. A
|
|
43
|
+
candidate that is not close is never recommended or ok, so when no close candidate
|
|
44
|
+
fits, the headline names the usable candidate with the lowest KLD that fits instead.
|
|
45
|
+
|
|
46
|
+
Caveats do not change a status but sit next to it: a KLD interval near a bar (a rerun
|
|
47
|
+
could move the candidate across it), and a reliable suite whose estimate is more than
|
|
48
|
+
TASK_MARGIN points below the reference without being significant.
|
|
49
|
+
|
|
50
|
+
A candidate is never described as better than the reference. A significant gain is
|
|
51
|
+
reported as a warning sign about the suite or the reference, not as a win.
|
|
52
|
+
|
|
53
|
+
Ranking: status (recommended, ok, usable, inconclusive, avoid, failed), then mean KLD
|
|
54
|
+
ascending, size ascending, and input order.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
from __future__ import annotations
|
|
58
|
+
|
|
59
|
+
import itertools
|
|
60
|
+
import math
|
|
61
|
+
import os.path
|
|
62
|
+
import unicodedata
|
|
63
|
+
from collections.abc import Iterable, Sequence
|
|
64
|
+
from dataclasses import dataclass, field
|
|
65
|
+
from typing import Final, Literal
|
|
66
|
+
|
|
67
|
+
from quantdiff.stats import (
|
|
68
|
+
Interval,
|
|
69
|
+
bootstrap_mean,
|
|
70
|
+
cases_to_bound_loss,
|
|
71
|
+
items_to_bound_below,
|
|
72
|
+
mcnemar_exact,
|
|
73
|
+
paired_proportion_diff,
|
|
74
|
+
)
|
|
75
|
+
from quantdiff.suites import load_builtin, load_scoring_prompts
|
|
76
|
+
from quantdiff.types import (
|
|
77
|
+
CandidateResult,
|
|
78
|
+
LogitMetrics,
|
|
79
|
+
PreflightFinding,
|
|
80
|
+
Report,
|
|
81
|
+
RunSettings,
|
|
82
|
+
ScoringPrompt,
|
|
83
|
+
TaskKind,
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
__all__ = [
|
|
87
|
+
"CALIBRATED_TOP_K",
|
|
88
|
+
"CLOSE_KLD",
|
|
89
|
+
"FULL_VOCAB_CLOSE",
|
|
90
|
+
"FULL_VOCAB_LARGE",
|
|
91
|
+
"FULL_VOCAB_NEAR_LOSSLESS",
|
|
92
|
+
"LABEL_SEPARATORS",
|
|
93
|
+
"LARGE_KLD",
|
|
94
|
+
"MIN_CASES",
|
|
95
|
+
"MIN_PROMPTS",
|
|
96
|
+
"MIN_SHARED_PREFIX",
|
|
97
|
+
"NEAR_BAR",
|
|
98
|
+
"NEAR_LOSSLESS_KLD",
|
|
99
|
+
"SCORED_TASK_KINDS",
|
|
100
|
+
"TASK_MARGIN",
|
|
101
|
+
"TOP_K_FRACTION",
|
|
102
|
+
"CandidateVerdict",
|
|
103
|
+
"Interval",
|
|
104
|
+
"KldBand",
|
|
105
|
+
"KldThresholds",
|
|
106
|
+
"ServerFinding",
|
|
107
|
+
"Status",
|
|
108
|
+
"TaskDelta",
|
|
109
|
+
"Verdict",
|
|
110
|
+
"describe_delta",
|
|
111
|
+
"display_labels",
|
|
112
|
+
"format_size",
|
|
113
|
+
"judge",
|
|
114
|
+
"kld_band",
|
|
115
|
+
"kld_thresholds",
|
|
116
|
+
"mostly_non_latin",
|
|
117
|
+
"top_k_fraction",
|
|
118
|
+
]
|
|
119
|
+
|
|
120
|
+
Status = Literal["recommended", "ok", "usable", "avoid", "inconclusive", "failed"]
|
|
121
|
+
"""recommended: the one to run. ok: also close to the reference, but not the pick (usually
|
|
122
|
+
larger). usable: a measured but moderate loss; a sound choice when nothing closer fits.
|
|
123
|
+
avoid: a large loss or measured task breakage. inconclusive: not enough evidence either way.
|
|
124
|
+
failed: the candidate produced no usable metrics."""
|
|
125
|
+
|
|
126
|
+
KldBand = Literal["near-lossless", "small", "moderate", "large"]
|
|
127
|
+
|
|
128
|
+
# Thresholds. Every number the rules use is here, so a calibration can change them in one
|
|
129
|
+
# place; docs/methodology.md, docs/calibration.md and README.md quote them.
|
|
130
|
+
#
|
|
131
|
+
# The KLD bars are set on llama.cpp's full-vocabulary scale and converted to quantdiff's
|
|
132
|
+
# top-k lower bound, which reads a fixed fraction of the full value. docs/calibration.md
|
|
133
|
+
# measured that fraction on identical positions: 0.24 at k=1, 0.57 at k=5, 0.67 at k=10 and
|
|
134
|
+
# 0.76 at k=20 (llama-perplexity --kl-divergence as the full-vocabulary truth).
|
|
135
|
+
FULL_VOCAB_NEAR_LOSSLESS: Final = 0.015
|
|
136
|
+
"""Full-vocabulary KLD below this is near-lossless (typical of Q8_0)."""
|
|
137
|
+
FULL_VOCAB_CLOSE: Final = 0.06
|
|
138
|
+
"""Full-vocabulary closeness bar: about Q4_K_M on a 7 to 8B model."""
|
|
139
|
+
FULL_VOCAB_LARGE: Final = 0.15
|
|
140
|
+
"""Full-vocabulary KLD at or above this is a large loss (Q2_K territory)."""
|
|
141
|
+
TOP_K_FRACTION: Final[tuple[tuple[int, float], ...]] = (
|
|
142
|
+
(1, 0.24),
|
|
143
|
+
(5, 0.57),
|
|
144
|
+
(10, 0.67),
|
|
145
|
+
(20, 0.76),
|
|
146
|
+
)
|
|
147
|
+
"""Measured top-k lower bound as a fraction of full-vocabulary KLD, by k."""
|
|
148
|
+
CALIBRATED_TOP_K: Final = 10
|
|
149
|
+
"""The default --top-k, at which the module-level KLD constants below apply."""
|
|
150
|
+
MIN_PROMPTS: Final = 8
|
|
151
|
+
"""Fewer paired prompts than this never prove a candidate close or far on logits: a
|
|
152
|
+
bootstrap over a handful of prompts has almost no distinct resamples."""
|
|
153
|
+
MIN_CASES: Final = 20
|
|
154
|
+
"""Fewer pooled paired task cases than this never prove a candidate close on tasks."""
|
|
155
|
+
TASK_MARGIN: Final = 10.0
|
|
156
|
+
"""Points of pass rate the pooled task interval may reach below the reference while the
|
|
157
|
+
candidate still counts as close."""
|
|
158
|
+
NEAR_BAR: Final = 0.1
|
|
159
|
+
"""A KLD interval that straddles a bar, or ends within this fraction of it, is near it."""
|
|
160
|
+
|
|
161
|
+
SCORED_TASK_KINDS: Final[tuple[TaskKind, ...]] = ("json", "tools", "code")
|
|
162
|
+
"""Task kinds with a pass/fail check. Chat is scored by agreement instead."""
|
|
163
|
+
LABEL_SEPARATORS: Final = ":-_/."
|
|
164
|
+
MIN_SHARED_PREFIX: Final = 8
|
|
165
|
+
|
|
166
|
+
_ALPHA: Final = 0.05
|
|
167
|
+
_RELIABLE_REFERENCE_RATE: Final = 0.5
|
|
168
|
+
_STATUS_ORDER: Final[dict[Status, int]] = {
|
|
169
|
+
"recommended": 0,
|
|
170
|
+
"ok": 1,
|
|
171
|
+
"usable": 2,
|
|
172
|
+
"inconclusive": 3,
|
|
173
|
+
"avoid": 4,
|
|
174
|
+
"failed": 5,
|
|
175
|
+
}
|
|
176
|
+
_MAX_CASES_TO_RESOLVE: Final = 5000
|
|
177
|
+
"""Larger estimates of the cases that would resolve a task loss are not worth quoting."""
|
|
178
|
+
_NON_LATIN_SHARE: Final = 0.5
|
|
179
|
+
"""Prompts whose letters are more than this share non-Latin are mostly non-Latin script."""
|
|
180
|
+
_SIZE_DIGITS: Final = 3
|
|
181
|
+
"""Significant digits of the sizes and budgets that sentences quote."""
|
|
182
|
+
_CONFIRM_EXACT: Final = "Confirm with llama-server (exact token ids) before ruling out a download."
|
|
183
|
+
_SAME_SIZE: Final = 0.005
|
|
184
|
+
"""Size changes under half a percent are reported as the same size."""
|
|
185
|
+
_ROUND_NEEDED_TO: Final = 10
|
|
186
|
+
"""Evidence estimates are rough, so they are rounded up to a multiple of this."""
|
|
187
|
+
_QUANTIZED_REFERENCE: Final = (
|
|
188
|
+
"usually a sign the suite is too small or the reference is quantized itself"
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
@dataclass(frozen=True, slots=True)
|
|
193
|
+
class TaskDelta:
|
|
194
|
+
"""Candidate minus reference pass rate for one suite, paired case by case, in points."""
|
|
195
|
+
|
|
196
|
+
kind: TaskKind
|
|
197
|
+
cases: int
|
|
198
|
+
candidate_rate: float
|
|
199
|
+
reference_rate: float
|
|
200
|
+
delta: Interval
|
|
201
|
+
significant: bool
|
|
202
|
+
"""True when the paired test (exact McNemar) rejects equality at the 5% level."""
|
|
203
|
+
reference_reliable: bool
|
|
204
|
+
"""False when the reference itself passes under half the cases, so the suite says little
|
|
205
|
+
about this model; such suites are shown but never used to judge a candidate."""
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
@dataclass(frozen=True, slots=True)
|
|
209
|
+
class CandidateVerdict:
|
|
210
|
+
label: str
|
|
211
|
+
status: Status
|
|
212
|
+
rank: int | None
|
|
213
|
+
"""1 is best; None for failed candidates."""
|
|
214
|
+
kld_band: KldBand | None
|
|
215
|
+
size_bytes: int | None
|
|
216
|
+
size_change: float | None
|
|
217
|
+
"""Fractional size change against the reference, e.g. -0.25 for 25% smaller."""
|
|
218
|
+
task_deltas: tuple[TaskDelta, ...]
|
|
219
|
+
reasons: tuple[str, ...]
|
|
220
|
+
"""Short plain-English sentences that justify the status, most important first."""
|
|
221
|
+
caveats: tuple[str, ...] = ()
|
|
222
|
+
"""Unresolved concerns that do not change the status but a reader must see next to it,
|
|
223
|
+
e.g. "tools -17 unresolved (95% CI -42 to +6); rerun with --max-cases 60"."""
|
|
224
|
+
near_bar: bool = False
|
|
225
|
+
"""True when the KLD interval straddles a band edge closely enough that a rerun could
|
|
226
|
+
change the status."""
|
|
227
|
+
fits_budget: bool | None = None
|
|
228
|
+
"""Whether the download fits the --max-size budget; None when no budget was given or
|
|
229
|
+
the size is unknown."""
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
@dataclass(frozen=True, slots=True)
|
|
233
|
+
class ServerFinding:
|
|
234
|
+
"""A pre-flight finding, with whether it could have changed the scores on this card."""
|
|
235
|
+
|
|
236
|
+
finding: PreflightFinding
|
|
237
|
+
labels: tuple[str, ...]
|
|
238
|
+
"""Models it applies to; every model when it is a server-wide issue."""
|
|
239
|
+
affects_scores: bool
|
|
240
|
+
impact: str
|
|
241
|
+
"""One sentence, e.g. "does not affect these scores: the longest prompt is ~900 tokens"."""
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
@dataclass(frozen=True, slots=True)
|
|
245
|
+
class Verdict:
|
|
246
|
+
headline: str
|
|
247
|
+
"""One short sentence that answers "which should I run?", e.g. "Run q4_K_M: 25% smaller
|
|
248
|
+
than q8_0, close on logits (KLD 0.03, CI up to 0.036) on 41 prompts.", "Best that fits
|
|
249
|
+
6 GB: q4_K_M, moderate loss (KLD 0.044)." or an honest "Keep q8_0 for now: no candidate
|
|
250
|
+
is shown to be close on 12 prompts and 24 cases."."""
|
|
251
|
+
details: tuple[str, ...]
|
|
252
|
+
"""Supporting sentences: the numbers behind the headline, what to avoid and why. An
|
|
253
|
+
unresolved task loss, when there is one, comes first."""
|
|
254
|
+
candidates: tuple[CandidateVerdict, ...]
|
|
255
|
+
"""Every candidate, in rank order, failed ones last."""
|
|
256
|
+
findings: tuple[ServerFinding, ...]
|
|
257
|
+
cases_needed: int | None = None
|
|
258
|
+
"""When nothing is recommended and a rerun would likely decide: the value of
|
|
259
|
+
--max-cases (cases per suite and scoring prompts, rounded up to a multiple of 10)."""
|
|
260
|
+
remedy: str | None = None
|
|
261
|
+
"""When nothing is recommended: one sentence with the next step, e.g. "Rerun with
|
|
262
|
+
--max-cases 40 to decide." Kept out of the headline and the details."""
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
# Public helpers ---------------------------------------------------------------------------
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
@dataclass(frozen=True, slots=True)
|
|
269
|
+
class KldThresholds:
|
|
270
|
+
"""The KLD bars on quantdiff's top-k scale for one --top-k setting."""
|
|
271
|
+
|
|
272
|
+
near_lossless: float
|
|
273
|
+
close: float
|
|
274
|
+
"""A candidate is close on logits when its KLD interval ends below this, and usable at
|
|
275
|
+
best when the interval starts above it. Also the top of the small band."""
|
|
276
|
+
large: float
|
|
277
|
+
"""Mean KLD at or above this is the large band; avoided when the interval starts at or
|
|
278
|
+
above it too."""
|
|
279
|
+
|
|
280
|
+
@property
|
|
281
|
+
def bar(self) -> str:
|
|
282
|
+
"""The closeness bar as sentences quote it."""
|
|
283
|
+
return f"{self.close:.2g}"
|
|
284
|
+
|
|
285
|
+
@property
|
|
286
|
+
def large_bar(self) -> str:
|
|
287
|
+
"""The large-loss bar as sentences quote it."""
|
|
288
|
+
return f"{self.large:.2f}"
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def top_k_fraction(top_k: int) -> float:
|
|
292
|
+
"""The measured fraction of full-vocabulary KLD that a top-k lower bound reads,
|
|
293
|
+
interpolated linearly between calibrated k values and held flat outside them."""
|
|
294
|
+
points = TOP_K_FRACTION
|
|
295
|
+
if top_k <= points[0][0]:
|
|
296
|
+
return points[0][1]
|
|
297
|
+
for (k_low, f_low), (k_high, f_high) in itertools.pairwise(points):
|
|
298
|
+
if top_k <= k_high:
|
|
299
|
+
return f_low + (f_high - f_low) * (top_k - k_low) / (k_high - k_low)
|
|
300
|
+
return points[-1][1]
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def kld_thresholds(top_k: int = CALIBRATED_TOP_K) -> KldThresholds:
|
|
304
|
+
"""The KLD bars for a run with this --top-k, rounded to the precision cards print."""
|
|
305
|
+
fraction = top_k_fraction(top_k)
|
|
306
|
+
return KldThresholds(
|
|
307
|
+
near_lossless=round(FULL_VOCAB_NEAR_LOSSLESS * fraction, 3),
|
|
308
|
+
close=round(FULL_VOCAB_CLOSE * fraction, 3),
|
|
309
|
+
large=round(FULL_VOCAB_LARGE * fraction, 2),
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
_DEFAULT_THRESHOLDS: Final = kld_thresholds()
|
|
314
|
+
NEAR_LOSSLESS_KLD: Final = _DEFAULT_THRESHOLDS.near_lossless
|
|
315
|
+
"""Near-lossless bar at the default --top-k (0.01)."""
|
|
316
|
+
CLOSE_KLD: Final = _DEFAULT_THRESHOLDS.close
|
|
317
|
+
"""Closeness bar at the default --top-k (0.04)."""
|
|
318
|
+
LARGE_KLD: Final = _DEFAULT_THRESHOLDS.large
|
|
319
|
+
"""Large-loss bar at the default --top-k (0.10)."""
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def kld_band(kld_mean: float, thresholds: KldThresholds = _DEFAULT_THRESHOLDS) -> KldBand:
|
|
323
|
+
"""quantdiff's band for a mean KLD; see docs/calibration.md for the anchors."""
|
|
324
|
+
if kld_mean < thresholds.near_lossless:
|
|
325
|
+
return "near-lossless"
|
|
326
|
+
if kld_mean < thresholds.close:
|
|
327
|
+
return "small"
|
|
328
|
+
if kld_mean < thresholds.large:
|
|
329
|
+
return "moderate"
|
|
330
|
+
return "large"
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def display_labels(report: Report) -> dict[str, str]:
|
|
334
|
+
"""Short names for sentences: each full label mapped to the part that tells models apart.
|
|
335
|
+
|
|
336
|
+
Quant labels of one model usually differ only after a long common stem, such as
|
|
337
|
+
qwen2.5:7b-instruct-q4_K_M and qwen2.5:7b-instruct-q8_0. When every label (reference
|
|
338
|
+
included) shares a prefix that ends at one of LABEL_SEPARATORS, is at least
|
|
339
|
+
MIN_SHARED_PREFIX characters, and leaves something on every label, the prefix is
|
|
340
|
+
dropped. Otherwise labels are used as they are.
|
|
341
|
+
"""
|
|
342
|
+
labels = [result.spec.label for result in (report.reference, *report.candidates)]
|
|
343
|
+
prefix = ""
|
|
344
|
+
if len(labels) >= 2:
|
|
345
|
+
common = os.path.commonprefix(labels)
|
|
346
|
+
end = max((i + 1 for i, char in enumerate(common) if char in LABEL_SEPARATORS), default=0)
|
|
347
|
+
prefix = common[:end]
|
|
348
|
+
if len(prefix) < MIN_SHARED_PREFIX or any(len(label) == len(prefix) for label in labels):
|
|
349
|
+
prefix = ""
|
|
350
|
+
return {label: label[len(prefix) :] for label in labels}
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def describe_delta(delta: TaskDelta) -> str:
|
|
354
|
+
"""The delta as a reader should take it, e.g. "-30 points (95% CI -48 to -12)".
|
|
355
|
+
|
|
356
|
+
A gain that is not significant reads "= reference (within noise)", so a lucky case or
|
|
357
|
+
two is never shown as a candidate beating the reference.
|
|
358
|
+
"""
|
|
359
|
+
estimate = round(delta.delta.estimate)
|
|
360
|
+
if not delta.significant:
|
|
361
|
+
if estimate >= 0:
|
|
362
|
+
return "= reference (within noise)"
|
|
363
|
+
return f"{estimate} points (within noise)"
|
|
364
|
+
if estimate > 0:
|
|
365
|
+
return f"+{estimate} points vs reference; {_QUANTIZED_REFERENCE}"
|
|
366
|
+
return f"{estimate} points ({_interval_points(delta.delta)})"
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def format_size(size_bytes: int, *, round_up: bool = False) -> str:
|
|
370
|
+
"""A download size or budget to three significant digits in decimal units, e.g.
|
|
371
|
+
"7.1 GB", "6.25 GB" or "398 MB". With `round_up`, never less than `size_bytes`, so the
|
|
372
|
+
text can be passed back as a --max-size that the download fits."""
|
|
373
|
+
for unit, scale in (("TB", 1e12), ("GB", 1e9), ("MB", 1e6), ("KB", 1e3)):
|
|
374
|
+
if size_bytes >= scale:
|
|
375
|
+
value = size_bytes / scale
|
|
376
|
+
decimals = max(0, _SIZE_DIGITS - len(str(int(value))))
|
|
377
|
+
step = 10**decimals
|
|
378
|
+
if round_up:
|
|
379
|
+
value = math.ceil(value * step) / step
|
|
380
|
+
text = f"{value:.{decimals}f}"
|
|
381
|
+
return (text.rstrip("0").rstrip(".") if decimals else text) + f" {unit}"
|
|
382
|
+
return f"{size_bytes} bytes"
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def mostly_non_latin(texts: Iterable[str]) -> bool:
|
|
386
|
+
"""True when more than half the letters in `texts` are outside the Latin script.
|
|
387
|
+
|
|
388
|
+
Only letters count, so digits, punctuation, code symbols and whitespace never tip the
|
|
389
|
+
balance. Text forcing re-tokenizes the reference's text, which is least faithful for
|
|
390
|
+
scripts that byte-level tokenizers split into many pieces.
|
|
391
|
+
"""
|
|
392
|
+
letters = non_latin = 0
|
|
393
|
+
for text in texts:
|
|
394
|
+
for char in text:
|
|
395
|
+
if not char.isalpha():
|
|
396
|
+
continue
|
|
397
|
+
letters += 1
|
|
398
|
+
if not (char.isascii() or unicodedata.name(char, "").startswith("LATIN ")):
|
|
399
|
+
non_latin += 1
|
|
400
|
+
return letters > 0 and non_latin > letters * _NON_LATIN_SHARE
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
# Evidence ---------------------------------------------------------------------------------
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
@dataclass(frozen=True, slots=True)
|
|
407
|
+
class _LogitEvidence:
|
|
408
|
+
"""Mean per-prompt KLD against the reference, with its bootstrap interval over prompts."""
|
|
409
|
+
|
|
410
|
+
prompts: int
|
|
411
|
+
"""Prompts with a per-prompt KLD, the unit the interval resamples."""
|
|
412
|
+
mean: float
|
|
413
|
+
interval: Interval | None
|
|
414
|
+
"""None when the report has no per-prompt results (reports before schema version 2)."""
|
|
415
|
+
thresholds: KldThresholds
|
|
416
|
+
exact: bool
|
|
417
|
+
"""True when the backend scored by token id; False for text forcing."""
|
|
418
|
+
|
|
419
|
+
def _bounded(self) -> Interval | None:
|
|
420
|
+
"""The interval, when there are enough prompts for it to decide anything."""
|
|
421
|
+
return self.interval if self.prompts >= MIN_PROMPTS else None
|
|
422
|
+
|
|
423
|
+
@property
|
|
424
|
+
def close(self) -> bool:
|
|
425
|
+
interval = self._bounded()
|
|
426
|
+
return interval is not None and interval.high < self.thresholds.close
|
|
427
|
+
|
|
428
|
+
@property
|
|
429
|
+
def moderate(self) -> bool:
|
|
430
|
+
interval = self._bounded()
|
|
431
|
+
return (
|
|
432
|
+
interval is not None
|
|
433
|
+
and interval.low > self.thresholds.close
|
|
434
|
+
and self.mean < self.thresholds.large
|
|
435
|
+
)
|
|
436
|
+
|
|
437
|
+
@property
|
|
438
|
+
def large(self) -> bool:
|
|
439
|
+
interval = self._bounded()
|
|
440
|
+
return (
|
|
441
|
+
interval is not None
|
|
442
|
+
and self.mean >= self.thresholds.large
|
|
443
|
+
and interval.low >= self.thresholds.large
|
|
444
|
+
)
|
|
445
|
+
|
|
446
|
+
@property
|
|
447
|
+
def near_bar(self) -> str | None:
|
|
448
|
+
"""The name of the bar the interval is near ("closeness" or "large-loss"), if any."""
|
|
449
|
+
interval = self._bounded()
|
|
450
|
+
if interval is None:
|
|
451
|
+
return None
|
|
452
|
+
for name, bar in (
|
|
453
|
+
("closeness", self.thresholds.close),
|
|
454
|
+
("large-loss", self.thresholds.large),
|
|
455
|
+
):
|
|
456
|
+
nearest = min(abs(interval.low - bar), abs(interval.high - bar))
|
|
457
|
+
if interval.low <= bar <= interval.high or nearest <= NEAR_BAR * bar:
|
|
458
|
+
return name
|
|
459
|
+
return None
|
|
460
|
+
|
|
461
|
+
@property
|
|
462
|
+
def prompts_needed(self) -> int | None:
|
|
463
|
+
"""Prompts that would likely settle which side of the nearest bar the mean is on;
|
|
464
|
+
None if more will not.
|
|
465
|
+
|
|
466
|
+
Below the closeness bar that is the interval's upper end dropping under it. At or
|
|
467
|
+
above a bar it is the lower end clearing that bar, estimated on the mirrored
|
|
468
|
+
interval with the same square-root scaling.
|
|
469
|
+
"""
|
|
470
|
+
interval = self.interval
|
|
471
|
+
if interval is None:
|
|
472
|
+
return None
|
|
473
|
+
if self.mean < self.thresholds.close:
|
|
474
|
+
needed = items_to_bound_below(interval, self.prompts, self.thresholds.close)
|
|
475
|
+
else:
|
|
476
|
+
bar = (
|
|
477
|
+
self.thresholds.large
|
|
478
|
+
if self.mean >= self.thresholds.large
|
|
479
|
+
else self.thresholds.close
|
|
480
|
+
)
|
|
481
|
+
mirrored = Interval(-interval.estimate, -interval.high, -interval.low)
|
|
482
|
+
needed = items_to_bound_below(mirrored, self.prompts, -bar)
|
|
483
|
+
return None if needed is None else max(MIN_PROMPTS, needed)
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
@dataclass(frozen=True, slots=True)
|
|
487
|
+
class _TaskEvidence:
|
|
488
|
+
"""Every reliable suite's paired outcomes pooled into one pass-rate comparison."""
|
|
489
|
+
|
|
490
|
+
kinds: tuple[TaskKind, ...]
|
|
491
|
+
cases: int
|
|
492
|
+
delta: Interval
|
|
493
|
+
"""Candidate minus reference pass rate in points, with a Newcombe 95% interval."""
|
|
494
|
+
pooled_needed: int | None
|
|
495
|
+
"""Pooled cases at which the interval's lower end would likely clear -TASK_MARGIN (the
|
|
496
|
+
current count when it already does); None if more cases will not."""
|
|
497
|
+
p_value: float
|
|
498
|
+
"""Exact McNemar p-value of the pooled comparison."""
|
|
499
|
+
|
|
500
|
+
@property
|
|
501
|
+
def close(self) -> bool:
|
|
502
|
+
return self.cases >= MIN_CASES and self.delta.low >= -TASK_MARGIN
|
|
503
|
+
|
|
504
|
+
@property
|
|
505
|
+
def significant_loss(self) -> bool:
|
|
506
|
+
return self.p_value < _ALPHA and self.delta.estimate < 0.0
|
|
507
|
+
|
|
508
|
+
@property
|
|
509
|
+
def cases_needed(self) -> int | None:
|
|
510
|
+
"""The pooled requirement spread over the suites, as cases per suite."""
|
|
511
|
+
if self.pooled_needed is None:
|
|
512
|
+
return None
|
|
513
|
+
return math.ceil(max(self.pooled_needed, MIN_CASES) / len(self.kinds))
|
|
514
|
+
|
|
515
|
+
|
|
516
|
+
def _logit_evidence(logit: LogitMetrics | None, thresholds: KldThresholds) -> _LogitEvidence | None:
|
|
517
|
+
if logit is None or logit.prompts == 0:
|
|
518
|
+
return None
|
|
519
|
+
values = [p.kld_mean for p in logit.per_prompt if p.kld_mean is not None]
|
|
520
|
+
if not values:
|
|
521
|
+
return _LogitEvidence(0, logit.kld_mean, None, thresholds, logit.exact_token_ids)
|
|
522
|
+
interval = bootstrap_mean(values)
|
|
523
|
+
return _LogitEvidence(
|
|
524
|
+
len(values), interval.estimate, interval, thresholds, logit.exact_token_ids
|
|
525
|
+
)
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
def _task_evidence(
|
|
529
|
+
reference: dict[TaskKind, dict[str, bool]],
|
|
530
|
+
candidate: dict[TaskKind, dict[str, bool]],
|
|
531
|
+
reliable: frozenset[TaskKind],
|
|
532
|
+
) -> _TaskEvidence | None:
|
|
533
|
+
kinds: list[TaskKind] = []
|
|
534
|
+
ref_pass: list[bool] = []
|
|
535
|
+
cand_pass: list[bool] = []
|
|
536
|
+
for kind in SCORED_TASK_KINDS:
|
|
537
|
+
if kind not in reliable:
|
|
538
|
+
continue
|
|
539
|
+
ours, theirs = _pair_cases(reference[kind], candidate[kind])
|
|
540
|
+
if ours:
|
|
541
|
+
kinds.append(kind)
|
|
542
|
+
ref_pass += ours
|
|
543
|
+
cand_pass += theirs
|
|
544
|
+
if not kinds:
|
|
545
|
+
return None
|
|
546
|
+
return _TaskEvidence(
|
|
547
|
+
kinds=tuple(kinds),
|
|
548
|
+
cases=len(ref_pass),
|
|
549
|
+
delta=paired_proportion_diff(ref_pass, cand_pass),
|
|
550
|
+
pooled_needed=cases_to_bound_loss(ref_pass, cand_pass, TASK_MARGIN),
|
|
551
|
+
p_value=mcnemar_exact(ref_pass, cand_pass),
|
|
552
|
+
)
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def _outcomes(result: CandidateResult) -> dict[TaskKind, dict[str, bool]]:
|
|
556
|
+
joined: dict[TaskKind, dict[str, bool]] = {kind: {} for kind in SCORED_TASK_KINDS}
|
|
557
|
+
for outcome in result.outcomes:
|
|
558
|
+
if outcome.kind in joined and outcome.passed is not None:
|
|
559
|
+
joined[outcome.kind][outcome.case_id] = outcome.passed
|
|
560
|
+
return joined
|
|
561
|
+
|
|
562
|
+
|
|
563
|
+
def _pair_cases(first: dict[str, bool], second: dict[str, bool]) -> tuple[list[bool], list[bool]]:
|
|
564
|
+
shared = [case_id for case_id in first if case_id in second]
|
|
565
|
+
return [first[c] for c in shared], [second[c] for c in shared]
|
|
566
|
+
|
|
567
|
+
|
|
568
|
+
# Per-candidate measurements ---------------------------------------------------------------
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
@dataclass(frozen=True, slots=True)
|
|
572
|
+
class _Candidate:
|
|
573
|
+
result: CandidateResult
|
|
574
|
+
index: int
|
|
575
|
+
name: str
|
|
576
|
+
kld: float | None
|
|
577
|
+
"""Mean KLD over all scored positions, as the card shows it."""
|
|
578
|
+
logit: _LogitEvidence | None
|
|
579
|
+
tasks: _TaskEvidence | None
|
|
580
|
+
size: int | None
|
|
581
|
+
size_change: float | None
|
|
582
|
+
deltas: tuple[TaskDelta, ...]
|
|
583
|
+
has_task_mean: bool
|
|
584
|
+
unresolved: tuple[tuple[TaskDelta, int | None], ...]
|
|
585
|
+
"""Reliable suites more than TASK_MARGIN points below the reference whose interval still
|
|
586
|
+
reaches zero and whose loss is not significant, each with the paired cases that would
|
|
587
|
+
likely make the loss significant."""
|
|
588
|
+
|
|
589
|
+
@property
|
|
590
|
+
def label(self) -> str:
|
|
591
|
+
return self.result.spec.label
|
|
592
|
+
|
|
593
|
+
@property
|
|
594
|
+
def measured(self) -> bool:
|
|
595
|
+
agreement = self.result.agreement
|
|
596
|
+
return (
|
|
597
|
+
self.kld is not None
|
|
598
|
+
or self.has_task_mean
|
|
599
|
+
or (agreement is not None and agreement.cases > 0)
|
|
600
|
+
)
|
|
601
|
+
|
|
602
|
+
@property
|
|
603
|
+
def regressions(self) -> list[TaskDelta]:
|
|
604
|
+
return sorted(
|
|
605
|
+
(
|
|
606
|
+
d
|
|
607
|
+
for d in self.deltas
|
|
608
|
+
if d.reference_reliable and d.significant and d.delta.estimate < 0
|
|
609
|
+
),
|
|
610
|
+
key=lambda d: d.delta.estimate,
|
|
611
|
+
)
|
|
612
|
+
|
|
613
|
+
@property
|
|
614
|
+
def gains(self) -> list[TaskDelta]:
|
|
615
|
+
return [d for d in self.deltas if d.significant and d.delta.estimate > 0]
|
|
616
|
+
|
|
617
|
+
def rank_key(self) -> tuple[bool, float, bool, int, int]:
|
|
618
|
+
return (
|
|
619
|
+
self.kld is None,
|
|
620
|
+
self.kld or 0.0,
|
|
621
|
+
self.size is None,
|
|
622
|
+
self.size or 0,
|
|
623
|
+
self.index,
|
|
624
|
+
)
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
def _measure(
|
|
628
|
+
result: CandidateResult,
|
|
629
|
+
index: int,
|
|
630
|
+
name: str,
|
|
631
|
+
*,
|
|
632
|
+
reference: CandidateResult,
|
|
633
|
+
reference_outcomes: dict[TaskKind, dict[str, bool]],
|
|
634
|
+
reliable: frozenset[TaskKind],
|
|
635
|
+
thresholds: KldThresholds,
|
|
636
|
+
) -> _Candidate:
|
|
637
|
+
logit = result.logit if result.logit is not None and result.logit.prompts > 0 else None
|
|
638
|
+
size = None if result.info is None else result.info.size_bytes
|
|
639
|
+
reference_size = None if reference.info is None else reference.info.size_bytes
|
|
640
|
+
outcomes = _outcomes(result)
|
|
641
|
+
deltas = []
|
|
642
|
+
unresolved = []
|
|
643
|
+
for kind in SCORED_TASK_KINDS:
|
|
644
|
+
ref_pass, cand_pass = _pair_cases(reference_outcomes[kind], outcomes[kind])
|
|
645
|
+
if not ref_pass:
|
|
646
|
+
continue
|
|
647
|
+
delta = TaskDelta(
|
|
648
|
+
kind=kind,
|
|
649
|
+
cases=len(ref_pass),
|
|
650
|
+
candidate_rate=sum(cand_pass) / len(cand_pass),
|
|
651
|
+
reference_rate=sum(ref_pass) / len(ref_pass),
|
|
652
|
+
delta=paired_proportion_diff(ref_pass, cand_pass),
|
|
653
|
+
significant=mcnemar_exact(ref_pass, cand_pass) < _ALPHA,
|
|
654
|
+
reference_reliable=kind in reliable,
|
|
655
|
+
)
|
|
656
|
+
deltas.append(delta)
|
|
657
|
+
if (
|
|
658
|
+
delta.reference_reliable
|
|
659
|
+
and not delta.significant
|
|
660
|
+
and delta.delta.estimate < -TASK_MARGIN
|
|
661
|
+
and delta.delta.high >= 0.0
|
|
662
|
+
):
|
|
663
|
+
unresolved.append((delta, _cases_to_resolve(ref_pass, cand_pass)))
|
|
664
|
+
return _Candidate(
|
|
665
|
+
result=result,
|
|
666
|
+
index=index,
|
|
667
|
+
name=name,
|
|
668
|
+
kld=None if logit is None else logit.kld_mean,
|
|
669
|
+
logit=_logit_evidence(logit, thresholds),
|
|
670
|
+
tasks=_task_evidence(reference_outcomes, outcomes, reliable),
|
|
671
|
+
size=size,
|
|
672
|
+
size_change=(None if size is None or not reference_size else size / reference_size - 1.0),
|
|
673
|
+
deltas=tuple(deltas),
|
|
674
|
+
has_task_mean=any(
|
|
675
|
+
task.kind in SCORED_TASK_KINDS and task.rate is not None for task in result.tasks
|
|
676
|
+
),
|
|
677
|
+
unresolved=tuple(unresolved),
|
|
678
|
+
)
|
|
679
|
+
|
|
680
|
+
|
|
681
|
+
def _cases_to_resolve(ref: Sequence[bool], cand: Sequence[bool]) -> int | None:
|
|
682
|
+
"""Roughly how many paired cases would make an observed loss significant.
|
|
683
|
+
|
|
684
|
+
The observed shares of the four paired outcomes are held fixed while the case count
|
|
685
|
+
grows, as for the task estimate in stats.cases_to_bound_loss, and the exact McNemar
|
|
686
|
+
test is rerun on the scaled counts. None past _MAX_CASES_TO_RESOLVE cases.
|
|
687
|
+
"""
|
|
688
|
+
n = len(ref)
|
|
689
|
+
lost = sum(1 for r, c in zip(ref, cand, strict=True) if r and not c)
|
|
690
|
+
gained = sum(1 for r, c in zip(ref, cand, strict=True) if c and not r)
|
|
691
|
+
|
|
692
|
+
def significant(cases: int) -> bool:
|
|
693
|
+
scaled_lost = round(lost * cases / n)
|
|
694
|
+
scaled_gained = round(gained * cases / n)
|
|
695
|
+
reference = [True] * scaled_lost + [False] * scaled_gained
|
|
696
|
+
candidate = [False] * scaled_lost + [True] * scaled_gained
|
|
697
|
+
return mcnemar_exact(reference, candidate) < _ALPHA
|
|
698
|
+
|
|
699
|
+
low, high = n, 2 * n
|
|
700
|
+
while not significant(high):
|
|
701
|
+
if high >= _MAX_CASES_TO_RESOLVE:
|
|
702
|
+
return None
|
|
703
|
+
low, high = high, 2 * high
|
|
704
|
+
while high - low > 1:
|
|
705
|
+
middle = (low + high) // 2
|
|
706
|
+
if significant(middle):
|
|
707
|
+
high = middle
|
|
708
|
+
else:
|
|
709
|
+
low = middle
|
|
710
|
+
return high
|
|
711
|
+
|
|
712
|
+
|
|
713
|
+
# Judging ----------------------------------------------------------------------------------
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
@dataclass(slots=True)
|
|
717
|
+
class _Call:
|
|
718
|
+
"""Status and reasons for one candidate while the verdict is being built."""
|
|
719
|
+
|
|
720
|
+
candidate: _Candidate
|
|
721
|
+
status: Status
|
|
722
|
+
reasons: list[str]
|
|
723
|
+
"""Phrases, most important first; turned into sentences at the end."""
|
|
724
|
+
detail: str | None = None
|
|
725
|
+
"""The sentence this candidate contributes to Verdict.details, if any."""
|
|
726
|
+
prompts_needed: int | None = None
|
|
727
|
+
cases_needed: int | None = None
|
|
728
|
+
"""For an inconclusive candidate that more evidence could decide: scoring prompts and
|
|
729
|
+
cases per suite, rounded up. Both None when no rerun is likely to decide."""
|
|
730
|
+
fits: bool | None = None
|
|
731
|
+
"""Whether the download fits the --max-size budget; None without a budget or a size."""
|
|
732
|
+
caveats: list[str] = field(default_factory=list)
|
|
733
|
+
task_caveat: str | None = None
|
|
734
|
+
"""The first unresolved task loss, as a caveat phrase."""
|
|
735
|
+
|
|
736
|
+
@property
|
|
737
|
+
def name(self) -> str:
|
|
738
|
+
return self.candidate.name
|
|
739
|
+
|
|
740
|
+
|
|
741
|
+
def judge(report: Report, *, scoring: Sequence[ScoringPrompt] = ()) -> Verdict:
|
|
742
|
+
"""Build the recommendation for `report`. Deterministic: same report, same verdict.
|
|
743
|
+
|
|
744
|
+
`scoring` is the text of the scoring prompts the run used, which reports do not store.
|
|
745
|
+
When it is mostly non-Latin script and a text-forced candidate reads above the
|
|
746
|
+
closeness bar, the details say that text forcing may read high on such prompts.
|
|
747
|
+
"""
|
|
748
|
+
names = display_labels(report)
|
|
749
|
+
reference = report.reference
|
|
750
|
+
ref_name = names[reference.spec.label]
|
|
751
|
+
ref_outcomes = _outcomes(reference)
|
|
752
|
+
settings = report.settings
|
|
753
|
+
thresholds = kld_thresholds(settings.top_k)
|
|
754
|
+
reliable = frozenset(
|
|
755
|
+
kind
|
|
756
|
+
for kind in SCORED_TASK_KINDS
|
|
757
|
+
if ref_outcomes[kind]
|
|
758
|
+
and sum(ref_outcomes[kind].values()) / len(ref_outcomes[kind]) >= _RELIABLE_REFERENCE_RATE
|
|
759
|
+
)
|
|
760
|
+
candidates = [
|
|
761
|
+
_measure(
|
|
762
|
+
result,
|
|
763
|
+
index,
|
|
764
|
+
names[result.spec.label],
|
|
765
|
+
reference=reference,
|
|
766
|
+
reference_outcomes=ref_outcomes,
|
|
767
|
+
reliable=reliable,
|
|
768
|
+
thresholds=thresholds,
|
|
769
|
+
)
|
|
770
|
+
for index, result in enumerate(report.candidates)
|
|
771
|
+
]
|
|
772
|
+
calls = [_assess(candidate, ref_name) for candidate in candidates]
|
|
773
|
+
for call in calls:
|
|
774
|
+
_annotate(call, settings)
|
|
775
|
+
_pick(calls, ref_name, settings.max_size_bytes)
|
|
776
|
+
for call in calls:
|
|
777
|
+
if call.status != "failed":
|
|
778
|
+
call.reasons += [
|
|
779
|
+
f"scored higher than the reference on {d.kind}; {_QUANTIZED_REFERENCE}"
|
|
780
|
+
for d in call.candidate.gains
|
|
781
|
+
]
|
|
782
|
+
|
|
783
|
+
ordered = sorted(
|
|
784
|
+
calls, key=lambda call: (_STATUS_ORDER[call.status], call.candidate.rank_key())
|
|
785
|
+
)
|
|
786
|
+
verdicts = tuple(
|
|
787
|
+
CandidateVerdict(
|
|
788
|
+
label=call.candidate.label,
|
|
789
|
+
status=call.status,
|
|
790
|
+
rank=None if call.status == "failed" else rank,
|
|
791
|
+
kld_band=(
|
|
792
|
+
None if call.candidate.kld is None else kld_band(call.candidate.kld, thresholds)
|
|
793
|
+
),
|
|
794
|
+
size_bytes=call.candidate.size,
|
|
795
|
+
size_change=call.candidate.size_change,
|
|
796
|
+
task_deltas=call.candidate.deltas,
|
|
797
|
+
reasons=tuple(_sentence(reason) for reason in call.reasons),
|
|
798
|
+
caveats=tuple(call.caveats),
|
|
799
|
+
near_bar=call.candidate.logit is not None and call.candidate.logit.near_bar is not None,
|
|
800
|
+
fits_budget=call.fits,
|
|
801
|
+
)
|
|
802
|
+
for rank, call in enumerate(ordered, start=1)
|
|
803
|
+
)
|
|
804
|
+
summary = _summarize(
|
|
805
|
+
ordered, ref_name, settings, non_latin=mostly_non_latin(p.text for p in scoring)
|
|
806
|
+
)
|
|
807
|
+
return Verdict(
|
|
808
|
+
headline=summary.headline,
|
|
809
|
+
details=summary.details,
|
|
810
|
+
candidates=verdicts,
|
|
811
|
+
findings=_findings(report, names),
|
|
812
|
+
cases_needed=summary.cases_needed,
|
|
813
|
+
remedy=summary.remedy,
|
|
814
|
+
)
|
|
815
|
+
|
|
816
|
+
|
|
817
|
+
def _assess(candidate: _Candidate, ref_name: str) -> _Call:
|
|
818
|
+
"""The candidate's status on its own evidence; close candidates come back as ok."""
|
|
819
|
+
if not candidate.measured:
|
|
820
|
+
return _Call(candidate, "failed", [_failure_reason(candidate.result)])
|
|
821
|
+
avoid = _avoid_reasons(candidate, ref_name)
|
|
822
|
+
if avoid:
|
|
823
|
+
return _Call(
|
|
824
|
+
candidate, "avoid", avoid, f"Avoid {candidate.name}: {' and '.join(avoid[:2])}."
|
|
825
|
+
)
|
|
826
|
+
logit, tasks = candidate.logit, candidate.tasks
|
|
827
|
+
# Significant task losses were handled above, so with logit evidence only KLD decides.
|
|
828
|
+
close = logit.close if logit is not None else tasks is not None and tasks.close
|
|
829
|
+
if close:
|
|
830
|
+
return _Call(candidate, "ok", _close_reasons(candidate, ref_name))
|
|
831
|
+
if logit is not None and logit.moderate and logit.interval is not None:
|
|
832
|
+
return _usable(candidate, logit.interval, ref_name)
|
|
833
|
+
return _inconclusive(candidate, ref_name)
|
|
834
|
+
|
|
835
|
+
|
|
836
|
+
def _annotate(call: _Call, settings: RunSettings) -> None:
|
|
837
|
+
"""Budget fit and caveats, which sit next to the status without changing it."""
|
|
838
|
+
candidate = call.candidate
|
|
839
|
+
budget = settings.max_size_bytes
|
|
840
|
+
if budget is not None and candidate.size is not None:
|
|
841
|
+
call.fits = candidate.size <= budget
|
|
842
|
+
if call.status == "failed":
|
|
843
|
+
return
|
|
844
|
+
near = None if candidate.logit is None else candidate.logit.near_bar
|
|
845
|
+
if near is not None:
|
|
846
|
+
call.caveats.append(f"near the {near} bar; a rerun could change this")
|
|
847
|
+
tasks = [_task_caveat(delta, needed, settings) for delta, needed in candidate.unresolved]
|
|
848
|
+
call.caveats += tasks
|
|
849
|
+
call.task_caveat = tasks[0] if tasks else None
|
|
850
|
+
|
|
851
|
+
|
|
852
|
+
def _task_caveat(delta: TaskDelta, needed: int | None, settings: RunSettings) -> str:
|
|
853
|
+
"""E.g. "tools -17 unresolved (95% CI -42 to +6); rerun with --max-cases 60"."""
|
|
854
|
+
text = (
|
|
855
|
+
f"{delta.kind} {round(delta.delta.estimate)} unresolved ({_interval_points(delta.delta)})"
|
|
856
|
+
)
|
|
857
|
+
if needed is None:
|
|
858
|
+
return text
|
|
859
|
+
cases = _round_up(needed)
|
|
860
|
+
if settings.prompts_file is None and cases <= len(load_builtin(delta.kind)):
|
|
861
|
+
return f"{text}; rerun with --max-cases {cases}"
|
|
862
|
+
return f"{text}; about {cases} {delta.kind} cases would settle it"
|
|
863
|
+
|
|
864
|
+
|
|
865
|
+
def _failure_reason(result: CandidateResult) -> str:
|
|
866
|
+
if result.errors:
|
|
867
|
+
return "no metrics: " + " ".join(result.errors[0].split()).rstrip(".")
|
|
868
|
+
return "no metrics were produced"
|
|
869
|
+
|
|
870
|
+
|
|
871
|
+
def _avoid_reasons(candidate: _Candidate, ref_name: str) -> list[str]:
|
|
872
|
+
reasons = [
|
|
873
|
+
f"{d.kind} drops {-round(d.delta.estimate)} points vs {ref_name} "
|
|
874
|
+
f"({_interval_points(d.delta)})"
|
|
875
|
+
for d in candidate.regressions
|
|
876
|
+
]
|
|
877
|
+
tasks = candidate.tasks
|
|
878
|
+
if not reasons and tasks is not None and tasks.significant_loss:
|
|
879
|
+
reasons.append(
|
|
880
|
+
f"task scores drop {_points(-tasks.delta.estimate)} points overall vs {ref_name} "
|
|
881
|
+
f"({_interval_points(tasks.delta)} on {_plural(tasks.cases, 'case')})"
|
|
882
|
+
)
|
|
883
|
+
logit = candidate.logit
|
|
884
|
+
if logit is not None and logit.large and logit.interval is not None:
|
|
885
|
+
reasons.append(f"KLD is large ({_kld(logit.mean)}, {_interval_kld(logit.interval)})")
|
|
886
|
+
return reasons
|
|
887
|
+
|
|
888
|
+
|
|
889
|
+
def _close_reasons(candidate: _Candidate, ref_name: str) -> list[str]:
|
|
890
|
+
reasons = []
|
|
891
|
+
logit, tasks = candidate.logit, candidate.tasks
|
|
892
|
+
if logit is not None and logit.interval is not None:
|
|
893
|
+
reasons.append(
|
|
894
|
+
f"KLD {_kld(logit.mean)} ({_interval_kld(logit.interval)}) "
|
|
895
|
+
f"on {_plural(logit.prompts, 'prompt')}"
|
|
896
|
+
)
|
|
897
|
+
if tasks is not None:
|
|
898
|
+
reasons.append(
|
|
899
|
+
_tasks_within(tasks, ref_name) if tasks.close else _no_task_loss(tasks, ref_name)
|
|
900
|
+
)
|
|
901
|
+
note = _evidence_note(candidate, ref_name)
|
|
902
|
+
if note:
|
|
903
|
+
reasons.append(f"rests on {note}")
|
|
904
|
+
return reasons
|
|
905
|
+
|
|
906
|
+
|
|
907
|
+
def _usable(candidate: _Candidate, interval: Interval, ref_name: str) -> _Call:
|
|
908
|
+
loss = f"moderate loss: KLD {_kld(interval.estimate)} ({_interval_kld(interval)})"
|
|
909
|
+
reasons = [loss]
|
|
910
|
+
tasks = candidate.tasks
|
|
911
|
+
if tasks is not None:
|
|
912
|
+
reasons.append(
|
|
913
|
+
_tasks_within(tasks, ref_name) if tasks.close else _no_task_loss(tasks, ref_name)
|
|
914
|
+
)
|
|
915
|
+
return _Call(candidate, "usable", reasons, f"{candidate.name} has a {loss}.")
|
|
916
|
+
|
|
917
|
+
|
|
918
|
+
def _evidence_note(candidate: _Candidate, ref_name: str) -> str | None:
|
|
919
|
+
"""Which evidence a close call rests on, when it is only one kind."""
|
|
920
|
+
if candidate.tasks is None:
|
|
921
|
+
if candidate.deltas:
|
|
922
|
+
return (
|
|
923
|
+
f"logit evidence only; {ref_name} passes under half of every task suite, "
|
|
924
|
+
"so the suites cannot judge it"
|
|
925
|
+
)
|
|
926
|
+
return "logit evidence only; no task suites ran"
|
|
927
|
+
if candidate.logit is None:
|
|
928
|
+
return "task evidence only; no logit metrics"
|
|
929
|
+
return None
|
|
930
|
+
|
|
931
|
+
|
|
932
|
+
def _inconclusive(candidate: _Candidate, ref_name: str) -> _Call:
|
|
933
|
+
logit, tasks = candidate.logit, candidate.tasks
|
|
934
|
+
reasons: list[str] = []
|
|
935
|
+
gaps: list[str] = []
|
|
936
|
+
needs: list[int | None] = []
|
|
937
|
+
prompts_needed = cases_needed = None
|
|
938
|
+
if logit is not None:
|
|
939
|
+
if logit.close and logit.interval is not None:
|
|
940
|
+
reasons.append(
|
|
941
|
+
f"KLD {_kld(logit.mean)} is under the {logit.thresholds.bar} closeness bar "
|
|
942
|
+
f"({_interval_kld(logit.interval)})"
|
|
943
|
+
)
|
|
944
|
+
else:
|
|
945
|
+
gaps.append(_logit_gap(logit))
|
|
946
|
+
prompts_needed = logit.prompts_needed
|
|
947
|
+
needs.append(prompts_needed)
|
|
948
|
+
if tasks is not None:
|
|
949
|
+
if tasks.close:
|
|
950
|
+
reasons.append(_tasks_within(tasks, ref_name))
|
|
951
|
+
else:
|
|
952
|
+
gaps.append(_task_gap(tasks, ref_name))
|
|
953
|
+
cases_needed = tasks.cases_needed
|
|
954
|
+
needs.append(cases_needed)
|
|
955
|
+
if logit is None and tasks is None:
|
|
956
|
+
gaps.append("no logit metrics and no reliable task results to judge it by")
|
|
957
|
+
needs.append(None)
|
|
958
|
+
call = _Call(candidate, "inconclusive", gaps + reasons)
|
|
959
|
+
if all(need is not None for need in needs):
|
|
960
|
+
call.prompts_needed = None if prompts_needed is None else _round_up(prompts_needed)
|
|
961
|
+
call.cases_needed = None if cases_needed is None else _round_up(cases_needed)
|
|
962
|
+
call.reasons.append(f"{_wanted(call)} would likely decide")
|
|
963
|
+
above = logit is not None and logit.mean >= logit.thresholds.close
|
|
964
|
+
state = "is undecided" if above else "is not shown to be close"
|
|
965
|
+
call.detail = f"{candidate.name} {state}: {gaps[0]}."
|
|
966
|
+
return call
|
|
967
|
+
|
|
968
|
+
|
|
969
|
+
def _logit_gap(logit: _LogitEvidence) -> str:
|
|
970
|
+
kld = _kld(logit.mean)
|
|
971
|
+
interval = logit.interval
|
|
972
|
+
if interval is None:
|
|
973
|
+
return f"KLD {kld}, but the report has no per-prompt KLD to bound it"
|
|
974
|
+
if logit.prompts < MIN_PROMPTS:
|
|
975
|
+
return _thin_logit_gap(logit)
|
|
976
|
+
thresholds = logit.thresholds
|
|
977
|
+
bar = thresholds.bar
|
|
978
|
+
if logit.mean >= thresholds.large:
|
|
979
|
+
return (
|
|
980
|
+
f"KLD {kld} may be a large loss, but its 95% CI starts at {_kld(interval.low)}, "
|
|
981
|
+
f"under the {thresholds.large_bar} large-loss bar"
|
|
982
|
+
)
|
|
983
|
+
if logit.mean >= thresholds.close:
|
|
984
|
+
return (
|
|
985
|
+
f"KLD {kld} is above the {bar} closeness bar, but its 95% CI reaches down to "
|
|
986
|
+
f"{_kld(interval.low)}"
|
|
987
|
+
)
|
|
988
|
+
return f"KLD {kld}, but its 95% CI reaches {_kld(interval.high)}, above the {bar} closeness bar"
|
|
989
|
+
|
|
990
|
+
|
|
991
|
+
def _thin_logit_gap(logit: _LogitEvidence) -> str:
|
|
992
|
+
"""Why fewer than MIN_PROMPTS prompts decide nothing about this KLD."""
|
|
993
|
+
kld = _kld(logit.mean)
|
|
994
|
+
bar = logit.thresholds.bar
|
|
995
|
+
if logit.mean >= logit.thresholds.large:
|
|
996
|
+
return f"KLD {kld} looks large on {_plural(logit.prompts, 'prompt')}, too few to call it"
|
|
997
|
+
if logit.mean >= logit.thresholds.close:
|
|
998
|
+
return f"KLD {kld} is above the {bar} closeness bar"
|
|
999
|
+
return f"KLD {kld}, but {_plural(logit.prompts, 'prompt')} cannot bound it below {bar}"
|
|
1000
|
+
|
|
1001
|
+
|
|
1002
|
+
def _task_gap(tasks: _TaskEvidence, ref_name: str) -> str:
|
|
1003
|
+
margin = _points(TASK_MARGIN)
|
|
1004
|
+
cases = _plural(tasks.cases, "case")
|
|
1005
|
+
if tasks.delta.estimate <= -TASK_MARGIN:
|
|
1006
|
+
return f"task scores are {_points(-tasks.delta.estimate)} points lower than {ref_name}"
|
|
1007
|
+
if tasks.cases < MIN_CASES:
|
|
1008
|
+
return f"{cases} cannot show task scores within {margin} points of {ref_name}"
|
|
1009
|
+
return (
|
|
1010
|
+
f"task scores could be up to {_points(-tasks.delta.low)} points lower than {ref_name} "
|
|
1011
|
+
f"({_interval_points(tasks.delta)} on {cases})"
|
|
1012
|
+
)
|
|
1013
|
+
|
|
1014
|
+
|
|
1015
|
+
def _no_task_loss(tasks: _TaskEvidence, ref_name: str) -> str:
|
|
1016
|
+
return (
|
|
1017
|
+
f"no significant task loss vs {ref_name} on {_plural(tasks.cases, 'case')} "
|
|
1018
|
+
f"({_interval_points(tasks.delta)})"
|
|
1019
|
+
)
|
|
1020
|
+
|
|
1021
|
+
|
|
1022
|
+
def _tasks_within(tasks: _TaskEvidence, ref_name: str) -> str:
|
|
1023
|
+
return (
|
|
1024
|
+
f"task scores within {_points(TASK_MARGIN)} points of {ref_name} "
|
|
1025
|
+
f"on {_plural(tasks.cases, 'case')}"
|
|
1026
|
+
)
|
|
1027
|
+
|
|
1028
|
+
|
|
1029
|
+
def _wanted(call: _Call) -> str:
|
|
1030
|
+
"""The evidence an inconclusive candidate needs, e.g. "about 20 scoring prompts"."""
|
|
1031
|
+
parts = []
|
|
1032
|
+
if call.prompts_needed is not None:
|
|
1033
|
+
parts.append(f"{call.prompts_needed} scoring prompts")
|
|
1034
|
+
if call.cases_needed is not None:
|
|
1035
|
+
parts.append(f"{call.cases_needed} cases per suite")
|
|
1036
|
+
return "about " + " and ".join(parts)
|
|
1037
|
+
|
|
1038
|
+
|
|
1039
|
+
def _pick(calls: Sequence[_Call], ref_name: str, budget: int | None) -> None:
|
|
1040
|
+
"""Recommend the smallest close candidate that fits the budget; the other close ones
|
|
1041
|
+
stay ok, and those over the budget say so."""
|
|
1042
|
+
close = [call for call in calls if call.status == "ok"]
|
|
1043
|
+
if budget is not None:
|
|
1044
|
+
for call in close:
|
|
1045
|
+
if not call.fits:
|
|
1046
|
+
call.reasons.insert(0, _over_budget(call, ref_name, budget))
|
|
1047
|
+
call.detail = f"{call.name} is close but {_needs(call)}."
|
|
1048
|
+
eligible = [call for call in close if budget is None or call.fits]
|
|
1049
|
+
if not eligible:
|
|
1050
|
+
return
|
|
1051
|
+
if all(call.candidate.size is not None for call in eligible):
|
|
1052
|
+
pick = min(eligible, key=lambda call: (call.candidate.size, call.candidate.rank_key()))
|
|
1053
|
+
else:
|
|
1054
|
+
pick = min(eligible, key=lambda call: call.candidate.rank_key())
|
|
1055
|
+
pick.status = "recommended"
|
|
1056
|
+
if len(eligible) > 1:
|
|
1057
|
+
pick.reasons.append(f"smallest download that is close to {ref_name}")
|
|
1058
|
+
for call in eligible:
|
|
1059
|
+
if call is pick:
|
|
1060
|
+
continue
|
|
1061
|
+
reason = _also_close(call.candidate, pick.candidate, ref_name)
|
|
1062
|
+
call.reasons.insert(0, reason)
|
|
1063
|
+
call.detail = f"{call.name} is {reason}."
|
|
1064
|
+
|
|
1065
|
+
|
|
1066
|
+
def _also_close(candidate: _Candidate, pick: _Candidate, ref_name: str) -> str:
|
|
1067
|
+
if candidate.size is not None and pick.size:
|
|
1068
|
+
larger = candidate.size / pick.size - 1.0
|
|
1069
|
+
if larger > _SAME_SIZE:
|
|
1070
|
+
return f"also close to {ref_name} but {_percent(larger)} larger"
|
|
1071
|
+
return f"also close to {ref_name}"
|
|
1072
|
+
|
|
1073
|
+
|
|
1074
|
+
def _over_budget(call: _Call, ref_name: str, budget: int) -> str:
|
|
1075
|
+
size = call.candidate.size
|
|
1076
|
+
if size is None:
|
|
1077
|
+
return (
|
|
1078
|
+
f"close to {ref_name}, but its size is unknown, so it cannot be checked against "
|
|
1079
|
+
f"the {format_size(budget)} budget"
|
|
1080
|
+
)
|
|
1081
|
+
return (
|
|
1082
|
+
f"close to {ref_name} but needs {format_size(size)}, over the {format_size(budget)} budget"
|
|
1083
|
+
)
|
|
1084
|
+
|
|
1085
|
+
|
|
1086
|
+
def _needs(call: _Call) -> str:
|
|
1087
|
+
size = call.candidate.size
|
|
1088
|
+
return "its size is unknown" if size is None else f"needs {format_size(size)}"
|
|
1089
|
+
|
|
1090
|
+
|
|
1091
|
+
# Headline and details ---------------------------------------------------------------------
|
|
1092
|
+
|
|
1093
|
+
|
|
1094
|
+
@dataclass(slots=True)
|
|
1095
|
+
class _Summary:
|
|
1096
|
+
headline: str
|
|
1097
|
+
details: tuple[str, ...] = ()
|
|
1098
|
+
remedy: str | None = None
|
|
1099
|
+
cases_needed: int | None = None
|
|
1100
|
+
|
|
1101
|
+
|
|
1102
|
+
def _summarize(
|
|
1103
|
+
ordered: Sequence[_Call], ref_name: str, settings: RunSettings, *, non_latin: bool
|
|
1104
|
+
) -> _Summary:
|
|
1105
|
+
if not ordered:
|
|
1106
|
+
return _Summary("No candidates were evaluated.")
|
|
1107
|
+
live = [call for call in ordered if call.status != "failed"]
|
|
1108
|
+
failed = [call.name for call in ordered if call.status == "failed"]
|
|
1109
|
+
if not live:
|
|
1110
|
+
return _Summary("No candidate produced metrics; see the errors for each candidate.")
|
|
1111
|
+
lead = _lead(live, ref_name, settings.max_size_bytes)
|
|
1112
|
+
summary = _Summary(lead.headline)
|
|
1113
|
+
if live[0].status != "recommended":
|
|
1114
|
+
summary.remedy, summary.cases_needed = _next_step(
|
|
1115
|
+
live, ref_name, settings, non_latin=non_latin
|
|
1116
|
+
)
|
|
1117
|
+
details = list(lead.lines)
|
|
1118
|
+
caveated = [lead.about, *(call for call in live if call is not lead.about)]
|
|
1119
|
+
unresolved = next((call for call in caveated if call.task_caveat), None)
|
|
1120
|
+
if unresolved is not None:
|
|
1121
|
+
# The headline's own candidate first, right under the headline; another one's
|
|
1122
|
+
# unresolved loss after the numbers behind the headline.
|
|
1123
|
+
at = 0 if unresolved is lead.about else len(details)
|
|
1124
|
+
details.insert(at, f"{unresolved.name}: {unresolved.task_caveat}.")
|
|
1125
|
+
details += [
|
|
1126
|
+
call.detail
|
|
1127
|
+
for call in live
|
|
1128
|
+
if call.detail and not (call is lead.about and lead.replaces_detail)
|
|
1129
|
+
]
|
|
1130
|
+
if non_latin and _text_forced_above_bar(live):
|
|
1131
|
+
note = "Most scoring prompts are not Latin script; text forcing may read higher there."
|
|
1132
|
+
details.append(note if summary.remedy == _CONFIRM_EXACT else f"{note} {_CONFIRM_EXACT}")
|
|
1133
|
+
details += [
|
|
1134
|
+
f"{call.name} scored higher than the reference on {d.kind}; {_QUANTIZED_REFERENCE}."
|
|
1135
|
+
for call in live
|
|
1136
|
+
for d in call.candidate.gains
|
|
1137
|
+
]
|
|
1138
|
+
if failed:
|
|
1139
|
+
details.append(f"{_join(failed)} produced no metrics; see the errors.")
|
|
1140
|
+
summary.details = tuple(details)
|
|
1141
|
+
return summary
|
|
1142
|
+
|
|
1143
|
+
|
|
1144
|
+
@dataclass(frozen=True, slots=True)
|
|
1145
|
+
class _Lead:
|
|
1146
|
+
headline: str
|
|
1147
|
+
about: _Call
|
|
1148
|
+
"""The candidate the headline is about."""
|
|
1149
|
+
lines: tuple[str, ...] = ()
|
|
1150
|
+
"""The numbers behind the headline, first in the details."""
|
|
1151
|
+
replaces_detail: bool = False
|
|
1152
|
+
"""True when the headline or the lines already say what the candidate's detail says."""
|
|
1153
|
+
|
|
1154
|
+
|
|
1155
|
+
def _lead(live: Sequence[_Call], ref_name: str, budget: int | None) -> _Lead:
|
|
1156
|
+
"""The headline for candidates in rank order, at least one of them not failed."""
|
|
1157
|
+
top = live[0]
|
|
1158
|
+
if top.status == "recommended":
|
|
1159
|
+
return _Lead(
|
|
1160
|
+
_run_headline(top.candidate, ref_name),
|
|
1161
|
+
top,
|
|
1162
|
+
(_run_detail(top.candidate, ref_name),),
|
|
1163
|
+
replaces_detail=True,
|
|
1164
|
+
)
|
|
1165
|
+
close = [call for call in live if call.status == "ok"]
|
|
1166
|
+
usable = [
|
|
1167
|
+
(call, call.candidate.logit)
|
|
1168
|
+
for call in live
|
|
1169
|
+
if call.status == "usable" and call.candidate.logit is not None
|
|
1170
|
+
]
|
|
1171
|
+
fitting = [(call, logit) for call, logit in usable if budget is None or call.fits]
|
|
1172
|
+
if fitting:
|
|
1173
|
+
best, logit = fitting[0]
|
|
1174
|
+
if budget is None:
|
|
1175
|
+
unsettled = any(call.status == "inconclusive" for call in live)
|
|
1176
|
+
headline = _smallest_loss_headline(best, logit, ref_name, unsettled=unsettled)
|
|
1177
|
+
else:
|
|
1178
|
+
headline = (
|
|
1179
|
+
f"Best that fits {format_size(budget)}: {best.name}, moderate loss "
|
|
1180
|
+
f"(KLD {_kld(logit.mean)})."
|
|
1181
|
+
)
|
|
1182
|
+
return _Lead(headline, best)
|
|
1183
|
+
if budget is not None and (close or usable):
|
|
1184
|
+
nearest = close[0] if close else usable[0][0]
|
|
1185
|
+
state = "is close" if close else "has a moderate loss"
|
|
1186
|
+
headline = (
|
|
1187
|
+
f"Nothing that fits {format_size(budget)} is close to {ref_name}; "
|
|
1188
|
+
f"{nearest.name} {state} but {_needs(nearest)}."
|
|
1189
|
+
)
|
|
1190
|
+
return _Lead(headline, nearest, replaces_detail=bool(close))
|
|
1191
|
+
if top.status == "inconclusive":
|
|
1192
|
+
closest = _closest_detail(top.candidate, ref_name)
|
|
1193
|
+
return _Lead(
|
|
1194
|
+
_keep_for_now_headline(top.candidate, ref_name),
|
|
1195
|
+
top,
|
|
1196
|
+
() if closest is None else (closest,),
|
|
1197
|
+
replaces_detail=True,
|
|
1198
|
+
)
|
|
1199
|
+
who = f"{top.name} shows" if len(live) == 1 else "every candidate shows"
|
|
1200
|
+
return _Lead(f"Keep {ref_name}: {who} a measured loss.", top)
|
|
1201
|
+
|
|
1202
|
+
|
|
1203
|
+
def _text_forced_above_bar(live: Sequence[_Call]) -> bool:
|
|
1204
|
+
"""Whether a text-forced candidate reads above the closeness bar."""
|
|
1205
|
+
return any(
|
|
1206
|
+
call.candidate.logit is not None
|
|
1207
|
+
and not call.candidate.logit.exact
|
|
1208
|
+
and call.candidate.logit.mean > call.candidate.logit.thresholds.close
|
|
1209
|
+
for call in live
|
|
1210
|
+
)
|
|
1211
|
+
|
|
1212
|
+
|
|
1213
|
+
def _next_step(
|
|
1214
|
+
live: Sequence[_Call], ref_name: str, settings: RunSettings, *, non_latin: bool
|
|
1215
|
+
) -> tuple[str, int | None]:
|
|
1216
|
+
"""What to do when nothing is recommended, and the --max-cases value of a rerun."""
|
|
1217
|
+
for call in live:
|
|
1218
|
+
if call.status == "inconclusive":
|
|
1219
|
+
rerun = _remedy(call, settings)
|
|
1220
|
+
if rerun is not None:
|
|
1221
|
+
return rerun
|
|
1222
|
+
return _advice(live, ref_name, settings, non_latin=non_latin), None
|
|
1223
|
+
|
|
1224
|
+
|
|
1225
|
+
def _advice(live: Sequence[_Call], ref_name: str, settings: RunSettings, *, non_latin: bool) -> str:
|
|
1226
|
+
"""The next step when no rerun of the same kind is likely to decide."""
|
|
1227
|
+
text_forced = _text_forced_above_bar(live)
|
|
1228
|
+
budget = settings.max_size_bytes
|
|
1229
|
+
over = sorted(
|
|
1230
|
+
(call.candidate.size, call.name)
|
|
1231
|
+
for call in live
|
|
1232
|
+
if call.status == "ok" and call.candidate.size is not None
|
|
1233
|
+
)
|
|
1234
|
+
if non_latin and text_forced:
|
|
1235
|
+
return _CONFIRM_EXACT
|
|
1236
|
+
if budget is None and any(call.status == "usable" for call in live):
|
|
1237
|
+
return (
|
|
1238
|
+
"Pass --max-size with the memory you can spare, for example --max-size 6GB, to "
|
|
1239
|
+
"pick the best download that fits."
|
|
1240
|
+
)
|
|
1241
|
+
if budget is not None and over:
|
|
1242
|
+
size, name = over[0]
|
|
1243
|
+
flag = format_size(size, round_up=True).replace(" ", "")
|
|
1244
|
+
return f"Rerun with --max-size {flag} to run {name}, which is close."
|
|
1245
|
+
if text_forced:
|
|
1246
|
+
return _CONFIRM_EXACT
|
|
1247
|
+
return _missing_evidence(live, ref_name)
|
|
1248
|
+
|
|
1249
|
+
|
|
1250
|
+
def _missing_evidence(live: Sequence[_Call], ref_name: str) -> str:
|
|
1251
|
+
undecided = [call.candidate for call in live if call.status == "inconclusive"]
|
|
1252
|
+
if any(c.logit is None and c.tasks is None for c in undecided):
|
|
1253
|
+
return (
|
|
1254
|
+
"Rerun on a server that returns logprobs, or with json or tools cases the "
|
|
1255
|
+
"reference passes, so there is evidence to judge by."
|
|
1256
|
+
)
|
|
1257
|
+
if any(c.logit is not None and c.logit.interval is None for c in undecided):
|
|
1258
|
+
return (
|
|
1259
|
+
"Rerun with this version of quantdiff, which records the per-prompt KLD needed "
|
|
1260
|
+
"to decide."
|
|
1261
|
+
)
|
|
1262
|
+
return f"Try a larger quant against {ref_name}; nothing tested here is close."
|
|
1263
|
+
|
|
1264
|
+
|
|
1265
|
+
def _run_headline(pick: _Candidate, ref_name: str) -> str:
|
|
1266
|
+
logit = pick.logit
|
|
1267
|
+
if logit is not None and logit.interval is not None:
|
|
1268
|
+
evidence = (
|
|
1269
|
+
f"close on logits (KLD {_kld(logit.mean)}, CI up to {_kld(logit.interval.high)}) "
|
|
1270
|
+
f"on {_plural(logit.prompts, 'prompt')}"
|
|
1271
|
+
)
|
|
1272
|
+
else:
|
|
1273
|
+
cases = 0 if pick.tasks is None else pick.tasks.cases
|
|
1274
|
+
evidence = f"close on task scores on {_plural(cases, 'case')} (no logit metrics)"
|
|
1275
|
+
change = pick.size_change
|
|
1276
|
+
if change is not None and change < -_SAME_SIZE:
|
|
1277
|
+
return f"Run {pick.name}: {_percent(-change)} smaller than {ref_name}, {evidence}."
|
|
1278
|
+
if change is not None and change > _SAME_SIZE:
|
|
1279
|
+
return f"Run {pick.name}: {_percent(change)} larger than {ref_name}, {evidence}."
|
|
1280
|
+
return f"Run {pick.name}: {evidence}."
|
|
1281
|
+
|
|
1282
|
+
|
|
1283
|
+
def _smallest_loss_headline(
|
|
1284
|
+
best: _Call, logit: _LogitEvidence, ref_name: str, *, unsettled: bool
|
|
1285
|
+
) -> str:
|
|
1286
|
+
"""The headline when nothing is close but a candidate has a moderate loss."""
|
|
1287
|
+
close = "shown to be close" if unsettled else "close"
|
|
1288
|
+
change = best.candidate.size_change
|
|
1289
|
+
among = " among the smaller downloads" if change is not None and change < -_SAME_SIZE else ""
|
|
1290
|
+
return (
|
|
1291
|
+
f"No download is {close} to {ref_name}; {best.name} has the smallest loss{among} "
|
|
1292
|
+
f"(moderate, KLD {_kld(logit.mean)})."
|
|
1293
|
+
)
|
|
1294
|
+
|
|
1295
|
+
|
|
1296
|
+
def _run_detail(pick: _Candidate, ref_name: str) -> str:
|
|
1297
|
+
"""The numbers behind a Run headline, as one sentence."""
|
|
1298
|
+
logit, tasks = pick.logit, pick.tasks
|
|
1299
|
+
parts = []
|
|
1300
|
+
if logit is not None and logit.interval is not None:
|
|
1301
|
+
parts.append(f"KLD {_kld(logit.mean)} ({_interval_kld(logit.interval)})")
|
|
1302
|
+
if tasks is not None:
|
|
1303
|
+
parts.append(
|
|
1304
|
+
_tasks_within(tasks, ref_name) if tasks.close else _no_task_loss(tasks, ref_name)
|
|
1305
|
+
)
|
|
1306
|
+
sentence = f"{pick.name}: {' and '.join(parts)}"
|
|
1307
|
+
note = _evidence_note(pick, ref_name)
|
|
1308
|
+
return f"{sentence}; {note}." if note else f"{sentence}."
|
|
1309
|
+
|
|
1310
|
+
|
|
1311
|
+
def _keep_for_now_headline(closest: _Candidate, ref_name: str) -> str:
|
|
1312
|
+
logit, tasks = closest.logit, closest.tasks
|
|
1313
|
+
if logit is None and tasks is None:
|
|
1314
|
+
return f"Keep {ref_name} for now: no candidate has logit or reliable task results."
|
|
1315
|
+
scope = []
|
|
1316
|
+
if logit is not None and logit.prompts:
|
|
1317
|
+
scope.append(_plural(logit.prompts, "prompt"))
|
|
1318
|
+
if tasks is not None:
|
|
1319
|
+
scope.append(_plural(tasks.cases, "case"))
|
|
1320
|
+
on = f" on {' and '.join(scope)}" if scope else ""
|
|
1321
|
+
return f"Keep {ref_name} for now: no candidate is shown to be close{on}."
|
|
1322
|
+
|
|
1323
|
+
|
|
1324
|
+
def _closest_detail(closest: _Candidate, ref_name: str) -> str | None:
|
|
1325
|
+
"""Why the candidate that looks closest is still not proven close."""
|
|
1326
|
+
logit, tasks = closest.logit, closest.tasks
|
|
1327
|
+
looks = []
|
|
1328
|
+
if logit is not None:
|
|
1329
|
+
if logit.interval is None:
|
|
1330
|
+
looks.append(f"KLD {_kld(logit.mean)}, with no per-prompt results to bound it")
|
|
1331
|
+
elif logit.prompts >= MIN_PROMPTS:
|
|
1332
|
+
looks.append(f"KLD {_kld(logit.mean)}, 95% CI up to {_kld(logit.interval.high)}")
|
|
1333
|
+
else:
|
|
1334
|
+
looks.append(f"KLD {_kld(logit.mean)} on {_plural(logit.prompts, 'prompt')}")
|
|
1335
|
+
if tasks is not None:
|
|
1336
|
+
lower = round(-tasks.delta.estimate)
|
|
1337
|
+
level = f"{lower} points lower" if lower > 0 else f"level with {ref_name}"
|
|
1338
|
+
looks.append(f"task scores {level}, 95% CI down to {_signed(tasks.delta.low)}")
|
|
1339
|
+
if not looks:
|
|
1340
|
+
return None
|
|
1341
|
+
return f"{closest.name} looks closest: {'; '.join(looks)}."
|
|
1342
|
+
|
|
1343
|
+
|
|
1344
|
+
def _remedy(call: _Call, settings: RunSettings) -> tuple[str, int] | None:
|
|
1345
|
+
"""What to rerun with to decide an inconclusive candidate, and the --max-cases value."""
|
|
1346
|
+
prompts, cases = call.prompts_needed, call.cases_needed
|
|
1347
|
+
if prompts is None and cases is None:
|
|
1348
|
+
return None
|
|
1349
|
+
flag = max(prompts or 0, cases or 0)
|
|
1350
|
+
wanted = _sentence(_wanted(call)).removesuffix(".")
|
|
1351
|
+
if settings.prompts_file is not None:
|
|
1352
|
+
return f"{wanted} would decide; add them to {settings.prompts_file}.", flag
|
|
1353
|
+
kinds = () if cases is None or call.candidate.tasks is None else call.candidate.tasks.kinds
|
|
1354
|
+
room = [len(load_builtin(kind)) for kind in kinds]
|
|
1355
|
+
if prompts is not None:
|
|
1356
|
+
room.append(len(load_scoring_prompts()))
|
|
1357
|
+
if flag <= min(room):
|
|
1358
|
+
return f"Rerun with --max-cases {flag} to decide.", flag
|
|
1359
|
+
return (
|
|
1360
|
+
f"{wanted} would decide. That is more than the built-in suites hold, so add your "
|
|
1361
|
+
"own with --prompts.",
|
|
1362
|
+
flag,
|
|
1363
|
+
)
|
|
1364
|
+
|
|
1365
|
+
|
|
1366
|
+
# Server findings --------------------------------------------------------------------------
|
|
1367
|
+
|
|
1368
|
+
|
|
1369
|
+
def _findings(report: Report, names: dict[str, str]) -> tuple[ServerFinding, ...]:
|
|
1370
|
+
results = (report.reference, *report.candidates)
|
|
1371
|
+
grouped: dict[PreflightFinding, list[str]] = {}
|
|
1372
|
+
for result in results:
|
|
1373
|
+
failed = {f.check for f in result.preflight if f.severity == "fail"}
|
|
1374
|
+
for finding in result.preflight:
|
|
1375
|
+
if finding.severity not in ("warn", "fail"):
|
|
1376
|
+
continue
|
|
1377
|
+
if finding.severity == "warn" and finding.check in failed:
|
|
1378
|
+
continue
|
|
1379
|
+
labels = grouped.setdefault(finding, [])
|
|
1380
|
+
if result.spec.label not in labels:
|
|
1381
|
+
labels.append(result.spec.label)
|
|
1382
|
+
by_label = {result.spec.label: result for result in results}
|
|
1383
|
+
findings = [
|
|
1384
|
+
ServerFinding(finding, tuple(labels), *_impact(finding, labels, by_label, report))
|
|
1385
|
+
for finding, labels in grouped.items()
|
|
1386
|
+
]
|
|
1387
|
+
findings += _same_weights(report, names)
|
|
1388
|
+
severity = {"fail": 0, "warn": 1}
|
|
1389
|
+
findings.sort(key=lambda f: (not f.affects_scores, severity.get(f.finding.severity, 2)))
|
|
1390
|
+
return tuple(findings)
|
|
1391
|
+
|
|
1392
|
+
|
|
1393
|
+
def _impact(
|
|
1394
|
+
finding: PreflightFinding,
|
|
1395
|
+
labels: Sequence[str],
|
|
1396
|
+
by_label: dict[str, CandidateResult],
|
|
1397
|
+
report: Report,
|
|
1398
|
+
) -> tuple[bool, str]:
|
|
1399
|
+
if finding.check == "context":
|
|
1400
|
+
return _context_impact(labels, by_label, report.settings.longest_prompt_tokens)
|
|
1401
|
+
return _FIXED_IMPACT.get(
|
|
1402
|
+
finding.check, (True, "may affect scores: a serving problem can lower a model's results")
|
|
1403
|
+
)
|
|
1404
|
+
|
|
1405
|
+
|
|
1406
|
+
_FIXED_IMPACT: Final[dict[str, tuple[bool, str]]] = {
|
|
1407
|
+
"template": (True, "affects these scores: every task case runs through the chat template"),
|
|
1408
|
+
"tokenizer": (True, "affects these scores: logit positions do not line up with the reference"),
|
|
1409
|
+
"logprobs": (
|
|
1410
|
+
False,
|
|
1411
|
+
"does not affect these scores: logit metrics are missing, so only task results count",
|
|
1412
|
+
),
|
|
1413
|
+
}
|
|
1414
|
+
|
|
1415
|
+
|
|
1416
|
+
def _context_impact(
|
|
1417
|
+
labels: Sequence[str], by_label: dict[str, CandidateResult], longest: int | None
|
|
1418
|
+
) -> tuple[bool, str]:
|
|
1419
|
+
if longest is None:
|
|
1420
|
+
return True, "may affect scores: the length of the longest prompt is not known"
|
|
1421
|
+
infos = [by_label[label].info for label in labels]
|
|
1422
|
+
known = [info.context_length for info in infos if info and info.context_length]
|
|
1423
|
+
if len(known) < len(infos):
|
|
1424
|
+
return True, "may affect scores: the context window is not known"
|
|
1425
|
+
window = min(known)
|
|
1426
|
+
if window >= longest:
|
|
1427
|
+
return False, (
|
|
1428
|
+
f"does not affect these scores: the longest prompt is ~{longest} tokens and "
|
|
1429
|
+
f"the context window is {window}"
|
|
1430
|
+
)
|
|
1431
|
+
return True, (
|
|
1432
|
+
f"may affect scores: the longest prompt is ~{longest} tokens but the context "
|
|
1433
|
+
f"window is {window}"
|
|
1434
|
+
)
|
|
1435
|
+
|
|
1436
|
+
|
|
1437
|
+
def _same_weights(report: Report, names: dict[str, str]) -> list[ServerFinding]:
|
|
1438
|
+
info = report.reference.info
|
|
1439
|
+
reference_id = None if info is None else info.weights_id
|
|
1440
|
+
if reference_id is None:
|
|
1441
|
+
return []
|
|
1442
|
+
return [
|
|
1443
|
+
ServerFinding(
|
|
1444
|
+
finding=PreflightFinding(
|
|
1445
|
+
check="same-weights",
|
|
1446
|
+
severity="warn",
|
|
1447
|
+
message=f"{names[result.spec.label]} serves the same weights as the reference",
|
|
1448
|
+
fix="Point the reference at a different download, such as the BF16 or Q8_0 file.",
|
|
1449
|
+
),
|
|
1450
|
+
labels=(result.spec.label,),
|
|
1451
|
+
affects_scores=True,
|
|
1452
|
+
impact="affects these scores: the candidate is being compared with itself",
|
|
1453
|
+
)
|
|
1454
|
+
for result in report.candidates
|
|
1455
|
+
if result.info is not None and result.info.weights_id == reference_id
|
|
1456
|
+
]
|
|
1457
|
+
|
|
1458
|
+
|
|
1459
|
+
# Formatting -------------------------------------------------------------------------------
|
|
1460
|
+
|
|
1461
|
+
|
|
1462
|
+
def _sentence(phrase: str) -> str:
|
|
1463
|
+
# Suite names are identifiers ("json", "tools") and stay lowercase at a sentence start.
|
|
1464
|
+
starts_with_suite = phrase.split(" ", 1)[0] in SCORED_TASK_KINDS
|
|
1465
|
+
text = phrase if starts_with_suite else phrase[:1].upper() + phrase[1:]
|
|
1466
|
+
return text if text.endswith(".") else text + "."
|
|
1467
|
+
|
|
1468
|
+
|
|
1469
|
+
def _percent(fraction: float) -> str:
|
|
1470
|
+
return f"{fraction * 100:.0f}%"
|
|
1471
|
+
|
|
1472
|
+
|
|
1473
|
+
def _kld(value: float) -> str:
|
|
1474
|
+
"""Up to three decimals from 0.01 up, which keeps values near the 0.05 bar readable;
|
|
1475
|
+
two significant digits below that."""
|
|
1476
|
+
if value >= 10:
|
|
1477
|
+
return f"{value:.0f}"
|
|
1478
|
+
if value >= 0.01:
|
|
1479
|
+
return f"{value:.3f}".rstrip("0").rstrip(".")
|
|
1480
|
+
if 0 < value < 0.0001:
|
|
1481
|
+
return "under 0.0001"
|
|
1482
|
+
return f"{value:.2g}"
|
|
1483
|
+
|
|
1484
|
+
|
|
1485
|
+
def _interval_kld(interval: Interval) -> str:
|
|
1486
|
+
return f"95% CI {_kld(interval.low)} to {_kld(interval.high)}"
|
|
1487
|
+
|
|
1488
|
+
|
|
1489
|
+
def _interval_points(interval: Interval) -> str:
|
|
1490
|
+
return f"95% CI {_signed(interval.low)} to {_signed(interval.high)}"
|
|
1491
|
+
|
|
1492
|
+
|
|
1493
|
+
def _signed(points: float) -> str:
|
|
1494
|
+
rounded = round(points)
|
|
1495
|
+
return f"+{rounded}" if rounded > 0 else str(rounded)
|
|
1496
|
+
|
|
1497
|
+
|
|
1498
|
+
def _points(points: float) -> str:
|
|
1499
|
+
return f"{points:.0f}"
|
|
1500
|
+
|
|
1501
|
+
|
|
1502
|
+
def _round_up(count: int) -> int:
|
|
1503
|
+
return math.ceil(count / _ROUND_NEEDED_TO) * _ROUND_NEEDED_TO
|
|
1504
|
+
|
|
1505
|
+
|
|
1506
|
+
def _plural(count: int, noun: str) -> str:
|
|
1507
|
+
return f"{count} {noun}" if count == 1 else f"{count} {noun}s"
|
|
1508
|
+
|
|
1509
|
+
|
|
1510
|
+
def _join(items: Sequence[str]) -> str:
|
|
1511
|
+
if len(items) <= 2:
|
|
1512
|
+
return " and ".join(items)
|
|
1513
|
+
return ", ".join(items[:-1]) + " and " + items[-1]
|