quantdiff 0.1.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. quantdiff/__init__.py +53 -0
  2. quantdiff/__main__.py +5 -0
  3. quantdiff/_http.py +151 -0
  4. quantdiff/_text.py +13 -0
  5. quantdiff/_version.py +1 -0
  6. quantdiff/api.py +340 -0
  7. quantdiff/backends/__init__.py +28 -0
  8. quantdiff/backends/_common.py +342 -0
  9. quantdiff/backends/base.py +91 -0
  10. quantdiff/backends/llamacpp.py +428 -0
  11. quantdiff/backends/ollama.py +359 -0
  12. quantdiff/backends/openai_compat.py +338 -0
  13. quantdiff/cache.py +240 -0
  14. quantdiff/card.py +1664 -0
  15. quantdiff/cli.py +377 -0
  16. quantdiff/discover.py +488 -0
  17. quantdiff/errors.py +45 -0
  18. quantdiff/metrics/__init__.py +36 -0
  19. quantdiff/metrics/codeexec.py +428 -0
  20. quantdiff/metrics/jsonschema.py +610 -0
  21. quantdiff/metrics/logit.py +214 -0
  22. quantdiff/metrics/tasks.py +114 -0
  23. quantdiff/metrics/textsim.py +66 -0
  24. quantdiff/metrics/toolcheck.py +99 -0
  25. quantdiff/png.py +360 -0
  26. quantdiff/preflight.py +365 -0
  27. quantdiff/progress.py +283 -0
  28. quantdiff/py.typed +0 -0
  29. quantdiff/report.py +780 -0
  30. quantdiff/runner.py +492 -0
  31. quantdiff/spec.py +154 -0
  32. quantdiff/stats.py +226 -0
  33. quantdiff/suites/__init__.py +462 -0
  34. quantdiff/suites/data/chat.jsonl +22 -0
  35. quantdiff/suites/data/code.jsonl +32 -0
  36. quantdiff/suites/data/json.jsonl +34 -0
  37. quantdiff/suites/data/scoring.jsonl +41 -0
  38. quantdiff/suites/data/tools.jsonl +32 -0
  39. quantdiff/types.py +322 -0
  40. quantdiff/verdict.py +1513 -0
  41. quantdiff-0.1.0rc1.dist-info/METADATA +514 -0
  42. quantdiff-0.1.0rc1.dist-info/RECORD +45 -0
  43. quantdiff-0.1.0rc1.dist-info/WHEEL +4 -0
  44. quantdiff-0.1.0rc1.dist-info/entry_points.txt +2 -0
  45. quantdiff-0.1.0rc1.dist-info/licenses/LICENSE +202 -0
quantdiff/card.py ADDED
@@ -0,0 +1,1664 @@
1
+ """Scorecards: one recommendation and one table, rendered for a terminal, markdown, and HTML.
2
+
3
+ Every card answers "which download should I run?" first: the verdict headline leads, and
4
+ each candidate row carries the status the verdict engine gave it (RUN, OK, USABLE, UNSURE,
5
+ AVOID, FAILED), in the engine's order, under the reference row, with any caveats the engine
6
+ attached. All three renderers share one
7
+ column model so a number never differs between formats. Columns with nothing to show are
8
+ left out. The HTML card is a single self-contained document with no scripts and no
9
+ external assets, so it can be opened offline, screenshotted, and attached anywhere.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import html
15
+ import textwrap
16
+ from collections.abc import Callable, Sequence
17
+ from dataclasses import dataclass
18
+ from pathlib import PureWindowsPath
19
+ from typing import Final, Literal
20
+
21
+ from quantdiff.report import format_kld, format_rate, format_top1, task_for, text_forced_note
22
+ from quantdiff.suites import BUILTIN_SUITES, load_builtin
23
+ from quantdiff.types import BackendKind, CandidateResult, Report, TaskKind
24
+ from quantdiff.verdict import (
25
+ CALIBRATED_TOP_K,
26
+ LABEL_SEPARATORS,
27
+ CandidateVerdict,
28
+ KldThresholds,
29
+ ServerFinding,
30
+ Status,
31
+ TaskDelta,
32
+ Verdict,
33
+ display_labels,
34
+ judge,
35
+ kld_band,
36
+ kld_thresholds,
37
+ )
38
+
39
+ MISSING: Final = "n/a"
40
+ BASELINE: Final = "-"
41
+ """Shown in the reference row for metrics that compare a model against the reference."""
42
+ TERMINAL_WIDTH: Final = 100
43
+ P99_MIN_POSITIONS: Final = 1000
44
+ """Fewer scored positions than this make a 99th percentile little more than the maximum."""
45
+ SIGNIFICANT_MARK: Final = "*"
46
+ SIGNIFICANT_FOOTNOTE: Final = "* significant (paired, 95%)"
47
+ REFERENCE_CHIP: Final = "REF"
48
+ NOTE_CHIP: Final = "NOTE"
49
+ """Chip for a server finding that does not affect the scores on the card."""
50
+ USER_PROMPTS: Final = "your prompts"
51
+ STATUS_CHIPS: Final[dict[Status, str]] = {
52
+ "recommended": "RUN",
53
+ "ok": "OK",
54
+ "usable": "USABLE",
55
+ "avoid": "AVOID",
56
+ "inconclusive": "UNSURE",
57
+ "failed": "FAILED",
58
+ }
59
+ NEAR_BAR: Final = "near bar"
60
+ """Tag next to the chip of a candidate whose KLD sits close enough to a band edge that a
61
+ rerun could change its status."""
62
+ CAVEAT_PREFIX: Final = "caveat:"
63
+ OVER_BUDGET: Final = "over budget"
64
+ OVER_BUDGET_SHORT: Final = "(over)"
65
+ CACHED: Final = "cached"
66
+ """Marker on a reference speed that comes from an earlier run, not this one."""
67
+ _LABEL_MIN_WIDTH: Final = 12
68
+ _GAP: Final = " "
69
+ _ELLIPSIS: Final = "..."
70
+ _SEVERITY_ORDER: Final = ("fail", "warn", "skip", "ok")
71
+
72
+ Group = Literal["logit", "task"]
73
+ Bar = Literal["rate", "kld"]
74
+ Align = Literal["left", "right"]
75
+ SpeedMode = Literal["server", "wall"]
76
+
77
+
78
+ # Column model -----------------------------------------------------------------------------
79
+
80
+
81
+ @dataclass(frozen=True, slots=True)
82
+ class _Metric:
83
+ short: str
84
+ """Terminal header."""
85
+ name: str
86
+ """Markdown and HTML header."""
87
+ group: Group
88
+ value: Callable[[CandidateResult], float | None]
89
+ text: Callable[[float], str]
90
+ bar: Bar
91
+ vs_reference: bool
92
+ """The metric compares a model with the reference, so the reference row omits it."""
93
+ task: TaskKind | None = None
94
+ """The suite of a pass-rate column, whose candidate cells carry the paired delta."""
95
+
96
+
97
+ def _top1(result: CandidateResult) -> float | None:
98
+ return None if result.logit is None else result.logit.top1_agreement
99
+
100
+
101
+ def _kld_mean(result: CandidateResult) -> float | None:
102
+ return None if result.logit is None else result.logit.kld_mean
103
+
104
+
105
+ def _kld_p99(result: CandidateResult) -> float | None:
106
+ logit = result.logit
107
+ if logit is None or logit.positions < P99_MIN_POSITIONS:
108
+ return None
109
+ return logit.kld_p99
110
+
111
+
112
+ def _task_rate(kind: TaskKind) -> Callable[[CandidateResult], float | None]:
113
+ def read(result: CandidateResult) -> float | None:
114
+ task = task_for(result, kind)
115
+ return None if task is None else task.rate
116
+
117
+ return read
118
+
119
+
120
+ def _agreement(result: CandidateResult) -> float | None:
121
+ return None if result.agreement is None else result.agreement.mean_similarity
122
+
123
+
124
+ def _task_metric(kind: TaskKind, name: str) -> _Metric:
125
+ return _Metric(
126
+ name, name, "task", _task_rate(kind), format_rate, "rate", vs_reference=False, task=kind
127
+ )
128
+
129
+
130
+ _METRICS: Final = (
131
+ _Metric("Top-1", "Top-1", "logit", _top1, format_top1, "rate", vs_reference=True),
132
+ _Metric("KLD", "KLD mean", "logit", _kld_mean, format_kld, "kld", vs_reference=True),
133
+ _Metric("KLD99", "KLD p99", "logit", _kld_p99, format_kld, "kld", vs_reference=True),
134
+ _task_metric("json", "JSON"),
135
+ _task_metric("tools", "Tools"),
136
+ _task_metric("code", "Code"),
137
+ _Metric("Agree", "Agree", "task", _agreement, format_rate, "rate", vs_reference=True),
138
+ )
139
+ _SUITE_NAMES: Final[dict[TaskKind, str]] = {
140
+ "json": "JSON",
141
+ "tools": "Tools",
142
+ "code": "Code",
143
+ "chat": "Chat",
144
+ }
145
+
146
+
147
+ def format_size(size_bytes: int) -> str:
148
+ """Decimal megabytes or gigabytes, the way model hubs and Ollama print file sizes."""
149
+ if size_bytes >= 1_000_000_000:
150
+ return f"{size_bytes / 1e9:.1f} GB"
151
+ megabytes = size_bytes / 1e6
152
+ return f"{megabytes:.1f} MB" if megabytes < 10 else f"{megabytes:.0f} MB"
153
+
154
+
155
+ def format_size_change(change: float) -> str:
156
+ """A fractional size change as a signed whole percentage, such as "-25%" or "0%"."""
157
+ percent = round(change * 100)
158
+ return f"{percent:+d}%" if percent else "0%"
159
+
160
+
161
+ def _points(value: float) -> str:
162
+ rounded = round(value)
163
+ return f"{rounded:+d}" if rounded else "0"
164
+
165
+
166
+ def format_delta(delta: TaskDelta) -> str:
167
+ """Compact paired delta in points, such as "-30*" (significant) or "-5".
168
+
169
+ A gain or tie that is not significant reads "=", so a lucky case or two never looks like
170
+ a candidate beating the reference.
171
+ """
172
+ if delta.significant:
173
+ return _points(delta.delta.estimate) + SIGNIFICANT_MARK
174
+ return "=" if round(delta.delta.estimate) >= 0 else _points(delta.delta.estimate)
175
+
176
+
177
+ def format_interval(delta: TaskDelta) -> str:
178
+ """The compact delta with its 95% interval, such as "-30* [-48, -12]" or "= [-5, +25]"."""
179
+ interval = delta.delta
180
+ return f"{format_delta(delta)} [{_points(interval.low)}, {_points(interval.high)}]"
181
+
182
+
183
+ def is_regression(delta: TaskDelta) -> bool:
184
+ """A significant loss against a reference that is reliable on this suite."""
185
+ return delta.significant and delta.reference_reliable and delta.delta.estimate < 0
186
+
187
+
188
+ def _format_speed(value: float) -> str:
189
+ return f"{value:.0f}" if value >= 100 else f"{value:.1f}"
190
+
191
+
192
+ @dataclass(frozen=True, slots=True)
193
+ class _Row:
194
+ """One table row: the reference, or a candidate with the verdict engine's call on it."""
195
+
196
+ result: CandidateResult
197
+ label: str
198
+ call: CandidateVerdict | None
199
+ """None for the reference row."""
200
+
201
+ @property
202
+ def is_reference(self) -> bool:
203
+ return self.call is None
204
+
205
+ @property
206
+ def chip(self) -> str:
207
+ return REFERENCE_CHIP if self.call is None else STATUS_CHIPS[self.call.status]
208
+
209
+ @property
210
+ def status(self) -> Status | None:
211
+ return None if self.call is None else self.call.status
212
+
213
+ def delta(self, kind: TaskKind | None) -> TaskDelta | None:
214
+ if self.call is None or kind is None:
215
+ return None
216
+ return next((d for d in self.call.task_deltas if d.kind == kind), None)
217
+
218
+ @property
219
+ def near_bar(self) -> bool:
220
+ return self.call is not None and self.call.near_bar
221
+
222
+ @property
223
+ def over_budget(self) -> bool:
224
+ return self.call is not None and self.call.fits_budget is False
225
+
226
+ @property
227
+ def size_bytes(self) -> int | None:
228
+ if self.call is not None and self.call.size_bytes is not None:
229
+ return self.call.size_bytes
230
+ return None if self.result.info is None else self.result.info.size_bytes
231
+
232
+
233
+ @dataclass(frozen=True, slots=True)
234
+ class _Cell:
235
+ text: str
236
+ value: float | None = None
237
+ """The number behind `text`, for bars; None for placeholders."""
238
+ delta: TaskDelta | None = None
239
+ muted: bool = False
240
+ marker: str = ""
241
+ """A short word shown after the value, such as "cached"."""
242
+
243
+
244
+ @dataclass(frozen=True, slots=True)
245
+ class _View:
246
+ """Everything the three renderers share for one report."""
247
+
248
+ report: Report
249
+ verdict: Verdict
250
+ rows: tuple[_Row, ...]
251
+ prefix: str
252
+ backend: BackendKind | None
253
+ """The backend every model ran on, or None when they differ."""
254
+ metrics: tuple[_Metric, ...]
255
+ """Metric columns that have at least one value to show."""
256
+ unreliable: frozenset[TaskKind]
257
+ """Suites where the reference passes too few cases to judge a regression."""
258
+ speed: SpeedMode | None
259
+ show_size: bool
260
+
261
+ @property
262
+ def reference(self) -> _Row:
263
+ return self.rows[0]
264
+
265
+ @property
266
+ def candidates(self) -> tuple[_Row, ...]:
267
+ return self.rows[1:]
268
+
269
+ @property
270
+ def thresholds(self) -> KldThresholds:
271
+ return kld_thresholds(self.report.settings.top_k)
272
+
273
+ @property
274
+ def model_name(self) -> str:
275
+ return self.prefix.rstrip(LABEL_SEPARATORS)
276
+
277
+ @property
278
+ def has_significant(self) -> bool:
279
+ return any(d.significant for row in self.candidates for d in _row_deltas(self, row))
280
+
281
+ def context(self) -> list[tuple[str, str, str]]:
282
+ """Header facts as (name, value, backend) triples; backend is "" when not shown."""
283
+ reference = self.reference.result.spec
284
+ items = []
285
+ if self.prefix:
286
+ items.append(("Model", self.model_name, ""))
287
+ items.append(("Reference", reference.label, "" if self.backend else reference.kind))
288
+ if self.backend:
289
+ items.append(("Backend", self.backend, ""))
290
+ budget = budget_text(self.report)
291
+ if budget is not None:
292
+ items.append(("Budget", budget, ""))
293
+ return items
294
+
295
+ def cell(self, metric: _Metric, row: _Row) -> _Cell:
296
+ if row.is_reference and metric.vs_reference:
297
+ return _Cell(BASELINE)
298
+ value = metric.value(row.result)
299
+ if value is None:
300
+ return _Cell(MISSING)
301
+ muted = metric.task is not None and metric.task in self.unreliable
302
+ return _Cell(metric.text(value), value, row.delta(metric.task), muted)
303
+
304
+ def speed_cell(self, row: _Row) -> _Cell:
305
+ perf = row.result.perf
306
+ if perf is None or perf.tokens_per_second is None:
307
+ return _Cell(MISSING)
308
+ text = _format_speed(perf.tokens_per_second)
309
+ if row.is_reference and perf.source == "cached":
310
+ return _Cell(text, perf.tokens_per_second, muted=True, marker=CACHED)
311
+ return _Cell(text, perf.tokens_per_second, muted=self.speed == "wall")
312
+
313
+ def size_cell(self, row: _Row) -> tuple[str, str]:
314
+ """Size text and its change against the reference, such as ("398 MB", "-25%")."""
315
+ size = row.size_bytes
316
+ if size is None:
317
+ return MISSING, ""
318
+ change = None if row.call is None else row.call.size_change
319
+ return format_size(size), "" if change is None else format_size_change(change)
320
+
321
+ def covers_everyone(self, finding: ServerFinding) -> bool:
322
+ """True when a finding applies to every model on a card with more than one."""
323
+ everyone = {row.result.spec.label for row in self.rows}
324
+ return len(everyone) > 1 and everyone <= set(finding.labels)
325
+
326
+ def who(self, finding: ServerFinding) -> str:
327
+ """The models a finding applies to: "both models" or "all N models" when it is every
328
+ one of them, otherwise their short labels."""
329
+ if self.covers_everyone(finding):
330
+ count = len(self.rows)
331
+ return "both models" if count == 2 else f"all {count} models"
332
+ names = {row.result.spec.label: row.label for row in self.rows}
333
+ return ", ".join(names.get(label, label) for label in dict.fromkeys(finding.labels))
334
+
335
+
336
+ def _row_deltas(view: _View, row: _Row) -> list[TaskDelta]:
337
+ """Deltas the table actually shows for this row."""
338
+ return [delta for metric in view.metrics if (delta := view.cell(metric, row).delta) is not None]
339
+
340
+
341
+ def _view(report: Report, verdict: Verdict) -> _View:
342
+ results = (report.reference, *report.candidates)
343
+ names = display_labels(report)
344
+ reference = report.reference.spec.label
345
+ # display_labels drops the stem every label shares; the header states it once instead.
346
+ prefix = reference[: len(reference) - len(names[reference])]
347
+ by_label = {result.spec.label: result for result in report.candidates}
348
+ rows = [_Row(report.reference, names[reference], None)]
349
+ rows += [
350
+ _Row(by_label[call.label], names[call.label], call)
351
+ for call in verdict.candidates
352
+ if call.label in by_label
353
+ ]
354
+ backends = {result.spec.kind for result in results}
355
+ unreliable = frozenset(
356
+ delta.kind
357
+ for call in verdict.candidates
358
+ for delta in call.task_deltas
359
+ if not delta.reference_reliable
360
+ )
361
+ metrics = tuple(
362
+ metric
363
+ for metric in _METRICS
364
+ if any(
365
+ metric.value(row.result) is not None
366
+ for row in rows
367
+ if not (row.is_reference and metric.vs_reference)
368
+ )
369
+ )
370
+ return _View(
371
+ report=report,
372
+ verdict=verdict,
373
+ rows=tuple(rows),
374
+ prefix=prefix,
375
+ backend=backends.pop() if len(backends) == 1 else None,
376
+ metrics=metrics,
377
+ unreliable=unreliable,
378
+ speed=_speed_mode(rows),
379
+ show_size=any(row.size_bytes is not None for row in rows),
380
+ )
381
+
382
+
383
+ def _speed_mode(rows: Sequence[_Row]) -> SpeedMode | None:
384
+ """ "server" when every shown speed is the server's own decode timing, "wall" when any is
385
+ wall-clock or a cached candidate speed, None when there is nothing to show. A cached
386
+ reference speed is shown muted and marked "cached", and does not set the mode."""
387
+ sources = [
388
+ (row.is_reference, perf.source)
389
+ for row in rows
390
+ if (perf := row.result.perf) is not None and perf.tokens_per_second is not None
391
+ ]
392
+ if not sources:
393
+ return None
394
+ fresh = {source for reference, source in sources if not (reference and source == "cached")}
395
+ return "server" if fresh <= {"server"} else "wall"
396
+
397
+
398
+ def _cached_reference(view: _View) -> bool:
399
+ perf = view.reference.result.perf
400
+ return (
401
+ view.speed is not None
402
+ and perf is not None
403
+ and perf.tokens_per_second is not None
404
+ and perf.source == "cached"
405
+ )
406
+
407
+
408
+ def _speed_header(view: _View) -> str:
409
+ return "tok/s (wall)" if view.speed == "wall" else "tok/s"
410
+
411
+
412
+ def finding_chip(item: ServerFinding) -> str:
413
+ """FAIL or WARN for a finding that may have changed the scores, NOTE for one that did
414
+ not, so a harmless finding never looks like a failure on a shared card."""
415
+ return item.finding.severity.upper() if item.affects_scores else NOTE_CHIP
416
+
417
+
418
+ def _findings(view: _View) -> list[ServerFinding]:
419
+ """Findings that could change the scores first, then by severity."""
420
+ return sorted(
421
+ view.verdict.findings,
422
+ key=lambda f: (not f.affects_scores, _SEVERITY_ORDER.index(f.finding.severity)),
423
+ )
424
+
425
+
426
+ @dataclass(frozen=True, slots=True)
427
+ class _Lead:
428
+ """The verdict box: the headline, the remedy for an inconclusive verdict, the rest."""
429
+
430
+ headline: str
431
+ remedy: str | None
432
+ details: tuple[str, ...]
433
+
434
+
435
+ def _lead(verdict: Verdict) -> _Lead:
436
+ """The headline, the remedy (what to rerun to settle an inconclusive verdict) on its own
437
+ line, and the remaining details."""
438
+ return _Lead(verdict.headline, verdict.remedy, verdict.details)
439
+
440
+
441
+ def _score_alerts(view: _View) -> list[str]:
442
+ """One line per finding that may have changed the scores, for the verdict box."""
443
+ return [
444
+ f"Server check on {view.who(f)}: {_sentence(f.finding.message, capitalize=False)} "
445
+ "This may affect the scores; see Server checks."
446
+ for f in _findings(view)
447
+ if f.affects_scores
448
+ ]
449
+
450
+
451
+ def _case_counts(report: Report) -> list[tuple[str, int]]:
452
+ counts: dict[str, int] = {}
453
+ for result in (report.reference, *report.candidates):
454
+ for task in result.tasks:
455
+ counts[task.kind] = max(counts.get(task.kind, 0), task.total)
456
+ if result.agreement is not None:
457
+ counts["chat"] = max(counts.get("chat", 0), result.agreement.cases)
458
+ order = ("json", "tools", "code", "chat")
459
+ return [(kind, counts[kind]) for kind in order if kind in counts]
460
+
461
+
462
+ def _plural(count: int, noun: str) -> str:
463
+ return f"{count} {noun}" if count == 1 else f"{count} {noun}s"
464
+
465
+
466
+ def summary_line(report: Report) -> str:
467
+ """One line of scale, such as "5 models, 115 cases, 24 scoring prompts"."""
468
+ parts = [_plural(1 + len(report.candidates), "model")]
469
+ cases = sum(count for _, count in _case_counts(report))
470
+ if cases:
471
+ parts.append(_plural(cases, "case"))
472
+ prompts = max((c.logit.prompts for c in report.candidates if c.logit is not None), default=0)
473
+ if prompts:
474
+ parts.append(_plural(prompts, "scoring prompt"))
475
+ return ", ".join(parts)
476
+
477
+
478
+ def prompts_file_name(value: str) -> str:
479
+ """The base name of a prompts file. Cards are shared publicly, so never a full path,
480
+ whichever separator the path was written with."""
481
+ return PureWindowsPath(value).name
482
+
483
+
484
+ def _user_case_count(report: Report) -> int | None:
485
+ """How many cases came from the user's prompts file, or None when the report does not
486
+ say. Built-in suites and the prompts file never share a case id, so the user's cases
487
+ are the ones outside every built-in suite that ran."""
488
+ if not report.settings.suites:
489
+ return sum(count for _, count in _case_counts(report)) or None
490
+ seen: set[str] = set()
491
+ for result in (report.reference, *report.candidates):
492
+ seen.update(outcome.case_id for outcome in result.outcomes)
493
+ if result.agreement is not None:
494
+ seen.update(case_id for case_id, _ in result.agreement.per_case)
495
+ if not seen:
496
+ return None
497
+ builtin = {
498
+ case.id
499
+ for name in report.settings.suites
500
+ if name in BUILTIN_SUITES
501
+ for case in load_builtin(name)
502
+ }
503
+ return len(seen - builtin)
504
+
505
+
506
+ def _suites_text(report: Report) -> str:
507
+ """The suites that ran, such as "json, tools + your prompts (12)"."""
508
+ settings = report.settings
509
+ builtin = ", ".join(settings.suites)
510
+ if not settings.prompts_file:
511
+ return builtin or "none"
512
+ count = _user_case_count(report)
513
+ mine = USER_PROMPTS if count is None else f"{USER_PROMPTS} ({count})"
514
+ return f"{builtin} + {mine}" if builtin else mine
515
+
516
+
517
+ def budget_text(report: Report) -> str | None:
518
+ """The --max-size budget as a size, such as "6.0 GB", or None when none was given."""
519
+ budget = report.settings.max_size_bytes
520
+ return None if budget is None else format_size(budget)
521
+
522
+
523
+ def _settings_items(report: Report) -> list[tuple[str, str]]:
524
+ settings = report.settings
525
+ cases = ", ".join(f"{kind} {count}" for kind, count in _case_counts(report))
526
+ items = [
527
+ ("Suites", _suites_text(report)),
528
+ ("Top-k", str(settings.top_k)),
529
+ ("Score tokens", str(settings.score_tokens)),
530
+ ("Cases", cases or "none"),
531
+ ("Code execution", "on" if settings.allow_code_exec else "off"),
532
+ ("Seed", str(settings.seed)),
533
+ ]
534
+ budget = budget_text(report)
535
+ if budget is not None:
536
+ items.append(("Budget", budget))
537
+ if settings.prompts_file:
538
+ items.append(("Prompts file", prompts_file_name(settings.prompts_file)))
539
+ return items
540
+
541
+
542
+ def _settings_line(report: Report) -> str:
543
+ return "; ".join(f"{name} {value}" for name, value in _settings_items(report))
544
+
545
+
546
+ def _notes(view: _View) -> list[str]:
547
+ """Short footnotes that explain how to read this particular card."""
548
+ report = view.report
549
+ shown = {metric.name for metric in view.metrics}
550
+ notes = []
551
+ if any(row.delta(metric.task) for row in view.candidates for metric in view.metrics):
552
+ notes.append(
553
+ "Signed numbers are percentage points against the reference on the same cases."
554
+ )
555
+ if shown & {"KLD mean", "KLD p99"}:
556
+ notes.append(
557
+ f"KLD is a lower bound computed from the top {report.settings.top_k} tokens; "
558
+ "lower means closer to the reference."
559
+ )
560
+ notes.append(kld_band_note(report.settings.top_k))
561
+ approximate = [
562
+ row
563
+ for row in view.candidates
564
+ if row.result.logit is not None and not row.result.logit.exact_token_ids
565
+ ]
566
+ if approximate:
567
+ ollama = all(row.result.spec.kind == "ollama" for row in approximate)
568
+ notes.append(text_forced_note([row.label for row in approximate], ollama=ollama))
569
+ for kind in sorted(view.unreliable):
570
+ name = _SUITE_NAMES[kind]
571
+ if name in shown:
572
+ notes.append(
573
+ f"{name} says little here: the reference itself passes under half of these "
574
+ "cases, so this suite is not used to judge the candidates."
575
+ )
576
+ if "Agree" in shown:
577
+ notes.append("Agree is how similar chat answers are to the reference's answers.")
578
+ if view.speed == "wall":
579
+ notes.append(
580
+ "tok/s (wall) is timed from outside the server and includes prompt processing; "
581
+ "treat it as rough."
582
+ )
583
+ if _cached_reference(view):
584
+ notes.append(
585
+ "The reference tok/s marked cached comes from an earlier run; compare it loosely."
586
+ )
587
+ code = [task_for(row.result, "code") for row in view.rows]
588
+ if not report.settings.allow_code_exec and any(t is not None and t.skipped for t in code):
589
+ notes.append("Code cases were skipped; add --allow-code-exec to score them.")
590
+ notes.extend(report.notes)
591
+ return notes
592
+
593
+
594
+ def kld_band_note(top_k: int = CALIBRATED_TOP_K) -> str:
595
+ """The verdict engine's KLD bands for this --top-k in words, built from its own
596
+ thresholds so the card can never disagree with the verdict."""
597
+ bars = kld_thresholds(top_k)
598
+ parts = []
599
+ floor = 0.0
600
+ for limit in (bars.near_lossless, bars.close, bars.large):
601
+ tag = " (the bar for RUN)" if limit == bars.close else ""
602
+ parts.append(f"under {limit:g} {kld_band(floor, bars)}{tag}")
603
+ floor = limit
604
+ return (
605
+ f"KLD bands (top-{top_k}, calibrated against llama.cpp full-vocabulary KLD): "
606
+ f"{', '.join(parts)}, else {kld_band(floor, bars)}."
607
+ )
608
+
609
+
610
+ def _reasons(view: _View) -> list[tuple[_Row, CandidateVerdict]]:
611
+ """Candidates whose call comes with reasons or caveats, in verdict order."""
612
+ return [
613
+ (row, row.call)
614
+ for row in view.candidates
615
+ if row.call is not None and (row.call.reasons or row.call.caveats)
616
+ ]
617
+
618
+
619
+ def _tagged_chip(row: _Row) -> str:
620
+ """The chip with the near-bar tag, for the plain-text formats."""
621
+ return f"{row.chip} ({NEAR_BAR})" if row.near_bar else row.chip
622
+
623
+
624
+ def _errors(view: _View) -> list[tuple[str, str]]:
625
+ return [(row.label, error) for row in view.candidates for error in row.result.errors]
626
+
627
+
628
+ # Terminal ---------------------------------------------------------------------------------
629
+
630
+ _RESET: Final = "\x1b[0m"
631
+ _BOLD: Final = "1"
632
+ _DIM: Final = "2"
633
+ _ITALIC: Final = "3"
634
+ _GREEN: Final = "32"
635
+ _YELLOW: Final = "33"
636
+ _RED: Final = "31"
637
+ _BLUE: Final = "34"
638
+ _STATUS_CODES: Final[dict[Status, tuple[str, ...]]] = {
639
+ "recommended": (_BOLD, _GREEN),
640
+ "ok": (),
641
+ "usable": (_BOLD, _BLUE),
642
+ "avoid": (_BOLD, _RED),
643
+ "inconclusive": (_YELLOW,),
644
+ "failed": (_RED,),
645
+ }
646
+ _SEVERITY_CODES: Final[dict[str, tuple[str, ...]]] = {
647
+ "fail": (_RED,),
648
+ "warn": (_YELLOW,),
649
+ "skip": (_DIM,),
650
+ "ok": (_DIM,),
651
+ }
652
+
653
+
654
+ @dataclass(frozen=True, slots=True)
655
+ class _Ansi:
656
+ enabled: bool
657
+
658
+ def __call__(self, text: str, *codes: str) -> str:
659
+ if not self.enabled or not codes:
660
+ return text
661
+ return f"\x1b[{';'.join(codes)}m{text}{_RESET}"
662
+
663
+
664
+ @dataclass(frozen=True, slots=True)
665
+ class _TermCell:
666
+ text: str
667
+ delta: str = ""
668
+ codes: tuple[str, ...] = ()
669
+ delta_codes: tuple[str, ...] = ()
670
+
671
+
672
+ @dataclass(frozen=True, slots=True)
673
+ class _TermColumn:
674
+ header: str
675
+ align: Align
676
+ cells: tuple[_TermCell, ...]
677
+
678
+ @property
679
+ def text_width(self) -> int:
680
+ return max((len(cell.text) for cell in self.cells), default=0)
681
+
682
+ @property
683
+ def delta_width(self) -> int:
684
+ return max((len(cell.delta) for cell in self.cells), default=0)
685
+
686
+ @property
687
+ def width(self) -> int:
688
+ delta = self.delta_width
689
+ return max(len(self.header), self.text_width + (delta + 1 if delta else 0))
690
+
691
+ def render(self, cell: _TermCell, width: int, paint: _Ansi, row_codes: tuple[str, ...]) -> str:
692
+ if not self.delta_width:
693
+ text = _pad(_truncate(cell.text, width), width, self.align)
694
+ return paint(text, *row_codes, *cell.codes)
695
+ # Values and deltas get sub-columns so the numbers line up across rows.
696
+ used = self.text_width + 1 + self.delta_width
697
+ lead = " " * (width - used) + cell.text.rjust(self.text_width)
698
+ tail = cell.delta.ljust(self.delta_width)
699
+ return (
700
+ paint(lead, *row_codes, *cell.codes) + " " + paint(tail, *row_codes, *cell.delta_codes)
701
+ )
702
+
703
+
704
+ def _pack(facts: Sequence[str], width: int = TERMINAL_WIDTH, gap: str = " ") -> list[str]:
705
+ """Lay header facts out on as few lines as fit, never splitting one fact across lines."""
706
+ lines: list[str] = []
707
+ for fact in facts:
708
+ if len(fact) > width:
709
+ lines += textwrap.wrap(fact, width, break_on_hyphens=False)
710
+ elif lines and len(lines[-1]) + len(gap) + len(fact) <= width:
711
+ lines[-1] += gap + fact
712
+ else:
713
+ lines.append(fact)
714
+ return lines
715
+
716
+
717
+ def render_terminal(report: Report, *, color: bool = False, verdict: Verdict | None = None) -> str:
718
+ """Render a fixed-width scorecard that fits a 100-column terminal."""
719
+ view = _view(report, verdict or judge(report))
720
+ paint = _Ansi(color)
721
+ facts = [
722
+ f"{name}: {_plain(value)}" + (f" ({backend})" if backend else "")
723
+ for name, value, backend in view.context()
724
+ ]
725
+ lines = [
726
+ paint(_plain(report.title), _BOLD),
727
+ paint(summary_line(report), _DIM),
728
+ *(paint(line, _DIM) for line in _pack(facts)),
729
+ "",
730
+ ]
731
+ lead = _lead(view.verdict)
732
+ lines += [paint(line, _BOLD) for line in _wrap(_plain(lead.headline), first="> ")]
733
+ if lead.remedy is not None:
734
+ lines += [paint(line, _BOLD, _YELLOW) for line in _wrap(_plain(lead.remedy), first=" ")]
735
+ for sentence in (*lead.details, *_score_alerts(view)):
736
+ lines += _wrap(_plain(sentence), first=" ", rest=" ")
737
+ lines += ["", *_terminal_table(view, paint)]
738
+ if view.has_significant:
739
+ lines.append(paint(SIGNIFICANT_FOOTNOTE, _DIM))
740
+ lines += _terminal_reasons(view, paint)
741
+ findings = _findings(view)
742
+ if findings:
743
+ lines += ["", paint("Server checks", _BOLD)]
744
+ for finding in findings:
745
+ lines += _terminal_finding(view, finding, paint)
746
+ errors = _errors(view)
747
+ if errors:
748
+ lines += ["", paint("Errors", _BOLD)]
749
+ for label, error in errors:
750
+ lines += _wrap(f"{_plain(label)}: {_plain(error)}", first=" ", rest=" ")
751
+ lines += ["", *_wrap(f"Settings: {_plain(_settings_line(report))}")]
752
+ notes = _notes(view)
753
+ if notes:
754
+ lines += ["", paint("Notes", _BOLD)]
755
+ for number, note in enumerate(notes, start=1):
756
+ lines += _wrap(_plain(note), first=f" {number}. ", rest=" ")
757
+ return "\n".join(lines) + "\n"
758
+
759
+
760
+ def _terminal_reasons(view: _View, paint: _Ansi) -> list[str]:
761
+ reasons = _reasons(view)
762
+ if not reasons:
763
+ return []
764
+ width = max(len(_tagged_chip(row)) for row, _ in reasons)
765
+ indent = " " * (width + 4)
766
+ lines = ["", paint("Why", _BOLD)]
767
+ for row, call in reasons:
768
+ tag = paint(_tagged_chip(row).ljust(width), *_STATUS_CODES[call.status])
769
+ body = f"{_plain(row.label)}: {_plain(' '.join(call.reasons))}".rstrip(": ")
770
+ lines += _wrap(body, first=f" {tag} ", rest=indent)
771
+ for caveat in call.caveats:
772
+ text = f"{CAVEAT_PREFIX} {_plain(_sentence(caveat, capitalize=False))}"
773
+ lines += [paint(line, _YELLOW) for line in _wrap(text, first=indent, rest=indent)]
774
+ return lines
775
+
776
+
777
+ def _terminal_finding(view: _View, item: ServerFinding, paint: _Ansi) -> list[str]:
778
+ finding = item.finding
779
+ codes = _SEVERITY_CODES[finding.severity] if item.affects_scores else (_DIM,)
780
+ tag = paint(finding_chip(item).ljust(4), *codes)
781
+ message = _sentence(_plain(finding.message), capitalize=False)
782
+ head = f"{_plain(view.who(item))}, {_plain(finding.check)}: {message}"
783
+ indent = " "
784
+ lines = _wrap(head, first=f" {tag} ", rest=indent)
785
+ impact = _wrap(_sentence(_plain(item.impact)), first=indent, rest=indent)
786
+ lines += [paint(line, _DIM) for line in impact]
787
+ if finding.fix:
788
+ lines += _wrap(f"Fix: {_plain(finding.fix)}", first=indent, rest=indent)
789
+ return lines
790
+
791
+
792
+ def _terminal_table(view: _View, paint: _Ansi) -> list[str]:
793
+ columns = _terminal_columns(view, backend=view.backend is None)
794
+ widths = _fit(columns)
795
+ if widths is None:
796
+ # Mixed backends and long labels: the label is worth more than the backend name,
797
+ # which the markdown and HTML cards still show.
798
+ columns = _terminal_columns(view, backend=False)
799
+ widths = _fit(columns) or _squeezed(columns)
800
+ header = _GAP.join(_pad(c.header, w, c.align) for c, w in zip(columns, widths, strict=True))
801
+ lines = [paint(header, _BOLD), paint("-" * len(header), _DIM)]
802
+ for index, row in enumerate(view.rows):
803
+ row_codes = (_DIM, _ITALIC) if row.is_reference else ()
804
+ cells = [
805
+ column.render(column.cells[index], width, paint, row_codes)
806
+ for column, width in zip(columns, widths, strict=True)
807
+ ]
808
+ lines.append(_GAP.join(cells).rstrip())
809
+ if not view.candidates:
810
+ lines.append("(no candidates)")
811
+ return lines
812
+
813
+
814
+ def _terminal_columns(view: _View, *, backend: bool) -> list[_TermColumn]:
815
+ rows = view.rows
816
+ columns = [
817
+ _TermColumn(
818
+ "Call",
819
+ "left",
820
+ tuple(
821
+ _TermCell(row.chip, codes=() if row.status is None else _STATUS_CODES[row.status])
822
+ for row in rows
823
+ ),
824
+ ),
825
+ _TermColumn("Label", "left", tuple(_TermCell(_plain(row.label)) for row in rows)),
826
+ ]
827
+ if backend:
828
+ columns.append(
829
+ _TermColumn("Backend", "left", tuple(_TermCell(row.result.spec.kind) for row in rows))
830
+ )
831
+ if view.show_size:
832
+ columns.append(_TermColumn("Size", "right", tuple(_term_size_cell(view, r) for r in rows)))
833
+ for metric in view.metrics:
834
+ cells = tuple(_term_metric_cell(view.cell(metric, row)) for row in rows)
835
+ columns.append(_TermColumn(metric.short, "right", cells))
836
+ if view.speed is not None:
837
+ cells = tuple(_term_metric_cell(view.speed_cell(row)) for row in rows)
838
+ columns.append(_TermColumn(_speed_header(view), "right", cells))
839
+ return columns
840
+
841
+
842
+ def _term_size_cell(view: _View, row: _Row) -> _TermCell:
843
+ size, change = view.size_cell(row)
844
+ if not row.over_budget:
845
+ return _TermCell(size, change)
846
+ marked = f"{change} {OVER_BUDGET_SHORT}" if change else OVER_BUDGET_SHORT
847
+ return _TermCell(size, marked, (_DIM,), (_DIM,))
848
+
849
+
850
+ def _term_metric_cell(cell: _Cell) -> _TermCell:
851
+ codes = (_DIM,) if cell.muted else ()
852
+ if cell.marker:
853
+ return _TermCell(cell.text, cell.marker, codes, (_DIM,))
854
+ if cell.delta is None:
855
+ return _TermCell(cell.text, codes=codes)
856
+ delta_codes = (_RED,) if is_regression(cell.delta) else (_DIM,)
857
+ return _TermCell(cell.text, format_delta(cell.delta), codes, delta_codes)
858
+
859
+
860
+ def _label_room(columns: Sequence[_TermColumn]) -> int:
861
+ others = sum(column.width for column in columns[2:]) + columns[0].width
862
+ return TERMINAL_WIDTH - others - len(_GAP) * (len(columns) - 1)
863
+
864
+
865
+ def _fit(columns: Sequence[_TermColumn]) -> list[int] | None:
866
+ """Column widths that fit the terminal, or None when the label would get too narrow."""
867
+ needed = columns[1].width
868
+ room = _label_room(columns)
869
+ if room < min(needed, _LABEL_MIN_WIDTH):
870
+ return None
871
+ return [columns[0].width, min(needed, room), *(column.width for column in columns[2:])]
872
+
873
+
874
+ def _squeezed(columns: Sequence[_TermColumn]) -> list[int]:
875
+ label = max(_label_room(columns), len(_ELLIPSIS) + 2)
876
+ return [columns[0].width, label, *(column.width for column in columns[2:])]
877
+
878
+
879
+ def _pad(text: str, width: int, align: Align) -> str:
880
+ return text.rjust(width) if align == "right" else text.ljust(width)
881
+
882
+
883
+ def _truncate(text: str, width: int) -> str:
884
+ """Shorten from the middle: quant labels differ at the end (q4_K_M vs q2_K), not the start."""
885
+ if len(text) <= width:
886
+ return text
887
+ keep = width - len(_ELLIPSIS)
888
+ head = keep // 3
889
+ return text[:head] + _ELLIPSIS + text[len(text) - (keep - head) :]
890
+
891
+
892
+ def _plain(text: str) -> str:
893
+ """Neutralize control characters so labels cannot inject terminal escape sequences."""
894
+ return "".join(char if char.isprintable() else " " for char in text)
895
+
896
+
897
+ def _sentence(text: str, *, capitalize: bool = True) -> str:
898
+ """End with a period, and capitalize unless told not to, so phrases read as sentences."""
899
+ text = " ".join(text.split())
900
+ if not text:
901
+ return text
902
+ if capitalize:
903
+ text = text[0].upper() + text[1:]
904
+ return text if text.endswith((".", "!", "?")) else text + "."
905
+
906
+
907
+ def _wrap(text: str, *, first: str = "", rest: str = " ") -> list[str]:
908
+ return textwrap.wrap(
909
+ text,
910
+ width=TERMINAL_WIDTH,
911
+ initial_indent=first,
912
+ subsequent_indent=rest,
913
+ break_on_hyphens=False,
914
+ ) or [first.rstrip()]
915
+
916
+
917
+ # Markdown ---------------------------------------------------------------------------------
918
+
919
+ _MARKDOWN_SPECIAL: Final = frozenset("\\`*_[]<>|~")
920
+
921
+
922
+ def render_markdown(report: Report, *, verdict: Verdict | None = None) -> str:
923
+ """Render a GitHub and Reddit friendly markdown scorecard."""
924
+ view = _view(report, verdict or judge(report))
925
+ context = "; ".join(
926
+ f"{name}: **{_md(value)}**" + (f" ({backend})" if backend else "")
927
+ for name, value, backend in view.context()
928
+ )
929
+ lines = [
930
+ f"## {_md(report.title)}",
931
+ "",
932
+ f"*{_md(summary_line(report))}*",
933
+ "",
934
+ context,
935
+ "",
936
+ ]
937
+ lead = _lead(view.verdict)
938
+ lines.append(f"> **{_md(lead.headline)}**")
939
+ if lead.remedy is not None:
940
+ lines += [">", f"> **{_md(lead.remedy)}**"]
941
+ for sentence in (*lead.details, *_score_alerts(view)):
942
+ lines += [">", f"> {_md(sentence)}"]
943
+ lines += ["", *_markdown_table(view)]
944
+ if view.has_significant:
945
+ lines += ["", _md(SIGNIFICANT_FOOTNOTE)]
946
+ reasons = _reasons(view)
947
+ if reasons:
948
+ lines += ["", "**Why**", ""]
949
+ for row, call in reasons:
950
+ body = f"{_md(row.label)}: {_md(' '.join(call.reasons))}".rstrip(": ")
951
+ lines.append(f"- **{_tagged_chip(row)}** {body}")
952
+ lines += [
953
+ f" - {CAVEAT_PREFIX} {_md(_sentence(caveat, capitalize=False))}"
954
+ for caveat in call.caveats
955
+ ]
956
+ findings = _findings(view)
957
+ if findings:
958
+ lines += ["", "### Server checks", ""]
959
+ lines += [_markdown_finding(view, finding) for finding in findings]
960
+ errors = _errors(view)
961
+ if errors:
962
+ lines += ["", "### Errors", ""]
963
+ lines += [f"- {_md(label)}: {_md(error)}" for label, error in errors]
964
+ lines += ["", f"**Settings:** {_md(_settings_line(report))}"]
965
+ notes = _notes(view)
966
+ if notes:
967
+ lines += ["", "### Notes", ""]
968
+ lines += [f"{number}. {_md(note)}" for number, note in enumerate(notes, start=1)]
969
+ lines += ["", f"Generated with quantdiff v{_md(report.quantdiff_version)}"]
970
+ return "\n".join(lines) + "\n"
971
+
972
+
973
+ def _markdown_table(view: _View) -> list[str]:
974
+ show_backend = view.backend is None
975
+ headers = ["Call", "Label"]
976
+ aligns = [":--", ":--"]
977
+ if show_backend:
978
+ headers.append("Backend")
979
+ aligns.append(":--")
980
+ if view.show_size:
981
+ headers.append("Size")
982
+ aligns.append("--:")
983
+ headers += [metric.name for metric in view.metrics]
984
+ aligns += ["--:" for _ in view.metrics]
985
+ if view.speed is not None:
986
+ headers.append(_speed_header(view))
987
+ aligns.append("--:")
988
+ lines = [_md_row(headers), _md_row(aligns)]
989
+ for row in view.rows:
990
+ label = _md(row.label)
991
+ chip = f"**{row.chip}**" if row.status == "recommended" else row.chip
992
+ if row.near_bar:
993
+ chip += f" *{NEAR_BAR}*"
994
+ cells = [chip, f"{label} (reference)" if row.is_reference else label]
995
+ if show_backend:
996
+ cells.append(row.result.spec.kind)
997
+ if view.show_size:
998
+ cells.append(_markdown_size(view, row))
999
+ cells += [_markdown_cell(view.cell(metric, row)) for metric in view.metrics]
1000
+ if view.speed is not None:
1001
+ cells.append(_markdown_cell(view.speed_cell(row)))
1002
+ lines.append(_md_row(cells))
1003
+ return lines
1004
+
1005
+
1006
+ def _markdown_size(view: _View, row: _Row) -> str:
1007
+ size, change = view.size_cell(row)
1008
+ if not row.over_budget:
1009
+ return f"{size} ({change})" if change else size
1010
+ marked = f"{size} ({change}) {OVER_BUDGET_SHORT}" if change else f"{size} {OVER_BUDGET_SHORT}"
1011
+ return f"*{marked}*"
1012
+
1013
+
1014
+ def _markdown_cell(cell: _Cell) -> str:
1015
+ text = cell.text if cell.delta is None else f"{cell.text} ({_md(format_delta(cell.delta))})"
1016
+ if cell.marker:
1017
+ text = f"{text} ({cell.marker})"
1018
+ return f"*{text}*" if cell.muted and cell.value is not None else text
1019
+
1020
+
1021
+ def _markdown_finding(view: _View, item: ServerFinding) -> str:
1022
+ finding = item.finding
1023
+ chip = finding_chip(item)
1024
+ tag = f"**{chip}**" if item.affects_scores else chip
1025
+ message = _sentence(finding.message, capitalize=False)
1026
+ line = (
1027
+ f"- {tag} {_md(view.who(item))}, {_md(finding.check)}: {_md(message)} "
1028
+ f"*{_md(_sentence(item.impact))}*"
1029
+ )
1030
+ if finding.fix:
1031
+ line += f" Fix: {_md(finding.fix)}"
1032
+ return line
1033
+
1034
+
1035
+ def _md_row(cells: Sequence[str]) -> str:
1036
+ return "| " + " | ".join(cells) + " |"
1037
+
1038
+
1039
+ def _md(text: str) -> str:
1040
+ """Escape markdown syntax and flatten newlines so text stays inside its table cell."""
1041
+ flat = " ".join(text.split())
1042
+ return "".join(f"\\{char}" if char in _MARKDOWN_SPECIAL else char for char in flat)
1043
+
1044
+
1045
+ # HTML -------------------------------------------------------------------------------------
1046
+
1047
+ _CSS: Final = """
1048
+ :root {
1049
+ --bg: #f3f4f1;
1050
+ --surface: #ffffff;
1051
+ --text: #16181d;
1052
+ --muted: #5d6470;
1053
+ --faint: #8a909a;
1054
+ --border: #e3e5e8;
1055
+ --row-pick: #f0f8f3;
1056
+ --row-ref: #f7f8fa;
1057
+ --accent: #1f7a4d;
1058
+ --accent-soft: #e6f3eb;
1059
+ --neutral-soft: #f1f2f4;
1060
+ --track: #eceef1;
1061
+ --fill: #6f7b8a;
1062
+ --fill-ref: #b4bac3;
1063
+ --kld-good: #3f9a6b;
1064
+ --kld-moderate: #d29a2a;
1065
+ --kld-large: #cc4b3c;
1066
+ --usable: #1d64b0;
1067
+ --usable-soft: #e4eefa;
1068
+ --usable-line: #9dbde3;
1069
+ --down: #c03a2b;
1070
+ --warn: #9a6200;
1071
+ --warn-soft: #fdf1d8;
1072
+ --fail: #b42318;
1073
+ --fail-soft: #fde5e2;
1074
+ --t1: #3b5bdb;
1075
+ --t1-soft: #e6ebfc;
1076
+ --t2: #0b7285;
1077
+ --t2-soft: #dff3f6;
1078
+ --shadow: 0 1px 2px rgba(16, 24, 40, 0.06), 0 8px 24px rgba(16, 24, 40, 0.06);
1079
+ }
1080
+ @media (prefers-color-scheme: dark) {
1081
+ :root {
1082
+ --bg: #0e1013;
1083
+ --surface: #171a1f;
1084
+ --text: #e8eaed;
1085
+ --muted: #a3a9b3;
1086
+ --faint: #757c87;
1087
+ --border: #2a2f37;
1088
+ --row-pick: #16261d;
1089
+ --row-ref: #1c2026;
1090
+ --accent: #5fd08f;
1091
+ --accent-soft: #1a2e23;
1092
+ --neutral-soft: #1f232a;
1093
+ --track: #262b33;
1094
+ --fill: #8b95a3;
1095
+ --fill-ref: #5a616c;
1096
+ --kld-good: #5cc58e;
1097
+ --kld-moderate: #e6b450;
1098
+ --kld-large: #f07a6b;
1099
+ --usable: #7cb6f5;
1100
+ --usable-soft: #182a40;
1101
+ --usable-line: #2f5a8a;
1102
+ --down: #ff8a7d;
1103
+ --warn: #f2b84b;
1104
+ --warn-soft: #3a2c10;
1105
+ --fail: #ff7b6e;
1106
+ --fail-soft: #3d1a17;
1107
+ --t1: #91a7ff;
1108
+ --t1-soft: #1e2645;
1109
+ --t2: #66d9e8;
1110
+ --t2-soft: #0f3036;
1111
+ --shadow: none;
1112
+ }
1113
+ }
1114
+ * { box-sizing: border-box; }
1115
+ body {
1116
+ margin: 0;
1117
+ padding: 32px 16px;
1118
+ background: var(--bg);
1119
+ color: var(--text);
1120
+ font: 14px/1.5 system-ui, -apple-system, "Segoe UI", Roboto, "Helvetica Neue", Arial,
1121
+ sans-serif;
1122
+ font-variant-numeric: tabular-nums;
1123
+ -webkit-font-smoothing: antialiased;
1124
+ }
1125
+ .card {
1126
+ max-width: 960px;
1127
+ margin: 0 auto;
1128
+ background: var(--surface);
1129
+ border: 1px solid var(--border);
1130
+ border-radius: 14px;
1131
+ box-shadow: var(--shadow);
1132
+ overflow: hidden;
1133
+ }
1134
+ .section { padding: 20px 28px; border-top: 1px solid var(--border); }
1135
+ .head { padding: 26px 28px 18px; }
1136
+ .eyebrow {
1137
+ margin: 0 0 6px;
1138
+ color: var(--accent);
1139
+ font-size: 12px;
1140
+ font-weight: 600;
1141
+ letter-spacing: 0.08em;
1142
+ text-transform: uppercase;
1143
+ }
1144
+ h1 {
1145
+ margin: 0;
1146
+ font-size: 22px;
1147
+ line-height: 1.25;
1148
+ letter-spacing: -0.01em;
1149
+ overflow-wrap: anywhere;
1150
+ }
1151
+ h2 {
1152
+ margin: 0 0 12px;
1153
+ color: var(--muted);
1154
+ font-size: 12px;
1155
+ font-weight: 600;
1156
+ letter-spacing: 0.06em;
1157
+ text-transform: uppercase;
1158
+ }
1159
+ .stats { margin: 4px 0 0; color: var(--faint); font-size: 14px; }
1160
+ .meta {
1161
+ display: flex;
1162
+ flex-wrap: wrap;
1163
+ gap: 6px 22px;
1164
+ margin: 12px 0 0;
1165
+ padding: 0;
1166
+ list-style: none;
1167
+ color: var(--muted);
1168
+ font-size: 13px;
1169
+ }
1170
+ .meta li { min-width: 0; overflow-wrap: anywhere; }
1171
+ .meta .mono { color: var(--text); font-weight: 600; }
1172
+ .mono {
1173
+ font-family: ui-monospace, "SF Mono", "Cascadia Mono", "Segoe UI Mono", Consolas, monospace;
1174
+ font-size: 0.95em;
1175
+ overflow-wrap: anywhere;
1176
+ }
1177
+ .verdict {
1178
+ margin: 0 28px 22px;
1179
+ padding: 20px 24px;
1180
+ background: var(--accent-soft);
1181
+ border-left: 5px solid var(--accent);
1182
+ border-radius: 10px;
1183
+ }
1184
+ .verdict.open { background: var(--neutral-soft); border-left-color: var(--faint); }
1185
+ .verdict.fits { background: var(--usable-soft); border-left-color: var(--usable); }
1186
+ .verdict p { margin: 0; overflow-wrap: anywhere; }
1187
+ .verdict .kicker {
1188
+ margin-bottom: 6px;
1189
+ color: var(--accent);
1190
+ font-size: 12px;
1191
+ font-weight: 700;
1192
+ letter-spacing: 0.06em;
1193
+ text-transform: uppercase;
1194
+ }
1195
+ .verdict.open .kicker { color: var(--muted); }
1196
+ .verdict.fits .kicker { color: var(--usable); }
1197
+ .verdict .headline { font-size: 24px; font-weight: 700; line-height: 1.3; }
1198
+ .verdict ul { margin: 10px 0 0; padding-left: 18px; color: var(--muted); font-size: 14px; }
1199
+ .verdict li { overflow-wrap: anywhere; }
1200
+ .verdict li + li { margin-top: 3px; }
1201
+ .verdict .remedy {
1202
+ margin-top: 12px;
1203
+ padding: 10px 14px;
1204
+ border-left: 4px solid var(--warn);
1205
+ border-radius: 6px;
1206
+ background: var(--warn-soft);
1207
+ color: var(--text);
1208
+ font-size: 15px;
1209
+ font-weight: 600;
1210
+ }
1211
+ .verdict .alert { margin-top: 10px; color: var(--fail); font-size: 14px; font-weight: 600; }
1212
+ .table-wrap { overflow-x: auto; padding: 4px 14px 8px; }
1213
+ table { width: 100%; border-collapse: collapse; }
1214
+ th, td { padding: 10px 7px; text-align: right; vertical-align: top; }
1215
+ th {
1216
+ color: var(--muted);
1217
+ font-size: 11px;
1218
+ font-weight: 600;
1219
+ letter-spacing: 0.04em;
1220
+ text-transform: uppercase;
1221
+ white-space: nowrap;
1222
+ border-bottom: 1px solid var(--border);
1223
+ }
1224
+ thead tr.groups th { border-bottom: none; padding-bottom: 0; text-align: center; }
1225
+ td { border-bottom: 1px solid var(--border); font-size: 13px; white-space: nowrap; }
1226
+ tbody tr:last-child td { border-bottom: none; }
1227
+ th.left, td.left { text-align: left; }
1228
+ td.name { white-space: normal; }
1229
+ .name-box { min-width: 96px; max-width: 250px; }
1230
+ tr.pick td { background: var(--row-pick); }
1231
+ tr.baseline td { background: var(--row-ref); border-bottom: 2px solid var(--border); }
1232
+ .status {
1233
+ display: inline-block;
1234
+ min-width: 58px;
1235
+ padding: 2px 8px;
1236
+ border-radius: 999px;
1237
+ background: var(--track);
1238
+ color: var(--muted);
1239
+ font-size: 11px;
1240
+ font-weight: 700;
1241
+ letter-spacing: 0.05em;
1242
+ line-height: 18px;
1243
+ text-align: center;
1244
+ }
1245
+ .status.recommended { background: var(--accent); color: var(--surface); }
1246
+ .status.usable { background: var(--usable-soft); color: var(--usable); }
1247
+ .status.avoid { background: var(--fail-soft); color: var(--fail); }
1248
+ .status.inconclusive {
1249
+ background: var(--warn-soft);
1250
+ border: 1px solid var(--warn);
1251
+ color: var(--warn);
1252
+ line-height: 16px;
1253
+ }
1254
+ .status.failed { background: transparent; border: 1px solid var(--fail); color: var(--fail); }
1255
+ .status.reference {
1256
+ background: transparent;
1257
+ border: 1px solid var(--border);
1258
+ color: var(--faint);
1259
+ line-height: 16px;
1260
+ }
1261
+ .tag {
1262
+ display: block;
1263
+ margin-top: 4px;
1264
+ color: var(--warn);
1265
+ font-size: 10px;
1266
+ font-weight: 700;
1267
+ letter-spacing: 0.04em;
1268
+ text-align: center;
1269
+ text-transform: uppercase;
1270
+ white-space: nowrap;
1271
+ }
1272
+ .label { font-weight: 600; }
1273
+ .sub { display: block; color: var(--faint); font-size: 12px; font-weight: 400; }
1274
+ .why { display: block; margin-top: 3px; color: var(--muted); font-size: 12px; line-height: 1.35; }
1275
+ .caveat {
1276
+ display: block;
1277
+ position: relative;
1278
+ margin-top: 4px;
1279
+ padding-left: 18px;
1280
+ color: var(--warn);
1281
+ font-size: 12px;
1282
+ line-height: 1.35;
1283
+ }
1284
+ .caveat::before {
1285
+ content: "!";
1286
+ position: absolute;
1287
+ left: 0;
1288
+ top: 1px;
1289
+ width: 13px;
1290
+ height: 13px;
1291
+ border-radius: 50%;
1292
+ background: var(--warn-soft);
1293
+ border: 1px solid var(--warn);
1294
+ font-size: 9px;
1295
+ font-weight: 800;
1296
+ line-height: 11px;
1297
+ text-align: center;
1298
+ }
1299
+ .backend { color: var(--muted); }
1300
+ .na { color: var(--faint); }
1301
+ td.base { color: var(--faint); font-size: 12px; font-style: italic; text-align: center; }
1302
+ .sub.failed { color: var(--fail); font-family: inherit; }
1303
+ td.grey { opacity: 0.45; }
1304
+ td.wall { color: var(--muted); }
1305
+ td.over { color: var(--faint); }
1306
+ .change.over { color: var(--warn); font-weight: 600; }
1307
+ .marker { display: block; margin-top: 3px; color: var(--faint); font-size: 11px; }
1308
+ .bar {
1309
+ display: block;
1310
+ height: 4px;
1311
+ margin-top: 5px;
1312
+ margin-left: auto;
1313
+ width: 52px;
1314
+ border-radius: 2px;
1315
+ background: var(--track);
1316
+ overflow: hidden;
1317
+ }
1318
+ .bar span { display: block; height: 100%; border-radius: 2px; background: var(--fill); }
1319
+ .bar.near-lossless span, .bar.small span { background: var(--kld-good); }
1320
+ .bar.moderate span { background: var(--kld-moderate); }
1321
+ .bar.large span { background: var(--kld-large); }
1322
+ tr.baseline .bar span { background: var(--fill-ref); }
1323
+ .delta { display: block; margin-top: 3px; color: var(--faint); font-size: 11px; line-height: 1.2; }
1324
+ .delta.down { color: var(--down); font-weight: 700; }
1325
+ .change { display: block; margin-top: 3px; color: var(--muted); font-size: 11px; }
1326
+ .footnote { margin: 0; padding: 0 28px 14px; color: var(--faint); font-size: 12px; }
1327
+ .tier {
1328
+ display: inline-block;
1329
+ padding: 1px 7px;
1330
+ border-radius: 999px;
1331
+ font-size: 10px;
1332
+ font-weight: 700;
1333
+ letter-spacing: 0.04em;
1334
+ }
1335
+ .tier.t1 { background: var(--t1-soft); color: var(--t1); }
1336
+ .tier.t2 { background: var(--t2-soft); color: var(--t2); }
1337
+ .group-name { margin-left: 6px; color: var(--muted); font-weight: 600; }
1338
+ .chip {
1339
+ display: inline-block;
1340
+ min-width: 42px;
1341
+ padding: 1px 8px;
1342
+ border-radius: 999px;
1343
+ font-size: 11px;
1344
+ font-weight: 700;
1345
+ letter-spacing: 0.04em;
1346
+ text-align: center;
1347
+ flex: none;
1348
+ margin-top: 2px;
1349
+ }
1350
+ .chip.fail { background: var(--fail-soft); color: var(--fail); }
1351
+ .chip.warn { background: var(--warn-soft); color: var(--warn); }
1352
+ .chip.note { background: var(--track); color: var(--muted); }
1353
+ .columns { display: grid; grid-template-columns: 3fr 2fr; gap: 28px; }
1354
+ @media (max-width: 720px) { .columns { grid-template-columns: 1fr; } }
1355
+ .columns > section { min-width: 0; }
1356
+ .findings { margin: 0; padding: 0; list-style: none; }
1357
+ .findings li { display: flex; align-items: flex-start; gap: 12px; padding: 8px 0; }
1358
+ .findings li + li { border-top: 1px dashed var(--border); }
1359
+ .findings div { min-width: 0; }
1360
+ .findings p { margin: 0; overflow-wrap: anywhere; }
1361
+ .findings .impact { color: var(--muted); font-size: 13px; }
1362
+ .findings .fix { color: var(--muted); font-size: 13px; }
1363
+ .findings li.quiet p { color: var(--muted); }
1364
+ .findings li.quiet .impact, .findings li.quiet .fix { color: var(--faint); }
1365
+ .muted { margin: 0; color: var(--muted); }
1366
+ dl { display: grid; grid-template-columns: auto 1fr; gap: 6px 16px; margin: 0; }
1367
+ dt { color: var(--muted); }
1368
+ dd { margin: 0; overflow-wrap: anywhere; }
1369
+ .notes ol { margin: 0; padding-left: 20px; color: var(--muted); font-size: 13px; }
1370
+ .notes li + li { margin-top: 4px; }
1371
+ .errors { margin: 0; padding-left: 20px; color: var(--fail); font-size: 13px; }
1372
+ .errors li { overflow-wrap: anywhere; }
1373
+ footer {
1374
+ display: flex;
1375
+ justify-content: space-between;
1376
+ gap: 16px;
1377
+ padding: 14px 28px;
1378
+ border-top: 1px solid var(--border);
1379
+ color: var(--faint);
1380
+ font-size: 12px;
1381
+ }
1382
+ footer strong { color: var(--muted); }
1383
+ """
1384
+
1385
+
1386
+ def render_html(report: Report, *, verdict: Verdict | None = None) -> str:
1387
+ """Render a self-contained HTML scorecard: inline CSS, no scripts, no external assets."""
1388
+ view = _view(report, verdict or judge(report))
1389
+ footnote = f'<p class="footnote">{_e(SIGNIFICANT_FOOTNOTE)}</p>' if view.has_significant else ""
1390
+ sections = [
1391
+ '<header class="head">',
1392
+ '<p class="eyebrow">quantdiff scorecard</p>',
1393
+ f"<h1>{_e(report.title)}</h1>",
1394
+ f'<p class="stats">{_e(summary_line(report))}</p>',
1395
+ _html_context(view),
1396
+ "</header>",
1397
+ _html_verdict(view),
1398
+ _html_table(view),
1399
+ footnote,
1400
+ '<div class="section columns">',
1401
+ _html_findings(view),
1402
+ _html_settings(report),
1403
+ "</div>",
1404
+ _html_errors(view),
1405
+ _html_notes(view),
1406
+ "<footer>",
1407
+ f"<span>Generated with <strong>quantdiff v{_e(report.quantdiff_version)}</strong></span>",
1408
+ f'<time datetime="{_e(report.created_at)}">{_e(report.created_at)}</time>',
1409
+ "</footer>",
1410
+ ]
1411
+ return "\n".join(
1412
+ [
1413
+ "<!doctype html>",
1414
+ '<html lang="en">',
1415
+ "<head>",
1416
+ '<meta charset="utf-8">',
1417
+ '<meta name="viewport" content="width=device-width, initial-scale=1">',
1418
+ '<meta name="color-scheme" content="light dark">',
1419
+ f"<title>{_e(report.title)}</title>",
1420
+ f"<style>{_CSS}</style>",
1421
+ "</head>",
1422
+ "<body>",
1423
+ '<main class="card">',
1424
+ *(section for section in sections if section),
1425
+ "</main>",
1426
+ "</body>",
1427
+ "</html>",
1428
+ "",
1429
+ ]
1430
+ )
1431
+
1432
+
1433
+ def _html_context(view: _View) -> str:
1434
+ items = "".join(
1435
+ f'<li>{_e(name)} <span class="mono" title="{_e(value)}">{_e(value)}</span>'
1436
+ + (f" on {_e(backend)}" if backend else "")
1437
+ + "</li>"
1438
+ for name, value, backend in view.context()
1439
+ )
1440
+ return f'<ul class="meta">{items}</ul>'
1441
+
1442
+
1443
+ def _html_verdict(view: _View) -> str:
1444
+ verdict = view.verdict
1445
+ lead = _lead(verdict)
1446
+ remedy = "" if lead.remedy is None else f'<p class="remedy">{_e(lead.remedy)}</p>'
1447
+ items = "".join(f"<li>{_e(sentence)}</li>" for sentence in lead.details)
1448
+ alerts = "".join(f'<p class="alert">{_e(alert)}</p>' for alert in _score_alerts(view))
1449
+ return (
1450
+ f'<section class="verdict{_verdict_tone(verdict)}"><p class="kicker">Verdict</p>'
1451
+ f'<p class="headline">{_e(lead.headline)}</p>{remedy}'
1452
+ f"{f'<ul>{items}</ul>' if items else ''}{alerts}</section>"
1453
+ )
1454
+
1455
+
1456
+ def _verdict_tone(verdict: Verdict) -> str:
1457
+ """The verdict box's CSS modifier: "" (green) when a candidate within the budget is
1458
+ recommended, " fits" (blue) when the best that fits is a usable candidate, and " open"
1459
+ (grey) otherwise, which covers every "Keep <reference>" and all-avoid outcome."""
1460
+ if verdict.headline.startswith("Keep "):
1461
+ return " open"
1462
+ within = [call for call in verdict.candidates if call.fits_budget is not False]
1463
+ if any(call.status == "recommended" for call in within):
1464
+ return ""
1465
+ if any(call.status == "usable" for call in within):
1466
+ return " fits"
1467
+ return " open"
1468
+
1469
+
1470
+ def _html_table(view: _View) -> str:
1471
+ show_backend = view.backend is None
1472
+ lead = 2 + show_backend + view.show_size
1473
+ logit = sum(1 for metric in view.metrics if metric.group == "logit")
1474
+ task = len(view.metrics) - logit
1475
+ groups = f'<tr class="groups"><th colspan="{lead}"></th>'
1476
+ if logit:
1477
+ groups += (
1478
+ f'<th colspan="{logit}"><span class="tier t1">T1</span>'
1479
+ '<span class="group-name">Logit fidelity</span></th>'
1480
+ )
1481
+ if task:
1482
+ groups += (
1483
+ f'<th colspan="{task}"><span class="tier t2">T2</span>'
1484
+ '<span class="group-name">Task quality</span></th>'
1485
+ )
1486
+ if view.speed is not None:
1487
+ groups += "<th></th>"
1488
+ groups += "</tr>"
1489
+ header = '<tr><th class="left">Call</th><th class="left">Label</th>'
1490
+ if show_backend:
1491
+ header += '<th class="left">Backend</th>'
1492
+ if view.show_size:
1493
+ header += "<th>Size</th>"
1494
+ header += "".join(f"<th>{_e(metric.name)}</th>" for metric in view.metrics)
1495
+ if view.speed is not None:
1496
+ header += f"<th>{_e(_speed_header(view))}</th>"
1497
+ header += "</tr>"
1498
+ scales = [_column_max(view, metric) for metric in view.metrics]
1499
+ rows = [_html_row(view, row, scales, logit) for row in view.rows]
1500
+ if not view.candidates:
1501
+ span = lead + len(view.metrics) + (view.speed is not None)
1502
+ rows.append(f'<tr><td class="left na" colspan="{span}">No candidates</td></tr>')
1503
+ return (
1504
+ '<div class="table-wrap"><table>'
1505
+ f"<thead>{groups}{header}</thead><tbody>{''.join(rows)}</tbody>"
1506
+ "</table></div>"
1507
+ )
1508
+
1509
+
1510
+ def _column_max(view: _View, metric: _Metric) -> float:
1511
+ values = [v for row in view.candidates if (v := metric.value(row.result)) is not None]
1512
+ return max(values, default=0.0)
1513
+
1514
+
1515
+ def _html_lead_cells(row: _Row) -> list[str]:
1516
+ """The status chip and the label, with the reasons for the call under the label."""
1517
+ result = row.result
1518
+ sub_html = ""
1519
+ if row.status == "failed" and result.errors:
1520
+ sub_html = '<span class="sub failed">failed to run</span>'
1521
+ elif row.is_reference:
1522
+ sub_html = '<span class="sub">reference</span>'
1523
+ why = ""
1524
+ if row.call is not None and row.call.reasons:
1525
+ why = f'<span class="why">{_e(" ".join(row.call.reasons))}</span>'
1526
+ if row.call is not None:
1527
+ why += "".join(
1528
+ f'<span class="caveat">{_e(_sentence(caveat, capitalize=False))}</span>'
1529
+ for caveat in row.call.caveats
1530
+ )
1531
+ tag = f'<span class="tag">{NEAR_BAR}</span>' if row.near_bar else ""
1532
+ return [
1533
+ f'<td class="left"><span class="status {row.status or "reference"}">{row.chip}</span>'
1534
+ f"{tag}</td>",
1535
+ f'<td class="left name"><div class="name-box"><span class="label mono" '
1536
+ f'title="{_e(result.spec.label)}">{_e(row.label)}</span>{sub_html}{why}</div></td>',
1537
+ ]
1538
+
1539
+
1540
+ def _html_row(view: _View, row: _Row, scales: Sequence[float], logit_columns: int) -> str:
1541
+ spec = row.result.spec
1542
+ cells = _html_lead_cells(row)
1543
+ if view.backend is None:
1544
+ cells.append(f'<td class="left backend">{_e(spec.kind)}</td>')
1545
+ if view.show_size:
1546
+ cells.append(_html_size_cell(view, row))
1547
+ if row.is_reference and logit_columns:
1548
+ cells.append(f'<td class="base" colspan="{logit_columns}">baseline</td>')
1549
+ for metric, scale in zip(view.metrics, scales, strict=True):
1550
+ if row.is_reference and metric.group == "logit":
1551
+ continue
1552
+ cells.append(_html_metric_cell(view.cell(metric, row), metric, scale, view.thresholds))
1553
+ if view.speed is not None:
1554
+ cells.append(_html_speed_cell(view.speed_cell(row)))
1555
+ row_class = ""
1556
+ if row.is_reference:
1557
+ row_class = ' class="baseline"'
1558
+ elif row.status == "recommended":
1559
+ row_class = ' class="pick"'
1560
+ return f"<tr{row_class}>{''.join(cells)}</tr>"
1561
+
1562
+
1563
+ def _html_size_cell(view: _View, row: _Row) -> str:
1564
+ size, change = view.size_cell(row)
1565
+ change_html = f'<span class="change">{_e(change)}</span>' if change else ""
1566
+ if row.size_bytes is None:
1567
+ return f'<td class="na">{_e(size)}</td>'
1568
+ if row.over_budget:
1569
+ over = f'<span class="change over">{OVER_BUDGET}</span>'
1570
+ return f'<td class="over">{_e(size)}{change_html}{over}</td>'
1571
+ return f"<td>{_e(size)}{change_html}</td>"
1572
+
1573
+
1574
+ def _bar_class(metric: _Metric, value: float, thresholds: KldThresholds) -> str:
1575
+ """Rate bars are neutral. The KLD mean bar takes its band's colour, so a near-lossless or
1576
+ small KLD never reads as a warning; the p99 tail has no bands and stays neutral."""
1577
+ if metric.bar == "kld" and metric.short == "KLD":
1578
+ return f"bar {kld_band(value, thresholds)}"
1579
+ return "bar"
1580
+
1581
+
1582
+ def _html_metric_cell(cell: _Cell, metric: _Metric, scale: float, thresholds: KldThresholds) -> str:
1583
+ if cell.value is None:
1584
+ return f'<td class="na">{_e(cell.text)}</td>'
1585
+ fraction = cell.value if metric.bar == "rate" else cell.value / scale if scale > 0 else 0.0
1586
+ bar = _bar_html(_bar_class(metric, cell.value, thresholds), fraction)
1587
+ delta = ""
1588
+ if cell.delta is not None:
1589
+ direction = " down" if is_regression(cell.delta) else ""
1590
+ delta = f'<span class="delta{direction}">{_e(format_interval(cell.delta))}</span>'
1591
+ css = ' class="grey"' if cell.muted else ""
1592
+ return f'<td{css}><span class="val">{_e(cell.text)}</span>{bar}{delta}</td>'
1593
+
1594
+
1595
+ def _html_speed_cell(cell: _Cell) -> str:
1596
+ if cell.marker:
1597
+ marker = f'<span class="marker">{_e(cell.marker)}</span>'
1598
+ return f'<td class="wall">{_e(cell.text)}{marker}</td>'
1599
+ if cell.muted:
1600
+ return f'<td class="wall">{_e(cell.text)}</td>'
1601
+ if cell.text == MISSING:
1602
+ return f'<td class="na">{_e(cell.text)}</td>'
1603
+ return f"<td>{_e(cell.text)}</td>"
1604
+
1605
+
1606
+ def _bar_html(css: str, fraction: float) -> str:
1607
+ percent = min(max(fraction, 0.0), 1.0) * 100
1608
+ return f'<span class="{css}"><span style="width:{percent:.1f}%"></span></span>'
1609
+
1610
+
1611
+ def _html_findings(view: _View) -> str:
1612
+ findings = _findings(view)
1613
+ if not findings:
1614
+ return '<section><h2>Server checks</h2><p class="muted">No problems found.</p></section>'
1615
+ items = []
1616
+ for item in findings:
1617
+ finding = item.finding
1618
+ quiet = not item.affects_scores
1619
+ chip = "note" if quiet else finding.severity
1620
+ fix = f'<p class="fix">Fix: {_e(finding.fix)}</p>' if finding.fix else ""
1621
+ items.append(
1622
+ ('<li class="quiet">' if quiet else "<li>")
1623
+ + f'<span class="chip {chip}">{_e(finding_chip(item))}</span>'
1624
+ + f"<div><p>{_html_who(view, item)}, "
1625
+ f"{_e(finding.check)} check</p>"
1626
+ f"<p>{_e(_sentence(finding.message, capitalize=False))}</p>"
1627
+ f'<p class="impact">{_e(_sentence(item.impact))}</p>{fix}</div></li>'
1628
+ )
1629
+ return f'<section><h2>Server checks</h2><ul class="findings">{"".join(items)}</ul></section>'
1630
+
1631
+
1632
+ def _html_who(view: _View, item: ServerFinding) -> str:
1633
+ """Model labels in monospace; "both models" or "all N models" as plain words."""
1634
+ who = _e(view.who(item))
1635
+ return who if view.covers_everyone(item) else f'<span class="mono">{who}</span>'
1636
+
1637
+
1638
+ def _html_settings(report: Report) -> str:
1639
+ rows = "".join(
1640
+ f"<dt>{_e(name)}</dt><dd>{_e(value)}</dd>" for name, value in _settings_items(report)
1641
+ )
1642
+ return f"<section><h2>Settings</h2><dl>{rows}</dl></section>"
1643
+
1644
+
1645
+ def _html_errors(view: _View) -> str:
1646
+ items = "".join(
1647
+ f'<li><span class="mono">{_e(label)}</span>: {_e(error)}</li>'
1648
+ for label, error in _errors(view)
1649
+ )
1650
+ if not items:
1651
+ return ""
1652
+ return f'<section class="section"><h2>Errors</h2><ul class="errors">{items}</ul></section>'
1653
+
1654
+
1655
+ def _html_notes(view: _View) -> str:
1656
+ notes = _notes(view)
1657
+ if not notes:
1658
+ return ""
1659
+ items = "".join(f"<li>{_e(note)}</li>" for note in notes)
1660
+ return f'<section class="section notes"><h2>Notes</h2><ol>{items}</ol></section>'
1661
+
1662
+
1663
+ def _e(text: str) -> str:
1664
+ return html.escape(text, quote=True)