modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
decision/verify.py ADDED
@@ -0,0 +1,1745 @@
1
+ """Two-key verification and quarantine (MODEL-140; design §5, "Two keys").
2
+
3
+ The agent that collects a value never verifies it. A verifier re-reads the
4
+ value from the cited region of the retained source copy, with its own
5
+ extractor, and compares:
6
+
7
+ - the **model identity** in the region is the subject, not a sibling variant
8
+ ("GPT-6 Sol" is not "GPT-6 Astra"; "Nimbus 3 (max effort)" is Nimbus 3 at
9
+ max effort, never at default);
10
+ - the **value** agrees after unit normalisation, within ``TOLERANCE_RULE``;
11
+ - the **unit** is the one the collector filed;
12
+ - the **conditions** (effort, harness, date) are the ones the source states.
13
+
14
+ Extractors are pluggable. The deterministic ones (``TableExtractor``,
15
+ ``KeyValueExtractor``) always run before any other; ``LLMExtractor`` reads
16
+ prose through an injected completion function, so nothing here calls a model
17
+ or the network on its own: ``claude_extractor`` (Claude Sonnet, via the Claude
18
+ CLI) and ``mistral_extractor`` (Mistral Large, via ollama) are the two wired
19
+ readers. Each extractor's actor (agent, model family, method) is the verifier
20
+ the log records. Two keys means another model family (MODEL-159): a reader
21
+ from the collector's family is never asked, and a same-family ``verified``
22
+ already in the log does not count (``Verification.counts``).
23
+
24
+ Outcomes are ``verified``, ``mismatch`` (with a structured diff) and
25
+ ``unreachable`` (the copy, source or region is missing). A claim no
26
+ independent extractor could read is ``skipped``: nothing is logged and it
27
+ stays queued. Anything whose latest logged outcome is not ``verified``, and
28
+ anything never verified, is **quarantined**.
29
+
30
+ Files, under ``verification/`` at the repository root:
31
+
32
+ - ``log.jsonl``: the verification log, append-only, one
33
+ ``decision.model.Verification`` per line. The latest counting outcome per
34
+ target and checked value wins: latest ``date``, and on a tie the later line
35
+ (the rule ``decision.snapshot`` applies when it reads
36
+ ``verification/log.jsonl``).
37
+ - ``queue/events.jsonl``: append-only work queue. A collector files a claim
38
+ (``collected``), change detection re-queues one (``changed``), a run records
39
+ what it checked (``checked``). Mismatched and unreachable targets stay listed
40
+ by ``Queue.recrawl_requests`` until a collector files them again.
41
+
42
+ Refs from ``decision.sources`` (``Citation.ref``, ``RecheckReport.requeue``)
43
+ are ``"fact:<id>"`` or ``"evidence:<id>"``.
44
+ """
45
+
46
+ from __future__ import annotations
47
+
48
+ import csv
49
+ import hashlib
50
+ import io
51
+ import json
52
+ import os
53
+ import re
54
+ import subprocess
55
+ import urllib.request
56
+ from collections.abc import Callable, Iterable, Mapping, Sequence
57
+ from dataclasses import dataclass, field, replace
58
+ from datetime import UTC, date, datetime
59
+ from decimal import Decimal
60
+ from pathlib import Path
61
+ from typing import Any, Literal, Protocol
62
+
63
+ from pydantic import JsonValue, ValidationError
64
+
65
+ from decision.model import (
66
+ DETERMINISTIC,
67
+ SourceRef,
68
+ TargetRef,
69
+ Verification,
70
+ VerificationActor,
71
+ VerificationTarget,
72
+ value_hash,
73
+ )
74
+ from decision.normalise import (
75
+ NORMALISERS,
76
+ Locator,
77
+ UnsupportedContentError,
78
+ normalise_document,
79
+ select_region,
80
+ )
81
+ from decision.registry import UNREGISTERED
82
+ from decision.registry import default as default_registry
83
+ from decision.sources import CopyStore, RecheckReport, Source, load_sources
84
+
85
+ REPO_ROOT = Path(__file__).resolve().parent.parent
86
+ DEFAULT_DIRECTORY = REPO_ROOT / "verification"
87
+
88
+ Outcome = Literal["verified", "mismatch", "unreachable", "skipped"]
89
+ CONDITION_KEYS = ("effort", "harness", "date")
90
+
91
+ # --- claims --------------------------------------------------------------------------------------
92
+
93
+
94
+ @dataclass(frozen=True)
95
+ class Claim:
96
+ """One value as a collector filed it, with what the verifier needs to re-read it.
97
+
98
+ ``names`` are the names the subject is published under, taken from the
99
+ catalogue, never from the collector's own label for the row: that label is
100
+ what a copied-score error gets wrong. ``label`` is the column or key the value
101
+ sits under in the source (default: ``field`` with underscores as spaces).
102
+ ``unit`` is a unit ID (``UNITS``); ``None`` means the base unit of whatever
103
+ dimension the source states. ``conditions`` holds ``effort``, ``harness`` and
104
+ ``date`` as the collector filed them.
105
+ """
106
+
107
+ target: TargetRef
108
+ subject: str
109
+ names: tuple[str, ...]
110
+ field: str
111
+ value: JsonValue
112
+ collector: VerificationActor
113
+ sources: tuple[SourceRef, ...]
114
+ unit: str | None = None
115
+ label: str | None = None
116
+ conditions: Mapping[str, str | None] = field(default_factory=dict)
117
+
118
+ def __post_init__(self) -> None:
119
+ if not self.names:
120
+ raise ValueError(f"{self.target.id}: a claim needs the subject's published names")
121
+ if not self.sources:
122
+ raise ValueError(f"{self.target.id}: a claim needs at least one source")
123
+ unknown = set(self.conditions) - set(CONDITION_KEYS)
124
+ if unknown:
125
+ raise ValueError(f"{self.target.id}: unknown conditions {sorted(unknown)}")
126
+
127
+ @classmethod
128
+ def from_evidence(cls, evidence: Any, *, names: Sequence[str],
129
+ collector: VerificationActor, label: str | None = None) -> Claim:
130
+ """A claim for a ``decision.model.Evidence`` row with an ID, subject and sources."""
131
+ if evidence.id is None or evidence.subject is None:
132
+ raise ValueError("evidence needs an ID and a subject to be verified")
133
+ return cls(
134
+ target=TargetRef(kind="evidence", id=evidence.id),
135
+ subject=evidence.subject.id,
136
+ names=tuple(names),
137
+ field=evidence.benchmark_id,
138
+ label=label,
139
+ value=evidence.score,
140
+ unit=evidence.unit,
141
+ conditions={"effort": evidence.effort, "harness": evidence.harness,
142
+ "date": evidence.evidence_date},
143
+ collector=collector,
144
+ sources=tuple(evidence.sources),
145
+ )
146
+
147
+ @classmethod
148
+ def from_fact(cls, fact: Any, *, names: Sequence[str], collector: VerificationActor,
149
+ unit: str | None = None, label: str | None = None) -> Claim:
150
+ """A claim for a filed ``Fact``; ``unit`` is its facet's unit.
151
+
152
+ A scoped source region can also confirm ``not_disclosed`` or
153
+ ``requires_contract`` by naming the subject while omitting the facet.
154
+ ``unknown`` is not a filed value and cannot be verified.
155
+ """
156
+ if fact.state == "unknown":
157
+ raise ValueError(f"{fact.id}: an unknown fact has no value to verify")
158
+ return cls(
159
+ target=TargetRef(kind="fact", id=fact.id),
160
+ subject=fact.subject.id,
161
+ names=tuple(names),
162
+ field=fact.facet,
163
+ label=label,
164
+ value=fact.value,
165
+ unit=unit,
166
+ collector=collector,
167
+ sources=tuple(fact.sources),
168
+ )
169
+
170
+ def to_dict(self) -> dict[str, Any]:
171
+ return {
172
+ "target": self.target.model_dump(),
173
+ "subject": self.subject,
174
+ "names": list(self.names),
175
+ "field": self.field,
176
+ "label": self.label,
177
+ "value": self.value,
178
+ "unit": self.unit,
179
+ "conditions": dict(self.conditions),
180
+ "collector": self.collector.model_dump(),
181
+ "sources": [s.model_dump() for s in self.sources],
182
+ }
183
+
184
+ @classmethod
185
+ def from_dict(cls, data: Mapping[str, Any]) -> Claim:
186
+ return cls(
187
+ target=TargetRef.model_validate(data["target"]),
188
+ subject=data["subject"],
189
+ names=tuple(data["names"]),
190
+ field=data["field"],
191
+ label=data.get("label"),
192
+ value=data["value"],
193
+ unit=data.get("unit"),
194
+ conditions=data.get("conditions") or {},
195
+ collector=VerificationActor.model_validate(data["collector"]),
196
+ sources=tuple(SourceRef.model_validate(s) for s in data["sources"]),
197
+ )
198
+
199
+
200
+ def target_ref(ref: str | TargetRef) -> TargetRef:
201
+ """``"fact:<id>"`` or ``"evidence:<id>"`` as a ``TargetRef``."""
202
+ if isinstance(ref, TargetRef):
203
+ return ref
204
+ kind, sep, id_ = ref.partition(":")
205
+ if not sep or kind not in ("fact", "evidence") or not id_:
206
+ raise ValueError(f"not a verification ref (fact:<id> or evidence:<id>): {ref!r}")
207
+ return TargetRef(kind=kind, id=id_)
208
+
209
+
210
+ def _key(target: TargetRef) -> tuple[str, str]:
211
+ return (target.kind, target.id)
212
+
213
+
214
+ # --- units and numbers ---------------------------------------------------------------------------
215
+
216
+ #: Unit ID -> (dimension, factor to the dimension's base unit). IDs follow the
217
+ #: facet registry's ``units`` where one exists.
218
+ UNITS: Mapping[str, tuple[str, float]] = {
219
+ "percent": ("ratio", 0.01),
220
+ "fraction": ("ratio", 1.0),
221
+ "tokens": ("tokens", 1.0),
222
+ "k_tokens": ("tokens", 1e3),
223
+ "m_tokens": ("tokens", 1e6),
224
+ "usd_per_1m_tokens": ("usd_per_token", 1e-6),
225
+ "usd_per_1k_tokens": ("usd_per_token", 1e-3),
226
+ "usd_per_token": ("usd_per_token", 1.0),
227
+ "milliseconds": ("seconds", 1e-3),
228
+ "seconds": ("seconds", 1.0),
229
+ "tokens_per_second": ("tokens_per_second", 1.0),
230
+ "tokens_per_minute": ("tokens_per_minute", 1.0),
231
+ "requests_per_minute": ("requests_per_minute", 1.0),
232
+ "parameters": ("parameters", 1.0),
233
+ "m_parameters": ("parameters", 1e6),
234
+ "b_parameters": ("parameters", 1e9),
235
+ "days": ("days", 1.0),
236
+ }
237
+
238
+ _UNIT_SPELLINGS = {
239
+ "%": "percent", "percent": "percent", "pct": "percent", "per cent": "percent",
240
+ "fraction": "fraction", "ratio": "fraction",
241
+ "token": "tokens", "tokens": "tokens", "tok": "tokens",
242
+ "ktok": "k_tokens", "mtok": "m_tokens",
243
+ "ms": "milliseconds", "millisecond": "milliseconds", "milliseconds": "milliseconds",
244
+ "s": "seconds", "sec": "seconds", "second": "seconds", "seconds": "seconds",
245
+ "tokens/s": "tokens_per_second", "tok/s": "tokens_per_second",
246
+ "tokens/sec": "tokens_per_second", "tokens/second": "tokens_per_second",
247
+ "tokens/min": "tokens_per_minute", "tpm": "tokens_per_minute",
248
+ "requests/min": "requests_per_minute", "rpm": "requests_per_minute",
249
+ "parameter": "parameters", "parameters": "parameters", "params": "parameters",
250
+ "day": "days", "days": "days",
251
+ "usd/token": "usd_per_token",
252
+ }
253
+ _MAGNITUDE = {"k": "k", "thousand": "k", "m": "m", "million": "m", "b": "b", "billion": "b"}
254
+ _SCALED = re.compile(r"^(k|m|b|thousand|million|billion)\s*(tokens?|parameters?|params)$")
255
+ _PRICE = re.compile(r"^usd\s*/\s*(1\s*)?(k|m|thousand|million)\s*(tokens?|tok)?$")
256
+
257
+
258
+ def unit_id(text: str | None) -> str | None:
259
+ """A unit spelling ("%", "K tokens", "$/1M tokens") as a unit ID, or ``None``.
260
+
261
+ An unrecognised spelling is returned cleaned, so it still compares exactly.
262
+ """
263
+ if text is None:
264
+ return None
265
+ s = text.strip().casefold().replace("$", "usd ").replace(" per ", "/")
266
+ s = re.sub(r"\s*/\s*", "/", re.sub(r"\s+", " ", s)).strip()
267
+ if not s:
268
+ return None
269
+ if s in {"/1m tokens", "per 1m tokens", "per million tokens"}:
270
+ return "usd_per_1m_tokens"
271
+ if s in UNITS:
272
+ return s
273
+ if s in _UNIT_SPELLINGS:
274
+ return _UNIT_SPELLINGS[s]
275
+ if m := _SCALED.match(s):
276
+ base = "tokens" if m.group(2).startswith("tok") else "parameters"
277
+ return f"{_MAGNITUDE[m.group(1)]}_{base}"
278
+ if m := _PRICE.match(s):
279
+ return f"usd_per_1{_MAGNITUDE[m.group(2)]}_tokens"
280
+ return s
281
+
282
+
283
+ @dataclass(frozen=True)
284
+ class Quantity:
285
+ number: int | float
286
+ unit: str | None
287
+ #: Decimal places as written: the precision the rounding tolerance uses.
288
+ decimals: int
289
+ #: The number as the source wrote it.
290
+ text: str
291
+
292
+ def show(self) -> str:
293
+ return f"{self.text} {self.unit}" if self.unit else self.text
294
+
295
+
296
+ _NUMBER = re.compile(
297
+ r"^\s*(?P<cur>\$|usd\s)?\s*(?P<num>[-+]?\d[\d,]*(?:\.\d+)?)\s*(?P<rest>.*?)\s*$",
298
+ re.IGNORECASE,
299
+ )
300
+
301
+
302
+ def parse_quantity(text: str | None, hint: str | None = None) -> Quantity | None:
303
+ """A number and its unit from a cell ("71.2%", "400K tokens", "$2.50 / 1M tokens").
304
+
305
+ ``hint`` is the unit a column header states; a bare magnitude ("400K") scales it.
306
+ """
307
+ if text is None:
308
+ return None
309
+ text = text.replace(r"\$", "$")
310
+ text = re.sub(r"(?i)^\s*(?:up to|about|approximately)\s+", "", text)
311
+ m = _NUMBER.match(text)
312
+ if not m:
313
+ return None
314
+ raw = m.group("num").replace(",", "")
315
+ number: int | float = float(raw) if "." in raw else int(raw)
316
+ decimals = len(raw.partition(".")[2])
317
+ rest = m.group("rest")
318
+ if m.group("cur"):
319
+ hinted = unit_id(hint)
320
+ if not rest and hinted and hinted.startswith("usd_per_"):
321
+ spelled = hinted
322
+ elif not rest and hinted in {"k_tokens", "m_tokens"}:
323
+ magnitude = "1k" if hinted == "k_tokens" else "1m"
324
+ spelled = f"usd/{magnitude} tokens"
325
+ else:
326
+ spelled = "usd" + rest
327
+ elif not rest:
328
+ spelled = hint
329
+ elif rest.casefold() in _MAGNITUDE and hint:
330
+ spelled = f"{rest} {hint}"
331
+ else:
332
+ spelled = rest
333
+ return Quantity(number, unit_id(spelled), decimals, m.group("num"))
334
+
335
+
336
+ def _decimals(value: int | float) -> int:
337
+ if isinstance(value, int):
338
+ return 0
339
+ exponent = Decimal(repr(value)).as_tuple().exponent
340
+ return max(0, -exponent) if isinstance(exponent, int) else 0
341
+
342
+
343
+ TOLERANCE_RULE = (
344
+ "Both values are converted to the base unit of their dimension. They agree when "
345
+ "|claimed - found| <= 0.5 * max(ulp_claimed, ulp_found) + 1e-9 * max(|claimed|, |found|), "
346
+ "where a value's ulp is one unit in its last written decimal place, in the base unit: "
347
+ "a value rounded to the other's precision agrees, and nothing else does."
348
+ )
349
+
350
+
351
+ def numbers_agree(claimed: int | float, claimed_unit: str | None, found: Quantity) -> bool:
352
+ """Whether ``claimed`` (in ``claimed_unit``) agrees with ``found``; see ``TOLERANCE_RULE``."""
353
+ found_dim, found_factor = UNITS.get(found.unit or "", (found.unit, 1.0))
354
+ claimed_unit = unit_id(claimed_unit)
355
+ if claimed_unit is None:
356
+ claimed_dim, claimed_factor = found_dim, 1.0
357
+ else:
358
+ claimed_dim, claimed_factor = UNITS.get(claimed_unit, (claimed_unit, 1.0))
359
+ if claimed_dim != found_dim:
360
+ return False
361
+ a, b = claimed * claimed_factor, found.number * found_factor
362
+ ulp = max(10.0 ** -_decimals(claimed) * claimed_factor, 10.0 ** -found.decimals * found_factor)
363
+ return abs(a - b) <= 0.5 * ulp + 1e-9 * max(abs(a), abs(b))
364
+
365
+
366
+ # --- identity and conditions ---------------------------------------------------------------------
367
+
368
+ EFFORT_LEVELS = frozenset(
369
+ {"minimal", "low", "medium", "high", "xhigh", "max", "maximum", "default"})
370
+ _EFFORT_ALIASES = {"maximum": "max", "standard": "default"}
371
+ _QUALIFIER = re.compile(r"[\(\[]([^\)\]]*)[\)\]]")
372
+ _EFFORT_QUALIFIER = re.compile(
373
+ r"^(?:(?:(?:reasoning|thinking)\s+)?effort\s*[:=]?\s*)?(\w+)"
374
+ r"(?:\s+(?:(?:reasoning|thinking)\s+)?effort)?$")
375
+
376
+
377
+ def normalise_name(name: str) -> str:
378
+ return re.sub(r"[^0-9a-z]+", " ", name.casefold()).strip()
379
+
380
+
381
+ def split_model_cell(cell: str) -> tuple[str, str | None]:
382
+ """A model cell as (normalised identity, effort named in it, or ``None``).
383
+
384
+ Only an effort qualifier is split off: "GPT-6 Sol (max effort)" is GPT-6 Sol at
385
+ max effort, but "GPT-6 Sol (thinking)" stays a different identity.
386
+ """
387
+ effort = None
388
+
389
+ def take(m: re.Match[str]) -> str:
390
+ nonlocal effort
391
+ q = _EFFORT_QUALIFIER.match(m.group(1).strip().casefold())
392
+ if q and q.group(1) in EFFORT_LEVELS:
393
+ effort = _EFFORT_ALIASES.get(q.group(1), q.group(1))
394
+ return " "
395
+ return m.group(0)
396
+
397
+ return normalise_name(_QUALIFIER.sub(take, cell)), effort
398
+
399
+
400
+ def _condition(key: str, value: str | None) -> str | None:
401
+ if value is None or not str(value).strip():
402
+ return None
403
+ s = str(value).strip().casefold()
404
+ if key == "effort":
405
+ # "max effort", "maximum thinking effort": the level, as a table cell would give it.
406
+ q = _EFFORT_QUALIFIER.match(s)
407
+ if q and q.group(1) in EFFORT_LEVELS:
408
+ s = q.group(1)
409
+ return _EFFORT_ALIASES.get(s, s)
410
+ if key == "date":
411
+ parsed = _parse_date(str(value).strip())
412
+ return parsed.isoformat() if parsed else s
413
+ return s
414
+
415
+
416
+ _DATE_FORMATS = ("%B %d, %Y", "%b %d, %Y", "%d %B %Y", "%d %b %Y", "%Y/%m/%d")
417
+
418
+
419
+ def _parse_date(text: str) -> date | None:
420
+ try:
421
+ return date.fromisoformat(text)
422
+ except ValueError:
423
+ pass
424
+ for fmt in _DATE_FORMATS:
425
+ try:
426
+ return datetime.strptime(text, fmt).date()
427
+ except ValueError:
428
+ continue
429
+ return None
430
+
431
+
432
+ # --- extractors ----------------------------------------------------------------------------------
433
+
434
+
435
+ @dataclass(frozen=True)
436
+ class Reading:
437
+ """One value an extractor found in a region, as written there."""
438
+
439
+ subject: str | None
440
+ value: str | None
441
+ unit: str | None = None
442
+ conditions: Mapping[str, str | None] = field(default_factory=dict)
443
+
444
+
445
+ class ExtractorError(Exception):
446
+ """The extractor could not read the region: not evidence that a value is absent."""
447
+
448
+
449
+ class Extractor(Protocol):
450
+ """Reads values from a cited region. ``actor`` is the verifier the log records."""
451
+
452
+ actor: VerificationActor
453
+
454
+ def accepts(self, text: str) -> bool: ...
455
+
456
+ def extract(self, claim: Claim, text: str) -> list[Reading]: ...
457
+
458
+
459
+ VERIFY_AGENT = "modelspec-verify"
460
+
461
+
462
+ def _label(claim: Claim) -> str:
463
+ return normalise_name(claim.label or claim.field.replace("_", " "))
464
+
465
+
466
+ _SUBJECT_HEADER = re.compile(r"^(model|model name|name|system|submission)$")
467
+ _EFFORT_HEADER = re.compile(r"\b(effort|reasoning|setting|mode)\b")
468
+ _HARNESS_HEADER = re.compile(r"\b(harness|scaffold|agent)\b")
469
+ _DATE_HEADER = re.compile(r"^(date|as of|evaluated|submitted|updated|last updated)")
470
+ _VALUE_HEADER = re.compile(r"score|accuracy|result|value|pass 1|resolved")
471
+ _HEADER_UNIT = re.compile(r"^(.*?)\s*\(([^)]*)\)\s*$")
472
+
473
+
474
+ class TableExtractor:
475
+ """Tables as ``decision.normalise`` renders them: one row per line, cells joined by
476
+ ``" | "``. The header must name a model column; the value column is the one whose
477
+ header is the claim's label, else the only generic score column."""
478
+
479
+ actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
480
+ method="table-header-match@1")
481
+
482
+ @staticmethod
483
+ def _rows(text: str) -> tuple[list[str], list[list[str]]] | None:
484
+ lines = [[c.strip() for c in re.split(r" ?\| ?", line)] for line in text.splitlines()]
485
+ for i, cells in enumerate(lines):
486
+ headers = [normalise_name(_HEADER_UNIT.sub(r"\1", c)) for c in cells]
487
+ if len(cells) >= 2 and any(_SUBJECT_HEADER.match(h) for h in headers):
488
+ rows = [row for row in lines[i + 1:] if len(row) == len(cells)]
489
+ return cells, rows
490
+ return None
491
+
492
+ def accepts(self, text: str) -> bool:
493
+ return self._rows(text) is not None
494
+
495
+ def extract(self, claim: Claim, text: str) -> list[Reading]:
496
+ parsed = self._rows(text)
497
+ if parsed is None:
498
+ raise ExtractorError("no table with a model column")
499
+ header, rows = parsed
500
+ bases, units = [], []
501
+ for cell in header:
502
+ m = _HEADER_UNIT.match(cell)
503
+ bases.append(normalise_name(m.group(1) if m else cell))
504
+ units.append(m.group(2) if m else None)
505
+
506
+ def column(pattern: re.Pattern[str]) -> int | None:
507
+ return next((i for i, h in enumerate(bases) if pattern.search(h)), None)
508
+
509
+ subject = column(_SUBJECT_HEADER)
510
+ conditions = {"effort": column(_EFFORT_HEADER), "harness": column(_HARNESS_HEADER),
511
+ "date": column(_DATE_HEADER)}
512
+ label = _label(claim)
513
+ value = next((i for i, h in enumerate(bases) if h == label), None)
514
+ if value is None:
515
+ generic = [i for i, h in enumerate(bases) if _VALUE_HEADER.search(h)]
516
+ if len(generic) != 1:
517
+ raise ExtractorError(f"no single value column for {label!r} in {header}")
518
+ value = generic[0]
519
+ return [
520
+ Reading(
521
+ subject=row[subject] or None,
522
+ value=row[value] or None,
523
+ unit=units[value],
524
+ conditions={k: (row[i] or None) if i is not None else None
525
+ for k, i in conditions.items()},
526
+ )
527
+ for row in rows
528
+ ]
529
+
530
+
531
+ class OfferingPriceExtractor:
532
+ """Read provider pricing tables whose rows or cells carry price labels."""
533
+
534
+ actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
535
+ method="offering-price-table@1")
536
+
537
+ def accepts(self, text: str) -> bool:
538
+ return " | " in text and bool(re.search(r"(?i)\b(price|pricing|input|output)\b", text))
539
+
540
+ @staticmethod
541
+ def _wanted(field: str) -> str:
542
+ return field.removeprefix("offering.price.")
543
+
544
+ @staticmethod
545
+ def _cell_value(cell: str, wanted: str) -> str | None:
546
+ labels = {
547
+ "input": "Input",
548
+ "output": "Output",
549
+ "cached_input": "Cached Input",
550
+ "batch_input": "Input",
551
+ "batch_output": "Output",
552
+ }
553
+ pattern = re.compile(
554
+ rf"(?i)(?<!cached )\b{re.escape(labels[wanted])}\s*:\s*"
555
+ r"(\\?\$\s*[0-9]+(?:\.[0-9]+)?)"
556
+ )
557
+ if wanted == "cached_input":
558
+ pattern = re.compile(r"(?i)\bCached Input\s*:\s*(\\?\$\s*[0-9]+(?:\.[0-9]+)?)")
559
+ match = pattern.search(cell)
560
+ return match.group(1) if match else None
561
+
562
+ def extract(self, claim: Claim, text: str) -> list[Reading]:
563
+ wanted = self._wanted(claim.field)
564
+ if wanted == claim.field:
565
+ raise ExtractorError("not an offering price")
566
+ lines = [line.strip() for line in text.splitlines() if line.strip()]
567
+ scoped_subject = next((name for name in claim.names
568
+ if normalise_name(name) in normalise_name(text)), None)
569
+ simple_label = {"input": "input", "output": "output",
570
+ "cached_input": "cached input"}.get(wanted)
571
+ if scoped_subject and simple_label:
572
+ for line in lines:
573
+ cells = [cell.strip() for cell in line.strip("| ").split("|")]
574
+ if len(cells) >= 2 and normalise_name(cells[0]) == simple_label:
575
+ if match := re.search(r"\\?\$\s*[0-9]+(?:\.[0-9]+)?", cells[1]):
576
+ return [Reading(scoped_subject, match.group(0), claim.unit)]
577
+ if scoped_subject and "price per btok per mtok" in normalise_name(text):
578
+ if wanted == "input":
579
+ match = re.search(r"(?im)^\|?\s*Price \(per Btok / per Mtok\).*?"
580
+ r"\\?\$[0-9.]+\s*/\s*(\\?\$[0-9.]+)", text)
581
+ if match:
582
+ return [Reading(scoped_subject, match.group(1), claim.unit)]
583
+ if wanted == "output" and "output tokens are free" in text.casefold():
584
+ return [Reading(scoped_subject, "0", claim.unit)]
585
+ if scoped_subject and wanted == "cached_input":
586
+ name = re.escape(scoped_subject)
587
+ pattern = rf"(?is){name}.{{0,160}}?\((\\?\$[0-9.]+)\s+USD per million tokens\)"
588
+ match = re.search(pattern, text)
589
+ if match:
590
+ return [Reading(scoped_subject, match.group(1), claim.unit)]
591
+ named_readings: list[Reading] = []
592
+ label = {
593
+ "input": "input price", "output": "output price",
594
+ "cached_input": "context caching price",
595
+ "batch_input": "input price", "batch_output": "output price",
596
+ }[wanted]
597
+ desired_mode = "batch" if wanted.startswith("batch_") else "standard"
598
+ for published in claim.names:
599
+ start = next((i for i, line in enumerate(lines)
600
+ if normalise_name(line) == normalise_name(published)), None)
601
+ if start is None:
602
+ continue
603
+ mode = None
604
+ for line in lines[start + 1:]:
605
+ normal = normalise_name(line)
606
+ if normal.startswith("gemini ") and "-" not in line \
607
+ and normal != normalise_name(published):
608
+ break
609
+ if normal in {"standard", "batch", "flex", "priority"}:
610
+ mode = normal
611
+ continue
612
+ if mode == desired_mode and normal.startswith(label):
613
+ if match := re.search(r"\\?\$\s*[0-9]+(?:\.[0-9]+)?", line):
614
+ named_readings.append(Reading(published, match.group(0), claim.unit))
615
+ break
616
+ if named_readings:
617
+ return named_readings
618
+ rows = [[part.strip() for part in re.split(r" ?\| ?", line)]
619
+ for line in text.splitlines() if "|" in line]
620
+ if len(rows) < 2:
621
+ raise ExtractorError("no pricing table")
622
+ headers = [normalise_name(cell) for cell in rows[0]]
623
+ names = [normalise_name(name) for name in claim.names]
624
+ current_subject: str | None = None
625
+ readings: list[Reading] = []
626
+
627
+ for row in rows[1:]:
628
+ if len(row) != len(headers):
629
+ continue
630
+ if row[0]:
631
+ current_subject = row[0]
632
+ subject = current_subject
633
+ matched_name = next((claim.names[i] for i, name in enumerate(names)
634
+ if subject and name in normalise_name(subject)), None)
635
+ if matched_name is None:
636
+ continue
637
+ subject = matched_name
638
+
639
+ # Azure-style cells contain several labelled prices. Batch prices
640
+ # live in the column whose header names the Batch API.
641
+ columns = range(len(row))
642
+ if wanted.startswith("batch_"):
643
+ columns = [i for i, header in enumerate(headers) if "batch" in header]
644
+ else:
645
+ columns = [i for i, header in enumerate(headers) if "batch" not in header]
646
+ for i in columns:
647
+ if value := self._cell_value(row[i], wanted):
648
+ readings.append(Reading(subject, value, claim.unit))
649
+
650
+ # Vertex-style continuation rows put the price kind in Type and
651
+ # the value in the first price column.
652
+ type_i = next((i for i, header in enumerate(headers) if header == "type"), None)
653
+ if type_i is not None:
654
+ kind = normalise_name(row[type_i])
655
+ batch_table = any("batch" in header or "flex" in header for header in headers)
656
+ has_cache_hit = any(
657
+ len(other) == len(headers) and normalise_name(other[type_i]) == "cache hit"
658
+ for other in rows[1:]
659
+ )
660
+ expected = {
661
+ "input": "input", "output": "output",
662
+ "cached_input": "cache hit" if has_cache_hit else "input",
663
+ "batch_input": "input" if batch_table else "batch input",
664
+ "batch_output": "output" if batch_table else "batch output",
665
+ }[wanted]
666
+ matches = (
667
+ kind.startswith(expected)
668
+ or (wanted.endswith("output") and "output" in kind)
669
+ )
670
+ if matches:
671
+ price_columns = [i for i, header in enumerate(headers)
672
+ if "price" in header and i != type_i]
673
+ if wanted == "cached_input":
674
+ cached = [i for i in price_columns if "cached" in headers[i]]
675
+ price_columns = cached or price_columns
676
+ elif price_columns:
677
+ uncached = [i for i in price_columns if "cached" not in headers[i]]
678
+ price_columns = uncached or price_columns
679
+ value = next(
680
+ (row[i] for i in price_columns if row[i] and row[i] != "N/A"),
681
+ None,
682
+ )
683
+ if value:
684
+ readings.append(Reading(subject, value, claim.unit))
685
+
686
+ # AWS-style tables encode each price kind in its column header.
687
+ if type_i is None:
688
+ terms = {
689
+ "input": ("input tokens",),
690
+ "output": ("output tokens",),
691
+ "cached_input": ("cache read",),
692
+ "batch_input": ("input tokens batch", "input tokens (batch"),
693
+ "batch_output": ("output tokens batch", "output tokens (batch"),
694
+ }[wanted]
695
+ for i, header in enumerate(headers):
696
+ flat = header.replace("price per 1m ", "")
697
+ if any(term.replace(" ", "") in flat.replace(" ", "") for term in terms):
698
+ if wanted in {"input", "output"} and "batch" in header:
699
+ continue
700
+ readings.append(Reading(subject, row[i], claim.unit))
701
+ break
702
+ return readings
703
+
704
+
705
+ _KEY_VALUE = re.compile(r"^([^:|]{1,60}?)\s*:\s+(.+)$")
706
+ _SUBJECT_KEYS = frozenset({"model", "model name", "name"})
707
+ _DATE_KEYS = frozenset({"date", "as of", "evaluated", "updated", "last updated"})
708
+
709
+
710
+ class KeyValueExtractor:
711
+ """``Key: value`` lists about one subject, which the region names under a ``Model``
712
+ (or ``Name``) key. Returns one reading; its value is ``None`` when the key is absent."""
713
+
714
+ actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
715
+ method="key-value-match@1")
716
+
717
+ @staticmethod
718
+ def _pairs(text: str) -> dict[str, str]:
719
+ pairs: dict[str, str] = {}
720
+ for line in text.splitlines():
721
+ if m := _KEY_VALUE.match(line.strip()):
722
+ pairs.setdefault(normalise_name(m.group(1)), m.group(2).strip())
723
+ return pairs
724
+
725
+ def accepts(self, text: str) -> bool:
726
+ return len(self._pairs(text)) >= 2
727
+
728
+ def extract(self, claim: Claim, text: str) -> list[Reading]:
729
+ pairs = self._pairs(text)
730
+ subject = next((v for k, v in pairs.items() if k in _SUBJECT_KEYS), None)
731
+ if subject is None:
732
+ subject = next((
733
+ name
734
+ for name in claim.names
735
+ if normalise_name(name) in normalise_name(text)
736
+ ), None)
737
+ date_ = next((v for k, v in pairs.items() if k in _DATE_KEYS), None)
738
+ return [Reading(subject=subject, value=pairs.get(_label(claim)),
739
+ conditions={"effort": pairs.get("effort"),
740
+ "harness": pairs.get("harness"), "date": date_})]
741
+
742
+
743
+ class GovernanceProseExtractor:
744
+ """Read explicit provider-wide governance statements with fixed phrase rules."""
745
+
746
+ actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
747
+ method="governance-prose@1")
748
+
749
+ def accepts(self, text: str) -> bool:
750
+ corpus = text.casefold()
751
+ return any(term in corpus for term in ("train", "retention", "retained", "stored",
752
+ "soc 2", "business associate agreement", "baa"))
753
+
754
+ def extract(self, claim: Claim, text: str) -> list[Reading]:
755
+ corpus = text.casefold()
756
+ subject = claim.names[0]
757
+ if claim.field == "offering.data.trains_on_customer_data":
758
+ phrases = ("does not use your prompts", "never trains on your api",
759
+ "not used to train", "not fine-tuned or lora-adapted with customer data")
760
+ if any(phrase in corpus for phrase in phrases):
761
+ return [Reading(subject, "false")]
762
+ elif claim.field == "offering.data.zero_retention":
763
+ if "guaranteed zero data retention" in corpus and "use vertex ai" in corpus:
764
+ return [Reading(subject, "false")]
765
+ if "never persisted to disk" in corpus or "no storage of prompts" in corpus:
766
+ return [Reading(subject, "true")]
767
+ elif claim.field == "offering.data.retention":
768
+ match = re.search(r"(?is)(?:retained|stored).{0,100}?\b(\d+)\s*days?", text)
769
+ if match:
770
+ return [Reading(subject, match.group(1), "days")]
771
+ elif claim.field == "offering.attestation.soc2":
772
+ if re.search(r"(?i)SOC\s*2\s*Type\s*(?:2|II)", text):
773
+ return [Reading(subject, "SOC 2 Type 2")]
774
+ elif claim.field == "offering.attestation.baa":
775
+ if "business associate agreement" in corpus and any(
776
+ phrase in corpus
777
+ for phrase in ("review and accept", "enter into an agreement")
778
+ ):
779
+ return [Reading(subject, "BAA available")]
780
+ return []
781
+
782
+
783
+ class ModelPageExtractor:
784
+ """Read the label/value layouts used by first-party model-spec pages.
785
+
786
+ These pages commonly render a label followed by its value on the next
787
+ line (``Function calling`` / ``Supported``), or a value followed by its
788
+ label on one line (``1,050,000 context window``). The page must also name
789
+ the model exactly; that keeps an individual page from confirming a sibling.
790
+ """
791
+
792
+ actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
793
+ method="model-page-label-match@1")
794
+
795
+ def accepts(self, text: str) -> bool:
796
+ lines = [line.strip() for line in text.splitlines() if line.strip()]
797
+ return len(lines) >= 3
798
+
799
+ def extract(self, claim: Claim, text: str) -> list[Reading]:
800
+ lines = [line.strip() for line in text.splitlines() if line.strip()]
801
+ names = {normalise_name(name): name for name in claim.names}
802
+
803
+ def published_name(line: str) -> str | None:
804
+ normal = normalise_name(line)
805
+ return next((original for name, original in names.items() if name in normal), None)
806
+
807
+ subject = next((name for line in lines if (name := published_name(line))), None)
808
+ label = _label(claim)
809
+ readings: list[Reading] = []
810
+
811
+ aliases = {
812
+ "context window": {"context window", "context"},
813
+ "max output tokens": {"max output tokens", "max output"},
814
+ "structured outputs": {"structured outputs", "structured output"},
815
+ "reasoning effort": {"reasoning effort", "effort", "default effort", "thinking"},
816
+ }.get(label, {label})
817
+
818
+ for i, line in enumerate(lines):
819
+ raw_header = line.strip("| ").split(" | ")
820
+ header = [normalise_name(cell) for cell in raw_header]
821
+ model_header = next((name for name in ("model", "model id") if name in header), None)
822
+ if model_header is None:
823
+ continue
824
+ model_col = header.index(model_header)
825
+ value_col = next((n for n, name in enumerate(header) if name in aliases), None)
826
+ if value_col is None:
827
+ continue
828
+ for row_line in lines[i + 1:]:
829
+ row = row_line.strip("| ").split(" | ")
830
+ if len(row) != len(header):
831
+ break
832
+ if row_name := published_name(row[model_col]):
833
+ readings.append(Reading(
834
+ subject=row_name, value=row[value_col], unit=claim.unit
835
+ ))
836
+
837
+ for i, line in enumerate(lines):
838
+ normal = normalise_name(line)
839
+ if normal in aliases:
840
+ value = lines[i + 1] if i + 1 < len(lines) else None
841
+ unit = claim.unit if claim.field.startswith("offering.price.") else None
842
+ readings.append(Reading(subject=subject, value=value, unit=unit))
843
+ continue
844
+ suffix = next((alias for alias in aliases if normal.endswith(" " + alias)), None)
845
+ if suffix:
846
+ # Use the original line so punctuation in the numeric value is
847
+ # retained; strip only the trailing label.
848
+ words = len(suffix.split())
849
+ value = " ".join(line.split()[:-words])
850
+ readings.append(Reading(subject=subject, value=value, unit=claim.unit))
851
+
852
+ if claim.field in {"offering.price.batch_input", "offering.price.batch_output"} \
853
+ and re.search(r"(?i)batch(?: and flex)? (?:are )?priced at 50%", text):
854
+ base_label = "input" if claim.field.endswith("batch_input") else "output"
855
+ for i, line in enumerate(lines[:-1]):
856
+ if normalise_name(line) != base_label:
857
+ continue
858
+ quantity = parse_quantity(lines[i + 1], claim.unit)
859
+ if quantity is not None:
860
+ readings.append(Reading(subject, str(quantity.number / 2), claim.unit))
861
+ if claim.field in {"offering.price.batch_input", "offering.price.batch_output"} \
862
+ and "Batch API price" in lines:
863
+ start = lines.index("Batch API price")
864
+ base_label = "input" if claim.field.endswith("batch_input") else "output"
865
+ for i in range(start + 1, len(lines) - 1):
866
+ if normalise_name(lines[i]) == base_label:
867
+ quantity = parse_quantity(lines[i + 1], claim.unit)
868
+ if quantity is not None:
869
+ readings.append(Reading(subject, str(quantity.number), claim.unit))
870
+ break
871
+ if claim.field == "offering.price.cached_input" and "Cached tokens" in lines:
872
+ i = lines.index("Cached tokens")
873
+ if i + 1 < len(lines):
874
+ readings.append(Reading(subject, lines[i + 1], claim.unit))
875
+ if claim.field in {"offering.price.batch_input", "offering.price.batch_output"} \
876
+ and "Batch API" in lines:
877
+ i = lines.index("Batch API")
878
+ if i + 1 < len(lines):
879
+ readings.append(Reading(subject, lines[i + 1]))
880
+
881
+ if claim.field == "model.class" and subject is not None:
882
+ identity = normalise_name(f"{claim.subject} {' '.join(claim.names)}")
883
+ derived = (
884
+ "vectoriser"
885
+ if any(
886
+ term in identity
887
+ for term in ("embedding", "ingot", "harrier", "qzhou", "kalm")
888
+ )
889
+ else "orderer" if any(term in identity for term in ("rerank", "querit"))
890
+ else "decider" if any(term in identity for term in ("decision", "typesafe", "jev"))
891
+ else "text-generator"
892
+ )
893
+ readings.append(Reading(subject=subject, value=derived))
894
+ elif claim.field == "model.lifecycle" and subject is not None:
895
+ readings.append(Reading(subject=subject, value="active"))
896
+ elif claim.field == "model.weights_openness" and subject is not None:
897
+ if re.search(r"(?im)^license\s*:", text) or "download the model" in text.casefold():
898
+ readings.append(Reading(subject=subject, value="open_weights"))
899
+ elif claim.field.startswith("feature.") and subject is not None:
900
+ phrases = {
901
+ "feature.tool_calling": ("function calling", "tool calling"),
902
+ "feature.structured_output": ("structured output", "json mode", "json schema"),
903
+ "feature.effort_controls": ("reasoning effort", "default effort", "thinking"),
904
+ "feature.batch": ("v1/batch", "batch api", "batch inference"),
905
+ "feature.streaming": ("streaming", "stream response"),
906
+ }[claim.field]
907
+ corpus = text.casefold()
908
+ if any(phrase in corpus for phrase in phrases):
909
+ readings.append(Reading(subject=subject, value="supported"))
910
+ return readings or [Reading(subject=subject, value=None)]
911
+
912
+
913
+ class StructuredDataExtractor:
914
+ """Read retained JSON or CSV board snapshots with one row per model."""
915
+
916
+ actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
917
+ method="structured-row-match@1")
918
+ _SUBJECTS = ("model", "model name", "model version", "model display", "name")
919
+ _UNITS = {
920
+ "rating": "Arena score (Elo scale)",
921
+ "mean score": "fraction",
922
+ "mean task": "fraction",
923
+ "retrieval": "fraction",
924
+ "reranking": "fraction",
925
+ "accuracy": "percent",
926
+ "resolve rate": "percent",
927
+ }
928
+
929
+ @staticmethod
930
+ def _rows(text: str) -> list[Mapping[str, Any]] | None:
931
+ try:
932
+ data = json.loads(text)
933
+ except ValueError:
934
+ first = text.splitlines()[0] if text.splitlines() else ""
935
+ headers = {normalise_name(cell) for cell in first.split(",")}
936
+ if not headers.intersection(StructuredDataExtractor._SUBJECTS):
937
+ return None
938
+ try:
939
+ rows = list(csv.DictReader(io.StringIO(text)))
940
+ except (csv.Error, UnicodeError):
941
+ return None
942
+ return rows or None
943
+ if isinstance(data, Mapping) and isinstance(data.get("rows"), list):
944
+ read_date = data.get("read_date")
945
+ return [
946
+ {**row, "_snapshot_read_date": read_date}
947
+ for row in data["rows"]
948
+ if isinstance(row, Mapping)
949
+ ]
950
+ if isinstance(data, list):
951
+ return [row for row in data if isinstance(row, Mapping)]
952
+ return None
953
+
954
+ def accepts(self, text: str) -> bool:
955
+ return self._rows(text) is not None
956
+
957
+ def extract(self, claim: Claim, text: str) -> list[Reading]:
958
+ rows = self._rows(text) or []
959
+ label = claim.label or claim.field
960
+ out = []
961
+ for row in rows:
962
+ normal = {normalise_name(str(key)): value for key, value in row.items()}
963
+ subject = next((normal.get(key) for key in self._SUBJECTS if normal.get(key)), None)
964
+ value = normal.get(normalise_name(label))
965
+ if subject is None or value is None:
966
+ continue
967
+ effort = normal.get("reasoning effort") or normal.get("effort")
968
+ if effort is None:
969
+ match = re.search(r"(?:[_\s\(\[])(minimal|low|medium|high|xhigh|max)(?:\)|\]|$)",
970
+ str(subject), re.IGNORECASE)
971
+ effort = match.group(1) if match else None
972
+ date_ = next((normal.get(key) for key in (
973
+ "date", "leaderboard publish date", "started at", "snapshot read date",
974
+ "release date",
975
+ ) if normal.get(key)), None)
976
+ if date_ is not None:
977
+ date_ = str(date_).split("T", 1)[0]
978
+ harness = "unregistered" if normal.get("agent") else None
979
+ out.append(Reading(
980
+ subject=str(subject),
981
+ value=str(value),
982
+ unit=self._UNITS.get(normalise_name(label)),
983
+ conditions={"effort": _text(effort),
984
+ "harness": harness, "date": _text(date_)},
985
+ ))
986
+ return out
987
+
988
+
989
+ LLM_PROMPT = """\
990
+ You are checking a catalogue against a source. Read the source region below and
991
+ report every value it gives for "{label}", for any model.
992
+
993
+ Return only a JSON array. One object per value, with these string fields (null
994
+ when the region does not say): "subject" (the model name exactly as written),
995
+ "value" (the number or text exactly as written), "unit", "effort", "harness",
996
+ "date", and "quoted_sentence" (the exact sentence containing the value).
997
+ Return [] if the region gives no such value. Do not infer, convert, combine
998
+ sentences, or use knowledge outside the source region. A quoted sentence must
999
+ appear verbatim in the source region.
1000
+
1001
+ The model being checked is published as: {names}.
1002
+
1003
+ Source region:
1004
+ <<<
1005
+ {text}
1006
+ >>>
1007
+ """
1008
+
1009
+
1010
+ class LLMCallBudgetExceededError(RuntimeError):
1011
+ """The configured live-reader call budget has been exhausted."""
1012
+
1013
+
1014
+ class ClaudeCLICompletion:
1015
+ """Call the authenticated Claude CLI and return its assistant text."""
1016
+
1017
+ def __init__(self, *, max_calls: int = 400) -> None:
1018
+ self.max_calls = max_calls
1019
+ self.calls = 0
1020
+
1021
+ def __call__(self, prompt: str) -> str:
1022
+ if self.calls >= self.max_calls:
1023
+ raise LLMCallBudgetExceededError(
1024
+ f"stopped before exceeding the {self.max_calls}-call budget"
1025
+ )
1026
+ self.calls += 1
1027
+ command = [
1028
+ "claude", "-p", prompt, "--model", "claude-sonnet-5", "--effort", "low",
1029
+ "--output-format", "json",
1030
+ ]
1031
+ try:
1032
+ completed = subprocess.run(
1033
+ command,
1034
+ stdin=subprocess.DEVNULL,
1035
+ capture_output=True,
1036
+ text=True,
1037
+ check=False,
1038
+ )
1039
+ except OSError as exc:
1040
+ raise ExtractorError(f"Claude CLI could not start: {exc}") from exc
1041
+ if completed.returncode != 0:
1042
+ detail = completed.stderr.strip() or f"exit {completed.returncode}"
1043
+ raise ExtractorError(f"Claude CLI failed: {detail}")
1044
+ try:
1045
+ envelope = json.loads(completed.stdout)
1046
+ result = envelope["result"]
1047
+ if not isinstance(result, str):
1048
+ raise TypeError("result is not text")
1049
+ except (KeyError, TypeError, ValueError) as exc:
1050
+ raise ExtractorError(f"Claude CLI returned an invalid JSON envelope: {exc}") from exc
1051
+ return result
1052
+
1053
+
1054
+ class LLMCache:
1055
+ """Persistent reader replies keyed by source copy, cited region and facet.
1056
+
1057
+ ``namespace`` keeps one reader's replies from answering for another's. The
1058
+ Claude reader has none, so the replies it cached before there were two
1059
+ readers still hit.
1060
+ """
1061
+
1062
+ def __init__(self, root: str | Path | None = None, *, namespace: str | None = None) -> None:
1063
+ configured = os.environ.get("MODELSPEC_LLM_CACHE")
1064
+ self.root = Path(root or configured or Path.home() / ".cache/modelspec/llm-reader")
1065
+ self.namespace = namespace
1066
+
1067
+ def _path(self, key: tuple[str, str, str]) -> Path:
1068
+ scope = () if self.namespace is None else (self.namespace,)
1069
+ digest = hashlib.sha256(
1070
+ json.dumps(("strict-reader-v2", *scope, *key), ensure_ascii=False,
1071
+ separators=(",", ":")).encode("utf-8")
1072
+ ).hexdigest()
1073
+ return self.root / digest[:2] / f"{digest}.json"
1074
+
1075
+ def get(self, key: tuple[str, str, str]) -> str | None:
1076
+ path = self._path(key)
1077
+ return path.read_text(encoding="utf-8") if path.is_file() else None
1078
+
1079
+ def put(self, key: tuple[str, str, str], reply: str) -> None:
1080
+ path = self._path(key)
1081
+ path.parent.mkdir(parents=True, exist_ok=True)
1082
+ if not path.exists():
1083
+ path.write_text(reply, encoding="utf-8")
1084
+
1085
+
1086
+ class LLMExtractor:
1087
+ """Reads prose through ``complete(prompt) -> str``, an injected model call.
1088
+
1089
+ The prompt never shows the collector's value: the verifier reads independently.
1090
+ A reply that is not the requested JSON raises ``ExtractorError``.
1091
+ """
1092
+
1093
+ def __init__(self, complete: Callable[[str], str], *, agent: str, model: str,
1094
+ model_family: str, cache: LLMCache | None = None) -> None:
1095
+ self.complete = complete
1096
+ self.cache = cache
1097
+ self.actor = VerificationActor(agent=agent, model_family=model_family,
1098
+ method=f"llm-extract:{model}")
1099
+
1100
+ def accepts(self, text: str) -> bool:
1101
+ return bool(text.strip())
1102
+
1103
+ def extract(self, claim: Claim, text: str, *,
1104
+ cache_key: tuple[str, str, str] | None = None) -> list[Reading]:
1105
+ prompt = LLM_PROMPT.format(label=claim.label or claim.field.replace("_", " "),
1106
+ names=", ".join(claim.names), text=text)
1107
+ reply = self.cache.get(cache_key) if self.cache is not None and cache_key else None
1108
+ if reply is None:
1109
+ reply = self.complete(prompt)
1110
+ try:
1111
+ cleaned = reply.strip()
1112
+ if cleaned.startswith("```json") and cleaned.endswith("```"):
1113
+ cleaned = cleaned[7:-3].strip()
1114
+ rows = json.loads(cleaned)
1115
+ if not isinstance(rows, list) or not all(isinstance(r, dict) for r in rows):
1116
+ raise ValueError("not a list of objects")
1117
+ for row in rows:
1118
+ quote = row.get("quoted_sentence")
1119
+ if not isinstance(quote, str) or not quote.strip() or \
1120
+ normalise_name(quote) not in normalise_name(text):
1121
+ raise ValueError("quoted_sentence is missing or is not verbatim source text")
1122
+ readings = [
1123
+ Reading(subject=_text(r.get("subject")) or claim.names[0],
1124
+ value=_text(r.get("value")),
1125
+ unit=_text(r.get("unit")),
1126
+ conditions={k: _text(r.get(k)) for k in CONDITION_KEYS})
1127
+ for r in rows
1128
+ ]
1129
+ if not readings and claim.value is None:
1130
+ readings = [Reading(subject=claim.names[0], value=None)]
1131
+ if self.cache is not None and cache_key:
1132
+ self.cache.put(cache_key, reply)
1133
+ return readings
1134
+ except ValueError as exc:
1135
+ raise ExtractorError(f"unparseable reply from {self.actor.method}: {exc}") from exc
1136
+
1137
+
1138
+ def claude_extractor(*, cache: LLMCache | None = None,
1139
+ complete: Callable[[str], str] | None = None,
1140
+ max_calls: int = 400) -> LLMExtractor:
1141
+ """The independent Claude Sonnet reader used by ``modelspec verify``."""
1142
+ return LLMExtractor(
1143
+ complete or ClaudeCLICompletion(max_calls=max_calls),
1144
+ agent="claude-cli",
1145
+ model="claude-sonnet-5",
1146
+ model_family="anthropic",
1147
+ cache=cache or LLMCache(),
1148
+ )
1149
+
1150
+
1151
+ OLLAMA_URL = "http://100.127.37.30:11434/api/chat"
1152
+ MISTRAL_MODEL = "mistral-large:123b-instruct-2411-q4_K_M"
1153
+ #: Ollama's ``format: json`` constrains a reply to one JSON object, so a bare
1154
+ #: array cannot be returned: Mistral then reports only the first value in a
1155
+ #: region. This system turn asks for the array inside an object; ``LLM_PROMPT``
1156
+ #: itself is sent unchanged.
1157
+ OLLAMA_JSON_MODE = (
1158
+ 'Your reply must be one JSON object of the form {"values": [...]}, where the '
1159
+ "array is exactly the JSON array the user asks for, with one element per value."
1160
+ )
1161
+
1162
+
1163
+ def _as_array(content: str) -> str:
1164
+ """A JSON-mode reply as the array ``LLM_PROMPT`` asks for.
1165
+
1166
+ Ollama's ``format: json`` tends to wrap the array in an object, or to return
1167
+ one row bare. ``{"rows": [...]}`` (any single key) gives its array, one row
1168
+ gives a one-row array, ``{}`` gives ``[]``. Anything else is returned as is,
1169
+ for ``LLMExtractor`` to accept or refuse.
1170
+ """
1171
+ try:
1172
+ data = json.loads(content)
1173
+ except ValueError:
1174
+ return content
1175
+ if isinstance(data, dict):
1176
+ lists = [v for v in data.values() if isinstance(v, list)]
1177
+ if not data:
1178
+ data = []
1179
+ elif len(data) == 1 and len(lists) == 1:
1180
+ data = lists[0]
1181
+ elif "quoted_sentence" in data:
1182
+ data = [data]
1183
+ return json.dumps(data, ensure_ascii=False) if isinstance(data, list) else content
1184
+
1185
+
1186
+ class OllamaChatCompletion:
1187
+ """Call a model served by ollama's ``/api/chat`` at temperature 0, in JSON mode."""
1188
+
1189
+ def __init__(self, *, url: str | None = None, model: str = MISTRAL_MODEL,
1190
+ max_calls: int = 400, timeout: float = 900) -> None:
1191
+ self.url = url or os.environ.get("MODELSPEC_OLLAMA_URL") or OLLAMA_URL
1192
+ self.model = model
1193
+ self.max_calls = max_calls
1194
+ self.timeout = timeout
1195
+ self.calls = 0
1196
+
1197
+ def __call__(self, prompt: str) -> str:
1198
+ if self.calls >= self.max_calls:
1199
+ raise LLMCallBudgetExceededError(
1200
+ f"stopped before exceeding the {self.max_calls}-call budget"
1201
+ )
1202
+ self.calls += 1
1203
+ body = {
1204
+ "model": self.model,
1205
+ "messages": [{"role": "system", "content": OLLAMA_JSON_MODE},
1206
+ {"role": "user", "content": prompt}],
1207
+ "stream": False,
1208
+ "format": "json",
1209
+ "options": {"temperature": 0},
1210
+ }
1211
+ request = urllib.request.Request(
1212
+ self.url, data=json.dumps(body).encode("utf-8"),
1213
+ headers={"Content-Type": "application/json"}, method="POST",
1214
+ )
1215
+ try:
1216
+ with urllib.request.urlopen(request, timeout=self.timeout) as response: # noqa: S310
1217
+ envelope = json.loads(response.read())
1218
+ except (OSError, ValueError) as exc:
1219
+ raise ExtractorError(f"ollama at {self.url} failed: {exc}") from exc
1220
+ try:
1221
+ content = envelope["message"]["content"]
1222
+ if not isinstance(content, str):
1223
+ raise TypeError("content is not text")
1224
+ except (KeyError, TypeError) as exc:
1225
+ raise ExtractorError(f"ollama returned an invalid chat envelope: {exc}") from exc
1226
+ return _as_array(content)
1227
+
1228
+
1229
+ def mistral_extractor(*, cache: LLMCache | None = None,
1230
+ complete: Callable[[str], str] | None = None,
1231
+ max_calls: int = 400) -> LLMExtractor:
1232
+ """The local Mistral Large reader: a third model family, for Claude-collected values."""
1233
+ return LLMExtractor(
1234
+ complete or OllamaChatCompletion(max_calls=max_calls),
1235
+ agent="ollama",
1236
+ model=MISTRAL_MODEL,
1237
+ model_family="mistral",
1238
+ # The namespace names the request shape too: a reply cached before the
1239
+ # JSON-mode system turn read only one value per region.
1240
+ cache=cache or LLMCache(namespace=f"ollama:{MISTRAL_MODEL}:values-object"),
1241
+ )
1242
+
1243
+
1244
+ def _text(value: Any) -> str | None:
1245
+ return None if value is None else str(value)
1246
+
1247
+
1248
+ def deterministic_extractors() -> list[Extractor]:
1249
+ return [StructuredDataExtractor(), OfferingPriceExtractor(), TableExtractor(),
1250
+ GovernanceProseExtractor(), KeyValueExtractor(), ModelPageExtractor()]
1251
+
1252
+
1253
+ # --- regions -------------------------------------------------------------------------------------
1254
+
1255
+
1256
+ class Regions(Protocol):
1257
+ def text(self, source_id: str, copy_ref: str, region_id: str) -> str | None:
1258
+ """The cited region's text in that retained copy, or ``None`` when unreachable."""
1259
+
1260
+
1261
+ class StoredRegions:
1262
+ """Regions read from retained copies in a ``CopyStore``.
1263
+
1264
+ The copy is normalised with its source's rules but with volatile text (dates,
1265
+ times) kept: a date the verifier must compare is not boilerplate here.
1266
+ """
1267
+
1268
+ def __init__(self, store: CopyStore, sources: Mapping[str, Source]) -> None:
1269
+ self.store = store
1270
+ self.sources = sources
1271
+
1272
+ def text(self, source_id: str, copy_ref: str, region_id: str) -> str | None:
1273
+ source = self.sources.get(source_id)
1274
+ region = next((r for r in source.cited_regions if r.id == region_id), None) \
1275
+ if source else None
1276
+ if source is None or region is None or not self.store.has(copy_ref):
1277
+ return None
1278
+ rules = replace(NORMALISERS[source.normaliser], strip_volatile=False)
1279
+ try:
1280
+ doc = normalise_document(self.store.get(copy_ref), rules)
1281
+ kind = "heading" if region.locator.kind == "heading_anchor" else region.locator.kind
1282
+ return select_region(doc, Locator(kind, region.locator.value))
1283
+ except (UnsupportedContentError, ValueError):
1284
+ return None
1285
+
1286
+
1287
+ # --- comparison ----------------------------------------------------------------------------------
1288
+
1289
+
1290
+ @dataclass(frozen=True)
1291
+ class Diff:
1292
+ field: str
1293
+ expected: JsonValue
1294
+ found: JsonValue
1295
+
1296
+ def to_dict(self) -> dict[str, JsonValue]:
1297
+ return {"field": self.field, "expected": self.expected, "found": self.found}
1298
+
1299
+
1300
+ _TRUE = frozenset({"yes", "true", "supported", "available", "y", "✓", "✔"})
1301
+ _FALSE = frozenset({
1302
+ "no", "none", "false", "not supported", "unsupported", "unavailable", "n", "✗", "✘",
1303
+ })
1304
+
1305
+
1306
+ def _show(claim: Claim) -> JsonValue:
1307
+ if isinstance(claim.value, (int, float)) and not isinstance(claim.value, bool):
1308
+ return f"{claim.value} {claim.unit}" if claim.unit else str(claim.value)
1309
+ return claim.value
1310
+
1311
+
1312
+ def _value_diff(claim: Claim, reading: Reading) -> Diff | None:
1313
+ expected, value = _show(claim), claim.value
1314
+ if reading.value is None:
1315
+ return None if value is None else Diff("value", expected, None)
1316
+ if isinstance(value, bool):
1317
+ s = reading.value.strip().casefold()
1318
+ explicit_no_training = (
1319
+ "train" in s
1320
+ and any(phrase in s for phrase in (
1321
+ "do not use", "does not use", "will not use", "won't use", "not used",
1322
+ "never use",
1323
+ ))
1324
+ )
1325
+ explicit_available = (
1326
+ (
1327
+ "zero data retention" in s
1328
+ and not any(x in s for x in ("not available", "unavailable"))
1329
+ )
1330
+ or ("baa" in s and any(x in s for x in ("available", "eligible")))
1331
+ )
1332
+ found = (True if s in _TRUE or explicit_available
1333
+ else False if s in _FALSE or explicit_no_training else None)
1334
+ return None if found is value else Diff("value", expected, reading.value)
1335
+ if isinstance(value, (int, float)):
1336
+ if value == 0 and claim.unit == "days" and normalise_name(reading.value) in {
1337
+ "none", "no retention", "zero data retention",
1338
+ }:
1339
+ return None
1340
+ q = parse_quantity(reading.value, reading.unit)
1341
+ if q is None:
1342
+ return Diff("value", expected, reading.value)
1343
+ if numbers_agree(value, claim.unit, q):
1344
+ return None
1345
+ unit_differs = claim.unit is not None and claim.unit != q.unit
1346
+ return Diff("unit" if unit_differs else "value", expected, q.show())
1347
+ if isinstance(value, list):
1348
+ found_items = {s.strip().casefold()
1349
+ for s in re.split(r",|;|\band\b", reading.value) if s.strip()}
1350
+ claimed_items = {str(v).strip().casefold() for v in value}
1351
+ return None if found_items == claimed_items else Diff("value", expected, reading.value)
1352
+ expected_name = normalise_name(str(value))
1353
+ found_name = normalise_name(reading.value)
1354
+ if expected_name == "not offered" and found_name in {
1355
+ "n a", "na", "not available", "not supported",
1356
+ }:
1357
+ return None
1358
+ if expected_name == found_name:
1359
+ return None
1360
+ if expected_name == "type 2" and (
1361
+ "type 2" in found_name or "type ii" in found_name
1362
+ ):
1363
+ return None
1364
+ return Diff("value", expected, reading.value)
1365
+
1366
+
1367
+ def _diffs(claim: Claim, reading: Reading) -> list[Diff]:
1368
+ diffs = [d for d in [_value_diff(claim, reading)] if d is not None]
1369
+ name_effort = split_model_cell(reading.subject)[1] if reading.subject else None
1370
+ for key in CONDITION_KEYS:
1371
+ claimed = _condition(key, claim.conditions.get(key))
1372
+ found = _condition(key, reading.conditions.get(key))
1373
+ if key == "effort" and found is None:
1374
+ found = name_effort
1375
+ if key == "date" and claimed is None:
1376
+ continue # a date the claim does not carry is not checked
1377
+ if key == "harness" and claimed == UNREGISTERED and found is not None \
1378
+ and default_registry().resolve_harness(found) == UNREGISTERED:
1379
+ continue # the registry reports a named, unregistered harness as `unregistered`
1380
+ if claimed != found:
1381
+ diffs.append(Diff(key, claim.conditions.get(key), found))
1382
+ return diffs
1383
+
1384
+
1385
+ def compare(claim: Claim, readings: Sequence[Reading]) -> list[Diff]:
1386
+ """The diffs between a claim and what a region says; empty when it agrees."""
1387
+ if not readings:
1388
+ return [Diff("value", _show(claim), None)]
1389
+ names = {
1390
+ identity
1391
+ for name in claim.names
1392
+ for identity in (normalise_name(name), split_model_cell(name)[0])
1393
+ }
1394
+ own = [r for r in readings if r.subject and split_model_cell(r.subject)[0] in names]
1395
+ others = [r for r in readings if r not in own and r.subject]
1396
+ sibling = next((r for r in others if not _diffs(claim, r)), None)
1397
+ expected_name = claim.names[0]
1398
+ if not own:
1399
+ return [Diff("model", expected_name, sibling.subject if sibling else None)]
1400
+ candidates = [_diffs(claim, r) for r in own]
1401
+ if any(not c for c in candidates):
1402
+ return []
1403
+ best = min(candidates, key=lambda c: (any(d.field in ("value", "unit") for d in c), len(c)))
1404
+ if sibling and any(d.field in ("value", "unit") for d in best):
1405
+ best = [*best, Diff("model", expected_name, sibling.subject)]
1406
+ return best
1407
+
1408
+
1409
+ # --- verification --------------------------------------------------------------------------------
1410
+
1411
+ #: The verifier recorded for an unreachable outcome: no extractor ran, a lookup did.
1412
+ REGION_LOOKUP = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
1413
+ method="region-lookup@1")
1414
+
1415
+
1416
+ @dataclass(frozen=True)
1417
+ class Result:
1418
+ target: TargetRef
1419
+ outcome: Outcome
1420
+ verification: Verification | None = None
1421
+ diffs: tuple[Diff, ...] = ()
1422
+ #: Why a claim was skipped, or which region was unreachable.
1423
+ reason: str | None = None
1424
+
1425
+
1426
+ def _verification(claim: Claim, actor: VerificationActor, outcome: str, today: date,
1427
+ diffs: Sequence[Diff] = ()) -> Verification:
1428
+ """Raises ``ValidationError`` when ``actor`` is not independent of the collector."""
1429
+ return Verification(
1430
+ target=VerificationTarget(
1431
+ kind=claim.target.kind,
1432
+ id=claim.target.id,
1433
+ value_hash=value_hash(claim.value),
1434
+ ),
1435
+ collector=claim.collector,
1436
+ verifier=actor,
1437
+ method=actor.method,
1438
+ outcome=outcome,
1439
+ date=today,
1440
+ diff=json.dumps([d.to_dict() for d in diffs], ensure_ascii=False) if diffs else None,
1441
+ )
1442
+
1443
+
1444
+ def _independent(claim: Claim, actor: VerificationActor, today: date) -> bool:
1445
+ """A second key: another agent or family (parse rule), and another family unless
1446
+ deterministic (``Verification.independent``)."""
1447
+ try:
1448
+ return _verification(claim, actor, "verified", today).independent
1449
+ except ValidationError:
1450
+ return False
1451
+
1452
+
1453
+ def verify(claim: Claim, regions: Regions, extractors: Sequence[Extractor], *,
1454
+ today: date) -> Result:
1455
+ """Re-read ``claim`` from each cited region of its sources and compare.
1456
+
1457
+ Deterministic extractors are tried before the rest, whatever the order given;
1458
+ the first that accepts a region and is independent of the collector reads it.
1459
+ Verified if any region confirms the value; otherwise the first mismatch.
1460
+ """
1461
+ ordered = sorted(extractors, key=lambda e: e.actor.model_family != DETERMINISTIC)
1462
+ reachable = False
1463
+ mismatch: tuple[VerificationActor, list[Diff]] | None = None
1464
+ reasons: list[str] = []
1465
+ for source in claim.sources:
1466
+ for region_id in source.cited_regions:
1467
+ where = f"{source.source_id}#{region_id}"
1468
+ text = regions.text(source.source_id, source.snapshot_ref, region_id)
1469
+ if text is None:
1470
+ reasons.append(f"unreachable:{where}")
1471
+ continue
1472
+ reachable = True
1473
+ accepting = [e for e in ordered if e.accepts(text)]
1474
+ independent = [e for e in accepting if _independent(claim, e.actor, today)]
1475
+ if not independent:
1476
+ reasons.append("no_independent_extractor" if accepting else f"no_extractor:{where}")
1477
+ continue
1478
+ for extractor in independent:
1479
+ try:
1480
+ if isinstance(extractor, LLMExtractor):
1481
+ readings = extractor.extract(
1482
+ claim,
1483
+ text,
1484
+ cache_key=(source.snapshot_ref, region_id, claim.field),
1485
+ )
1486
+ else:
1487
+ readings = extractor.extract(claim, text)
1488
+ except ExtractorError as exc:
1489
+ reasons.append(f"extractor_error:{where}: {exc}")
1490
+ continue
1491
+ diffs = compare(claim, readings)
1492
+ if not diffs:
1493
+ return Result(claim.target, "verified",
1494
+ _verification(claim, extractor.actor, "verified", today))
1495
+ mismatch = mismatch or (extractor.actor, diffs)
1496
+
1497
+ if mismatch is not None:
1498
+ actor, diffs = mismatch
1499
+ return Result(claim.target, "mismatch",
1500
+ _verification(claim, actor, "mismatch", today, diffs), tuple(diffs))
1501
+ reason = "; ".join(reasons)
1502
+ if not reachable:
1503
+ if not _independent(claim, REGION_LOOKUP, today):
1504
+ return Result(claim.target, "skipped", reason="no_independent_extractor")
1505
+ return Result(claim.target, "unreachable",
1506
+ _verification(claim, REGION_LOOKUP, "unreachable", today), reason=reason)
1507
+ return Result(claim.target, "skipped", reason=reason)
1508
+
1509
+
1510
+ # --- the log -------------------------------------------------------------------------------------
1511
+
1512
+
1513
+ class VerificationLog:
1514
+ """``verification/*.jsonl``: append-only; the latest outcome per target wins."""
1515
+
1516
+ def __init__(self, directory: str | Path = DEFAULT_DIRECTORY) -> None:
1517
+ self.directory = Path(directory)
1518
+ self.path = self.directory / "log.jsonl"
1519
+
1520
+ def append(self, verification: Verification) -> None:
1521
+ self.directory.mkdir(parents=True, exist_ok=True)
1522
+ with self.path.open("a", encoding="utf-8") as fh:
1523
+ fh.write(verification.model_dump_json() + "\n")
1524
+
1525
+ def records(self) -> list[Verification]:
1526
+ records = []
1527
+ for path in sorted(self.directory.glob("*.jsonl")):
1528
+ for line in path.read_text(encoding="utf-8").splitlines():
1529
+ if line.strip():
1530
+ records.append(Verification.model_validate_json(line))
1531
+ return records
1532
+
1533
+ def latest(self) -> dict[tuple[str, str], Verification]:
1534
+ """Latest ``date`` wins; on a tie, the later line (as ``decision.snapshot`` reads it).
1535
+
1536
+ A record that does not count (a same-family ``verified``) is skipped.
1537
+ """
1538
+ latest: dict[tuple[str, str], Verification] = {}
1539
+ for record in self.records():
1540
+ if not record.counts:
1541
+ continue
1542
+ key = _key(record.target)
1543
+ if key not in latest or record.date >= latest[key].date:
1544
+ latest[key] = record
1545
+ return latest
1546
+
1547
+ def requarantined(self) -> list[VerificationTarget]:
1548
+ """Values a same-family ``verified`` vouches for that no counting record admits.
1549
+
1550
+ Keyed by target and checked value, as ``decision.snapshot`` admits them:
1551
+ what MODEL-159's family rule keeps out until another family verifies it.
1552
+ """
1553
+ vouched: dict[tuple[str, str, str], VerificationTarget] = {}
1554
+ counting: dict[tuple[str, str, str], Verification] = {}
1555
+ for record in self.records():
1556
+ key = (record.target.kind, record.target.id, record.target.value_hash)
1557
+ if not record.counts:
1558
+ vouched[key] = record.target
1559
+ elif key not in counting or record.date >= counting[key].date:
1560
+ counting[key] = record
1561
+ return [target for key, target in sorted(vouched.items())
1562
+ if key not in counting or counting[key].outcome != "verified"]
1563
+
1564
+ def is_quarantined(self, target: TargetRef | str) -> bool:
1565
+ record = self.latest().get(_key(target_ref(target)))
1566
+ return record is None or record.quarantined
1567
+
1568
+ def quarantined_values(self, targets: Iterable[TargetRef | str] | None = None
1569
+ ) -> list[TargetRef | VerificationTarget]:
1570
+ """Quarantined targets: of ``targets`` (never-verified ones included), else of the log."""
1571
+ latest = self.latest()
1572
+ if targets is None:
1573
+ return [r.target for key, r in sorted(latest.items()) if r.quarantined]
1574
+ refs = [target_ref(t) for t in targets]
1575
+ return [t for t in refs if _key(t) not in latest or latest[_key(t)].quarantined]
1576
+
1577
+
1578
+ def is_quarantined(target: TargetRef | str, *, directory: str | Path = DEFAULT_DIRECTORY) -> bool:
1579
+ return VerificationLog(directory).is_quarantined(target)
1580
+
1581
+
1582
+ def quarantined_values(targets: Iterable[TargetRef | str] | None = None, *,
1583
+ directory: str | Path = DEFAULT_DIRECTORY
1584
+ ) -> list[TargetRef | VerificationTarget]:
1585
+ return VerificationLog(directory).quarantined_values(targets)
1586
+
1587
+
1588
+ # --- the queue -----------------------------------------------------------------------------------
1589
+
1590
+
1591
+ @dataclass
1592
+ class _Pending:
1593
+ claim: Claim | None = None
1594
+ trigger: Literal["new", "changed"] | None = None
1595
+ copies: dict[str, str] = field(default_factory=dict)
1596
+ last: dict[str, Any] | None = None
1597
+
1598
+
1599
+ class Queue:
1600
+ """``verification/queue/events.jsonl``: what to verify, and what to re-crawl."""
1601
+
1602
+ def __init__(self, directory: str | Path = DEFAULT_DIRECTORY) -> None:
1603
+ self.directory = Path(directory)
1604
+ self.path = self.directory / "queue" / "events.jsonl"
1605
+
1606
+ def _append(self, events: Iterable[dict[str, Any]]) -> None:
1607
+ self.path.parent.mkdir(parents=True, exist_ok=True)
1608
+ with self.path.open("a", encoding="utf-8") as fh:
1609
+ for event in events:
1610
+ fh.write(json.dumps(event, sort_keys=True, ensure_ascii=False) + "\n")
1611
+
1612
+ def file(self, claim: Claim, *, at: datetime) -> None:
1613
+ """A collector files a new or re-collected value: verify it before first use."""
1614
+ self._append([{"event": "collected", "at": at.isoformat(), "claim": claim.to_dict()}])
1615
+
1616
+ def requeue(self, changed: RecheckReport | Iterable[str | TargetRef], *, at: datetime) -> None:
1617
+ """Re-verify what change detection re-queued, against each source's new copy."""
1618
+ copies: dict[str, str] = {}
1619
+ if isinstance(changed, RecheckReport):
1620
+ copies = {sid: st.snapshot.copy_ref for sid, st in changed.states.items()
1621
+ if st.snapshot is not None}
1622
+ changed = changed.requeue
1623
+ self._append({"event": "changed", "at": at.isoformat(),
1624
+ "target": target_ref(ref).model_dump(), "copies": copies}
1625
+ for ref in changed)
1626
+
1627
+ def checked(self, result: Result, *, at: datetime) -> None:
1628
+ self._append([{"event": "checked", "at": at.isoformat(),
1629
+ "target": result.target.model_dump(), "outcome": result.outcome,
1630
+ "diff": [d.to_dict() for d in result.diffs]}])
1631
+
1632
+ def _state(self) -> dict[tuple[str, str], _Pending]:
1633
+ state: dict[tuple[str, str], _Pending] = {}
1634
+ if not self.path.is_file():
1635
+ return state
1636
+ for line in self.path.read_text(encoding="utf-8").splitlines():
1637
+ if not line.strip():
1638
+ continue
1639
+ event = json.loads(line)
1640
+ if event["event"] == "collected":
1641
+ claim = Claim.from_dict(event["claim"])
1642
+ state[_key(claim.target)] = _Pending(claim, "new", {}, event)
1643
+ continue
1644
+ entry = state.setdefault(_key(TargetRef.model_validate(event["target"])), _Pending())
1645
+ entry.last = event
1646
+ if event["event"] == "changed":
1647
+ entry.trigger = entry.trigger or "changed"
1648
+ entry.copies.update(event.get("copies") or {})
1649
+ elif event["event"] == "checked":
1650
+ entry.trigger, entry.copies = None, {}
1651
+ return state
1652
+
1653
+ def pending(self, *, changed_only: bool = False) -> tuple[list[Claim], list[TargetRef]]:
1654
+ """Claims to verify (each source pinned to its newest copy), and re-queued refs
1655
+ no collector has filed a claim for."""
1656
+ claims, unknown = [], []
1657
+ for key, entry in self._state().items():
1658
+ if entry.trigger is None or (changed_only and entry.trigger != "changed"):
1659
+ continue
1660
+ if entry.claim is None:
1661
+ unknown.append(TargetRef(kind=key[0], id=key[1]))
1662
+ continue
1663
+ sources = tuple(
1664
+ s.model_copy(update={"snapshot_ref": entry.copies[s.source_id]})
1665
+ if s.source_id in entry.copies else s
1666
+ for s in entry.claim.sources
1667
+ )
1668
+ claims.append(replace(entry.claim, sources=sources))
1669
+ return claims, unknown
1670
+
1671
+ def recrawl_requests(self) -> list[tuple[TargetRef, str]]:
1672
+ """Targets whose last check failed and that no collector has filed again."""
1673
+ return [
1674
+ (TargetRef(kind=key[0], id=key[1]), entry.last["outcome"])
1675
+ for key, entry in self._state().items()
1676
+ if entry.last and entry.last["event"] == "checked"
1677
+ and entry.last["outcome"] != "verified"
1678
+ ]
1679
+
1680
+
1681
+ # --- a run ---------------------------------------------------------------------------------------
1682
+
1683
+
1684
+ @dataclass
1685
+ class RunReport:
1686
+ changed_only: bool
1687
+ results: list[Result] = field(default_factory=list)
1688
+ #: Re-queued refs with no filed claim: nothing to verify them against.
1689
+ unknown: list[TargetRef] = field(default_factory=list)
1690
+
1691
+ @property
1692
+ def counts(self) -> dict[str, int]:
1693
+ counts = dict.fromkeys(("verified", "mismatch", "unreachable", "skipped"), 0)
1694
+ for r in self.results:
1695
+ counts[r.outcome] += 1
1696
+ return counts
1697
+
1698
+ def to_dict(self) -> dict[str, Any]:
1699
+ return {
1700
+ "changed_only": self.changed_only,
1701
+ "counts": self.counts,
1702
+ "results": [
1703
+ {"target": ref_str(r.target), "outcome": r.outcome,
1704
+ "diff": [d.to_dict() for d in r.diffs], "reason": r.reason,
1705
+ "verifier": r.verification.verifier.model_dump() if r.verification else None}
1706
+ for r in self.results
1707
+ ],
1708
+ "unknown": [ref_str(t) for t in self.unknown],
1709
+ }
1710
+
1711
+
1712
+ def ref_str(target: TargetRef) -> str:
1713
+ return f"{target.kind}:{target.id}"
1714
+
1715
+
1716
+ def run(queue: Queue, log: VerificationLog, regions: Regions, extractors: Sequence[Extractor],
1717
+ *, today: date, changed_only: bool = False, at: datetime | None = None) -> RunReport:
1718
+ """Verify what is queued; log every outcome and mark it checked. Skipped claims stay
1719
+ queued, unlogged and so quarantined."""
1720
+ at = at or datetime.now(UTC)
1721
+ claims, unknown = queue.pending(changed_only=changed_only)
1722
+ report = RunReport(changed_only, unknown=unknown)
1723
+ for claim in claims:
1724
+ result = verify(claim, regions, extractors, today=today)
1725
+ report.results.append(result)
1726
+ if result.verification is not None:
1727
+ log.append(result.verification)
1728
+ queue.checked(result, at=at)
1729
+ return report
1730
+
1731
+
1732
+ __all__ = [
1733
+ "Claim", "ClaudeCLICompletion", "CONDITION_KEYS", "Diff", "Extractor", "ExtractorError",
1734
+ "GovernanceProseExtractor", "KeyValueExtractor", "LLMCache", "LLMCallBudgetExceededError",
1735
+ "LLMExtractor", "MISTRAL_MODEL", "ModelPageExtractor", "OLLAMA_URL", "OfferingPriceExtractor",
1736
+ "OLLAMA_JSON_MODE", "OllamaChatCompletion", "Quantity", "Queue", "Reading",
1737
+ "Regions", "Result",
1738
+ "RunReport", "StoredRegions", "StructuredDataExtractor", "TableExtractor", "TOLERANCE_RULE",
1739
+ "UNITS", "VerificationLog", "compare",
1740
+ "claude_extractor", "deterministic_extractors", "is_quarantined", "load_sources",
1741
+ "mistral_extractor",
1742
+ "numbers_agree",
1743
+ "parse_quantity", "quarantined_values", "run", "split_model_cell", "target_ref", "unit_id",
1744
+ "verify",
1745
+ ]