modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
decision/verify.py
ADDED
|
@@ -0,0 +1,1745 @@
|
|
|
1
|
+
"""Two-key verification and quarantine (MODEL-140; design §5, "Two keys").
|
|
2
|
+
|
|
3
|
+
The agent that collects a value never verifies it. A verifier re-reads the
|
|
4
|
+
value from the cited region of the retained source copy, with its own
|
|
5
|
+
extractor, and compares:
|
|
6
|
+
|
|
7
|
+
- the **model identity** in the region is the subject, not a sibling variant
|
|
8
|
+
("GPT-6 Sol" is not "GPT-6 Astra"; "Nimbus 3 (max effort)" is Nimbus 3 at
|
|
9
|
+
max effort, never at default);
|
|
10
|
+
- the **value** agrees after unit normalisation, within ``TOLERANCE_RULE``;
|
|
11
|
+
- the **unit** is the one the collector filed;
|
|
12
|
+
- the **conditions** (effort, harness, date) are the ones the source states.
|
|
13
|
+
|
|
14
|
+
Extractors are pluggable. The deterministic ones (``TableExtractor``,
|
|
15
|
+
``KeyValueExtractor``) always run before any other; ``LLMExtractor`` reads
|
|
16
|
+
prose through an injected completion function, so nothing here calls a model
|
|
17
|
+
or the network on its own: ``claude_extractor`` (Claude Sonnet, via the Claude
|
|
18
|
+
CLI) and ``mistral_extractor`` (Mistral Large, via ollama) are the two wired
|
|
19
|
+
readers. Each extractor's actor (agent, model family, method) is the verifier
|
|
20
|
+
the log records. Two keys means another model family (MODEL-159): a reader
|
|
21
|
+
from the collector's family is never asked, and a same-family ``verified``
|
|
22
|
+
already in the log does not count (``Verification.counts``).
|
|
23
|
+
|
|
24
|
+
Outcomes are ``verified``, ``mismatch`` (with a structured diff) and
|
|
25
|
+
``unreachable`` (the copy, source or region is missing). A claim no
|
|
26
|
+
independent extractor could read is ``skipped``: nothing is logged and it
|
|
27
|
+
stays queued. Anything whose latest logged outcome is not ``verified``, and
|
|
28
|
+
anything never verified, is **quarantined**.
|
|
29
|
+
|
|
30
|
+
Files, under ``verification/`` at the repository root:
|
|
31
|
+
|
|
32
|
+
- ``log.jsonl``: the verification log, append-only, one
|
|
33
|
+
``decision.model.Verification`` per line. The latest counting outcome per
|
|
34
|
+
target and checked value wins: latest ``date``, and on a tie the later line
|
|
35
|
+
(the rule ``decision.snapshot`` applies when it reads
|
|
36
|
+
``verification/log.jsonl``).
|
|
37
|
+
- ``queue/events.jsonl``: append-only work queue. A collector files a claim
|
|
38
|
+
(``collected``), change detection re-queues one (``changed``), a run records
|
|
39
|
+
what it checked (``checked``). Mismatched and unreachable targets stay listed
|
|
40
|
+
by ``Queue.recrawl_requests`` until a collector files them again.
|
|
41
|
+
|
|
42
|
+
Refs from ``decision.sources`` (``Citation.ref``, ``RecheckReport.requeue``)
|
|
43
|
+
are ``"fact:<id>"`` or ``"evidence:<id>"``.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
from __future__ import annotations
|
|
47
|
+
|
|
48
|
+
import csv
|
|
49
|
+
import hashlib
|
|
50
|
+
import io
|
|
51
|
+
import json
|
|
52
|
+
import os
|
|
53
|
+
import re
|
|
54
|
+
import subprocess
|
|
55
|
+
import urllib.request
|
|
56
|
+
from collections.abc import Callable, Iterable, Mapping, Sequence
|
|
57
|
+
from dataclasses import dataclass, field, replace
|
|
58
|
+
from datetime import UTC, date, datetime
|
|
59
|
+
from decimal import Decimal
|
|
60
|
+
from pathlib import Path
|
|
61
|
+
from typing import Any, Literal, Protocol
|
|
62
|
+
|
|
63
|
+
from pydantic import JsonValue, ValidationError
|
|
64
|
+
|
|
65
|
+
from decision.model import (
|
|
66
|
+
DETERMINISTIC,
|
|
67
|
+
SourceRef,
|
|
68
|
+
TargetRef,
|
|
69
|
+
Verification,
|
|
70
|
+
VerificationActor,
|
|
71
|
+
VerificationTarget,
|
|
72
|
+
value_hash,
|
|
73
|
+
)
|
|
74
|
+
from decision.normalise import (
|
|
75
|
+
NORMALISERS,
|
|
76
|
+
Locator,
|
|
77
|
+
UnsupportedContentError,
|
|
78
|
+
normalise_document,
|
|
79
|
+
select_region,
|
|
80
|
+
)
|
|
81
|
+
from decision.registry import UNREGISTERED
|
|
82
|
+
from decision.registry import default as default_registry
|
|
83
|
+
from decision.sources import CopyStore, RecheckReport, Source, load_sources
|
|
84
|
+
|
|
85
|
+
REPO_ROOT = Path(__file__).resolve().parent.parent
|
|
86
|
+
DEFAULT_DIRECTORY = REPO_ROOT / "verification"
|
|
87
|
+
|
|
88
|
+
Outcome = Literal["verified", "mismatch", "unreachable", "skipped"]
|
|
89
|
+
CONDITION_KEYS = ("effort", "harness", "date")
|
|
90
|
+
|
|
91
|
+
# --- claims --------------------------------------------------------------------------------------
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@dataclass(frozen=True)
|
|
95
|
+
class Claim:
|
|
96
|
+
"""One value as a collector filed it, with what the verifier needs to re-read it.
|
|
97
|
+
|
|
98
|
+
``names`` are the names the subject is published under, taken from the
|
|
99
|
+
catalogue, never from the collector's own label for the row: that label is
|
|
100
|
+
what a copied-score error gets wrong. ``label`` is the column or key the value
|
|
101
|
+
sits under in the source (default: ``field`` with underscores as spaces).
|
|
102
|
+
``unit`` is a unit ID (``UNITS``); ``None`` means the base unit of whatever
|
|
103
|
+
dimension the source states. ``conditions`` holds ``effort``, ``harness`` and
|
|
104
|
+
``date`` as the collector filed them.
|
|
105
|
+
"""
|
|
106
|
+
|
|
107
|
+
target: TargetRef
|
|
108
|
+
subject: str
|
|
109
|
+
names: tuple[str, ...]
|
|
110
|
+
field: str
|
|
111
|
+
value: JsonValue
|
|
112
|
+
collector: VerificationActor
|
|
113
|
+
sources: tuple[SourceRef, ...]
|
|
114
|
+
unit: str | None = None
|
|
115
|
+
label: str | None = None
|
|
116
|
+
conditions: Mapping[str, str | None] = field(default_factory=dict)
|
|
117
|
+
|
|
118
|
+
def __post_init__(self) -> None:
|
|
119
|
+
if not self.names:
|
|
120
|
+
raise ValueError(f"{self.target.id}: a claim needs the subject's published names")
|
|
121
|
+
if not self.sources:
|
|
122
|
+
raise ValueError(f"{self.target.id}: a claim needs at least one source")
|
|
123
|
+
unknown = set(self.conditions) - set(CONDITION_KEYS)
|
|
124
|
+
if unknown:
|
|
125
|
+
raise ValueError(f"{self.target.id}: unknown conditions {sorted(unknown)}")
|
|
126
|
+
|
|
127
|
+
@classmethod
|
|
128
|
+
def from_evidence(cls, evidence: Any, *, names: Sequence[str],
|
|
129
|
+
collector: VerificationActor, label: str | None = None) -> Claim:
|
|
130
|
+
"""A claim for a ``decision.model.Evidence`` row with an ID, subject and sources."""
|
|
131
|
+
if evidence.id is None or evidence.subject is None:
|
|
132
|
+
raise ValueError("evidence needs an ID and a subject to be verified")
|
|
133
|
+
return cls(
|
|
134
|
+
target=TargetRef(kind="evidence", id=evidence.id),
|
|
135
|
+
subject=evidence.subject.id,
|
|
136
|
+
names=tuple(names),
|
|
137
|
+
field=evidence.benchmark_id,
|
|
138
|
+
label=label,
|
|
139
|
+
value=evidence.score,
|
|
140
|
+
unit=evidence.unit,
|
|
141
|
+
conditions={"effort": evidence.effort, "harness": evidence.harness,
|
|
142
|
+
"date": evidence.evidence_date},
|
|
143
|
+
collector=collector,
|
|
144
|
+
sources=tuple(evidence.sources),
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
@classmethod
|
|
148
|
+
def from_fact(cls, fact: Any, *, names: Sequence[str], collector: VerificationActor,
|
|
149
|
+
unit: str | None = None, label: str | None = None) -> Claim:
|
|
150
|
+
"""A claim for a filed ``Fact``; ``unit`` is its facet's unit.
|
|
151
|
+
|
|
152
|
+
A scoped source region can also confirm ``not_disclosed`` or
|
|
153
|
+
``requires_contract`` by naming the subject while omitting the facet.
|
|
154
|
+
``unknown`` is not a filed value and cannot be verified.
|
|
155
|
+
"""
|
|
156
|
+
if fact.state == "unknown":
|
|
157
|
+
raise ValueError(f"{fact.id}: an unknown fact has no value to verify")
|
|
158
|
+
return cls(
|
|
159
|
+
target=TargetRef(kind="fact", id=fact.id),
|
|
160
|
+
subject=fact.subject.id,
|
|
161
|
+
names=tuple(names),
|
|
162
|
+
field=fact.facet,
|
|
163
|
+
label=label,
|
|
164
|
+
value=fact.value,
|
|
165
|
+
unit=unit,
|
|
166
|
+
collector=collector,
|
|
167
|
+
sources=tuple(fact.sources),
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
def to_dict(self) -> dict[str, Any]:
|
|
171
|
+
return {
|
|
172
|
+
"target": self.target.model_dump(),
|
|
173
|
+
"subject": self.subject,
|
|
174
|
+
"names": list(self.names),
|
|
175
|
+
"field": self.field,
|
|
176
|
+
"label": self.label,
|
|
177
|
+
"value": self.value,
|
|
178
|
+
"unit": self.unit,
|
|
179
|
+
"conditions": dict(self.conditions),
|
|
180
|
+
"collector": self.collector.model_dump(),
|
|
181
|
+
"sources": [s.model_dump() for s in self.sources],
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
@classmethod
|
|
185
|
+
def from_dict(cls, data: Mapping[str, Any]) -> Claim:
|
|
186
|
+
return cls(
|
|
187
|
+
target=TargetRef.model_validate(data["target"]),
|
|
188
|
+
subject=data["subject"],
|
|
189
|
+
names=tuple(data["names"]),
|
|
190
|
+
field=data["field"],
|
|
191
|
+
label=data.get("label"),
|
|
192
|
+
value=data["value"],
|
|
193
|
+
unit=data.get("unit"),
|
|
194
|
+
conditions=data.get("conditions") or {},
|
|
195
|
+
collector=VerificationActor.model_validate(data["collector"]),
|
|
196
|
+
sources=tuple(SourceRef.model_validate(s) for s in data["sources"]),
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def target_ref(ref: str | TargetRef) -> TargetRef:
|
|
201
|
+
"""``"fact:<id>"`` or ``"evidence:<id>"`` as a ``TargetRef``."""
|
|
202
|
+
if isinstance(ref, TargetRef):
|
|
203
|
+
return ref
|
|
204
|
+
kind, sep, id_ = ref.partition(":")
|
|
205
|
+
if not sep or kind not in ("fact", "evidence") or not id_:
|
|
206
|
+
raise ValueError(f"not a verification ref (fact:<id> or evidence:<id>): {ref!r}")
|
|
207
|
+
return TargetRef(kind=kind, id=id_)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _key(target: TargetRef) -> tuple[str, str]:
|
|
211
|
+
return (target.kind, target.id)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
# --- units and numbers ---------------------------------------------------------------------------
|
|
215
|
+
|
|
216
|
+
#: Unit ID -> (dimension, factor to the dimension's base unit). IDs follow the
|
|
217
|
+
#: facet registry's ``units`` where one exists.
|
|
218
|
+
UNITS: Mapping[str, tuple[str, float]] = {
|
|
219
|
+
"percent": ("ratio", 0.01),
|
|
220
|
+
"fraction": ("ratio", 1.0),
|
|
221
|
+
"tokens": ("tokens", 1.0),
|
|
222
|
+
"k_tokens": ("tokens", 1e3),
|
|
223
|
+
"m_tokens": ("tokens", 1e6),
|
|
224
|
+
"usd_per_1m_tokens": ("usd_per_token", 1e-6),
|
|
225
|
+
"usd_per_1k_tokens": ("usd_per_token", 1e-3),
|
|
226
|
+
"usd_per_token": ("usd_per_token", 1.0),
|
|
227
|
+
"milliseconds": ("seconds", 1e-3),
|
|
228
|
+
"seconds": ("seconds", 1.0),
|
|
229
|
+
"tokens_per_second": ("tokens_per_second", 1.0),
|
|
230
|
+
"tokens_per_minute": ("tokens_per_minute", 1.0),
|
|
231
|
+
"requests_per_minute": ("requests_per_minute", 1.0),
|
|
232
|
+
"parameters": ("parameters", 1.0),
|
|
233
|
+
"m_parameters": ("parameters", 1e6),
|
|
234
|
+
"b_parameters": ("parameters", 1e9),
|
|
235
|
+
"days": ("days", 1.0),
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
_UNIT_SPELLINGS = {
|
|
239
|
+
"%": "percent", "percent": "percent", "pct": "percent", "per cent": "percent",
|
|
240
|
+
"fraction": "fraction", "ratio": "fraction",
|
|
241
|
+
"token": "tokens", "tokens": "tokens", "tok": "tokens",
|
|
242
|
+
"ktok": "k_tokens", "mtok": "m_tokens",
|
|
243
|
+
"ms": "milliseconds", "millisecond": "milliseconds", "milliseconds": "milliseconds",
|
|
244
|
+
"s": "seconds", "sec": "seconds", "second": "seconds", "seconds": "seconds",
|
|
245
|
+
"tokens/s": "tokens_per_second", "tok/s": "tokens_per_second",
|
|
246
|
+
"tokens/sec": "tokens_per_second", "tokens/second": "tokens_per_second",
|
|
247
|
+
"tokens/min": "tokens_per_minute", "tpm": "tokens_per_minute",
|
|
248
|
+
"requests/min": "requests_per_minute", "rpm": "requests_per_minute",
|
|
249
|
+
"parameter": "parameters", "parameters": "parameters", "params": "parameters",
|
|
250
|
+
"day": "days", "days": "days",
|
|
251
|
+
"usd/token": "usd_per_token",
|
|
252
|
+
}
|
|
253
|
+
_MAGNITUDE = {"k": "k", "thousand": "k", "m": "m", "million": "m", "b": "b", "billion": "b"}
|
|
254
|
+
_SCALED = re.compile(r"^(k|m|b|thousand|million|billion)\s*(tokens?|parameters?|params)$")
|
|
255
|
+
_PRICE = re.compile(r"^usd\s*/\s*(1\s*)?(k|m|thousand|million)\s*(tokens?|tok)?$")
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def unit_id(text: str | None) -> str | None:
|
|
259
|
+
"""A unit spelling ("%", "K tokens", "$/1M tokens") as a unit ID, or ``None``.
|
|
260
|
+
|
|
261
|
+
An unrecognised spelling is returned cleaned, so it still compares exactly.
|
|
262
|
+
"""
|
|
263
|
+
if text is None:
|
|
264
|
+
return None
|
|
265
|
+
s = text.strip().casefold().replace("$", "usd ").replace(" per ", "/")
|
|
266
|
+
s = re.sub(r"\s*/\s*", "/", re.sub(r"\s+", " ", s)).strip()
|
|
267
|
+
if not s:
|
|
268
|
+
return None
|
|
269
|
+
if s in {"/1m tokens", "per 1m tokens", "per million tokens"}:
|
|
270
|
+
return "usd_per_1m_tokens"
|
|
271
|
+
if s in UNITS:
|
|
272
|
+
return s
|
|
273
|
+
if s in _UNIT_SPELLINGS:
|
|
274
|
+
return _UNIT_SPELLINGS[s]
|
|
275
|
+
if m := _SCALED.match(s):
|
|
276
|
+
base = "tokens" if m.group(2).startswith("tok") else "parameters"
|
|
277
|
+
return f"{_MAGNITUDE[m.group(1)]}_{base}"
|
|
278
|
+
if m := _PRICE.match(s):
|
|
279
|
+
return f"usd_per_1{_MAGNITUDE[m.group(2)]}_tokens"
|
|
280
|
+
return s
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
@dataclass(frozen=True)
|
|
284
|
+
class Quantity:
|
|
285
|
+
number: int | float
|
|
286
|
+
unit: str | None
|
|
287
|
+
#: Decimal places as written: the precision the rounding tolerance uses.
|
|
288
|
+
decimals: int
|
|
289
|
+
#: The number as the source wrote it.
|
|
290
|
+
text: str
|
|
291
|
+
|
|
292
|
+
def show(self) -> str:
|
|
293
|
+
return f"{self.text} {self.unit}" if self.unit else self.text
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
_NUMBER = re.compile(
|
|
297
|
+
r"^\s*(?P<cur>\$|usd\s)?\s*(?P<num>[-+]?\d[\d,]*(?:\.\d+)?)\s*(?P<rest>.*?)\s*$",
|
|
298
|
+
re.IGNORECASE,
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def parse_quantity(text: str | None, hint: str | None = None) -> Quantity | None:
|
|
303
|
+
"""A number and its unit from a cell ("71.2%", "400K tokens", "$2.50 / 1M tokens").
|
|
304
|
+
|
|
305
|
+
``hint`` is the unit a column header states; a bare magnitude ("400K") scales it.
|
|
306
|
+
"""
|
|
307
|
+
if text is None:
|
|
308
|
+
return None
|
|
309
|
+
text = text.replace(r"\$", "$")
|
|
310
|
+
text = re.sub(r"(?i)^\s*(?:up to|about|approximately)\s+", "", text)
|
|
311
|
+
m = _NUMBER.match(text)
|
|
312
|
+
if not m:
|
|
313
|
+
return None
|
|
314
|
+
raw = m.group("num").replace(",", "")
|
|
315
|
+
number: int | float = float(raw) if "." in raw else int(raw)
|
|
316
|
+
decimals = len(raw.partition(".")[2])
|
|
317
|
+
rest = m.group("rest")
|
|
318
|
+
if m.group("cur"):
|
|
319
|
+
hinted = unit_id(hint)
|
|
320
|
+
if not rest and hinted and hinted.startswith("usd_per_"):
|
|
321
|
+
spelled = hinted
|
|
322
|
+
elif not rest and hinted in {"k_tokens", "m_tokens"}:
|
|
323
|
+
magnitude = "1k" if hinted == "k_tokens" else "1m"
|
|
324
|
+
spelled = f"usd/{magnitude} tokens"
|
|
325
|
+
else:
|
|
326
|
+
spelled = "usd" + rest
|
|
327
|
+
elif not rest:
|
|
328
|
+
spelled = hint
|
|
329
|
+
elif rest.casefold() in _MAGNITUDE and hint:
|
|
330
|
+
spelled = f"{rest} {hint}"
|
|
331
|
+
else:
|
|
332
|
+
spelled = rest
|
|
333
|
+
return Quantity(number, unit_id(spelled), decimals, m.group("num"))
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def _decimals(value: int | float) -> int:
|
|
337
|
+
if isinstance(value, int):
|
|
338
|
+
return 0
|
|
339
|
+
exponent = Decimal(repr(value)).as_tuple().exponent
|
|
340
|
+
return max(0, -exponent) if isinstance(exponent, int) else 0
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
TOLERANCE_RULE = (
|
|
344
|
+
"Both values are converted to the base unit of their dimension. They agree when "
|
|
345
|
+
"|claimed - found| <= 0.5 * max(ulp_claimed, ulp_found) + 1e-9 * max(|claimed|, |found|), "
|
|
346
|
+
"where a value's ulp is one unit in its last written decimal place, in the base unit: "
|
|
347
|
+
"a value rounded to the other's precision agrees, and nothing else does."
|
|
348
|
+
)
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def numbers_agree(claimed: int | float, claimed_unit: str | None, found: Quantity) -> bool:
|
|
352
|
+
"""Whether ``claimed`` (in ``claimed_unit``) agrees with ``found``; see ``TOLERANCE_RULE``."""
|
|
353
|
+
found_dim, found_factor = UNITS.get(found.unit or "", (found.unit, 1.0))
|
|
354
|
+
claimed_unit = unit_id(claimed_unit)
|
|
355
|
+
if claimed_unit is None:
|
|
356
|
+
claimed_dim, claimed_factor = found_dim, 1.0
|
|
357
|
+
else:
|
|
358
|
+
claimed_dim, claimed_factor = UNITS.get(claimed_unit, (claimed_unit, 1.0))
|
|
359
|
+
if claimed_dim != found_dim:
|
|
360
|
+
return False
|
|
361
|
+
a, b = claimed * claimed_factor, found.number * found_factor
|
|
362
|
+
ulp = max(10.0 ** -_decimals(claimed) * claimed_factor, 10.0 ** -found.decimals * found_factor)
|
|
363
|
+
return abs(a - b) <= 0.5 * ulp + 1e-9 * max(abs(a), abs(b))
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
# --- identity and conditions ---------------------------------------------------------------------
|
|
367
|
+
|
|
368
|
+
EFFORT_LEVELS = frozenset(
|
|
369
|
+
{"minimal", "low", "medium", "high", "xhigh", "max", "maximum", "default"})
|
|
370
|
+
_EFFORT_ALIASES = {"maximum": "max", "standard": "default"}
|
|
371
|
+
_QUALIFIER = re.compile(r"[\(\[]([^\)\]]*)[\)\]]")
|
|
372
|
+
_EFFORT_QUALIFIER = re.compile(
|
|
373
|
+
r"^(?:(?:(?:reasoning|thinking)\s+)?effort\s*[:=]?\s*)?(\w+)"
|
|
374
|
+
r"(?:\s+(?:(?:reasoning|thinking)\s+)?effort)?$")
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def normalise_name(name: str) -> str:
|
|
378
|
+
return re.sub(r"[^0-9a-z]+", " ", name.casefold()).strip()
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def split_model_cell(cell: str) -> tuple[str, str | None]:
|
|
382
|
+
"""A model cell as (normalised identity, effort named in it, or ``None``).
|
|
383
|
+
|
|
384
|
+
Only an effort qualifier is split off: "GPT-6 Sol (max effort)" is GPT-6 Sol at
|
|
385
|
+
max effort, but "GPT-6 Sol (thinking)" stays a different identity.
|
|
386
|
+
"""
|
|
387
|
+
effort = None
|
|
388
|
+
|
|
389
|
+
def take(m: re.Match[str]) -> str:
|
|
390
|
+
nonlocal effort
|
|
391
|
+
q = _EFFORT_QUALIFIER.match(m.group(1).strip().casefold())
|
|
392
|
+
if q and q.group(1) in EFFORT_LEVELS:
|
|
393
|
+
effort = _EFFORT_ALIASES.get(q.group(1), q.group(1))
|
|
394
|
+
return " "
|
|
395
|
+
return m.group(0)
|
|
396
|
+
|
|
397
|
+
return normalise_name(_QUALIFIER.sub(take, cell)), effort
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def _condition(key: str, value: str | None) -> str | None:
|
|
401
|
+
if value is None or not str(value).strip():
|
|
402
|
+
return None
|
|
403
|
+
s = str(value).strip().casefold()
|
|
404
|
+
if key == "effort":
|
|
405
|
+
# "max effort", "maximum thinking effort": the level, as a table cell would give it.
|
|
406
|
+
q = _EFFORT_QUALIFIER.match(s)
|
|
407
|
+
if q and q.group(1) in EFFORT_LEVELS:
|
|
408
|
+
s = q.group(1)
|
|
409
|
+
return _EFFORT_ALIASES.get(s, s)
|
|
410
|
+
if key == "date":
|
|
411
|
+
parsed = _parse_date(str(value).strip())
|
|
412
|
+
return parsed.isoformat() if parsed else s
|
|
413
|
+
return s
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
_DATE_FORMATS = ("%B %d, %Y", "%b %d, %Y", "%d %B %Y", "%d %b %Y", "%Y/%m/%d")
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def _parse_date(text: str) -> date | None:
|
|
420
|
+
try:
|
|
421
|
+
return date.fromisoformat(text)
|
|
422
|
+
except ValueError:
|
|
423
|
+
pass
|
|
424
|
+
for fmt in _DATE_FORMATS:
|
|
425
|
+
try:
|
|
426
|
+
return datetime.strptime(text, fmt).date()
|
|
427
|
+
except ValueError:
|
|
428
|
+
continue
|
|
429
|
+
return None
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
# --- extractors ----------------------------------------------------------------------------------
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
@dataclass(frozen=True)
|
|
436
|
+
class Reading:
|
|
437
|
+
"""One value an extractor found in a region, as written there."""
|
|
438
|
+
|
|
439
|
+
subject: str | None
|
|
440
|
+
value: str | None
|
|
441
|
+
unit: str | None = None
|
|
442
|
+
conditions: Mapping[str, str | None] = field(default_factory=dict)
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
class ExtractorError(Exception):
|
|
446
|
+
"""The extractor could not read the region: not evidence that a value is absent."""
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
class Extractor(Protocol):
|
|
450
|
+
"""Reads values from a cited region. ``actor`` is the verifier the log records."""
|
|
451
|
+
|
|
452
|
+
actor: VerificationActor
|
|
453
|
+
|
|
454
|
+
def accepts(self, text: str) -> bool: ...
|
|
455
|
+
|
|
456
|
+
def extract(self, claim: Claim, text: str) -> list[Reading]: ...
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
VERIFY_AGENT = "modelspec-verify"
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _label(claim: Claim) -> str:
|
|
463
|
+
return normalise_name(claim.label or claim.field.replace("_", " "))
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
_SUBJECT_HEADER = re.compile(r"^(model|model name|name|system|submission)$")
|
|
467
|
+
_EFFORT_HEADER = re.compile(r"\b(effort|reasoning|setting|mode)\b")
|
|
468
|
+
_HARNESS_HEADER = re.compile(r"\b(harness|scaffold|agent)\b")
|
|
469
|
+
_DATE_HEADER = re.compile(r"^(date|as of|evaluated|submitted|updated|last updated)")
|
|
470
|
+
_VALUE_HEADER = re.compile(r"score|accuracy|result|value|pass 1|resolved")
|
|
471
|
+
_HEADER_UNIT = re.compile(r"^(.*?)\s*\(([^)]*)\)\s*$")
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
class TableExtractor:
|
|
475
|
+
"""Tables as ``decision.normalise`` renders them: one row per line, cells joined by
|
|
476
|
+
``" | "``. The header must name a model column; the value column is the one whose
|
|
477
|
+
header is the claim's label, else the only generic score column."""
|
|
478
|
+
|
|
479
|
+
actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
|
|
480
|
+
method="table-header-match@1")
|
|
481
|
+
|
|
482
|
+
@staticmethod
|
|
483
|
+
def _rows(text: str) -> tuple[list[str], list[list[str]]] | None:
|
|
484
|
+
lines = [[c.strip() for c in re.split(r" ?\| ?", line)] for line in text.splitlines()]
|
|
485
|
+
for i, cells in enumerate(lines):
|
|
486
|
+
headers = [normalise_name(_HEADER_UNIT.sub(r"\1", c)) for c in cells]
|
|
487
|
+
if len(cells) >= 2 and any(_SUBJECT_HEADER.match(h) for h in headers):
|
|
488
|
+
rows = [row for row in lines[i + 1:] if len(row) == len(cells)]
|
|
489
|
+
return cells, rows
|
|
490
|
+
return None
|
|
491
|
+
|
|
492
|
+
def accepts(self, text: str) -> bool:
|
|
493
|
+
return self._rows(text) is not None
|
|
494
|
+
|
|
495
|
+
def extract(self, claim: Claim, text: str) -> list[Reading]:
|
|
496
|
+
parsed = self._rows(text)
|
|
497
|
+
if parsed is None:
|
|
498
|
+
raise ExtractorError("no table with a model column")
|
|
499
|
+
header, rows = parsed
|
|
500
|
+
bases, units = [], []
|
|
501
|
+
for cell in header:
|
|
502
|
+
m = _HEADER_UNIT.match(cell)
|
|
503
|
+
bases.append(normalise_name(m.group(1) if m else cell))
|
|
504
|
+
units.append(m.group(2) if m else None)
|
|
505
|
+
|
|
506
|
+
def column(pattern: re.Pattern[str]) -> int | None:
|
|
507
|
+
return next((i for i, h in enumerate(bases) if pattern.search(h)), None)
|
|
508
|
+
|
|
509
|
+
subject = column(_SUBJECT_HEADER)
|
|
510
|
+
conditions = {"effort": column(_EFFORT_HEADER), "harness": column(_HARNESS_HEADER),
|
|
511
|
+
"date": column(_DATE_HEADER)}
|
|
512
|
+
label = _label(claim)
|
|
513
|
+
value = next((i for i, h in enumerate(bases) if h == label), None)
|
|
514
|
+
if value is None:
|
|
515
|
+
generic = [i for i, h in enumerate(bases) if _VALUE_HEADER.search(h)]
|
|
516
|
+
if len(generic) != 1:
|
|
517
|
+
raise ExtractorError(f"no single value column for {label!r} in {header}")
|
|
518
|
+
value = generic[0]
|
|
519
|
+
return [
|
|
520
|
+
Reading(
|
|
521
|
+
subject=row[subject] or None,
|
|
522
|
+
value=row[value] or None,
|
|
523
|
+
unit=units[value],
|
|
524
|
+
conditions={k: (row[i] or None) if i is not None else None
|
|
525
|
+
for k, i in conditions.items()},
|
|
526
|
+
)
|
|
527
|
+
for row in rows
|
|
528
|
+
]
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
class OfferingPriceExtractor:
|
|
532
|
+
"""Read provider pricing tables whose rows or cells carry price labels."""
|
|
533
|
+
|
|
534
|
+
actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
|
|
535
|
+
method="offering-price-table@1")
|
|
536
|
+
|
|
537
|
+
def accepts(self, text: str) -> bool:
|
|
538
|
+
return " | " in text and bool(re.search(r"(?i)\b(price|pricing|input|output)\b", text))
|
|
539
|
+
|
|
540
|
+
@staticmethod
|
|
541
|
+
def _wanted(field: str) -> str:
|
|
542
|
+
return field.removeprefix("offering.price.")
|
|
543
|
+
|
|
544
|
+
@staticmethod
|
|
545
|
+
def _cell_value(cell: str, wanted: str) -> str | None:
|
|
546
|
+
labels = {
|
|
547
|
+
"input": "Input",
|
|
548
|
+
"output": "Output",
|
|
549
|
+
"cached_input": "Cached Input",
|
|
550
|
+
"batch_input": "Input",
|
|
551
|
+
"batch_output": "Output",
|
|
552
|
+
}
|
|
553
|
+
pattern = re.compile(
|
|
554
|
+
rf"(?i)(?<!cached )\b{re.escape(labels[wanted])}\s*:\s*"
|
|
555
|
+
r"(\\?\$\s*[0-9]+(?:\.[0-9]+)?)"
|
|
556
|
+
)
|
|
557
|
+
if wanted == "cached_input":
|
|
558
|
+
pattern = re.compile(r"(?i)\bCached Input\s*:\s*(\\?\$\s*[0-9]+(?:\.[0-9]+)?)")
|
|
559
|
+
match = pattern.search(cell)
|
|
560
|
+
return match.group(1) if match else None
|
|
561
|
+
|
|
562
|
+
def extract(self, claim: Claim, text: str) -> list[Reading]:
|
|
563
|
+
wanted = self._wanted(claim.field)
|
|
564
|
+
if wanted == claim.field:
|
|
565
|
+
raise ExtractorError("not an offering price")
|
|
566
|
+
lines = [line.strip() for line in text.splitlines() if line.strip()]
|
|
567
|
+
scoped_subject = next((name for name in claim.names
|
|
568
|
+
if normalise_name(name) in normalise_name(text)), None)
|
|
569
|
+
simple_label = {"input": "input", "output": "output",
|
|
570
|
+
"cached_input": "cached input"}.get(wanted)
|
|
571
|
+
if scoped_subject and simple_label:
|
|
572
|
+
for line in lines:
|
|
573
|
+
cells = [cell.strip() for cell in line.strip("| ").split("|")]
|
|
574
|
+
if len(cells) >= 2 and normalise_name(cells[0]) == simple_label:
|
|
575
|
+
if match := re.search(r"\\?\$\s*[0-9]+(?:\.[0-9]+)?", cells[1]):
|
|
576
|
+
return [Reading(scoped_subject, match.group(0), claim.unit)]
|
|
577
|
+
if scoped_subject and "price per btok per mtok" in normalise_name(text):
|
|
578
|
+
if wanted == "input":
|
|
579
|
+
match = re.search(r"(?im)^\|?\s*Price \(per Btok / per Mtok\).*?"
|
|
580
|
+
r"\\?\$[0-9.]+\s*/\s*(\\?\$[0-9.]+)", text)
|
|
581
|
+
if match:
|
|
582
|
+
return [Reading(scoped_subject, match.group(1), claim.unit)]
|
|
583
|
+
if wanted == "output" and "output tokens are free" in text.casefold():
|
|
584
|
+
return [Reading(scoped_subject, "0", claim.unit)]
|
|
585
|
+
if scoped_subject and wanted == "cached_input":
|
|
586
|
+
name = re.escape(scoped_subject)
|
|
587
|
+
pattern = rf"(?is){name}.{{0,160}}?\((\\?\$[0-9.]+)\s+USD per million tokens\)"
|
|
588
|
+
match = re.search(pattern, text)
|
|
589
|
+
if match:
|
|
590
|
+
return [Reading(scoped_subject, match.group(1), claim.unit)]
|
|
591
|
+
named_readings: list[Reading] = []
|
|
592
|
+
label = {
|
|
593
|
+
"input": "input price", "output": "output price",
|
|
594
|
+
"cached_input": "context caching price",
|
|
595
|
+
"batch_input": "input price", "batch_output": "output price",
|
|
596
|
+
}[wanted]
|
|
597
|
+
desired_mode = "batch" if wanted.startswith("batch_") else "standard"
|
|
598
|
+
for published in claim.names:
|
|
599
|
+
start = next((i for i, line in enumerate(lines)
|
|
600
|
+
if normalise_name(line) == normalise_name(published)), None)
|
|
601
|
+
if start is None:
|
|
602
|
+
continue
|
|
603
|
+
mode = None
|
|
604
|
+
for line in lines[start + 1:]:
|
|
605
|
+
normal = normalise_name(line)
|
|
606
|
+
if normal.startswith("gemini ") and "-" not in line \
|
|
607
|
+
and normal != normalise_name(published):
|
|
608
|
+
break
|
|
609
|
+
if normal in {"standard", "batch", "flex", "priority"}:
|
|
610
|
+
mode = normal
|
|
611
|
+
continue
|
|
612
|
+
if mode == desired_mode and normal.startswith(label):
|
|
613
|
+
if match := re.search(r"\\?\$\s*[0-9]+(?:\.[0-9]+)?", line):
|
|
614
|
+
named_readings.append(Reading(published, match.group(0), claim.unit))
|
|
615
|
+
break
|
|
616
|
+
if named_readings:
|
|
617
|
+
return named_readings
|
|
618
|
+
rows = [[part.strip() for part in re.split(r" ?\| ?", line)]
|
|
619
|
+
for line in text.splitlines() if "|" in line]
|
|
620
|
+
if len(rows) < 2:
|
|
621
|
+
raise ExtractorError("no pricing table")
|
|
622
|
+
headers = [normalise_name(cell) for cell in rows[0]]
|
|
623
|
+
names = [normalise_name(name) for name in claim.names]
|
|
624
|
+
current_subject: str | None = None
|
|
625
|
+
readings: list[Reading] = []
|
|
626
|
+
|
|
627
|
+
for row in rows[1:]:
|
|
628
|
+
if len(row) != len(headers):
|
|
629
|
+
continue
|
|
630
|
+
if row[0]:
|
|
631
|
+
current_subject = row[0]
|
|
632
|
+
subject = current_subject
|
|
633
|
+
matched_name = next((claim.names[i] for i, name in enumerate(names)
|
|
634
|
+
if subject and name in normalise_name(subject)), None)
|
|
635
|
+
if matched_name is None:
|
|
636
|
+
continue
|
|
637
|
+
subject = matched_name
|
|
638
|
+
|
|
639
|
+
# Azure-style cells contain several labelled prices. Batch prices
|
|
640
|
+
# live in the column whose header names the Batch API.
|
|
641
|
+
columns = range(len(row))
|
|
642
|
+
if wanted.startswith("batch_"):
|
|
643
|
+
columns = [i for i, header in enumerate(headers) if "batch" in header]
|
|
644
|
+
else:
|
|
645
|
+
columns = [i for i, header in enumerate(headers) if "batch" not in header]
|
|
646
|
+
for i in columns:
|
|
647
|
+
if value := self._cell_value(row[i], wanted):
|
|
648
|
+
readings.append(Reading(subject, value, claim.unit))
|
|
649
|
+
|
|
650
|
+
# Vertex-style continuation rows put the price kind in Type and
|
|
651
|
+
# the value in the first price column.
|
|
652
|
+
type_i = next((i for i, header in enumerate(headers) if header == "type"), None)
|
|
653
|
+
if type_i is not None:
|
|
654
|
+
kind = normalise_name(row[type_i])
|
|
655
|
+
batch_table = any("batch" in header or "flex" in header for header in headers)
|
|
656
|
+
has_cache_hit = any(
|
|
657
|
+
len(other) == len(headers) and normalise_name(other[type_i]) == "cache hit"
|
|
658
|
+
for other in rows[1:]
|
|
659
|
+
)
|
|
660
|
+
expected = {
|
|
661
|
+
"input": "input", "output": "output",
|
|
662
|
+
"cached_input": "cache hit" if has_cache_hit else "input",
|
|
663
|
+
"batch_input": "input" if batch_table else "batch input",
|
|
664
|
+
"batch_output": "output" if batch_table else "batch output",
|
|
665
|
+
}[wanted]
|
|
666
|
+
matches = (
|
|
667
|
+
kind.startswith(expected)
|
|
668
|
+
or (wanted.endswith("output") and "output" in kind)
|
|
669
|
+
)
|
|
670
|
+
if matches:
|
|
671
|
+
price_columns = [i for i, header in enumerate(headers)
|
|
672
|
+
if "price" in header and i != type_i]
|
|
673
|
+
if wanted == "cached_input":
|
|
674
|
+
cached = [i for i in price_columns if "cached" in headers[i]]
|
|
675
|
+
price_columns = cached or price_columns
|
|
676
|
+
elif price_columns:
|
|
677
|
+
uncached = [i for i in price_columns if "cached" not in headers[i]]
|
|
678
|
+
price_columns = uncached or price_columns
|
|
679
|
+
value = next(
|
|
680
|
+
(row[i] for i in price_columns if row[i] and row[i] != "N/A"),
|
|
681
|
+
None,
|
|
682
|
+
)
|
|
683
|
+
if value:
|
|
684
|
+
readings.append(Reading(subject, value, claim.unit))
|
|
685
|
+
|
|
686
|
+
# AWS-style tables encode each price kind in its column header.
|
|
687
|
+
if type_i is None:
|
|
688
|
+
terms = {
|
|
689
|
+
"input": ("input tokens",),
|
|
690
|
+
"output": ("output tokens",),
|
|
691
|
+
"cached_input": ("cache read",),
|
|
692
|
+
"batch_input": ("input tokens batch", "input tokens (batch"),
|
|
693
|
+
"batch_output": ("output tokens batch", "output tokens (batch"),
|
|
694
|
+
}[wanted]
|
|
695
|
+
for i, header in enumerate(headers):
|
|
696
|
+
flat = header.replace("price per 1m ", "")
|
|
697
|
+
if any(term.replace(" ", "") in flat.replace(" ", "") for term in terms):
|
|
698
|
+
if wanted in {"input", "output"} and "batch" in header:
|
|
699
|
+
continue
|
|
700
|
+
readings.append(Reading(subject, row[i], claim.unit))
|
|
701
|
+
break
|
|
702
|
+
return readings
|
|
703
|
+
|
|
704
|
+
|
|
705
|
+
_KEY_VALUE = re.compile(r"^([^:|]{1,60}?)\s*:\s+(.+)$")
|
|
706
|
+
_SUBJECT_KEYS = frozenset({"model", "model name", "name"})
|
|
707
|
+
_DATE_KEYS = frozenset({"date", "as of", "evaluated", "updated", "last updated"})
|
|
708
|
+
|
|
709
|
+
|
|
710
|
+
class KeyValueExtractor:
|
|
711
|
+
"""``Key: value`` lists about one subject, which the region names under a ``Model``
|
|
712
|
+
(or ``Name``) key. Returns one reading; its value is ``None`` when the key is absent."""
|
|
713
|
+
|
|
714
|
+
actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
|
|
715
|
+
method="key-value-match@1")
|
|
716
|
+
|
|
717
|
+
@staticmethod
|
|
718
|
+
def _pairs(text: str) -> dict[str, str]:
|
|
719
|
+
pairs: dict[str, str] = {}
|
|
720
|
+
for line in text.splitlines():
|
|
721
|
+
if m := _KEY_VALUE.match(line.strip()):
|
|
722
|
+
pairs.setdefault(normalise_name(m.group(1)), m.group(2).strip())
|
|
723
|
+
return pairs
|
|
724
|
+
|
|
725
|
+
def accepts(self, text: str) -> bool:
|
|
726
|
+
return len(self._pairs(text)) >= 2
|
|
727
|
+
|
|
728
|
+
def extract(self, claim: Claim, text: str) -> list[Reading]:
|
|
729
|
+
pairs = self._pairs(text)
|
|
730
|
+
subject = next((v for k, v in pairs.items() if k in _SUBJECT_KEYS), None)
|
|
731
|
+
if subject is None:
|
|
732
|
+
subject = next((
|
|
733
|
+
name
|
|
734
|
+
for name in claim.names
|
|
735
|
+
if normalise_name(name) in normalise_name(text)
|
|
736
|
+
), None)
|
|
737
|
+
date_ = next((v for k, v in pairs.items() if k in _DATE_KEYS), None)
|
|
738
|
+
return [Reading(subject=subject, value=pairs.get(_label(claim)),
|
|
739
|
+
conditions={"effort": pairs.get("effort"),
|
|
740
|
+
"harness": pairs.get("harness"), "date": date_})]
|
|
741
|
+
|
|
742
|
+
|
|
743
|
+
class GovernanceProseExtractor:
|
|
744
|
+
"""Read explicit provider-wide governance statements with fixed phrase rules."""
|
|
745
|
+
|
|
746
|
+
actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
|
|
747
|
+
method="governance-prose@1")
|
|
748
|
+
|
|
749
|
+
def accepts(self, text: str) -> bool:
|
|
750
|
+
corpus = text.casefold()
|
|
751
|
+
return any(term in corpus for term in ("train", "retention", "retained", "stored",
|
|
752
|
+
"soc 2", "business associate agreement", "baa"))
|
|
753
|
+
|
|
754
|
+
def extract(self, claim: Claim, text: str) -> list[Reading]:
|
|
755
|
+
corpus = text.casefold()
|
|
756
|
+
subject = claim.names[0]
|
|
757
|
+
if claim.field == "offering.data.trains_on_customer_data":
|
|
758
|
+
phrases = ("does not use your prompts", "never trains on your api",
|
|
759
|
+
"not used to train", "not fine-tuned or lora-adapted with customer data")
|
|
760
|
+
if any(phrase in corpus for phrase in phrases):
|
|
761
|
+
return [Reading(subject, "false")]
|
|
762
|
+
elif claim.field == "offering.data.zero_retention":
|
|
763
|
+
if "guaranteed zero data retention" in corpus and "use vertex ai" in corpus:
|
|
764
|
+
return [Reading(subject, "false")]
|
|
765
|
+
if "never persisted to disk" in corpus or "no storage of prompts" in corpus:
|
|
766
|
+
return [Reading(subject, "true")]
|
|
767
|
+
elif claim.field == "offering.data.retention":
|
|
768
|
+
match = re.search(r"(?is)(?:retained|stored).{0,100}?\b(\d+)\s*days?", text)
|
|
769
|
+
if match:
|
|
770
|
+
return [Reading(subject, match.group(1), "days")]
|
|
771
|
+
elif claim.field == "offering.attestation.soc2":
|
|
772
|
+
if re.search(r"(?i)SOC\s*2\s*Type\s*(?:2|II)", text):
|
|
773
|
+
return [Reading(subject, "SOC 2 Type 2")]
|
|
774
|
+
elif claim.field == "offering.attestation.baa":
|
|
775
|
+
if "business associate agreement" in corpus and any(
|
|
776
|
+
phrase in corpus
|
|
777
|
+
for phrase in ("review and accept", "enter into an agreement")
|
|
778
|
+
):
|
|
779
|
+
return [Reading(subject, "BAA available")]
|
|
780
|
+
return []
|
|
781
|
+
|
|
782
|
+
|
|
783
|
+
class ModelPageExtractor:
|
|
784
|
+
"""Read the label/value layouts used by first-party model-spec pages.
|
|
785
|
+
|
|
786
|
+
These pages commonly render a label followed by its value on the next
|
|
787
|
+
line (``Function calling`` / ``Supported``), or a value followed by its
|
|
788
|
+
label on one line (``1,050,000 context window``). The page must also name
|
|
789
|
+
the model exactly; that keeps an individual page from confirming a sibling.
|
|
790
|
+
"""
|
|
791
|
+
|
|
792
|
+
actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
|
|
793
|
+
method="model-page-label-match@1")
|
|
794
|
+
|
|
795
|
+
def accepts(self, text: str) -> bool:
|
|
796
|
+
lines = [line.strip() for line in text.splitlines() if line.strip()]
|
|
797
|
+
return len(lines) >= 3
|
|
798
|
+
|
|
799
|
+
def extract(self, claim: Claim, text: str) -> list[Reading]:
|
|
800
|
+
lines = [line.strip() for line in text.splitlines() if line.strip()]
|
|
801
|
+
names = {normalise_name(name): name for name in claim.names}
|
|
802
|
+
|
|
803
|
+
def published_name(line: str) -> str | None:
|
|
804
|
+
normal = normalise_name(line)
|
|
805
|
+
return next((original for name, original in names.items() if name in normal), None)
|
|
806
|
+
|
|
807
|
+
subject = next((name for line in lines if (name := published_name(line))), None)
|
|
808
|
+
label = _label(claim)
|
|
809
|
+
readings: list[Reading] = []
|
|
810
|
+
|
|
811
|
+
aliases = {
|
|
812
|
+
"context window": {"context window", "context"},
|
|
813
|
+
"max output tokens": {"max output tokens", "max output"},
|
|
814
|
+
"structured outputs": {"structured outputs", "structured output"},
|
|
815
|
+
"reasoning effort": {"reasoning effort", "effort", "default effort", "thinking"},
|
|
816
|
+
}.get(label, {label})
|
|
817
|
+
|
|
818
|
+
for i, line in enumerate(lines):
|
|
819
|
+
raw_header = line.strip("| ").split(" | ")
|
|
820
|
+
header = [normalise_name(cell) for cell in raw_header]
|
|
821
|
+
model_header = next((name for name in ("model", "model id") if name in header), None)
|
|
822
|
+
if model_header is None:
|
|
823
|
+
continue
|
|
824
|
+
model_col = header.index(model_header)
|
|
825
|
+
value_col = next((n for n, name in enumerate(header) if name in aliases), None)
|
|
826
|
+
if value_col is None:
|
|
827
|
+
continue
|
|
828
|
+
for row_line in lines[i + 1:]:
|
|
829
|
+
row = row_line.strip("| ").split(" | ")
|
|
830
|
+
if len(row) != len(header):
|
|
831
|
+
break
|
|
832
|
+
if row_name := published_name(row[model_col]):
|
|
833
|
+
readings.append(Reading(
|
|
834
|
+
subject=row_name, value=row[value_col], unit=claim.unit
|
|
835
|
+
))
|
|
836
|
+
|
|
837
|
+
for i, line in enumerate(lines):
|
|
838
|
+
normal = normalise_name(line)
|
|
839
|
+
if normal in aliases:
|
|
840
|
+
value = lines[i + 1] if i + 1 < len(lines) else None
|
|
841
|
+
unit = claim.unit if claim.field.startswith("offering.price.") else None
|
|
842
|
+
readings.append(Reading(subject=subject, value=value, unit=unit))
|
|
843
|
+
continue
|
|
844
|
+
suffix = next((alias for alias in aliases if normal.endswith(" " + alias)), None)
|
|
845
|
+
if suffix:
|
|
846
|
+
# Use the original line so punctuation in the numeric value is
|
|
847
|
+
# retained; strip only the trailing label.
|
|
848
|
+
words = len(suffix.split())
|
|
849
|
+
value = " ".join(line.split()[:-words])
|
|
850
|
+
readings.append(Reading(subject=subject, value=value, unit=claim.unit))
|
|
851
|
+
|
|
852
|
+
if claim.field in {"offering.price.batch_input", "offering.price.batch_output"} \
|
|
853
|
+
and re.search(r"(?i)batch(?: and flex)? (?:are )?priced at 50%", text):
|
|
854
|
+
base_label = "input" if claim.field.endswith("batch_input") else "output"
|
|
855
|
+
for i, line in enumerate(lines[:-1]):
|
|
856
|
+
if normalise_name(line) != base_label:
|
|
857
|
+
continue
|
|
858
|
+
quantity = parse_quantity(lines[i + 1], claim.unit)
|
|
859
|
+
if quantity is not None:
|
|
860
|
+
readings.append(Reading(subject, str(quantity.number / 2), claim.unit))
|
|
861
|
+
if claim.field in {"offering.price.batch_input", "offering.price.batch_output"} \
|
|
862
|
+
and "Batch API price" in lines:
|
|
863
|
+
start = lines.index("Batch API price")
|
|
864
|
+
base_label = "input" if claim.field.endswith("batch_input") else "output"
|
|
865
|
+
for i in range(start + 1, len(lines) - 1):
|
|
866
|
+
if normalise_name(lines[i]) == base_label:
|
|
867
|
+
quantity = parse_quantity(lines[i + 1], claim.unit)
|
|
868
|
+
if quantity is not None:
|
|
869
|
+
readings.append(Reading(subject, str(quantity.number), claim.unit))
|
|
870
|
+
break
|
|
871
|
+
if claim.field == "offering.price.cached_input" and "Cached tokens" in lines:
|
|
872
|
+
i = lines.index("Cached tokens")
|
|
873
|
+
if i + 1 < len(lines):
|
|
874
|
+
readings.append(Reading(subject, lines[i + 1], claim.unit))
|
|
875
|
+
if claim.field in {"offering.price.batch_input", "offering.price.batch_output"} \
|
|
876
|
+
and "Batch API" in lines:
|
|
877
|
+
i = lines.index("Batch API")
|
|
878
|
+
if i + 1 < len(lines):
|
|
879
|
+
readings.append(Reading(subject, lines[i + 1]))
|
|
880
|
+
|
|
881
|
+
if claim.field == "model.class" and subject is not None:
|
|
882
|
+
identity = normalise_name(f"{claim.subject} {' '.join(claim.names)}")
|
|
883
|
+
derived = (
|
|
884
|
+
"vectoriser"
|
|
885
|
+
if any(
|
|
886
|
+
term in identity
|
|
887
|
+
for term in ("embedding", "ingot", "harrier", "qzhou", "kalm")
|
|
888
|
+
)
|
|
889
|
+
else "orderer" if any(term in identity for term in ("rerank", "querit"))
|
|
890
|
+
else "decider" if any(term in identity for term in ("decision", "typesafe", "jev"))
|
|
891
|
+
else "text-generator"
|
|
892
|
+
)
|
|
893
|
+
readings.append(Reading(subject=subject, value=derived))
|
|
894
|
+
elif claim.field == "model.lifecycle" and subject is not None:
|
|
895
|
+
readings.append(Reading(subject=subject, value="active"))
|
|
896
|
+
elif claim.field == "model.weights_openness" and subject is not None:
|
|
897
|
+
if re.search(r"(?im)^license\s*:", text) or "download the model" in text.casefold():
|
|
898
|
+
readings.append(Reading(subject=subject, value="open_weights"))
|
|
899
|
+
elif claim.field.startswith("feature.") and subject is not None:
|
|
900
|
+
phrases = {
|
|
901
|
+
"feature.tool_calling": ("function calling", "tool calling"),
|
|
902
|
+
"feature.structured_output": ("structured output", "json mode", "json schema"),
|
|
903
|
+
"feature.effort_controls": ("reasoning effort", "default effort", "thinking"),
|
|
904
|
+
"feature.batch": ("v1/batch", "batch api", "batch inference"),
|
|
905
|
+
"feature.streaming": ("streaming", "stream response"),
|
|
906
|
+
}[claim.field]
|
|
907
|
+
corpus = text.casefold()
|
|
908
|
+
if any(phrase in corpus for phrase in phrases):
|
|
909
|
+
readings.append(Reading(subject=subject, value="supported"))
|
|
910
|
+
return readings or [Reading(subject=subject, value=None)]
|
|
911
|
+
|
|
912
|
+
|
|
913
|
+
class StructuredDataExtractor:
|
|
914
|
+
"""Read retained JSON or CSV board snapshots with one row per model."""
|
|
915
|
+
|
|
916
|
+
actor = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
|
|
917
|
+
method="structured-row-match@1")
|
|
918
|
+
_SUBJECTS = ("model", "model name", "model version", "model display", "name")
|
|
919
|
+
_UNITS = {
|
|
920
|
+
"rating": "Arena score (Elo scale)",
|
|
921
|
+
"mean score": "fraction",
|
|
922
|
+
"mean task": "fraction",
|
|
923
|
+
"retrieval": "fraction",
|
|
924
|
+
"reranking": "fraction",
|
|
925
|
+
"accuracy": "percent",
|
|
926
|
+
"resolve rate": "percent",
|
|
927
|
+
}
|
|
928
|
+
|
|
929
|
+
@staticmethod
|
|
930
|
+
def _rows(text: str) -> list[Mapping[str, Any]] | None:
|
|
931
|
+
try:
|
|
932
|
+
data = json.loads(text)
|
|
933
|
+
except ValueError:
|
|
934
|
+
first = text.splitlines()[0] if text.splitlines() else ""
|
|
935
|
+
headers = {normalise_name(cell) for cell in first.split(",")}
|
|
936
|
+
if not headers.intersection(StructuredDataExtractor._SUBJECTS):
|
|
937
|
+
return None
|
|
938
|
+
try:
|
|
939
|
+
rows = list(csv.DictReader(io.StringIO(text)))
|
|
940
|
+
except (csv.Error, UnicodeError):
|
|
941
|
+
return None
|
|
942
|
+
return rows or None
|
|
943
|
+
if isinstance(data, Mapping) and isinstance(data.get("rows"), list):
|
|
944
|
+
read_date = data.get("read_date")
|
|
945
|
+
return [
|
|
946
|
+
{**row, "_snapshot_read_date": read_date}
|
|
947
|
+
for row in data["rows"]
|
|
948
|
+
if isinstance(row, Mapping)
|
|
949
|
+
]
|
|
950
|
+
if isinstance(data, list):
|
|
951
|
+
return [row for row in data if isinstance(row, Mapping)]
|
|
952
|
+
return None
|
|
953
|
+
|
|
954
|
+
def accepts(self, text: str) -> bool:
|
|
955
|
+
return self._rows(text) is not None
|
|
956
|
+
|
|
957
|
+
def extract(self, claim: Claim, text: str) -> list[Reading]:
|
|
958
|
+
rows = self._rows(text) or []
|
|
959
|
+
label = claim.label or claim.field
|
|
960
|
+
out = []
|
|
961
|
+
for row in rows:
|
|
962
|
+
normal = {normalise_name(str(key)): value for key, value in row.items()}
|
|
963
|
+
subject = next((normal.get(key) for key in self._SUBJECTS if normal.get(key)), None)
|
|
964
|
+
value = normal.get(normalise_name(label))
|
|
965
|
+
if subject is None or value is None:
|
|
966
|
+
continue
|
|
967
|
+
effort = normal.get("reasoning effort") or normal.get("effort")
|
|
968
|
+
if effort is None:
|
|
969
|
+
match = re.search(r"(?:[_\s\(\[])(minimal|low|medium|high|xhigh|max)(?:\)|\]|$)",
|
|
970
|
+
str(subject), re.IGNORECASE)
|
|
971
|
+
effort = match.group(1) if match else None
|
|
972
|
+
date_ = next((normal.get(key) for key in (
|
|
973
|
+
"date", "leaderboard publish date", "started at", "snapshot read date",
|
|
974
|
+
"release date",
|
|
975
|
+
) if normal.get(key)), None)
|
|
976
|
+
if date_ is not None:
|
|
977
|
+
date_ = str(date_).split("T", 1)[0]
|
|
978
|
+
harness = "unregistered" if normal.get("agent") else None
|
|
979
|
+
out.append(Reading(
|
|
980
|
+
subject=str(subject),
|
|
981
|
+
value=str(value),
|
|
982
|
+
unit=self._UNITS.get(normalise_name(label)),
|
|
983
|
+
conditions={"effort": _text(effort),
|
|
984
|
+
"harness": harness, "date": _text(date_)},
|
|
985
|
+
))
|
|
986
|
+
return out
|
|
987
|
+
|
|
988
|
+
|
|
989
|
+
LLM_PROMPT = """\
|
|
990
|
+
You are checking a catalogue against a source. Read the source region below and
|
|
991
|
+
report every value it gives for "{label}", for any model.
|
|
992
|
+
|
|
993
|
+
Return only a JSON array. One object per value, with these string fields (null
|
|
994
|
+
when the region does not say): "subject" (the model name exactly as written),
|
|
995
|
+
"value" (the number or text exactly as written), "unit", "effort", "harness",
|
|
996
|
+
"date", and "quoted_sentence" (the exact sentence containing the value).
|
|
997
|
+
Return [] if the region gives no such value. Do not infer, convert, combine
|
|
998
|
+
sentences, or use knowledge outside the source region. A quoted sentence must
|
|
999
|
+
appear verbatim in the source region.
|
|
1000
|
+
|
|
1001
|
+
The model being checked is published as: {names}.
|
|
1002
|
+
|
|
1003
|
+
Source region:
|
|
1004
|
+
<<<
|
|
1005
|
+
{text}
|
|
1006
|
+
>>>
|
|
1007
|
+
"""
|
|
1008
|
+
|
|
1009
|
+
|
|
1010
|
+
class LLMCallBudgetExceededError(RuntimeError):
|
|
1011
|
+
"""The configured live-reader call budget has been exhausted."""
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
class ClaudeCLICompletion:
|
|
1015
|
+
"""Call the authenticated Claude CLI and return its assistant text."""
|
|
1016
|
+
|
|
1017
|
+
def __init__(self, *, max_calls: int = 400) -> None:
|
|
1018
|
+
self.max_calls = max_calls
|
|
1019
|
+
self.calls = 0
|
|
1020
|
+
|
|
1021
|
+
def __call__(self, prompt: str) -> str:
|
|
1022
|
+
if self.calls >= self.max_calls:
|
|
1023
|
+
raise LLMCallBudgetExceededError(
|
|
1024
|
+
f"stopped before exceeding the {self.max_calls}-call budget"
|
|
1025
|
+
)
|
|
1026
|
+
self.calls += 1
|
|
1027
|
+
command = [
|
|
1028
|
+
"claude", "-p", prompt, "--model", "claude-sonnet-5", "--effort", "low",
|
|
1029
|
+
"--output-format", "json",
|
|
1030
|
+
]
|
|
1031
|
+
try:
|
|
1032
|
+
completed = subprocess.run(
|
|
1033
|
+
command,
|
|
1034
|
+
stdin=subprocess.DEVNULL,
|
|
1035
|
+
capture_output=True,
|
|
1036
|
+
text=True,
|
|
1037
|
+
check=False,
|
|
1038
|
+
)
|
|
1039
|
+
except OSError as exc:
|
|
1040
|
+
raise ExtractorError(f"Claude CLI could not start: {exc}") from exc
|
|
1041
|
+
if completed.returncode != 0:
|
|
1042
|
+
detail = completed.stderr.strip() or f"exit {completed.returncode}"
|
|
1043
|
+
raise ExtractorError(f"Claude CLI failed: {detail}")
|
|
1044
|
+
try:
|
|
1045
|
+
envelope = json.loads(completed.stdout)
|
|
1046
|
+
result = envelope["result"]
|
|
1047
|
+
if not isinstance(result, str):
|
|
1048
|
+
raise TypeError("result is not text")
|
|
1049
|
+
except (KeyError, TypeError, ValueError) as exc:
|
|
1050
|
+
raise ExtractorError(f"Claude CLI returned an invalid JSON envelope: {exc}") from exc
|
|
1051
|
+
return result
|
|
1052
|
+
|
|
1053
|
+
|
|
1054
|
+
class LLMCache:
|
|
1055
|
+
"""Persistent reader replies keyed by source copy, cited region and facet.
|
|
1056
|
+
|
|
1057
|
+
``namespace`` keeps one reader's replies from answering for another's. The
|
|
1058
|
+
Claude reader has none, so the replies it cached before there were two
|
|
1059
|
+
readers still hit.
|
|
1060
|
+
"""
|
|
1061
|
+
|
|
1062
|
+
def __init__(self, root: str | Path | None = None, *, namespace: str | None = None) -> None:
|
|
1063
|
+
configured = os.environ.get("MODELSPEC_LLM_CACHE")
|
|
1064
|
+
self.root = Path(root or configured or Path.home() / ".cache/modelspec/llm-reader")
|
|
1065
|
+
self.namespace = namespace
|
|
1066
|
+
|
|
1067
|
+
def _path(self, key: tuple[str, str, str]) -> Path:
|
|
1068
|
+
scope = () if self.namespace is None else (self.namespace,)
|
|
1069
|
+
digest = hashlib.sha256(
|
|
1070
|
+
json.dumps(("strict-reader-v2", *scope, *key), ensure_ascii=False,
|
|
1071
|
+
separators=(",", ":")).encode("utf-8")
|
|
1072
|
+
).hexdigest()
|
|
1073
|
+
return self.root / digest[:2] / f"{digest}.json"
|
|
1074
|
+
|
|
1075
|
+
def get(self, key: tuple[str, str, str]) -> str | None:
|
|
1076
|
+
path = self._path(key)
|
|
1077
|
+
return path.read_text(encoding="utf-8") if path.is_file() else None
|
|
1078
|
+
|
|
1079
|
+
def put(self, key: tuple[str, str, str], reply: str) -> None:
|
|
1080
|
+
path = self._path(key)
|
|
1081
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
1082
|
+
if not path.exists():
|
|
1083
|
+
path.write_text(reply, encoding="utf-8")
|
|
1084
|
+
|
|
1085
|
+
|
|
1086
|
+
class LLMExtractor:
|
|
1087
|
+
"""Reads prose through ``complete(prompt) -> str``, an injected model call.
|
|
1088
|
+
|
|
1089
|
+
The prompt never shows the collector's value: the verifier reads independently.
|
|
1090
|
+
A reply that is not the requested JSON raises ``ExtractorError``.
|
|
1091
|
+
"""
|
|
1092
|
+
|
|
1093
|
+
def __init__(self, complete: Callable[[str], str], *, agent: str, model: str,
|
|
1094
|
+
model_family: str, cache: LLMCache | None = None) -> None:
|
|
1095
|
+
self.complete = complete
|
|
1096
|
+
self.cache = cache
|
|
1097
|
+
self.actor = VerificationActor(agent=agent, model_family=model_family,
|
|
1098
|
+
method=f"llm-extract:{model}")
|
|
1099
|
+
|
|
1100
|
+
def accepts(self, text: str) -> bool:
|
|
1101
|
+
return bool(text.strip())
|
|
1102
|
+
|
|
1103
|
+
def extract(self, claim: Claim, text: str, *,
|
|
1104
|
+
cache_key: tuple[str, str, str] | None = None) -> list[Reading]:
|
|
1105
|
+
prompt = LLM_PROMPT.format(label=claim.label or claim.field.replace("_", " "),
|
|
1106
|
+
names=", ".join(claim.names), text=text)
|
|
1107
|
+
reply = self.cache.get(cache_key) if self.cache is not None and cache_key else None
|
|
1108
|
+
if reply is None:
|
|
1109
|
+
reply = self.complete(prompt)
|
|
1110
|
+
try:
|
|
1111
|
+
cleaned = reply.strip()
|
|
1112
|
+
if cleaned.startswith("```json") and cleaned.endswith("```"):
|
|
1113
|
+
cleaned = cleaned[7:-3].strip()
|
|
1114
|
+
rows = json.loads(cleaned)
|
|
1115
|
+
if not isinstance(rows, list) or not all(isinstance(r, dict) for r in rows):
|
|
1116
|
+
raise ValueError("not a list of objects")
|
|
1117
|
+
for row in rows:
|
|
1118
|
+
quote = row.get("quoted_sentence")
|
|
1119
|
+
if not isinstance(quote, str) or not quote.strip() or \
|
|
1120
|
+
normalise_name(quote) not in normalise_name(text):
|
|
1121
|
+
raise ValueError("quoted_sentence is missing or is not verbatim source text")
|
|
1122
|
+
readings = [
|
|
1123
|
+
Reading(subject=_text(r.get("subject")) or claim.names[0],
|
|
1124
|
+
value=_text(r.get("value")),
|
|
1125
|
+
unit=_text(r.get("unit")),
|
|
1126
|
+
conditions={k: _text(r.get(k)) for k in CONDITION_KEYS})
|
|
1127
|
+
for r in rows
|
|
1128
|
+
]
|
|
1129
|
+
if not readings and claim.value is None:
|
|
1130
|
+
readings = [Reading(subject=claim.names[0], value=None)]
|
|
1131
|
+
if self.cache is not None and cache_key:
|
|
1132
|
+
self.cache.put(cache_key, reply)
|
|
1133
|
+
return readings
|
|
1134
|
+
except ValueError as exc:
|
|
1135
|
+
raise ExtractorError(f"unparseable reply from {self.actor.method}: {exc}") from exc
|
|
1136
|
+
|
|
1137
|
+
|
|
1138
|
+
def claude_extractor(*, cache: LLMCache | None = None,
|
|
1139
|
+
complete: Callable[[str], str] | None = None,
|
|
1140
|
+
max_calls: int = 400) -> LLMExtractor:
|
|
1141
|
+
"""The independent Claude Sonnet reader used by ``modelspec verify``."""
|
|
1142
|
+
return LLMExtractor(
|
|
1143
|
+
complete or ClaudeCLICompletion(max_calls=max_calls),
|
|
1144
|
+
agent="claude-cli",
|
|
1145
|
+
model="claude-sonnet-5",
|
|
1146
|
+
model_family="anthropic",
|
|
1147
|
+
cache=cache or LLMCache(),
|
|
1148
|
+
)
|
|
1149
|
+
|
|
1150
|
+
|
|
1151
|
+
OLLAMA_URL = "http://100.127.37.30:11434/api/chat"
|
|
1152
|
+
MISTRAL_MODEL = "mistral-large:123b-instruct-2411-q4_K_M"
|
|
1153
|
+
#: Ollama's ``format: json`` constrains a reply to one JSON object, so a bare
|
|
1154
|
+
#: array cannot be returned: Mistral then reports only the first value in a
|
|
1155
|
+
#: region. This system turn asks for the array inside an object; ``LLM_PROMPT``
|
|
1156
|
+
#: itself is sent unchanged.
|
|
1157
|
+
OLLAMA_JSON_MODE = (
|
|
1158
|
+
'Your reply must be one JSON object of the form {"values": [...]}, where the '
|
|
1159
|
+
"array is exactly the JSON array the user asks for, with one element per value."
|
|
1160
|
+
)
|
|
1161
|
+
|
|
1162
|
+
|
|
1163
|
+
def _as_array(content: str) -> str:
|
|
1164
|
+
"""A JSON-mode reply as the array ``LLM_PROMPT`` asks for.
|
|
1165
|
+
|
|
1166
|
+
Ollama's ``format: json`` tends to wrap the array in an object, or to return
|
|
1167
|
+
one row bare. ``{"rows": [...]}`` (any single key) gives its array, one row
|
|
1168
|
+
gives a one-row array, ``{}`` gives ``[]``. Anything else is returned as is,
|
|
1169
|
+
for ``LLMExtractor`` to accept or refuse.
|
|
1170
|
+
"""
|
|
1171
|
+
try:
|
|
1172
|
+
data = json.loads(content)
|
|
1173
|
+
except ValueError:
|
|
1174
|
+
return content
|
|
1175
|
+
if isinstance(data, dict):
|
|
1176
|
+
lists = [v for v in data.values() if isinstance(v, list)]
|
|
1177
|
+
if not data:
|
|
1178
|
+
data = []
|
|
1179
|
+
elif len(data) == 1 and len(lists) == 1:
|
|
1180
|
+
data = lists[0]
|
|
1181
|
+
elif "quoted_sentence" in data:
|
|
1182
|
+
data = [data]
|
|
1183
|
+
return json.dumps(data, ensure_ascii=False) if isinstance(data, list) else content
|
|
1184
|
+
|
|
1185
|
+
|
|
1186
|
+
class OllamaChatCompletion:
|
|
1187
|
+
"""Call a model served by ollama's ``/api/chat`` at temperature 0, in JSON mode."""
|
|
1188
|
+
|
|
1189
|
+
def __init__(self, *, url: str | None = None, model: str = MISTRAL_MODEL,
|
|
1190
|
+
max_calls: int = 400, timeout: float = 900) -> None:
|
|
1191
|
+
self.url = url or os.environ.get("MODELSPEC_OLLAMA_URL") or OLLAMA_URL
|
|
1192
|
+
self.model = model
|
|
1193
|
+
self.max_calls = max_calls
|
|
1194
|
+
self.timeout = timeout
|
|
1195
|
+
self.calls = 0
|
|
1196
|
+
|
|
1197
|
+
def __call__(self, prompt: str) -> str:
|
|
1198
|
+
if self.calls >= self.max_calls:
|
|
1199
|
+
raise LLMCallBudgetExceededError(
|
|
1200
|
+
f"stopped before exceeding the {self.max_calls}-call budget"
|
|
1201
|
+
)
|
|
1202
|
+
self.calls += 1
|
|
1203
|
+
body = {
|
|
1204
|
+
"model": self.model,
|
|
1205
|
+
"messages": [{"role": "system", "content": OLLAMA_JSON_MODE},
|
|
1206
|
+
{"role": "user", "content": prompt}],
|
|
1207
|
+
"stream": False,
|
|
1208
|
+
"format": "json",
|
|
1209
|
+
"options": {"temperature": 0},
|
|
1210
|
+
}
|
|
1211
|
+
request = urllib.request.Request(
|
|
1212
|
+
self.url, data=json.dumps(body).encode("utf-8"),
|
|
1213
|
+
headers={"Content-Type": "application/json"}, method="POST",
|
|
1214
|
+
)
|
|
1215
|
+
try:
|
|
1216
|
+
with urllib.request.urlopen(request, timeout=self.timeout) as response: # noqa: S310
|
|
1217
|
+
envelope = json.loads(response.read())
|
|
1218
|
+
except (OSError, ValueError) as exc:
|
|
1219
|
+
raise ExtractorError(f"ollama at {self.url} failed: {exc}") from exc
|
|
1220
|
+
try:
|
|
1221
|
+
content = envelope["message"]["content"]
|
|
1222
|
+
if not isinstance(content, str):
|
|
1223
|
+
raise TypeError("content is not text")
|
|
1224
|
+
except (KeyError, TypeError) as exc:
|
|
1225
|
+
raise ExtractorError(f"ollama returned an invalid chat envelope: {exc}") from exc
|
|
1226
|
+
return _as_array(content)
|
|
1227
|
+
|
|
1228
|
+
|
|
1229
|
+
def mistral_extractor(*, cache: LLMCache | None = None,
|
|
1230
|
+
complete: Callable[[str], str] | None = None,
|
|
1231
|
+
max_calls: int = 400) -> LLMExtractor:
|
|
1232
|
+
"""The local Mistral Large reader: a third model family, for Claude-collected values."""
|
|
1233
|
+
return LLMExtractor(
|
|
1234
|
+
complete or OllamaChatCompletion(max_calls=max_calls),
|
|
1235
|
+
agent="ollama",
|
|
1236
|
+
model=MISTRAL_MODEL,
|
|
1237
|
+
model_family="mistral",
|
|
1238
|
+
# The namespace names the request shape too: a reply cached before the
|
|
1239
|
+
# JSON-mode system turn read only one value per region.
|
|
1240
|
+
cache=cache or LLMCache(namespace=f"ollama:{MISTRAL_MODEL}:values-object"),
|
|
1241
|
+
)
|
|
1242
|
+
|
|
1243
|
+
|
|
1244
|
+
def _text(value: Any) -> str | None:
|
|
1245
|
+
return None if value is None else str(value)
|
|
1246
|
+
|
|
1247
|
+
|
|
1248
|
+
def deterministic_extractors() -> list[Extractor]:
|
|
1249
|
+
return [StructuredDataExtractor(), OfferingPriceExtractor(), TableExtractor(),
|
|
1250
|
+
GovernanceProseExtractor(), KeyValueExtractor(), ModelPageExtractor()]
|
|
1251
|
+
|
|
1252
|
+
|
|
1253
|
+
# --- regions -------------------------------------------------------------------------------------
|
|
1254
|
+
|
|
1255
|
+
|
|
1256
|
+
class Regions(Protocol):
|
|
1257
|
+
def text(self, source_id: str, copy_ref: str, region_id: str) -> str | None:
|
|
1258
|
+
"""The cited region's text in that retained copy, or ``None`` when unreachable."""
|
|
1259
|
+
|
|
1260
|
+
|
|
1261
|
+
class StoredRegions:
|
|
1262
|
+
"""Regions read from retained copies in a ``CopyStore``.
|
|
1263
|
+
|
|
1264
|
+
The copy is normalised with its source's rules but with volatile text (dates,
|
|
1265
|
+
times) kept: a date the verifier must compare is not boilerplate here.
|
|
1266
|
+
"""
|
|
1267
|
+
|
|
1268
|
+
def __init__(self, store: CopyStore, sources: Mapping[str, Source]) -> None:
|
|
1269
|
+
self.store = store
|
|
1270
|
+
self.sources = sources
|
|
1271
|
+
|
|
1272
|
+
def text(self, source_id: str, copy_ref: str, region_id: str) -> str | None:
|
|
1273
|
+
source = self.sources.get(source_id)
|
|
1274
|
+
region = next((r for r in source.cited_regions if r.id == region_id), None) \
|
|
1275
|
+
if source else None
|
|
1276
|
+
if source is None or region is None or not self.store.has(copy_ref):
|
|
1277
|
+
return None
|
|
1278
|
+
rules = replace(NORMALISERS[source.normaliser], strip_volatile=False)
|
|
1279
|
+
try:
|
|
1280
|
+
doc = normalise_document(self.store.get(copy_ref), rules)
|
|
1281
|
+
kind = "heading" if region.locator.kind == "heading_anchor" else region.locator.kind
|
|
1282
|
+
return select_region(doc, Locator(kind, region.locator.value))
|
|
1283
|
+
except (UnsupportedContentError, ValueError):
|
|
1284
|
+
return None
|
|
1285
|
+
|
|
1286
|
+
|
|
1287
|
+
# --- comparison ----------------------------------------------------------------------------------
|
|
1288
|
+
|
|
1289
|
+
|
|
1290
|
+
@dataclass(frozen=True)
|
|
1291
|
+
class Diff:
|
|
1292
|
+
field: str
|
|
1293
|
+
expected: JsonValue
|
|
1294
|
+
found: JsonValue
|
|
1295
|
+
|
|
1296
|
+
def to_dict(self) -> dict[str, JsonValue]:
|
|
1297
|
+
return {"field": self.field, "expected": self.expected, "found": self.found}
|
|
1298
|
+
|
|
1299
|
+
|
|
1300
|
+
_TRUE = frozenset({"yes", "true", "supported", "available", "y", "✓", "✔"})
|
|
1301
|
+
_FALSE = frozenset({
|
|
1302
|
+
"no", "none", "false", "not supported", "unsupported", "unavailable", "n", "✗", "✘",
|
|
1303
|
+
})
|
|
1304
|
+
|
|
1305
|
+
|
|
1306
|
+
def _show(claim: Claim) -> JsonValue:
|
|
1307
|
+
if isinstance(claim.value, (int, float)) and not isinstance(claim.value, bool):
|
|
1308
|
+
return f"{claim.value} {claim.unit}" if claim.unit else str(claim.value)
|
|
1309
|
+
return claim.value
|
|
1310
|
+
|
|
1311
|
+
|
|
1312
|
+
def _value_diff(claim: Claim, reading: Reading) -> Diff | None:
|
|
1313
|
+
expected, value = _show(claim), claim.value
|
|
1314
|
+
if reading.value is None:
|
|
1315
|
+
return None if value is None else Diff("value", expected, None)
|
|
1316
|
+
if isinstance(value, bool):
|
|
1317
|
+
s = reading.value.strip().casefold()
|
|
1318
|
+
explicit_no_training = (
|
|
1319
|
+
"train" in s
|
|
1320
|
+
and any(phrase in s for phrase in (
|
|
1321
|
+
"do not use", "does not use", "will not use", "won't use", "not used",
|
|
1322
|
+
"never use",
|
|
1323
|
+
))
|
|
1324
|
+
)
|
|
1325
|
+
explicit_available = (
|
|
1326
|
+
(
|
|
1327
|
+
"zero data retention" in s
|
|
1328
|
+
and not any(x in s for x in ("not available", "unavailable"))
|
|
1329
|
+
)
|
|
1330
|
+
or ("baa" in s and any(x in s for x in ("available", "eligible")))
|
|
1331
|
+
)
|
|
1332
|
+
found = (True if s in _TRUE or explicit_available
|
|
1333
|
+
else False if s in _FALSE or explicit_no_training else None)
|
|
1334
|
+
return None if found is value else Diff("value", expected, reading.value)
|
|
1335
|
+
if isinstance(value, (int, float)):
|
|
1336
|
+
if value == 0 and claim.unit == "days" and normalise_name(reading.value) in {
|
|
1337
|
+
"none", "no retention", "zero data retention",
|
|
1338
|
+
}:
|
|
1339
|
+
return None
|
|
1340
|
+
q = parse_quantity(reading.value, reading.unit)
|
|
1341
|
+
if q is None:
|
|
1342
|
+
return Diff("value", expected, reading.value)
|
|
1343
|
+
if numbers_agree(value, claim.unit, q):
|
|
1344
|
+
return None
|
|
1345
|
+
unit_differs = claim.unit is not None and claim.unit != q.unit
|
|
1346
|
+
return Diff("unit" if unit_differs else "value", expected, q.show())
|
|
1347
|
+
if isinstance(value, list):
|
|
1348
|
+
found_items = {s.strip().casefold()
|
|
1349
|
+
for s in re.split(r",|;|\band\b", reading.value) if s.strip()}
|
|
1350
|
+
claimed_items = {str(v).strip().casefold() for v in value}
|
|
1351
|
+
return None if found_items == claimed_items else Diff("value", expected, reading.value)
|
|
1352
|
+
expected_name = normalise_name(str(value))
|
|
1353
|
+
found_name = normalise_name(reading.value)
|
|
1354
|
+
if expected_name == "not offered" and found_name in {
|
|
1355
|
+
"n a", "na", "not available", "not supported",
|
|
1356
|
+
}:
|
|
1357
|
+
return None
|
|
1358
|
+
if expected_name == found_name:
|
|
1359
|
+
return None
|
|
1360
|
+
if expected_name == "type 2" and (
|
|
1361
|
+
"type 2" in found_name or "type ii" in found_name
|
|
1362
|
+
):
|
|
1363
|
+
return None
|
|
1364
|
+
return Diff("value", expected, reading.value)
|
|
1365
|
+
|
|
1366
|
+
|
|
1367
|
+
def _diffs(claim: Claim, reading: Reading) -> list[Diff]:
|
|
1368
|
+
diffs = [d for d in [_value_diff(claim, reading)] if d is not None]
|
|
1369
|
+
name_effort = split_model_cell(reading.subject)[1] if reading.subject else None
|
|
1370
|
+
for key in CONDITION_KEYS:
|
|
1371
|
+
claimed = _condition(key, claim.conditions.get(key))
|
|
1372
|
+
found = _condition(key, reading.conditions.get(key))
|
|
1373
|
+
if key == "effort" and found is None:
|
|
1374
|
+
found = name_effort
|
|
1375
|
+
if key == "date" and claimed is None:
|
|
1376
|
+
continue # a date the claim does not carry is not checked
|
|
1377
|
+
if key == "harness" and claimed == UNREGISTERED and found is not None \
|
|
1378
|
+
and default_registry().resolve_harness(found) == UNREGISTERED:
|
|
1379
|
+
continue # the registry reports a named, unregistered harness as `unregistered`
|
|
1380
|
+
if claimed != found:
|
|
1381
|
+
diffs.append(Diff(key, claim.conditions.get(key), found))
|
|
1382
|
+
return diffs
|
|
1383
|
+
|
|
1384
|
+
|
|
1385
|
+
def compare(claim: Claim, readings: Sequence[Reading]) -> list[Diff]:
|
|
1386
|
+
"""The diffs between a claim and what a region says; empty when it agrees."""
|
|
1387
|
+
if not readings:
|
|
1388
|
+
return [Diff("value", _show(claim), None)]
|
|
1389
|
+
names = {
|
|
1390
|
+
identity
|
|
1391
|
+
for name in claim.names
|
|
1392
|
+
for identity in (normalise_name(name), split_model_cell(name)[0])
|
|
1393
|
+
}
|
|
1394
|
+
own = [r for r in readings if r.subject and split_model_cell(r.subject)[0] in names]
|
|
1395
|
+
others = [r for r in readings if r not in own and r.subject]
|
|
1396
|
+
sibling = next((r for r in others if not _diffs(claim, r)), None)
|
|
1397
|
+
expected_name = claim.names[0]
|
|
1398
|
+
if not own:
|
|
1399
|
+
return [Diff("model", expected_name, sibling.subject if sibling else None)]
|
|
1400
|
+
candidates = [_diffs(claim, r) for r in own]
|
|
1401
|
+
if any(not c for c in candidates):
|
|
1402
|
+
return []
|
|
1403
|
+
best = min(candidates, key=lambda c: (any(d.field in ("value", "unit") for d in c), len(c)))
|
|
1404
|
+
if sibling and any(d.field in ("value", "unit") for d in best):
|
|
1405
|
+
best = [*best, Diff("model", expected_name, sibling.subject)]
|
|
1406
|
+
return best
|
|
1407
|
+
|
|
1408
|
+
|
|
1409
|
+
# --- verification --------------------------------------------------------------------------------
|
|
1410
|
+
|
|
1411
|
+
#: The verifier recorded for an unreachable outcome: no extractor ran, a lookup did.
|
|
1412
|
+
REGION_LOOKUP = VerificationActor(agent=VERIFY_AGENT, model_family=DETERMINISTIC,
|
|
1413
|
+
method="region-lookup@1")
|
|
1414
|
+
|
|
1415
|
+
|
|
1416
|
+
@dataclass(frozen=True)
|
|
1417
|
+
class Result:
|
|
1418
|
+
target: TargetRef
|
|
1419
|
+
outcome: Outcome
|
|
1420
|
+
verification: Verification | None = None
|
|
1421
|
+
diffs: tuple[Diff, ...] = ()
|
|
1422
|
+
#: Why a claim was skipped, or which region was unreachable.
|
|
1423
|
+
reason: str | None = None
|
|
1424
|
+
|
|
1425
|
+
|
|
1426
|
+
def _verification(claim: Claim, actor: VerificationActor, outcome: str, today: date,
|
|
1427
|
+
diffs: Sequence[Diff] = ()) -> Verification:
|
|
1428
|
+
"""Raises ``ValidationError`` when ``actor`` is not independent of the collector."""
|
|
1429
|
+
return Verification(
|
|
1430
|
+
target=VerificationTarget(
|
|
1431
|
+
kind=claim.target.kind,
|
|
1432
|
+
id=claim.target.id,
|
|
1433
|
+
value_hash=value_hash(claim.value),
|
|
1434
|
+
),
|
|
1435
|
+
collector=claim.collector,
|
|
1436
|
+
verifier=actor,
|
|
1437
|
+
method=actor.method,
|
|
1438
|
+
outcome=outcome,
|
|
1439
|
+
date=today,
|
|
1440
|
+
diff=json.dumps([d.to_dict() for d in diffs], ensure_ascii=False) if diffs else None,
|
|
1441
|
+
)
|
|
1442
|
+
|
|
1443
|
+
|
|
1444
|
+
def _independent(claim: Claim, actor: VerificationActor, today: date) -> bool:
|
|
1445
|
+
"""A second key: another agent or family (parse rule), and another family unless
|
|
1446
|
+
deterministic (``Verification.independent``)."""
|
|
1447
|
+
try:
|
|
1448
|
+
return _verification(claim, actor, "verified", today).independent
|
|
1449
|
+
except ValidationError:
|
|
1450
|
+
return False
|
|
1451
|
+
|
|
1452
|
+
|
|
1453
|
+
def verify(claim: Claim, regions: Regions, extractors: Sequence[Extractor], *,
|
|
1454
|
+
today: date) -> Result:
|
|
1455
|
+
"""Re-read ``claim`` from each cited region of its sources and compare.
|
|
1456
|
+
|
|
1457
|
+
Deterministic extractors are tried before the rest, whatever the order given;
|
|
1458
|
+
the first that accepts a region and is independent of the collector reads it.
|
|
1459
|
+
Verified if any region confirms the value; otherwise the first mismatch.
|
|
1460
|
+
"""
|
|
1461
|
+
ordered = sorted(extractors, key=lambda e: e.actor.model_family != DETERMINISTIC)
|
|
1462
|
+
reachable = False
|
|
1463
|
+
mismatch: tuple[VerificationActor, list[Diff]] | None = None
|
|
1464
|
+
reasons: list[str] = []
|
|
1465
|
+
for source in claim.sources:
|
|
1466
|
+
for region_id in source.cited_regions:
|
|
1467
|
+
where = f"{source.source_id}#{region_id}"
|
|
1468
|
+
text = regions.text(source.source_id, source.snapshot_ref, region_id)
|
|
1469
|
+
if text is None:
|
|
1470
|
+
reasons.append(f"unreachable:{where}")
|
|
1471
|
+
continue
|
|
1472
|
+
reachable = True
|
|
1473
|
+
accepting = [e for e in ordered if e.accepts(text)]
|
|
1474
|
+
independent = [e for e in accepting if _independent(claim, e.actor, today)]
|
|
1475
|
+
if not independent:
|
|
1476
|
+
reasons.append("no_independent_extractor" if accepting else f"no_extractor:{where}")
|
|
1477
|
+
continue
|
|
1478
|
+
for extractor in independent:
|
|
1479
|
+
try:
|
|
1480
|
+
if isinstance(extractor, LLMExtractor):
|
|
1481
|
+
readings = extractor.extract(
|
|
1482
|
+
claim,
|
|
1483
|
+
text,
|
|
1484
|
+
cache_key=(source.snapshot_ref, region_id, claim.field),
|
|
1485
|
+
)
|
|
1486
|
+
else:
|
|
1487
|
+
readings = extractor.extract(claim, text)
|
|
1488
|
+
except ExtractorError as exc:
|
|
1489
|
+
reasons.append(f"extractor_error:{where}: {exc}")
|
|
1490
|
+
continue
|
|
1491
|
+
diffs = compare(claim, readings)
|
|
1492
|
+
if not diffs:
|
|
1493
|
+
return Result(claim.target, "verified",
|
|
1494
|
+
_verification(claim, extractor.actor, "verified", today))
|
|
1495
|
+
mismatch = mismatch or (extractor.actor, diffs)
|
|
1496
|
+
|
|
1497
|
+
if mismatch is not None:
|
|
1498
|
+
actor, diffs = mismatch
|
|
1499
|
+
return Result(claim.target, "mismatch",
|
|
1500
|
+
_verification(claim, actor, "mismatch", today, diffs), tuple(diffs))
|
|
1501
|
+
reason = "; ".join(reasons)
|
|
1502
|
+
if not reachable:
|
|
1503
|
+
if not _independent(claim, REGION_LOOKUP, today):
|
|
1504
|
+
return Result(claim.target, "skipped", reason="no_independent_extractor")
|
|
1505
|
+
return Result(claim.target, "unreachable",
|
|
1506
|
+
_verification(claim, REGION_LOOKUP, "unreachable", today), reason=reason)
|
|
1507
|
+
return Result(claim.target, "skipped", reason=reason)
|
|
1508
|
+
|
|
1509
|
+
|
|
1510
|
+
# --- the log -------------------------------------------------------------------------------------
|
|
1511
|
+
|
|
1512
|
+
|
|
1513
|
+
class VerificationLog:
|
|
1514
|
+
"""``verification/*.jsonl``: append-only; the latest outcome per target wins."""
|
|
1515
|
+
|
|
1516
|
+
def __init__(self, directory: str | Path = DEFAULT_DIRECTORY) -> None:
|
|
1517
|
+
self.directory = Path(directory)
|
|
1518
|
+
self.path = self.directory / "log.jsonl"
|
|
1519
|
+
|
|
1520
|
+
def append(self, verification: Verification) -> None:
|
|
1521
|
+
self.directory.mkdir(parents=True, exist_ok=True)
|
|
1522
|
+
with self.path.open("a", encoding="utf-8") as fh:
|
|
1523
|
+
fh.write(verification.model_dump_json() + "\n")
|
|
1524
|
+
|
|
1525
|
+
def records(self) -> list[Verification]:
|
|
1526
|
+
records = []
|
|
1527
|
+
for path in sorted(self.directory.glob("*.jsonl")):
|
|
1528
|
+
for line in path.read_text(encoding="utf-8").splitlines():
|
|
1529
|
+
if line.strip():
|
|
1530
|
+
records.append(Verification.model_validate_json(line))
|
|
1531
|
+
return records
|
|
1532
|
+
|
|
1533
|
+
def latest(self) -> dict[tuple[str, str], Verification]:
|
|
1534
|
+
"""Latest ``date`` wins; on a tie, the later line (as ``decision.snapshot`` reads it).
|
|
1535
|
+
|
|
1536
|
+
A record that does not count (a same-family ``verified``) is skipped.
|
|
1537
|
+
"""
|
|
1538
|
+
latest: dict[tuple[str, str], Verification] = {}
|
|
1539
|
+
for record in self.records():
|
|
1540
|
+
if not record.counts:
|
|
1541
|
+
continue
|
|
1542
|
+
key = _key(record.target)
|
|
1543
|
+
if key not in latest or record.date >= latest[key].date:
|
|
1544
|
+
latest[key] = record
|
|
1545
|
+
return latest
|
|
1546
|
+
|
|
1547
|
+
def requarantined(self) -> list[VerificationTarget]:
|
|
1548
|
+
"""Values a same-family ``verified`` vouches for that no counting record admits.
|
|
1549
|
+
|
|
1550
|
+
Keyed by target and checked value, as ``decision.snapshot`` admits them:
|
|
1551
|
+
what MODEL-159's family rule keeps out until another family verifies it.
|
|
1552
|
+
"""
|
|
1553
|
+
vouched: dict[tuple[str, str, str], VerificationTarget] = {}
|
|
1554
|
+
counting: dict[tuple[str, str, str], Verification] = {}
|
|
1555
|
+
for record in self.records():
|
|
1556
|
+
key = (record.target.kind, record.target.id, record.target.value_hash)
|
|
1557
|
+
if not record.counts:
|
|
1558
|
+
vouched[key] = record.target
|
|
1559
|
+
elif key not in counting or record.date >= counting[key].date:
|
|
1560
|
+
counting[key] = record
|
|
1561
|
+
return [target for key, target in sorted(vouched.items())
|
|
1562
|
+
if key not in counting or counting[key].outcome != "verified"]
|
|
1563
|
+
|
|
1564
|
+
def is_quarantined(self, target: TargetRef | str) -> bool:
|
|
1565
|
+
record = self.latest().get(_key(target_ref(target)))
|
|
1566
|
+
return record is None or record.quarantined
|
|
1567
|
+
|
|
1568
|
+
def quarantined_values(self, targets: Iterable[TargetRef | str] | None = None
|
|
1569
|
+
) -> list[TargetRef | VerificationTarget]:
|
|
1570
|
+
"""Quarantined targets: of ``targets`` (never-verified ones included), else of the log."""
|
|
1571
|
+
latest = self.latest()
|
|
1572
|
+
if targets is None:
|
|
1573
|
+
return [r.target for key, r in sorted(latest.items()) if r.quarantined]
|
|
1574
|
+
refs = [target_ref(t) for t in targets]
|
|
1575
|
+
return [t for t in refs if _key(t) not in latest or latest[_key(t)].quarantined]
|
|
1576
|
+
|
|
1577
|
+
|
|
1578
|
+
def is_quarantined(target: TargetRef | str, *, directory: str | Path = DEFAULT_DIRECTORY) -> bool:
|
|
1579
|
+
return VerificationLog(directory).is_quarantined(target)
|
|
1580
|
+
|
|
1581
|
+
|
|
1582
|
+
def quarantined_values(targets: Iterable[TargetRef | str] | None = None, *,
|
|
1583
|
+
directory: str | Path = DEFAULT_DIRECTORY
|
|
1584
|
+
) -> list[TargetRef | VerificationTarget]:
|
|
1585
|
+
return VerificationLog(directory).quarantined_values(targets)
|
|
1586
|
+
|
|
1587
|
+
|
|
1588
|
+
# --- the queue -----------------------------------------------------------------------------------
|
|
1589
|
+
|
|
1590
|
+
|
|
1591
|
+
@dataclass
|
|
1592
|
+
class _Pending:
|
|
1593
|
+
claim: Claim | None = None
|
|
1594
|
+
trigger: Literal["new", "changed"] | None = None
|
|
1595
|
+
copies: dict[str, str] = field(default_factory=dict)
|
|
1596
|
+
last: dict[str, Any] | None = None
|
|
1597
|
+
|
|
1598
|
+
|
|
1599
|
+
class Queue:
|
|
1600
|
+
"""``verification/queue/events.jsonl``: what to verify, and what to re-crawl."""
|
|
1601
|
+
|
|
1602
|
+
def __init__(self, directory: str | Path = DEFAULT_DIRECTORY) -> None:
|
|
1603
|
+
self.directory = Path(directory)
|
|
1604
|
+
self.path = self.directory / "queue" / "events.jsonl"
|
|
1605
|
+
|
|
1606
|
+
def _append(self, events: Iterable[dict[str, Any]]) -> None:
|
|
1607
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
1608
|
+
with self.path.open("a", encoding="utf-8") as fh:
|
|
1609
|
+
for event in events:
|
|
1610
|
+
fh.write(json.dumps(event, sort_keys=True, ensure_ascii=False) + "\n")
|
|
1611
|
+
|
|
1612
|
+
def file(self, claim: Claim, *, at: datetime) -> None:
|
|
1613
|
+
"""A collector files a new or re-collected value: verify it before first use."""
|
|
1614
|
+
self._append([{"event": "collected", "at": at.isoformat(), "claim": claim.to_dict()}])
|
|
1615
|
+
|
|
1616
|
+
def requeue(self, changed: RecheckReport | Iterable[str | TargetRef], *, at: datetime) -> None:
|
|
1617
|
+
"""Re-verify what change detection re-queued, against each source's new copy."""
|
|
1618
|
+
copies: dict[str, str] = {}
|
|
1619
|
+
if isinstance(changed, RecheckReport):
|
|
1620
|
+
copies = {sid: st.snapshot.copy_ref for sid, st in changed.states.items()
|
|
1621
|
+
if st.snapshot is not None}
|
|
1622
|
+
changed = changed.requeue
|
|
1623
|
+
self._append({"event": "changed", "at": at.isoformat(),
|
|
1624
|
+
"target": target_ref(ref).model_dump(), "copies": copies}
|
|
1625
|
+
for ref in changed)
|
|
1626
|
+
|
|
1627
|
+
def checked(self, result: Result, *, at: datetime) -> None:
|
|
1628
|
+
self._append([{"event": "checked", "at": at.isoformat(),
|
|
1629
|
+
"target": result.target.model_dump(), "outcome": result.outcome,
|
|
1630
|
+
"diff": [d.to_dict() for d in result.diffs]}])
|
|
1631
|
+
|
|
1632
|
+
def _state(self) -> dict[tuple[str, str], _Pending]:
|
|
1633
|
+
state: dict[tuple[str, str], _Pending] = {}
|
|
1634
|
+
if not self.path.is_file():
|
|
1635
|
+
return state
|
|
1636
|
+
for line in self.path.read_text(encoding="utf-8").splitlines():
|
|
1637
|
+
if not line.strip():
|
|
1638
|
+
continue
|
|
1639
|
+
event = json.loads(line)
|
|
1640
|
+
if event["event"] == "collected":
|
|
1641
|
+
claim = Claim.from_dict(event["claim"])
|
|
1642
|
+
state[_key(claim.target)] = _Pending(claim, "new", {}, event)
|
|
1643
|
+
continue
|
|
1644
|
+
entry = state.setdefault(_key(TargetRef.model_validate(event["target"])), _Pending())
|
|
1645
|
+
entry.last = event
|
|
1646
|
+
if event["event"] == "changed":
|
|
1647
|
+
entry.trigger = entry.trigger or "changed"
|
|
1648
|
+
entry.copies.update(event.get("copies") or {})
|
|
1649
|
+
elif event["event"] == "checked":
|
|
1650
|
+
entry.trigger, entry.copies = None, {}
|
|
1651
|
+
return state
|
|
1652
|
+
|
|
1653
|
+
def pending(self, *, changed_only: bool = False) -> tuple[list[Claim], list[TargetRef]]:
|
|
1654
|
+
"""Claims to verify (each source pinned to its newest copy), and re-queued refs
|
|
1655
|
+
no collector has filed a claim for."""
|
|
1656
|
+
claims, unknown = [], []
|
|
1657
|
+
for key, entry in self._state().items():
|
|
1658
|
+
if entry.trigger is None or (changed_only and entry.trigger != "changed"):
|
|
1659
|
+
continue
|
|
1660
|
+
if entry.claim is None:
|
|
1661
|
+
unknown.append(TargetRef(kind=key[0], id=key[1]))
|
|
1662
|
+
continue
|
|
1663
|
+
sources = tuple(
|
|
1664
|
+
s.model_copy(update={"snapshot_ref": entry.copies[s.source_id]})
|
|
1665
|
+
if s.source_id in entry.copies else s
|
|
1666
|
+
for s in entry.claim.sources
|
|
1667
|
+
)
|
|
1668
|
+
claims.append(replace(entry.claim, sources=sources))
|
|
1669
|
+
return claims, unknown
|
|
1670
|
+
|
|
1671
|
+
def recrawl_requests(self) -> list[tuple[TargetRef, str]]:
|
|
1672
|
+
"""Targets whose last check failed and that no collector has filed again."""
|
|
1673
|
+
return [
|
|
1674
|
+
(TargetRef(kind=key[0], id=key[1]), entry.last["outcome"])
|
|
1675
|
+
for key, entry in self._state().items()
|
|
1676
|
+
if entry.last and entry.last["event"] == "checked"
|
|
1677
|
+
and entry.last["outcome"] != "verified"
|
|
1678
|
+
]
|
|
1679
|
+
|
|
1680
|
+
|
|
1681
|
+
# --- a run ---------------------------------------------------------------------------------------
|
|
1682
|
+
|
|
1683
|
+
|
|
1684
|
+
@dataclass
|
|
1685
|
+
class RunReport:
|
|
1686
|
+
changed_only: bool
|
|
1687
|
+
results: list[Result] = field(default_factory=list)
|
|
1688
|
+
#: Re-queued refs with no filed claim: nothing to verify them against.
|
|
1689
|
+
unknown: list[TargetRef] = field(default_factory=list)
|
|
1690
|
+
|
|
1691
|
+
@property
|
|
1692
|
+
def counts(self) -> dict[str, int]:
|
|
1693
|
+
counts = dict.fromkeys(("verified", "mismatch", "unreachable", "skipped"), 0)
|
|
1694
|
+
for r in self.results:
|
|
1695
|
+
counts[r.outcome] += 1
|
|
1696
|
+
return counts
|
|
1697
|
+
|
|
1698
|
+
def to_dict(self) -> dict[str, Any]:
|
|
1699
|
+
return {
|
|
1700
|
+
"changed_only": self.changed_only,
|
|
1701
|
+
"counts": self.counts,
|
|
1702
|
+
"results": [
|
|
1703
|
+
{"target": ref_str(r.target), "outcome": r.outcome,
|
|
1704
|
+
"diff": [d.to_dict() for d in r.diffs], "reason": r.reason,
|
|
1705
|
+
"verifier": r.verification.verifier.model_dump() if r.verification else None}
|
|
1706
|
+
for r in self.results
|
|
1707
|
+
],
|
|
1708
|
+
"unknown": [ref_str(t) for t in self.unknown],
|
|
1709
|
+
}
|
|
1710
|
+
|
|
1711
|
+
|
|
1712
|
+
def ref_str(target: TargetRef) -> str:
|
|
1713
|
+
return f"{target.kind}:{target.id}"
|
|
1714
|
+
|
|
1715
|
+
|
|
1716
|
+
def run(queue: Queue, log: VerificationLog, regions: Regions, extractors: Sequence[Extractor],
|
|
1717
|
+
*, today: date, changed_only: bool = False, at: datetime | None = None) -> RunReport:
|
|
1718
|
+
"""Verify what is queued; log every outcome and mark it checked. Skipped claims stay
|
|
1719
|
+
queued, unlogged and so quarantined."""
|
|
1720
|
+
at = at or datetime.now(UTC)
|
|
1721
|
+
claims, unknown = queue.pending(changed_only=changed_only)
|
|
1722
|
+
report = RunReport(changed_only, unknown=unknown)
|
|
1723
|
+
for claim in claims:
|
|
1724
|
+
result = verify(claim, regions, extractors, today=today)
|
|
1725
|
+
report.results.append(result)
|
|
1726
|
+
if result.verification is not None:
|
|
1727
|
+
log.append(result.verification)
|
|
1728
|
+
queue.checked(result, at=at)
|
|
1729
|
+
return report
|
|
1730
|
+
|
|
1731
|
+
|
|
1732
|
+
__all__ = [
|
|
1733
|
+
"Claim", "ClaudeCLICompletion", "CONDITION_KEYS", "Diff", "Extractor", "ExtractorError",
|
|
1734
|
+
"GovernanceProseExtractor", "KeyValueExtractor", "LLMCache", "LLMCallBudgetExceededError",
|
|
1735
|
+
"LLMExtractor", "MISTRAL_MODEL", "ModelPageExtractor", "OLLAMA_URL", "OfferingPriceExtractor",
|
|
1736
|
+
"OLLAMA_JSON_MODE", "OllamaChatCompletion", "Quantity", "Queue", "Reading",
|
|
1737
|
+
"Regions", "Result",
|
|
1738
|
+
"RunReport", "StoredRegions", "StructuredDataExtractor", "TableExtractor", "TOLERANCE_RULE",
|
|
1739
|
+
"UNITS", "VerificationLog", "compare",
|
|
1740
|
+
"claude_extractor", "deterministic_extractors", "is_quarantined", "load_sources",
|
|
1741
|
+
"mistral_extractor",
|
|
1742
|
+
"numbers_agree",
|
|
1743
|
+
"parse_quantity", "quarantined_values", "run", "split_model_cell", "target_ref", "unit_id",
|
|
1744
|
+
"verify",
|
|
1745
|
+
]
|