modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
decision/snapshot.py
ADDED
|
@@ -0,0 +1,1483 @@
|
|
|
1
|
+
"""The snapshot: verified facts and evidence, compiled, hashed and signed (MODEL-138).
|
|
2
|
+
|
|
3
|
+
A decision reads one snapshot and nothing else (design §4.2, §6). This module
|
|
4
|
+
builds it, gates it and loads it:
|
|
5
|
+
|
|
6
|
+
* ``build_snapshot`` compiles models, offerings and evidence into a columnar
|
|
7
|
+
snapshot. Only values whose latest verification is ``verified`` enter, and
|
|
8
|
+
only when every source they name resolves to a registered URL that is not an
|
|
9
|
+
excluded source. A ``verified`` from the collector's own model family is not
|
|
10
|
+
a second key (MODEL-159): it is skipped as if never logged, while a
|
|
11
|
+
same-family mismatch still counts. Retired models and their offerings go to
|
|
12
|
+
a separate ``archive`` section. With a premier list, the ``lineup`` holds only the
|
|
13
|
+
premier models and their offerings; other active models are counted in
|
|
14
|
+
``out_of_lineup`` and left out (MODEL-157). The legacy flat
|
|
15
|
+
``benchmarks.scores`` block is never read.
|
|
16
|
+
* The **completeness gate** fails the build when a guaranteed facet is unknown
|
|
17
|
+
or unverified for a premier model or one of its offerings, naming the
|
|
18
|
+
subject, the facet and the source. Computed facets (``computed_by`` in the
|
|
19
|
+
registry, such as ``estimate.capability``) are skipped: slice 1 does not
|
|
20
|
+
compute them.
|
|
21
|
+
* ``Snapshot.write`` serialises canonical JSON, gzips it with a fixed header,
|
|
22
|
+
and signs the content hash with HMAC-SHA256 when ``MODELSPEC_SNAPSHOT_KEY`` is
|
|
23
|
+
set. The same inputs give the same bytes.
|
|
24
|
+
* ``load_snapshot`` checks the hash and, when a key is available, the
|
|
25
|
+
signature, then builds the in-memory index: three-valued bitsets over the
|
|
26
|
+
candidates, per facet value.
|
|
27
|
+
|
|
28
|
+
Inputs are MODEL-134's records (``decision.model``) or their serialised dicts;
|
|
29
|
+
the builder reads them by field name, so either works.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import gzip
|
|
35
|
+
import hashlib
|
|
36
|
+
import hmac
|
|
37
|
+
import io
|
|
38
|
+
import json
|
|
39
|
+
import math
|
|
40
|
+
import os
|
|
41
|
+
from bisect import bisect_left, bisect_right
|
|
42
|
+
from collections import Counter
|
|
43
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
44
|
+
from dataclasses import dataclass, field
|
|
45
|
+
from datetime import date
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
from typing import Any, Literal, Protocol, runtime_checkable
|
|
48
|
+
|
|
49
|
+
import yaml
|
|
50
|
+
|
|
51
|
+
from decision.excluded import ExcludedSources, excluded_sources
|
|
52
|
+
from decision.model import value_hash, verification_counts
|
|
53
|
+
|
|
54
|
+
FORMAT = "modelspec.decision-snapshot"
|
|
55
|
+
FORMAT_VERSION = 1
|
|
56
|
+
KEY_ENV = "MODELSPEC_SNAPSHOT_KEY"
|
|
57
|
+
SIGNATURE_ALG = "hmac-sha256"
|
|
58
|
+
|
|
59
|
+
FactState = Literal["known", "unknown", "not_disclosed", "requires_contract"]
|
|
60
|
+
Lifecycle = Literal["active", "deprecated", "retired"]
|
|
61
|
+
Directness = Literal["direct", "proxy"]
|
|
62
|
+
FACT_STATES = ("known", "unknown", "not_disclosed", "requires_contract")
|
|
63
|
+
LIFECYCLES = ("active", "deprecated", "retired")
|
|
64
|
+
|
|
65
|
+
#: A number facet may hold these literals instead of a number (MODEL-133).
|
|
66
|
+
UNBOUNDED = "unbounded"
|
|
67
|
+
NOT_OFFERED = "not_offered"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# ── errors ─────────────────────────────────────────────────────────────────
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class SnapshotError(ValueError):
|
|
74
|
+
"""A snapshot could not be built or read."""
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class SnapshotBuildError(SnapshotError):
|
|
78
|
+
"""The inputs cannot make a snapshot."""
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class SnapshotIntegrityError(SnapshotError):
|
|
82
|
+
"""A snapshot file failed its format, hash or signature check."""
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@dataclass(frozen=True)
|
|
86
|
+
class Gap:
|
|
87
|
+
"""One guaranteed facet a premier subject lacks."""
|
|
88
|
+
|
|
89
|
+
model: str
|
|
90
|
+
subject: str
|
|
91
|
+
facet: str
|
|
92
|
+
reason: str
|
|
93
|
+
sources: tuple[str, ...] = ()
|
|
94
|
+
|
|
95
|
+
def __str__(self) -> str:
|
|
96
|
+
where = ", ".join(self.sources) if self.sources else "no source recorded"
|
|
97
|
+
return f"{self.subject}: {self.facet} is {self.reason}; source: {where}"
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class CompletenessError(SnapshotBuildError):
|
|
101
|
+
"""The premier-set completeness gate failed (design §5)."""
|
|
102
|
+
|
|
103
|
+
def __init__(self, gaps: Sequence[Gap]):
|
|
104
|
+
self.gaps = tuple(gaps)
|
|
105
|
+
lines = "\n ".join(str(g) for g in self.gaps)
|
|
106
|
+
super().__init__(f"completeness gate: {len(self.gaps)} guaranteed fact(s) missing "
|
|
107
|
+
f"for the premier set:\n {lines}")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
# ── the values an index returns ────────────────────────────────────────────
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@dataclass(frozen=True)
|
|
114
|
+
class FactValue:
|
|
115
|
+
state: FactState
|
|
116
|
+
value: Any = None
|
|
117
|
+
sources: tuple[str, ...] = ()
|
|
118
|
+
record_id: str | None = field(default=None, compare=False)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
UNKNOWN = FactValue("unknown")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
@dataclass(frozen=True)
|
|
125
|
+
class EvidenceValue:
|
|
126
|
+
benchmark_id: str
|
|
127
|
+
version: str | None
|
|
128
|
+
subcategory: str | None
|
|
129
|
+
value: float
|
|
130
|
+
unit: str | None
|
|
131
|
+
measured_by: str | None
|
|
132
|
+
effort: str | None
|
|
133
|
+
harness: str | None
|
|
134
|
+
#: The evidence date; ``None`` when the source gave less than a full date.
|
|
135
|
+
date: date | None
|
|
136
|
+
source_ids: tuple[str, ...]
|
|
137
|
+
verified: bool = True
|
|
138
|
+
#: Set by ``evidence_for_domain`` only: how directly the benchmark measures
|
|
139
|
+
#: the domain asked about. An addition to the agreed field list.
|
|
140
|
+
directness: Directness | None = None
|
|
141
|
+
record_id: str | None = field(default=None, compare=False)
|
|
142
|
+
date_type: str | None = None
|
|
143
|
+
source_snapshot: str | None = None
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
@dataclass(frozen=True)
|
|
147
|
+
class CapabilityEstimateValue:
|
|
148
|
+
value: float
|
|
149
|
+
low: float
|
|
150
|
+
high: float
|
|
151
|
+
sd: float
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
@dataclass(frozen=True)
|
|
155
|
+
class CapabilityDriverValue:
|
|
156
|
+
record_id: str
|
|
157
|
+
benchmark_id: str
|
|
158
|
+
version: str | None
|
|
159
|
+
loading: float
|
|
160
|
+
weight: float
|
|
161
|
+
recency_weight: float
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
@dataclass(frozen=True)
|
|
165
|
+
class Bitset3:
|
|
166
|
+
"""Three disjoint bitsets over ``candidates()``: bit ``i`` is candidate ``i``."""
|
|
167
|
+
|
|
168
|
+
passing: int
|
|
169
|
+
failing: int
|
|
170
|
+
unknown: int
|
|
171
|
+
|
|
172
|
+
def __post_init__(self) -> None:
|
|
173
|
+
if self.passing & self.failing or self.passing & self.unknown or self.failing & self.unknown:
|
|
174
|
+
raise ValueError("a Bitset3's three sets must be disjoint")
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
@runtime_checkable
|
|
178
|
+
class SnapshotIndex(Protocol):
|
|
179
|
+
snapshot_id: str
|
|
180
|
+
|
|
181
|
+
def candidates(self) -> Sequence[str]: ...
|
|
182
|
+
|
|
183
|
+
def lifecycle(self, cid: str) -> Lifecycle: ...
|
|
184
|
+
|
|
185
|
+
def fact(self, cid: str, facet_id: str) -> FactValue: ...
|
|
186
|
+
|
|
187
|
+
def ids_where(self, facet_id: str, op: str, arg: Any) -> Bitset3: ...
|
|
188
|
+
|
|
189
|
+
def evidence(self, cid: str, benchmark_id: str, *, measured_by: set[str] | None = None,
|
|
190
|
+
effort: str | None = None, harness: str | None = None,
|
|
191
|
+
after: date | None = None) -> Sequence[EvidenceValue]: ...
|
|
192
|
+
|
|
193
|
+
def evidence_where(
|
|
194
|
+
self,
|
|
195
|
+
benchmark_id: str,
|
|
196
|
+
op: str,
|
|
197
|
+
arg: Any,
|
|
198
|
+
*,
|
|
199
|
+
measured_by: set[str] | None = None,
|
|
200
|
+
effort: str | None = None,
|
|
201
|
+
harness: str | None = None,
|
|
202
|
+
after: date | None = None,
|
|
203
|
+
direct: bool = False,
|
|
204
|
+
domains: Iterable[str] = (),
|
|
205
|
+
) -> Bitset3: ...
|
|
206
|
+
|
|
207
|
+
def direct_for(self, benchmark_id: str, domains: Iterable[str] = ()) -> bool: ...
|
|
208
|
+
|
|
209
|
+
def evidence_for_domain(self, cid: str, domain_id: str) -> Sequence[EvidenceValue]: ...
|
|
210
|
+
|
|
211
|
+
def capability_estimate(
|
|
212
|
+
self, cid: str, domain_id: str
|
|
213
|
+
) -> CapabilityEstimateValue | None: ...
|
|
214
|
+
|
|
215
|
+
def capability_drivers(
|
|
216
|
+
self, cid: str, domain_id: str
|
|
217
|
+
) -> Sequence[CapabilityDriverValue]: ...
|
|
218
|
+
|
|
219
|
+
def evidence_record(self, cid: str, record_id: str) -> EvidenceValue | None: ...
|
|
220
|
+
|
|
221
|
+
def kind(self, cid: str) -> Literal["model", "offering"]: ...
|
|
222
|
+
|
|
223
|
+
def model_of(self, cid: str) -> str: ...
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
@runtime_checkable
|
|
227
|
+
class ExplanationIndex(SnapshotIndex, Protocol):
|
|
228
|
+
"""Snapshot metadata and retained records needed to transport a decision."""
|
|
229
|
+
|
|
230
|
+
def require_explanation_records(self) -> None: ...
|
|
231
|
+
def source_url(self, source_id: str) -> str: ...
|
|
232
|
+
def record(self, record_id: str) -> Mapping[str, Any]: ...
|
|
233
|
+
def facet_ids(self) -> tuple[str, ...]: ...
|
|
234
|
+
def domain_ids(self) -> tuple[str, ...]: ...
|
|
235
|
+
def benchmark_ids(self) -> tuple[str, ...]: ...
|
|
236
|
+
def benchmark_domain_tags(self) -> dict[str, tuple[tuple[str, str], ...]]: ...
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
# ── inputs ─────────────────────────────────────────────────────────────────
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
@dataclass(frozen=True)
|
|
243
|
+
class SnapshotInputs:
|
|
244
|
+
"""What a snapshot is compiled from. Records may be MODEL-134 objects or dicts."""
|
|
245
|
+
|
|
246
|
+
models: Sequence[Any] = ()
|
|
247
|
+
offerings: Sequence[Any] = ()
|
|
248
|
+
evidence: Sequence[Any] = ()
|
|
249
|
+
#: Registered source ID -> URL.
|
|
250
|
+
sources: Mapping[str, str] = field(default_factory=dict)
|
|
251
|
+
#: Benchmark ID -> ((domain ID, directness), ...), from the benchmark pages.
|
|
252
|
+
benchmark_domains: Mapping[str, Sequence[Sequence[str]]] = field(default_factory=dict)
|
|
253
|
+
#: Benchmark measurement metadata used by the build-time capability fit.
|
|
254
|
+
benchmark_metadata: Mapping[str, Mapping[str, Any]] = field(default_factory=dict)
|
|
255
|
+
#: The verification log. The latest verification of a target wins, over an
|
|
256
|
+
#: inline one too.
|
|
257
|
+
verifications: Sequence[Any] = ()
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _as_dict(record: Any) -> dict[str, Any]:
|
|
261
|
+
if isinstance(record, Mapping):
|
|
262
|
+
return dict(record)
|
|
263
|
+
if hasattr(record, "model_dump"):
|
|
264
|
+
return record.model_dump(mode="json", by_alias=True)
|
|
265
|
+
raise SnapshotBuildError(f"not a record: {record!r}")
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _counts(v: Mapping[str, Any]) -> bool:
|
|
269
|
+
"""A same-family ``verified`` is not a second key: it neither admits nor displaces."""
|
|
270
|
+
return verification_counts(str(v["outcome"]), str(v["collector"]["model_family"]),
|
|
271
|
+
str(v["verifier"]["model_family"]))
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _offering_id(o: Mapping[str, Any]) -> str:
|
|
275
|
+
return f"{o['provider']}/{o['model']}/{o['region']}/{o['tier']}"
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
# ── canonical form, hash and signature ─────────────────────────────────────
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def canonical_json(value: Any) -> bytes:
|
|
282
|
+
return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False,
|
|
283
|
+
allow_nan=False).encode("utf-8")
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def content_hash(content: Mapping[str, Any]) -> str:
|
|
287
|
+
return "sha256:" + hashlib.sha256(canonical_json(content)).hexdigest()
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def snapshot_id_for(digest: str) -> str:
|
|
291
|
+
return "snap_" + digest.removeprefix("sha256:")[:16]
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _key_bytes(key: bytes | str | None) -> bytes | None:
|
|
295
|
+
if key is None or key == "" or key == b"":
|
|
296
|
+
return None
|
|
297
|
+
return key.encode("utf-8") if isinstance(key, str) else key
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _sign(digest: str, key: bytes) -> str:
|
|
301
|
+
return hmac.new(key, digest.encode("ascii"), hashlib.sha256).hexdigest()
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
_FROM_ENV: Any = object()
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def env_key() -> bytes | None:
|
|
308
|
+
"""The signing key from ``MODELSPEC_SNAPSHOT_KEY``, or ``None``."""
|
|
309
|
+
return _key_bytes(os.environ.get(KEY_ENV))
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _record_fields(
|
|
313
|
+
record: Mapping[str, Any], prefix: tuple[str, ...] = (),
|
|
314
|
+
) -> Iterable[tuple[tuple[str, ...], Any]]:
|
|
315
|
+
for key, value in record.items():
|
|
316
|
+
path = (*prefix, key)
|
|
317
|
+
if isinstance(value, dict) and value:
|
|
318
|
+
yield from _record_fields(value, path)
|
|
319
|
+
else:
|
|
320
|
+
yield path, value
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _pack_records(records: Mapping[str, Mapping[str, Any]]) -> dict[str, Any]:
|
|
324
|
+
"""Intern repeated provenance values, including verification and source data.
|
|
325
|
+
|
|
326
|
+
Nested dictionary paths preserve absent fields, explicit nulls and empty
|
|
327
|
+
dictionaries distinctly. Values stay JSON until a record is requested.
|
|
328
|
+
"""
|
|
329
|
+
flattened = {rid: dict(_record_fields(record)) for rid, record in records.items()}
|
|
330
|
+
fields = sorted({path for record in flattened.values() for path in record})
|
|
331
|
+
values: list[str] = []
|
|
332
|
+
positions: dict[bytes, int] = {}
|
|
333
|
+
rows = {}
|
|
334
|
+
for rid, record in sorted(flattened.items()):
|
|
335
|
+
row = []
|
|
336
|
+
for path in fields:
|
|
337
|
+
if path not in record:
|
|
338
|
+
row.append(None)
|
|
339
|
+
continue
|
|
340
|
+
encoded = canonical_json(record[path])
|
|
341
|
+
if encoded not in positions:
|
|
342
|
+
positions[encoded] = len(values)
|
|
343
|
+
values.append(encoded.decode("utf-8"))
|
|
344
|
+
row.append(positions[encoded])
|
|
345
|
+
rows[rid] = row
|
|
346
|
+
return {"fields": fields, "values": values, "rows": rows}
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
def _unpack_record(table: Mapping[str, Any], rid: str) -> dict[str, Any]:
|
|
350
|
+
record: dict[str, Any] = {}
|
|
351
|
+
for path, position in zip(table["fields"], table["rows"][rid]):
|
|
352
|
+
if position is None:
|
|
353
|
+
continue
|
|
354
|
+
target = record
|
|
355
|
+
for key in path[:-1]:
|
|
356
|
+
target = target.setdefault(key, {})
|
|
357
|
+
target[path[-1]] = json.loads(table["values"][position])
|
|
358
|
+
return record
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
# ── the built snapshot ─────────────────────────────────────────────────────
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
@dataclass(frozen=True)
|
|
365
|
+
class Snapshot:
|
|
366
|
+
content: Mapping[str, Any]
|
|
367
|
+
content_hash: str
|
|
368
|
+
snapshot_id: str
|
|
369
|
+
|
|
370
|
+
def envelope(self, key: bytes | str | None = _FROM_ENV) -> dict[str, Any]:
|
|
371
|
+
key = env_key() if key is _FROM_ENV else _key_bytes(key)
|
|
372
|
+
signature = None if key is None else {"alg": SIGNATURE_ALG,
|
|
373
|
+
"value": _sign(self.content_hash, key)}
|
|
374
|
+
return {"format": FORMAT, "format_version": FORMAT_VERSION,
|
|
375
|
+
"snapshot_id": self.snapshot_id, "content_hash": self.content_hash,
|
|
376
|
+
"signature": signature, "content": self.content}
|
|
377
|
+
|
|
378
|
+
def to_bytes(self, key: bytes | str | None = _FROM_ENV) -> bytes:
|
|
379
|
+
buf = io.BytesIO()
|
|
380
|
+
# A fixed mtime and no file name keep the gzip header deterministic.
|
|
381
|
+
with gzip.GzipFile(filename="", mode="wb", fileobj=buf, compresslevel=9, mtime=0) as gz:
|
|
382
|
+
gz.write(canonical_json(self.envelope(key)))
|
|
383
|
+
return buf.getvalue()
|
|
384
|
+
|
|
385
|
+
def write(self, path: str | Path, *, key: bytes | str | None = _FROM_ENV) -> Path:
|
|
386
|
+
path = Path(path)
|
|
387
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
388
|
+
path.write_bytes(self.to_bytes(key))
|
|
389
|
+
return path
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
# ── building ───────────────────────────────────────────────────────────────
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
class _Compiler:
|
|
396
|
+
def __init__(self, inputs: SnapshotInputs, registry: Any, guard: ExcludedSources | None):
|
|
397
|
+
self.inputs = inputs
|
|
398
|
+
self.registry = registry
|
|
399
|
+
self.guard = guard
|
|
400
|
+
self.sources = {str(k): str(v) for k, v in inputs.sources.items()}
|
|
401
|
+
#: subject id (``None`` when a record names none) -> why its records stayed out.
|
|
402
|
+
self.excluded: dict[str | None, Counter[str]] = {}
|
|
403
|
+
#: (subject, facet) -> (reason, source URLs), for the gate's messages.
|
|
404
|
+
self.rejected: dict[tuple[str, str], tuple[str, tuple[str, ...]]] = {}
|
|
405
|
+
self.log = self._verification_log(inputs.verifications)
|
|
406
|
+
#: subject id -> {"kind", "model", "lifecycle"}
|
|
407
|
+
self.subjects: dict[str, dict[str, Any]] = {}
|
|
408
|
+
#: subject id -> facet -> [state, value, source ids]
|
|
409
|
+
self.facts: dict[str, dict[str, list[Any]]] = {}
|
|
410
|
+
self.facet_subject: dict[str, str] = {}
|
|
411
|
+
self.evidence: dict[str, list[list[Any]]] = {}
|
|
412
|
+
self.records: dict[str, dict[str, Any]] = {}
|
|
413
|
+
self.fact_records: dict[str, dict[str, str]] = {}
|
|
414
|
+
|
|
415
|
+
# verification ------------------------------------------------------------
|
|
416
|
+
|
|
417
|
+
@staticmethod
|
|
418
|
+
def _verification_log(
|
|
419
|
+
rows: Iterable[Any],
|
|
420
|
+
) -> dict[tuple[str, str, str], tuple[str, int, dict]]:
|
|
421
|
+
latest: dict[tuple[str, str, str], tuple[str, int, dict]] = {}
|
|
422
|
+
for i, raw in enumerate(rows):
|
|
423
|
+
v = _as_dict(raw)
|
|
424
|
+
if not _counts(v):
|
|
425
|
+
continue
|
|
426
|
+
target = v["target"]
|
|
427
|
+
key = (target["kind"], target["id"], str(target.get("value_hash") or ""))
|
|
428
|
+
entry = (str(v["date"]), i + 1, v)
|
|
429
|
+
if key not in latest or entry[:2] >= latest[key][:2]:
|
|
430
|
+
latest[key] = entry
|
|
431
|
+
return latest
|
|
432
|
+
|
|
433
|
+
def _verification(self, kind: str, rid: Any, inline: Any, value: Any) -> dict | None:
|
|
434
|
+
"""The winning verification record, or ``None``."""
|
|
435
|
+
expected = value_hash(value)
|
|
436
|
+
inline_record = None if inline is None else _as_dict(inline)
|
|
437
|
+
best = (
|
|
438
|
+
None
|
|
439
|
+
if inline_record is None
|
|
440
|
+
or not _counts(inline_record)
|
|
441
|
+
or inline_record.get("target", {}).get("value_hash") != expected
|
|
442
|
+
else (str(inline_record["date"]), 0, inline_record)
|
|
443
|
+
)
|
|
444
|
+
logged = self.log.get((kind, str(rid), expected)) if rid is not None else None
|
|
445
|
+
if logged is not None and (best is None or logged[:2] >= best[:2]):
|
|
446
|
+
best = logged
|
|
447
|
+
return None if best is None else best[2]
|
|
448
|
+
|
|
449
|
+
def _outcome(self, kind: str, rid: Any, inline: Any, value: Any) -> str:
|
|
450
|
+
record = self._verification(kind, rid, inline, value)
|
|
451
|
+
return "unverified" if record is None else str(record["outcome"])
|
|
452
|
+
|
|
453
|
+
def _retain(self, kind: str, record: dict) -> str:
|
|
454
|
+
rid = str(record.get("id") or content_hash(record))
|
|
455
|
+
value = record.get("value") if kind == "fact" else record.get("score")
|
|
456
|
+
retained = {**record, "verification": self._verification(
|
|
457
|
+
kind, record.get("id"), record.get("verification"), value)}
|
|
458
|
+
if rid in self.records and self.records[rid] != retained:
|
|
459
|
+
raise SnapshotBuildError(f"duplicate record ID {rid}")
|
|
460
|
+
self.records[rid] = retained
|
|
461
|
+
return rid
|
|
462
|
+
|
|
463
|
+
# admission ---------------------------------------------------------------
|
|
464
|
+
|
|
465
|
+
def _source_ids(self, refs: Iterable[Any]) -> list[str]:
|
|
466
|
+
return sorted({str(_as_dict(r)["source_id"]) for r in refs or ()})
|
|
467
|
+
|
|
468
|
+
def _admit(self, kind: str, rid: Any, inline: Any, value: Any, source_ids: list[str],
|
|
469
|
+
extra_urls: Iterable[Any] = (), benchmark: Any = None) -> str | None:
|
|
470
|
+
"""Why a record stays out, or ``None`` when it enters."""
|
|
471
|
+
urls = [self.sources[s] for s in source_ids if s in self.sources]
|
|
472
|
+
if self.guard is not None and (any(self.guard.url(u) for u in [*urls, *extra_urls])
|
|
473
|
+
or (benchmark is not None
|
|
474
|
+
and self.guard.benchmark(benchmark))):
|
|
475
|
+
return "excluded_source"
|
|
476
|
+
outcome = self._outcome(kind, rid, inline, value)
|
|
477
|
+
if outcome != "verified":
|
|
478
|
+
return f"quarantined ({outcome})"
|
|
479
|
+
if not source_ids:
|
|
480
|
+
return "unsourced"
|
|
481
|
+
if len(urls) != len(source_ids):
|
|
482
|
+
return "unresolved_source"
|
|
483
|
+
return None
|
|
484
|
+
|
|
485
|
+
def _exclude(self, sid: str | None, reason: str) -> None:
|
|
486
|
+
kind = "quarantined" if reason.startswith("quarantined") else reason
|
|
487
|
+
self.excluded.setdefault(sid, Counter())[kind] += 1
|
|
488
|
+
|
|
489
|
+
def _reject(self, key: tuple[str, str], reason: str, source_ids: list[str]) -> None:
|
|
490
|
+
self._exclude(key[0], reason)
|
|
491
|
+
urls = tuple(self.sources.get(s, s) for s in source_ids)
|
|
492
|
+
self.rejected[key] = (reason, urls)
|
|
493
|
+
|
|
494
|
+
# subjects ----------------------------------------------------------------
|
|
495
|
+
|
|
496
|
+
def _check_facet(self, facet_id: str, kind: str) -> None:
|
|
497
|
+
if self.registry is not None:
|
|
498
|
+
try:
|
|
499
|
+
registered = self.registry.facet(facet_id)
|
|
500
|
+
except KeyError as exc:
|
|
501
|
+
raise SnapshotBuildError(f"facet {facet_id!r} is not registered") from exc
|
|
502
|
+
if getattr(registered, "computed_by", None):
|
|
503
|
+
raise SnapshotBuildError(
|
|
504
|
+
f"facet {facet_id!r} is computed ({registered.computed_by}), never authored")
|
|
505
|
+
seen = self.facet_subject.setdefault(facet_id, kind)
|
|
506
|
+
if seen != kind:
|
|
507
|
+
raise SnapshotBuildError(f"facet {facet_id!r} is used on both a {seen} and a {kind}")
|
|
508
|
+
|
|
509
|
+
def _add_facts(self, sid: str, kind: str, facts: Iterable[Any]) -> None:
|
|
510
|
+
row = self.facts.setdefault(sid, {})
|
|
511
|
+
for raw in facts or ():
|
|
512
|
+
f = _as_dict(raw)
|
|
513
|
+
facet_id, state = str(f["facet"]), str(f["state"])
|
|
514
|
+
self._check_facet(facet_id, kind)
|
|
515
|
+
if state not in FACT_STATES:
|
|
516
|
+
raise SnapshotBuildError(f"{sid}: {facet_id} has an unknown state {state!r}")
|
|
517
|
+
if facet_id in row or (sid, facet_id) in self.rejected:
|
|
518
|
+
raise SnapshotBuildError(f"{sid}: {facet_id} is stated twice")
|
|
519
|
+
source_ids = self._source_ids(f.get("sources"))
|
|
520
|
+
if state == "unknown":
|
|
521
|
+
self.rejected[(sid, facet_id)] = ("unknown", tuple(
|
|
522
|
+
self.sources.get(s, s) for s in source_ids))
|
|
523
|
+
continue
|
|
524
|
+
reason = self._admit(
|
|
525
|
+
"fact", f.get("id"), f.get("verification"), f.get("value"), source_ids
|
|
526
|
+
)
|
|
527
|
+
if reason is not None:
|
|
528
|
+
self._reject((sid, facet_id), reason, source_ids)
|
|
529
|
+
continue
|
|
530
|
+
self.fact_records.setdefault(sid, {})[facet_id] = self._retain("fact", f)
|
|
531
|
+
row[facet_id] = [state, f.get("value") if state == "known" else None, source_ids]
|
|
532
|
+
|
|
533
|
+
def add_model(self, raw: Any) -> None:
|
|
534
|
+
m = _as_dict(raw)
|
|
535
|
+
mid, lifecycle = str(m["id"]), str(m.get("lifecycle"))
|
|
536
|
+
if lifecycle not in LIFECYCLES:
|
|
537
|
+
raise SnapshotBuildError(f"{mid}: lifecycle {lifecycle!r} is not one of {LIFECYCLES}")
|
|
538
|
+
if mid in self.subjects:
|
|
539
|
+
raise SnapshotBuildError(f"model {mid} appears twice")
|
|
540
|
+
self.subjects[mid] = {"kind": "model", "model": mid, "lifecycle": lifecycle}
|
|
541
|
+
self._add_facts(mid, "model", m.get("facts"))
|
|
542
|
+
|
|
543
|
+
def add_offering(self, raw: Any) -> None:
|
|
544
|
+
o = _as_dict(raw)
|
|
545
|
+
oid, mid = _offering_id(o), str(o["model"])
|
|
546
|
+
if mid not in self.subjects:
|
|
547
|
+
raise SnapshotBuildError(f"offering {oid} names model {mid}, which is not in the catalogue")
|
|
548
|
+
if oid in self.subjects:
|
|
549
|
+
raise SnapshotBuildError(f"offering {oid} appears twice")
|
|
550
|
+
self.subjects[oid] = {"kind": "offering", "model": mid,
|
|
551
|
+
"lifecycle": self.subjects[mid]["lifecycle"]}
|
|
552
|
+
self._add_facts(oid, "offering", o.get("facts"))
|
|
553
|
+
# An offering's identity is its provider, region and tier: structural,
|
|
554
|
+
# not a sourced claim, so they carry no source.
|
|
555
|
+
row = self.facts[oid]
|
|
556
|
+
for part in ("provider", "region", "tier"):
|
|
557
|
+
facet_id = f"offering.{part}"
|
|
558
|
+
if facet_id not in row:
|
|
559
|
+
self._check_facet(facet_id, "offering")
|
|
560
|
+
row[facet_id] = ["known", str(o[part]), []]
|
|
561
|
+
|
|
562
|
+
def add_evidence(self, raw: Any) -> None:
|
|
563
|
+
e = _as_dict(raw)
|
|
564
|
+
subject = e.get("subject") or {}
|
|
565
|
+
sid = subject.get("id")
|
|
566
|
+
if sid is None:
|
|
567
|
+
self._exclude(None, "quarantined") # no subject: cannot be v2-verified
|
|
568
|
+
return
|
|
569
|
+
if sid not in self.subjects:
|
|
570
|
+
raise SnapshotBuildError(f"evidence {e.get('id')!r} names {sid}, which is not in the catalogue")
|
|
571
|
+
source_ids = self._source_ids(e.get("sources"))
|
|
572
|
+
reason = self._admit("evidence", e.get("id"), e.get("verification"), e.get("score"), source_ids,
|
|
573
|
+
extra_urls=[e.get("source_url")], benchmark=e.get("benchmark_id"))
|
|
574
|
+
if reason is not None:
|
|
575
|
+
self._exclude(sid, reason)
|
|
576
|
+
return
|
|
577
|
+
self.evidence.setdefault(sid, []).append([
|
|
578
|
+
str(e["benchmark_id"]), e.get("benchmark_version") or None, e.get("subcategory"),
|
|
579
|
+
float(e["score"]), e.get("unit"), e.get("measured_by"), e.get("effort"),
|
|
580
|
+
e.get("harness"), str(e.get("evidence_date") or "") or None, source_ids,
|
|
581
|
+
self._retain("evidence", e), e.get("date_type"),
|
|
582
|
+
next((r.get("snapshot_ref") for r in e.get("sources", [])
|
|
583
|
+
if r["source_id"] == source_ids[0]), None),
|
|
584
|
+
])
|
|
585
|
+
|
|
586
|
+
# output ------------------------------------------------------------------
|
|
587
|
+
|
|
588
|
+
def _section(self, ids: list[str]) -> dict[str, Any]:
|
|
589
|
+
index = {sid: i for i, sid in enumerate(ids)}
|
|
590
|
+
columns: dict[str, dict[str, list[Any]]] = {}
|
|
591
|
+
for sid in ids:
|
|
592
|
+
for facet_id, (state, value, sources) in self.facts.get(sid, {}).items():
|
|
593
|
+
col = columns.setdefault(facet_id, {"row": [], "state": [], "value": [],
|
|
594
|
+
"sources": []})
|
|
595
|
+
col["row"].append(index[sid])
|
|
596
|
+
col["state"].append(state)
|
|
597
|
+
col["value"].append(value)
|
|
598
|
+
col["sources"].append(sources)
|
|
599
|
+
return {
|
|
600
|
+
"candidates": [{"id": sid, **self.subjects[sid]} for sid in ids],
|
|
601
|
+
"facets": columns,
|
|
602
|
+
"evidence": {sid: sorted(self.evidence[sid], key=lambda r: (
|
|
603
|
+
r[0], r[8] or "", r[3], canonical_json(r))) for sid in ids if sid in self.evidence},
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
def content(self, as_of: date | None, premier: Iterable[str] | None = None) -> dict[str, Any]:
|
|
607
|
+
"""The snapshot content. With ``premier``, the lineup is the premier set.
|
|
608
|
+
|
|
609
|
+
Retired models always go to the archive. Active and deprecated models
|
|
610
|
+
outside the premier set leave the snapshot, with their offerings,
|
|
611
|
+
evidence and records; only their number is kept, as ``out_of_lineup``.
|
|
612
|
+
"""
|
|
613
|
+
wanted = None if premier is None else set(premier)
|
|
614
|
+
archive = sorted(s for s, v in self.subjects.items() if v["lifecycle"] == "retired")
|
|
615
|
+
lineup = sorted(s for s, v in self.subjects.items() if v["lifecycle"] != "retired"
|
|
616
|
+
and (wanted is None or v["model"] in wanted))
|
|
617
|
+
kept = {*lineup, *archive}
|
|
618
|
+
out_of_lineup = sum(1 for s, v in self.subjects.items()
|
|
619
|
+
if v["kind"] == "model" and s not in kept)
|
|
620
|
+
sources: set[str] = set()
|
|
621
|
+
record_ids: set[str] = set()
|
|
622
|
+
for sid in kept:
|
|
623
|
+
for _state, _value, source_ids in self.facts.get(sid, {}).values():
|
|
624
|
+
sources.update(source_ids)
|
|
625
|
+
record_ids.update(self.fact_records.get(sid, {}).values())
|
|
626
|
+
for row in self.evidence.get(sid, ()):
|
|
627
|
+
sources.update(row[9])
|
|
628
|
+
record_ids.add(row[10])
|
|
629
|
+
excluded: Counter[str] = Counter()
|
|
630
|
+
for sid, counts in self.excluded.items():
|
|
631
|
+
if sid is None or sid in kept:
|
|
632
|
+
excluded.update(counts)
|
|
633
|
+
domains = {}
|
|
634
|
+
for bench, tags in sorted(self.inputs.benchmark_domains.items()):
|
|
635
|
+
if self.guard is not None and self.guard.benchmark(bench):
|
|
636
|
+
continue
|
|
637
|
+
domains[str(bench)] = sorted([str(d), str(k)] for d, k in tags)
|
|
638
|
+
capability: dict[str, Any] = {}
|
|
639
|
+
if self.inputs.benchmark_metadata and as_of is not None:
|
|
640
|
+
from decision.capability import (
|
|
641
|
+
BenchmarkSpec,
|
|
642
|
+
CapabilityObservation,
|
|
643
|
+
Directness,
|
|
644
|
+
fit_capabilities,
|
|
645
|
+
)
|
|
646
|
+
|
|
647
|
+
observations = []
|
|
648
|
+
fitted_models = {self.subjects[sid]["model"] for sid in kept}
|
|
649
|
+
for subject, evidence_rows in sorted(self.evidence.items()):
|
|
650
|
+
model_id = self.subjects[subject]["model"]
|
|
651
|
+
if model_id not in fitted_models:
|
|
652
|
+
continue
|
|
653
|
+
for row in evidence_rows:
|
|
654
|
+
evidence_date = _date(row[8])
|
|
655
|
+
tag_rows: list[tuple[str, Directness]] = []
|
|
656
|
+
for domain_id, raw_directness in self.inputs.benchmark_domains.get(row[0], ()):
|
|
657
|
+
directness = str(raw_directness)
|
|
658
|
+
if directness not in ("direct", "proxy"):
|
|
659
|
+
raise SnapshotBuildError(
|
|
660
|
+
f"{row[0]}: invalid capability directness {directness!r}"
|
|
661
|
+
)
|
|
662
|
+
tag_rows.append((str(domain_id), directness))
|
|
663
|
+
tags = tuple(tag_rows)
|
|
664
|
+
if evidence_date is None or not tags:
|
|
665
|
+
continue
|
|
666
|
+
observations.append(CapabilityObservation(
|
|
667
|
+
model_id=model_id,
|
|
668
|
+
benchmark_id=row[0],
|
|
669
|
+
value=float(row[3]),
|
|
670
|
+
unit=row[4],
|
|
671
|
+
measured_by=str(row[5] or ""),
|
|
672
|
+
date=evidence_date,
|
|
673
|
+
record_id=row[10],
|
|
674
|
+
version=row[1],
|
|
675
|
+
domains=tags,
|
|
676
|
+
))
|
|
677
|
+
specs = {
|
|
678
|
+
benchmark: BenchmarkSpec(
|
|
679
|
+
random_baseline=metadata.get("random_baseline"),
|
|
680
|
+
sample_size=metadata.get("sample_size"),
|
|
681
|
+
direction=metadata.get("direction", "higher_is_better"),
|
|
682
|
+
)
|
|
683
|
+
for benchmark, metadata in self.inputs.benchmark_metadata.items()
|
|
684
|
+
}
|
|
685
|
+
fit = fit_capabilities(observations, specs, as_of=as_of)
|
|
686
|
+
capability = fit.to_payload(
|
|
687
|
+
self.subjects[sid]["model"] for sid in kept
|
|
688
|
+
if self.subjects[sid]["kind"] == "model"
|
|
689
|
+
)
|
|
690
|
+
return {
|
|
691
|
+
"format_version": FORMAT_VERSION,
|
|
692
|
+
"as_of": as_of.isoformat() if as_of else None,
|
|
693
|
+
"facet_subjects": dict(sorted(self.facet_subject.items())),
|
|
694
|
+
"lineup": self._section(lineup),
|
|
695
|
+
"archive": self._section(archive),
|
|
696
|
+
"out_of_lineup": out_of_lineup,
|
|
697
|
+
"benchmark_domains": domains,
|
|
698
|
+
"capability": capability,
|
|
699
|
+
"sources": {s: self.sources[s] for s in sorted(sources)},
|
|
700
|
+
"excluded": dict(sorted(excluded.items())),
|
|
701
|
+
"record_table": _pack_records({r: self.records[r] for r in record_ids}),
|
|
702
|
+
"fact_records": {sid: rows for sid, rows in self.fact_records.items() if sid in kept},
|
|
703
|
+
}
|
|
704
|
+
|
|
705
|
+
# the gate ----------------------------------------------------------------
|
|
706
|
+
|
|
707
|
+
def gaps(self, premier: Iterable[str]) -> list[Gap]:
|
|
708
|
+
if self.registry is None:
|
|
709
|
+
raise SnapshotBuildError("the completeness gate needs the facet registry")
|
|
710
|
+
guaranteed = [f for f in self.registry.facets()
|
|
711
|
+
if f.tier == "guaranteed" and not getattr(f, "computed_by", None)
|
|
712
|
+
and f.subject in ("model", "offering")]
|
|
713
|
+
out: list[Gap] = []
|
|
714
|
+
for mid in sorted(set(premier)):
|
|
715
|
+
subject = self.subjects.get(mid)
|
|
716
|
+
if subject is None:
|
|
717
|
+
out.append(Gap(mid, mid, "(model)", "not in the catalogue"))
|
|
718
|
+
continue
|
|
719
|
+
if subject["lifecycle"] == "retired":
|
|
720
|
+
continue # retired models leave the premier set (design §5)
|
|
721
|
+
offerings = sorted(s for s, v in self.subjects.items()
|
|
722
|
+
if v["kind"] == "offering" and v["model"] == mid)
|
|
723
|
+
for f in sorted(guaranteed, key=lambda f: f.id):
|
|
724
|
+
for sid in ([mid] if f.subject == "model" else offerings):
|
|
725
|
+
if f.id in self.facts.get(sid, {}):
|
|
726
|
+
continue
|
|
727
|
+
reason, urls = self.rejected.get((sid, f.id), ("unknown (no fact)", ()))
|
|
728
|
+
if reason == "unknown":
|
|
729
|
+
reason = "unknown (stated as unknown)"
|
|
730
|
+
out.append(Gap(mid, sid, f.id, reason, urls))
|
|
731
|
+
return out
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
def default_registry() -> Any:
|
|
735
|
+
try:
|
|
736
|
+
from decision import registry
|
|
737
|
+
except ImportError as exc: # MODEL-133 not present
|
|
738
|
+
raise SnapshotBuildError(f"the facet registry is not available: {exc}") from exc
|
|
739
|
+
return registry.default()
|
|
740
|
+
|
|
741
|
+
|
|
742
|
+
def build_snapshot(inputs: SnapshotInputs, *, registry: Any = None,
|
|
743
|
+
premier: Iterable[str] | None = None, as_of: date | None = None,
|
|
744
|
+
guard: ExcludedSources | None = None, gate: bool = True) -> Snapshot:
|
|
745
|
+
"""Compile ``inputs``. With ``premier``, the lineup is the premier set.
|
|
746
|
+
|
|
747
|
+
With ``premier`` and ``gate`` (the default), the completeness gate runs
|
|
748
|
+
first. ``gate=False`` keeps the premier lineup but skips the gate, for an
|
|
749
|
+
audit that must run while facts are still missing (MODEL-146).
|
|
750
|
+
``registry`` validates facet IDs and names the guaranteed facets; the gate
|
|
751
|
+
requires it. ``guard`` drops excluded sources and scans the output;
|
|
752
|
+
``build_from_repo`` always passes it.
|
|
753
|
+
"""
|
|
754
|
+
c = _Compiler(inputs, registry, guard)
|
|
755
|
+
for m in inputs.models:
|
|
756
|
+
c.add_model(m)
|
|
757
|
+
for o in inputs.offerings:
|
|
758
|
+
c.add_offering(o)
|
|
759
|
+
for e in inputs.evidence:
|
|
760
|
+
c.add_evidence(e)
|
|
761
|
+
premier = None if premier is None else tuple(premier)
|
|
762
|
+
if premier is not None and gate:
|
|
763
|
+
gaps = c.gaps(premier)
|
|
764
|
+
if gaps:
|
|
765
|
+
raise CompletenessError(gaps)
|
|
766
|
+
content = c.content(as_of, premier)
|
|
767
|
+
if guard is not None:
|
|
768
|
+
text = canonical_json(content).decode("utf-8")
|
|
769
|
+
hit = guard.text.search(text)
|
|
770
|
+
bad = [u for u in content["sources"].values() if guard.url(u)]
|
|
771
|
+
if hit or bad:
|
|
772
|
+
raise SnapshotBuildError(
|
|
773
|
+
f"excluded source in the snapshot output: {hit.group(0) if hit else bad[0]!r}")
|
|
774
|
+
digest = content_hash(content)
|
|
775
|
+
return Snapshot(content=content, content_hash=digest, snapshot_id=snapshot_id_for(digest))
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
# ── reading the repository ─────────────────────────────────────────────────
|
|
779
|
+
|
|
780
|
+
|
|
781
|
+
_LIFECYCLE_FROM_STATUS = {"deprecated": "deprecated", "sunset": "deprecated"}
|
|
782
|
+
|
|
783
|
+
|
|
784
|
+
def collect_repo(root: Path) -> SnapshotInputs:
|
|
785
|
+
"""Read the snapshot's inputs from a repository checkout.
|
|
786
|
+
|
|
787
|
+
* Cards (``models/``): ``lifecycle`` (else the v1 ``status``: deprecated and
|
|
788
|
+
sunset map to ``deprecated``, everything else to ``active``), v2 ``facts``,
|
|
789
|
+
and ``benchmarks.evidence`` rows. ``benchmarks.scores`` is never read.
|
|
790
|
+
* Offerings: ``offerings/<provider>/<lab>/<model>.yaml``, each a list.
|
|
791
|
+
* Sources: canonical ``registry/sources.yaml``; see ``verification/README.md``.
|
|
792
|
+
* Domains: each benchmark page's ``domains`` tags.
|
|
793
|
+
* The verification log: ``verification/log.jsonl``.
|
|
794
|
+
"""
|
|
795
|
+
from pipeline.load import load_benchmarks, load_models
|
|
796
|
+
|
|
797
|
+
root = Path(root)
|
|
798
|
+
models, evidence = [], []
|
|
799
|
+
for card in load_models(root):
|
|
800
|
+
mid = card.model_id
|
|
801
|
+
front = card.front
|
|
802
|
+
lifecycle = front.get("lifecycle") or _LIFECYCLE_FROM_STATUS.get(
|
|
803
|
+
str(front.get("status") or ""), "active")
|
|
804
|
+
facts = []
|
|
805
|
+
for f in front.get("facts") or []:
|
|
806
|
+
facts.append({"subject": {"kind": "model", "id": mid},
|
|
807
|
+
"id": f"{mid}#{f.get('facet')}", **f})
|
|
808
|
+
models.append({"id": mid, "lifecycle": lifecycle, "facts": facts})
|
|
809
|
+
for row in card.evidence:
|
|
810
|
+
evidence.append({"subject": {"kind": "model", "id": mid}, **row})
|
|
811
|
+
offerings = []
|
|
812
|
+
for path in sorted((root / "offerings").glob("*/*/*.yaml")):
|
|
813
|
+
rows = yaml.safe_load(path.read_text(encoding="utf-8")) or []
|
|
814
|
+
if not isinstance(rows, list):
|
|
815
|
+
raise SnapshotBuildError(f"{path}: an offering file is a list")
|
|
816
|
+
for o in rows:
|
|
817
|
+
oid = _offering_id(o)
|
|
818
|
+
o = dict(o)
|
|
819
|
+
o["facts"] = [{"subject": {"kind": "offering", "id": oid},
|
|
820
|
+
"id": f"{oid}#{f.get('facet')}", **f} for f in o.get("facts") or []]
|
|
821
|
+
offerings.append(o)
|
|
822
|
+
from decision.sources import load_sources
|
|
823
|
+
|
|
824
|
+
sources = {
|
|
825
|
+
source_id: str(source.url)
|
|
826
|
+
for source_id, source in load_sources(root / "registry" / "sources.yaml").items()
|
|
827
|
+
}
|
|
828
|
+
domains = {}
|
|
829
|
+
metadata = {}
|
|
830
|
+
for b in load_benchmarks(root):
|
|
831
|
+
tags = b.front.get("domains") or []
|
|
832
|
+
if tags:
|
|
833
|
+
domains[b.benchmark_id] = tuple((str(t["id"]), str(t["directness"])) for t in tags)
|
|
834
|
+
metric = b.front.get("metric") or {}
|
|
835
|
+
dataset = b.front.get("dataset") or {}
|
|
836
|
+
metadata[b.benchmark_id] = {
|
|
837
|
+
"random_baseline": metric.get("random_baseline"),
|
|
838
|
+
"sample_size": dataset.get("size"),
|
|
839
|
+
"direction": metric.get("direction", "higher_is_better"),
|
|
840
|
+
}
|
|
841
|
+
verifications = []
|
|
842
|
+
verification_log = root / "verification" / "log.jsonl"
|
|
843
|
+
if verification_log.is_file():
|
|
844
|
+
for line in verification_log.read_text(encoding="utf-8").splitlines():
|
|
845
|
+
if line.strip():
|
|
846
|
+
verifications.append(json.loads(line))
|
|
847
|
+
return SnapshotInputs(models=models, offerings=offerings, evidence=evidence, sources=sources,
|
|
848
|
+
benchmark_domains=domains, benchmark_metadata=metadata,
|
|
849
|
+
verifications=verifications)
|
|
850
|
+
|
|
851
|
+
|
|
852
|
+
def load_premier(path: str | Path) -> tuple[str, ...]:
|
|
853
|
+
"""The premier model IDs from a YAML file: a list, or ``models:`` a list.
|
|
854
|
+
|
|
855
|
+
Each item is an ID or a mapping with ``id`` or ``model_id``.
|
|
856
|
+
"""
|
|
857
|
+
path = Path(path)
|
|
858
|
+
try:
|
|
859
|
+
data = yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
860
|
+
except (OSError, yaml.YAMLError) as exc:
|
|
861
|
+
raise SnapshotBuildError(f"cannot read the premier list {path}: {exc}") from exc
|
|
862
|
+
items = data.get("models") if isinstance(data, Mapping) else data
|
|
863
|
+
ids = set()
|
|
864
|
+
for item in items or []:
|
|
865
|
+
mid = item.get("id") or item.get("model_id") if isinstance(item, Mapping) else item
|
|
866
|
+
if not isinstance(mid, str) or not mid:
|
|
867
|
+
raise SnapshotBuildError(f"{path}: premier entry {item!r} has no model id")
|
|
868
|
+
ids.add(mid)
|
|
869
|
+
if not ids:
|
|
870
|
+
raise SnapshotBuildError(f"{path}: no premier models listed")
|
|
871
|
+
return tuple(sorted(ids))
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
def build_from_repo(root: Path, *, premier: str | Path | None, as_of: date | None,
|
|
875
|
+
registry: Any = None, gate: bool = True) -> Snapshot:
|
|
876
|
+
"""The production build: collect, guard, gate and compile."""
|
|
877
|
+
root = Path(root)
|
|
878
|
+
return build_snapshot(
|
|
879
|
+
collect_repo(root),
|
|
880
|
+
registry=registry if registry is not None else default_registry(),
|
|
881
|
+
premier=load_premier(premier) if premier is not None else None,
|
|
882
|
+
as_of=as_of,
|
|
883
|
+
guard=excluded_sources(),
|
|
884
|
+
gate=gate,
|
|
885
|
+
)
|
|
886
|
+
|
|
887
|
+
|
|
888
|
+
# ── loading ────────────────────────────────────────────────────────────────
|
|
889
|
+
|
|
890
|
+
|
|
891
|
+
def _date(value: Any) -> date | None:
|
|
892
|
+
try:
|
|
893
|
+
return date.fromisoformat(str(value))
|
|
894
|
+
except (TypeError, ValueError):
|
|
895
|
+
return None
|
|
896
|
+
|
|
897
|
+
|
|
898
|
+
def _key(value: Any) -> tuple[str, Any]:
|
|
899
|
+
if isinstance(value, str):
|
|
900
|
+
return "str", value
|
|
901
|
+
if isinstance(value, bool):
|
|
902
|
+
return "bool", value
|
|
903
|
+
if isinstance(value, int | float):
|
|
904
|
+
return "number", value
|
|
905
|
+
return "json", json.dumps(value, sort_keys=True)
|
|
906
|
+
|
|
907
|
+
|
|
908
|
+
def _ordered(value: Any) -> Any:
|
|
909
|
+
if value == UNBOUNDED:
|
|
910
|
+
return math.inf
|
|
911
|
+
if isinstance(value, date):
|
|
912
|
+
return value.isoformat()
|
|
913
|
+
return value
|
|
914
|
+
|
|
915
|
+
|
|
916
|
+
def _holds(value: Any, op: str, arg: Any) -> bool:
|
|
917
|
+
"""Whether a known value satisfies ``op arg``."""
|
|
918
|
+
if op in ("=", "=="):
|
|
919
|
+
if isinstance(value, list):
|
|
920
|
+
return isinstance(arg, (list, tuple, set, frozenset)) and sorted(value) == sorted(arg)
|
|
921
|
+
return _ordered(value) == _ordered(arg)
|
|
922
|
+
if op == "!=":
|
|
923
|
+
return not _holds(value, "=", arg)
|
|
924
|
+
if op in ("in", "not_in"):
|
|
925
|
+
members = {_key(_ordered(a)) for a in arg}
|
|
926
|
+
found = (any(_key(v) in members for v in value) if isinstance(value, list)
|
|
927
|
+
else _key(_ordered(value)) in members)
|
|
928
|
+
return found if op == "in" else not found
|
|
929
|
+
if op == "contains":
|
|
930
|
+
return isinstance(value, list) and arg in value
|
|
931
|
+
if op == "contains_all":
|
|
932
|
+
return isinstance(value, list) and set(arg) <= set(value)
|
|
933
|
+
if op == "contains_any":
|
|
934
|
+
return isinstance(value, list) and bool(set(arg) & set(value))
|
|
935
|
+
if op in ("<", "<=", ">", ">=", "between"):
|
|
936
|
+
if value == NOT_OFFERED or isinstance(value, (list, bool)):
|
|
937
|
+
return False
|
|
938
|
+
v = _ordered(value)
|
|
939
|
+
try:
|
|
940
|
+
if op == "between":
|
|
941
|
+
low, high = arg
|
|
942
|
+
return _ordered(low) <= v <= _ordered(high)
|
|
943
|
+
a = _ordered(arg)
|
|
944
|
+
return {"<": v < a, "<=": v <= a, ">": v > a, ">=": v >= a}[op]
|
|
945
|
+
except TypeError as exc:
|
|
946
|
+
raise SnapshotError(f"cannot compare {value!r} {op} {arg!r}") from exc
|
|
947
|
+
raise SnapshotError(f"unknown operator {op!r}")
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
def _equality_key(value: Any) -> tuple[str, Any]:
|
|
951
|
+
ordered = _ordered(value)
|
|
952
|
+
if isinstance(ordered, list):
|
|
953
|
+
return "list", tuple(sorted(_key(_ordered(member)) for member in ordered))
|
|
954
|
+
return "scalar", _key(ordered)
|
|
955
|
+
|
|
956
|
+
|
|
957
|
+
class _FacetBitsets:
|
|
958
|
+
"""Bitsets for one facet, built once from its known values."""
|
|
959
|
+
|
|
960
|
+
def __init__(self, rows: Iterable[tuple[int, Any]], *, alternatives: bool = False):
|
|
961
|
+
self.alternatives = alternatives
|
|
962
|
+
self.known = 0
|
|
963
|
+
self.collections = 0
|
|
964
|
+
self.exact: dict[tuple[str, Any], int] = {}
|
|
965
|
+
self.members: dict[tuple[str, Any], int] = {}
|
|
966
|
+
self.contains: dict[tuple[str, Any], int] = {}
|
|
967
|
+
ordered: dict[Any, int] = {}
|
|
968
|
+
self._ordered_usable = True
|
|
969
|
+
for row, raw in rows:
|
|
970
|
+
bit = 1 << row
|
|
971
|
+
self.known |= bit
|
|
972
|
+
values = raw if alternatives else (raw,)
|
|
973
|
+
for value in values:
|
|
974
|
+
key = _equality_key(value)
|
|
975
|
+
self.exact[key] = self.exact.get(key, 0) | bit
|
|
976
|
+
members = value if isinstance(value, list) and not alternatives else (value,)
|
|
977
|
+
if isinstance(value, list) and not alternatives:
|
|
978
|
+
self.collections |= bit
|
|
979
|
+
for member in members:
|
|
980
|
+
member_key = _key(_ordered(member))
|
|
981
|
+
self.members[member_key] = self.members.get(member_key, 0) | bit
|
|
982
|
+
if isinstance(value, list) and not alternatives:
|
|
983
|
+
self.contains[member_key] = self.contains.get(member_key, 0) | bit
|
|
984
|
+
if value == NOT_OFFERED or isinstance(value, (list, bool)):
|
|
985
|
+
continue
|
|
986
|
+
try:
|
|
987
|
+
ordered_value = _ordered(value)
|
|
988
|
+
ordered[ordered_value] = ordered.get(ordered_value, 0) | bit
|
|
989
|
+
except (TypeError, ValueError):
|
|
990
|
+
self._ordered_usable = False
|
|
991
|
+
try:
|
|
992
|
+
self.ordered_values = tuple(sorted(ordered))
|
|
993
|
+
except TypeError:
|
|
994
|
+
self._ordered_usable = False
|
|
995
|
+
self.ordered_values = ()
|
|
996
|
+
prefixes = [0]
|
|
997
|
+
for value in self.ordered_values:
|
|
998
|
+
prefixes.append(prefixes[-1] | ordered[value])
|
|
999
|
+
self.prefixes = tuple(prefixes)
|
|
1000
|
+
|
|
1001
|
+
def _ordered(self, op: str, arg: Any) -> int | None:
|
|
1002
|
+
if not self._ordered_usable:
|
|
1003
|
+
return None
|
|
1004
|
+
try:
|
|
1005
|
+
if op == "between":
|
|
1006
|
+
low, high = (_ordered(value) for value in arg)
|
|
1007
|
+
left = bisect_left(self.ordered_values, low)
|
|
1008
|
+
right = bisect_right(self.ordered_values, high)
|
|
1009
|
+
return 0 if left > right else self.prefixes[right] & ~self.prefixes[left]
|
|
1010
|
+
value = _ordered(arg)
|
|
1011
|
+
if op == "<":
|
|
1012
|
+
return self.prefixes[bisect_left(self.ordered_values, value)]
|
|
1013
|
+
if op == "<=":
|
|
1014
|
+
return self.prefixes[bisect_right(self.ordered_values, value)]
|
|
1015
|
+
if op == ">":
|
|
1016
|
+
return self.prefixes[-1] & ~self.prefixes[bisect_right(self.ordered_values, value)]
|
|
1017
|
+
if op == ">=":
|
|
1018
|
+
return self.prefixes[-1] & ~self.prefixes[bisect_left(self.ordered_values, value)]
|
|
1019
|
+
except (TypeError, ValueError):
|
|
1020
|
+
return None
|
|
1021
|
+
return None
|
|
1022
|
+
|
|
1023
|
+
def passing(self, op: str, arg: Any) -> int | None:
|
|
1024
|
+
if op in ("=", "=="):
|
|
1025
|
+
return self.exact.get(_equality_key(arg), 0)
|
|
1026
|
+
if op == "!=":
|
|
1027
|
+
key = _equality_key(arg)
|
|
1028
|
+
if self.alternatives:
|
|
1029
|
+
passing = 0
|
|
1030
|
+
for value_key, bits in self.exact.items():
|
|
1031
|
+
if value_key != key:
|
|
1032
|
+
passing |= bits
|
|
1033
|
+
return passing
|
|
1034
|
+
return self.known & ~self.exact.get(key, 0)
|
|
1035
|
+
if op in ("in", "not_in"):
|
|
1036
|
+
hit = 0
|
|
1037
|
+
for value in arg:
|
|
1038
|
+
hit |= self.members.get(_key(_ordered(value)), 0)
|
|
1039
|
+
return self.known & (hit if op == "in" else ~hit)
|
|
1040
|
+
if op == "contains":
|
|
1041
|
+
return self.contains.get(_key(_ordered(arg)), 0)
|
|
1042
|
+
if op in ("contains_all", "contains_any"):
|
|
1043
|
+
values = tuple(arg)
|
|
1044
|
+
if op == "contains_all":
|
|
1045
|
+
if not values:
|
|
1046
|
+
return self.collections
|
|
1047
|
+
passing = self.known
|
|
1048
|
+
for value in values:
|
|
1049
|
+
passing &= self.contains.get(_key(_ordered(value)), 0)
|
|
1050
|
+
return passing
|
|
1051
|
+
passing = 0
|
|
1052
|
+
for value in values:
|
|
1053
|
+
passing |= self.contains.get(_key(_ordered(value)), 0)
|
|
1054
|
+
return passing
|
|
1055
|
+
if op in ("<", "<=", ">", ">=", "between"):
|
|
1056
|
+
return self._ordered(op, arg)
|
|
1057
|
+
return None
|
|
1058
|
+
|
|
1059
|
+
|
|
1060
|
+
class _Evidence(dict[str, tuple[EvidenceValue, ...]]):
|
|
1061
|
+
"""Materialise one candidate's evidence on first access.
|
|
1062
|
+
|
|
1063
|
+
An offering answers its model's evidence: capability belongs to the model,
|
|
1064
|
+
and an offering is that model as one provider sells it. The offering's own
|
|
1065
|
+
measurements of a benchmark, when it has any, replace its model's for that
|
|
1066
|
+
benchmark. The stored snapshot keeps evidence under its subject only.
|
|
1067
|
+
"""
|
|
1068
|
+
|
|
1069
|
+
def __init__(self, rows: Mapping[str, Sequence[Sequence[Any]]],
|
|
1070
|
+
model_of: Mapping[str, str]):
|
|
1071
|
+
super().__init__()
|
|
1072
|
+
self.rows = rows
|
|
1073
|
+
self.model_of = model_of
|
|
1074
|
+
self.records = {
|
|
1075
|
+
(model_of.get(cid, cid), row[10]): row
|
|
1076
|
+
for cid, candidate_rows in rows.items()
|
|
1077
|
+
for row in candidate_rows
|
|
1078
|
+
if len(row) > 10 and row[10] is not None
|
|
1079
|
+
}
|
|
1080
|
+
|
|
1081
|
+
@staticmethod
|
|
1082
|
+
def _value(row: Sequence[Any]) -> EvidenceValue:
|
|
1083
|
+
return EvidenceValue(
|
|
1084
|
+
benchmark_id=row[0], version=row[1], subcategory=row[2], value=row[3],
|
|
1085
|
+
unit=row[4], measured_by=row[5], effort=row[6], harness=row[7],
|
|
1086
|
+
date=_date(row[8]), source_ids=tuple(row[9]),
|
|
1087
|
+
record_id=row[10] if len(row) > 10 else None,
|
|
1088
|
+
date_type=row[11] if len(row) > 11 else None,
|
|
1089
|
+
source_snapshot=row[12] if len(row) > 12 else None,
|
|
1090
|
+
)
|
|
1091
|
+
|
|
1092
|
+
def record(self, cid: str, record_id: str) -> EvidenceValue | None:
|
|
1093
|
+
row = self.records.get((self.model_of.get(cid, cid), record_id))
|
|
1094
|
+
return None if row is None else self._value(row)
|
|
1095
|
+
|
|
1096
|
+
def _rows(self, cid: str) -> list[Sequence[Any]]:
|
|
1097
|
+
own = list(self.rows.get(cid, ()))
|
|
1098
|
+
model = self.model_of.get(cid, cid)
|
|
1099
|
+
if model == cid:
|
|
1100
|
+
return own
|
|
1101
|
+
measured = {r[0] for r in own}
|
|
1102
|
+
inherited = [r for r in self.rows.get(model, ()) if r[0] not in measured]
|
|
1103
|
+
return sorted(own + inherited, key=lambda r: (r[0], r[8] or "", r[3], canonical_json(r)))
|
|
1104
|
+
|
|
1105
|
+
def __missing__(self, cid: str) -> tuple[EvidenceValue, ...]:
|
|
1106
|
+
self[cid] = tuple(self._value(row) for row in self._rows(cid))
|
|
1107
|
+
return self[cid]
|
|
1108
|
+
|
|
1109
|
+
|
|
1110
|
+
class LoadedSnapshot:
|
|
1111
|
+
"""The in-memory index over one snapshot. Implements ``SnapshotIndex``."""
|
|
1112
|
+
|
|
1113
|
+
def __init__(self, envelope: Mapping[str, Any], *, include_archive: bool,
|
|
1114
|
+
signature_verified: bool):
|
|
1115
|
+
content = envelope["content"]
|
|
1116
|
+
self.snapshot_id: str = envelope["snapshot_id"]
|
|
1117
|
+
self.content_hash: str = envelope["content_hash"]
|
|
1118
|
+
self.signature_verified = signature_verified
|
|
1119
|
+
self.as_of = _date(content.get("as_of"))
|
|
1120
|
+
self.excluded: dict[str, int] = dict(content.get("excluded") or {})
|
|
1121
|
+
#: Active models the build left out because they are not in the premier set.
|
|
1122
|
+
self.out_of_lineup: int = int(content.get("out_of_lineup") or 0)
|
|
1123
|
+
self._sources: dict[str, str] = dict(content["sources"])
|
|
1124
|
+
self._records = content.get("records", {})
|
|
1125
|
+
self._record_table = content.get("record_table")
|
|
1126
|
+
self.explanation_rebuild_required = (
|
|
1127
|
+
None
|
|
1128
|
+
if "fact_records" in content and ("record_table" in content or "records" in content)
|
|
1129
|
+
else "snapshot predates retained verification records; rebuild it before explaining"
|
|
1130
|
+
)
|
|
1131
|
+
sections = [content["lineup"]] + ([content["archive"]] if include_archive else [])
|
|
1132
|
+
|
|
1133
|
+
rows: list[tuple[dict[str, Any], dict[str, Any], int]] = []
|
|
1134
|
+
for section in sections:
|
|
1135
|
+
for i, cand in enumerate(section["candidates"]):
|
|
1136
|
+
rows.append((cand, section, i))
|
|
1137
|
+
rows.sort(key=lambda r: r[0]["id"])
|
|
1138
|
+
self._ids: tuple[str, ...] = tuple(r[0]["id"] for r in rows)
|
|
1139
|
+
self._row = {cid: i for i, cid in enumerate(self._ids)}
|
|
1140
|
+
self._meta = {r[0]["id"]: r[0] for r in rows}
|
|
1141
|
+
self._all = (1 << len(self._ids)) - 1
|
|
1142
|
+
|
|
1143
|
+
facts: dict[str, dict[int, FactValue]] = {}
|
|
1144
|
+
for section in sections:
|
|
1145
|
+
local = [c["id"] for c in section["candidates"]]
|
|
1146
|
+
for facet_id, col in section["facets"].items():
|
|
1147
|
+
target = facts.setdefault(facet_id, {})
|
|
1148
|
+
for r, state, value, srcs in zip(col["row"], col["state"], col["value"],
|
|
1149
|
+
col["sources"]):
|
|
1150
|
+
target[self._row[local[r]]] = FactValue(
|
|
1151
|
+
state, value, tuple(srcs),
|
|
1152
|
+
content.get("fact_records", {}).get(local[r], {}).get(facet_id))
|
|
1153
|
+
# An offering is its model as sold: it answers its model's facets.
|
|
1154
|
+
subjects = content.get("facet_subjects") or {}
|
|
1155
|
+
for facet_id, by_row in facts.items():
|
|
1156
|
+
if subjects.get(facet_id) != "model":
|
|
1157
|
+
continue
|
|
1158
|
+
for cid, meta in self._meta.items():
|
|
1159
|
+
if meta["kind"] == "offering":
|
|
1160
|
+
row, model_row = self._row[cid], self._row.get(meta["model"])
|
|
1161
|
+
if row not in by_row and model_row in by_row:
|
|
1162
|
+
by_row[row] = by_row[model_row]
|
|
1163
|
+
self._facts = facts
|
|
1164
|
+
|
|
1165
|
+
self._facet_bits = {
|
|
1166
|
+
facet_id: _FacetBitsets(
|
|
1167
|
+
(row, fv.value) for row, fv in by_row.items() if fv.state == "known"
|
|
1168
|
+
)
|
|
1169
|
+
for facet_id, by_row in facts.items()
|
|
1170
|
+
}
|
|
1171
|
+
|
|
1172
|
+
self._evidence = _Evidence({cid: rows for section in sections
|
|
1173
|
+
for cid, rows in section["evidence"].items()},
|
|
1174
|
+
{cid: meta["model"] for cid, meta in self._meta.items()})
|
|
1175
|
+
self._evidence_bits: dict[tuple[Any, ...], _FacetBitsets] = {}
|
|
1176
|
+
self._benchmarks = tuple(sorted(content["benchmark_domains"]))
|
|
1177
|
+
capability = content.get("capability") or {}
|
|
1178
|
+
# Parse the learned lookup once. Domain objectives are the page's
|
|
1179
|
+
# default, so reconstructing these values throughout filtering,
|
|
1180
|
+
# optimisation and explanation made the first Worker request pay the
|
|
1181
|
+
# same JSON-to-object cost repeatedly.
|
|
1182
|
+
self._capability_estimates = {
|
|
1183
|
+
(model_id, domain_id): CapabilityEstimateValue(*map(float, row))
|
|
1184
|
+
for model_id, domains in (capability.get("estimates") or {}).items()
|
|
1185
|
+
for domain_id, row in domains.items()
|
|
1186
|
+
}
|
|
1187
|
+
self._capability_drivers = {
|
|
1188
|
+
(model_id, domain_id): tuple(
|
|
1189
|
+
CapabilityDriverValue(
|
|
1190
|
+
record_id=row[0], benchmark_id=row[1], version=row[2],
|
|
1191
|
+
loading=float(row[3]), weight=float(row[4]),
|
|
1192
|
+
recency_weight=float(row[5]),
|
|
1193
|
+
)
|
|
1194
|
+
for row in rows
|
|
1195
|
+
)
|
|
1196
|
+
for model_id, domains in (capability.get("drivers") or {}).items()
|
|
1197
|
+
for domain_id, rows in domains.items()
|
|
1198
|
+
}
|
|
1199
|
+
self.capability_method = capability.get("method")
|
|
1200
|
+
self.capability_items = capability.get("items") or {}
|
|
1201
|
+
self.capability_source_offsets = capability.get("source_offsets") or {}
|
|
1202
|
+
self._domains: dict[str, list[tuple[str, str]]] = {}
|
|
1203
|
+
for bench, tags in content["benchmark_domains"].items():
|
|
1204
|
+
for domain_id, directness in tags:
|
|
1205
|
+
self._domains.setdefault(domain_id, []).append((bench, directness))
|
|
1206
|
+
|
|
1207
|
+
# SnapshotIndex -----------------------------------------------------------
|
|
1208
|
+
|
|
1209
|
+
def candidates(self) -> Sequence[str]:
|
|
1210
|
+
return self._ids
|
|
1211
|
+
|
|
1212
|
+
def _check(self, cid: str) -> int:
|
|
1213
|
+
try:
|
|
1214
|
+
return self._row[cid]
|
|
1215
|
+
except KeyError:
|
|
1216
|
+
raise KeyError(f"{cid!r} is not a candidate in snapshot {self.snapshot_id}") from None
|
|
1217
|
+
|
|
1218
|
+
def lifecycle(self, cid: str) -> Lifecycle:
|
|
1219
|
+
self._check(cid)
|
|
1220
|
+
return self._meta[cid]["lifecycle"]
|
|
1221
|
+
|
|
1222
|
+
def fact(self, cid: str, facet_id: str) -> FactValue:
|
|
1223
|
+
return self._facts.get(facet_id, {}).get(self._check(cid), UNKNOWN)
|
|
1224
|
+
|
|
1225
|
+
def ids_where(self, facet_id: str, op: str, arg: Any) -> Bitset3:
|
|
1226
|
+
"""Candidates passing, failing, or unknown on ``facet op arg``.
|
|
1227
|
+
|
|
1228
|
+
Operators: ``=`` (or ``==``), ``!=``, ``<``, ``<=``, ``>``, ``>=``,
|
|
1229
|
+
``between`` (a (low, high) pair, inclusive), ``in``, ``not_in``,
|
|
1230
|
+
``contains``, ``contains_all``, ``contains_any`` (on set facets) and
|
|
1231
|
+
``known``. Any state but ``known`` is unknown; ``known`` itself is
|
|
1232
|
+
never unknown. ``unbounded`` exceeds every number; ``not_offered``
|
|
1233
|
+
fails every ordered comparison.
|
|
1234
|
+
"""
|
|
1235
|
+
column = self._facet_bits.get(facet_id)
|
|
1236
|
+
known = 0 if column is None else column.known
|
|
1237
|
+
if op == "known":
|
|
1238
|
+
return Bitset3(known, self._all & ~known, 0)
|
|
1239
|
+
passing = None if column is None else column.passing(op, arg)
|
|
1240
|
+
if passing is None:
|
|
1241
|
+
passing = 0
|
|
1242
|
+
by_row = self._facts.get(facet_id, {})
|
|
1243
|
+
for row in self._rows(known):
|
|
1244
|
+
if _holds(by_row[row].value, op, arg):
|
|
1245
|
+
passing |= 1 << row
|
|
1246
|
+
return Bitset3(passing, known & ~passing, self._all & ~known)
|
|
1247
|
+
|
|
1248
|
+
def evidence_where(
|
|
1249
|
+
self,
|
|
1250
|
+
benchmark_id: str,
|
|
1251
|
+
op: str,
|
|
1252
|
+
arg: Any,
|
|
1253
|
+
*,
|
|
1254
|
+
measured_by: set[str] | None = None,
|
|
1255
|
+
effort: str | None = None,
|
|
1256
|
+
harness: str | None = None,
|
|
1257
|
+
after: date | None = None,
|
|
1258
|
+
direct: bool = False,
|
|
1259
|
+
domains: Iterable[str] = (),
|
|
1260
|
+
) -> Bitset3:
|
|
1261
|
+
"""Three-valued evidence condition, indexed lazily per qualifier set.
|
|
1262
|
+
|
|
1263
|
+
``direct`` admits the benchmark only when it is direct for one of
|
|
1264
|
+
``domains``, the capabilities asked about (see ``direct_for``).
|
|
1265
|
+
"""
|
|
1266
|
+
admits = not direct or self.direct_for(benchmark_id, domains)
|
|
1267
|
+
key = (
|
|
1268
|
+
benchmark_id,
|
|
1269
|
+
None if measured_by is None else frozenset(measured_by),
|
|
1270
|
+
effort,
|
|
1271
|
+
harness,
|
|
1272
|
+
after,
|
|
1273
|
+
admits,
|
|
1274
|
+
)
|
|
1275
|
+
column = self._evidence_bits.get(key)
|
|
1276
|
+
if column is None:
|
|
1277
|
+
admitted = []
|
|
1278
|
+
for row, cid in enumerate(self._ids if admits else ()):
|
|
1279
|
+
values = tuple(
|
|
1280
|
+
evidence.value
|
|
1281
|
+
for evidence in self.evidence(
|
|
1282
|
+
cid,
|
|
1283
|
+
benchmark_id,
|
|
1284
|
+
measured_by=measured_by,
|
|
1285
|
+
effort=effort,
|
|
1286
|
+
harness=harness,
|
|
1287
|
+
after=after,
|
|
1288
|
+
)
|
|
1289
|
+
)
|
|
1290
|
+
if values:
|
|
1291
|
+
admitted.append((row, values))
|
|
1292
|
+
column = _FacetBitsets(admitted, alternatives=True)
|
|
1293
|
+
self._evidence_bits[key] = column
|
|
1294
|
+
passing = column.passing(op, arg)
|
|
1295
|
+
if passing is None:
|
|
1296
|
+
passing = 0
|
|
1297
|
+
for row, cid in enumerate(self._ids if admits else ()):
|
|
1298
|
+
values = self.evidence(
|
|
1299
|
+
cid,
|
|
1300
|
+
benchmark_id,
|
|
1301
|
+
measured_by=measured_by,
|
|
1302
|
+
effort=effort,
|
|
1303
|
+
harness=harness,
|
|
1304
|
+
after=after,
|
|
1305
|
+
)
|
|
1306
|
+
if any(_holds(value.value, op, arg) for value in values):
|
|
1307
|
+
passing |= 1 << row
|
|
1308
|
+
return Bitset3(passing, column.known & ~passing, self._all & ~column.known)
|
|
1309
|
+
|
|
1310
|
+
def evidence(self, cid: str, benchmark_id: str, *, measured_by: set[str] | None = None,
|
|
1311
|
+
effort: str | None = None, harness: str | None = None,
|
|
1312
|
+
after: date | None = None) -> Sequence[EvidenceValue]:
|
|
1313
|
+
"""Evidence for one benchmark; ``after`` is exclusive. Unknown dates never pass it."""
|
|
1314
|
+
self._check(cid)
|
|
1315
|
+
return tuple(
|
|
1316
|
+
e for e in self._evidence[cid]
|
|
1317
|
+
if e.benchmark_id == benchmark_id
|
|
1318
|
+
and (measured_by is None or e.measured_by in measured_by)
|
|
1319
|
+
and (effort is None or e.effort == effort)
|
|
1320
|
+
and (harness is None or e.harness == harness)
|
|
1321
|
+
and (after is None or (e.date is not None and e.date > after)))
|
|
1322
|
+
|
|
1323
|
+
def direct_for(self, benchmark_id: str, domains: Iterable[str] = ()) -> bool:
|
|
1324
|
+
"""Whether ``benchmark_id`` directly measures a capability asked about.
|
|
1325
|
+
|
|
1326
|
+
Directness is relative to the request (design §4.1): the benchmark must
|
|
1327
|
+
be tagged ``direct`` for one of ``domains``. With no domain asked
|
|
1328
|
+
about, a ``direct`` tag for any domain is enough.
|
|
1329
|
+
"""
|
|
1330
|
+
return any((benchmark_id, "direct") in self._domains.get(domain_id, ())
|
|
1331
|
+
for domain_id in (set(domains) or self._domains))
|
|
1332
|
+
|
|
1333
|
+
def evidence_for_domain(self, cid: str, domain_id: str) -> Sequence[EvidenceValue]:
|
|
1334
|
+
self._check(cid)
|
|
1335
|
+
directness = dict(self._domains.get(domain_id, ()))
|
|
1336
|
+
return tuple(
|
|
1337
|
+
EvidenceValue(**{**e.__dict__, "directness": directness[e.benchmark_id]})
|
|
1338
|
+
for e in self._evidence[cid] if e.benchmark_id in directness)
|
|
1339
|
+
|
|
1340
|
+
def capability_estimate(
|
|
1341
|
+
self, cid: str, domain_id: str
|
|
1342
|
+
) -> CapabilityEstimateValue | None:
|
|
1343
|
+
self._check(cid)
|
|
1344
|
+
model_id = self._meta[cid]["model"]
|
|
1345
|
+
return self._capability_estimates.get((model_id, domain_id))
|
|
1346
|
+
|
|
1347
|
+
def capability_drivers(
|
|
1348
|
+
self, cid: str, domain_id: str
|
|
1349
|
+
) -> Sequence[CapabilityDriverValue]:
|
|
1350
|
+
self._check(cid)
|
|
1351
|
+
model_id = self._meta[cid]["model"]
|
|
1352
|
+
return self._capability_drivers.get((model_id, domain_id), ())
|
|
1353
|
+
|
|
1354
|
+
def evidence_record(self, cid: str, record_id: str) -> EvidenceValue | None:
|
|
1355
|
+
"""Return retained model evidence by record ID in constant time."""
|
|
1356
|
+
self._check(cid)
|
|
1357
|
+
return self._evidence.record(self._meta[cid]["model"], record_id)
|
|
1358
|
+
|
|
1359
|
+
# beyond the protocol -----------------------------------------------------
|
|
1360
|
+
|
|
1361
|
+
def require_explanation_records(self) -> None:
|
|
1362
|
+
"""Refuse explanations from a snapshot built before provenance retention."""
|
|
1363
|
+
if self.explanation_rebuild_required is not None:
|
|
1364
|
+
raise SnapshotError(self.explanation_rebuild_required)
|
|
1365
|
+
|
|
1366
|
+
def kind(self, cid: str) -> Literal["model", "offering"]:
|
|
1367
|
+
self._check(cid)
|
|
1368
|
+
return self._meta[cid]["kind"]
|
|
1369
|
+
|
|
1370
|
+
def model_of(self, cid: str) -> str:
|
|
1371
|
+
self._check(cid)
|
|
1372
|
+
return self._meta[cid]["model"]
|
|
1373
|
+
|
|
1374
|
+
def record(self, record_id: str) -> Mapping[str, Any]:
|
|
1375
|
+
"""The admitted record with its winning verification, retained verbatim."""
|
|
1376
|
+
if record_id not in self._records and self._record_table is not None:
|
|
1377
|
+
self._records[record_id] = _unpack_record(self._record_table, record_id)
|
|
1378
|
+
return self._records[record_id]
|
|
1379
|
+
|
|
1380
|
+
def facet_ids(self) -> tuple[str, ...]:
|
|
1381
|
+
return tuple(sorted(self._facts))
|
|
1382
|
+
|
|
1383
|
+
def domain_ids(self) -> tuple[str, ...]:
|
|
1384
|
+
return tuple(sorted(self._domains))
|
|
1385
|
+
|
|
1386
|
+
def benchmark_ids(self) -> tuple[str, ...]:
|
|
1387
|
+
return self._benchmarks
|
|
1388
|
+
|
|
1389
|
+
def benchmark_domain_tags(self) -> dict[str, tuple[tuple[str, str], ...]]:
|
|
1390
|
+
"""Each benchmark's (domain, directness) tags, sorted by domain."""
|
|
1391
|
+
tags: dict[str, list[tuple[str, str]]] = {b: [] for b in self._benchmarks}
|
|
1392
|
+
for domain_id, rows in self._domains.items():
|
|
1393
|
+
for bench, directness in rows:
|
|
1394
|
+
tags.setdefault(bench, []).append((domain_id, directness))
|
|
1395
|
+
return {b: tuple(sorted(t)) for b, t in tags.items()}
|
|
1396
|
+
|
|
1397
|
+
def source_url(self, source_id: str) -> str:
|
|
1398
|
+
return self._sources[source_id]
|
|
1399
|
+
|
|
1400
|
+
def ids(self, bits: int) -> tuple[str, ...]:
|
|
1401
|
+
"""The candidate IDs a bitset names."""
|
|
1402
|
+
return tuple(self._ids[r] for r in self._rows(bits))
|
|
1403
|
+
|
|
1404
|
+
@staticmethod
|
|
1405
|
+
def _rows(bits: int) -> Iterable[int]:
|
|
1406
|
+
row = 0
|
|
1407
|
+
while bits:
|
|
1408
|
+
if bits & 1:
|
|
1409
|
+
yield row
|
|
1410
|
+
bits >>= 1
|
|
1411
|
+
row += 1
|
|
1412
|
+
|
|
1413
|
+
|
|
1414
|
+
def load_snapshot_bytes(
|
|
1415
|
+
data: bytes,
|
|
1416
|
+
*,
|
|
1417
|
+
key: bytes | str | None = _FROM_ENV,
|
|
1418
|
+
include_archive: bool = False,
|
|
1419
|
+
source: str = "snapshot bytes",
|
|
1420
|
+
) -> LoadedSnapshot:
|
|
1421
|
+
"""Check and index a gzipped snapshot already held in memory."""
|
|
1422
|
+
try:
|
|
1423
|
+
text = gzip.decompress(data).decode("utf-8")
|
|
1424
|
+
stored_digest = None
|
|
1425
|
+
if text.startswith('{"content":{'):
|
|
1426
|
+
# raw_decode finds the JSON boundary, including escaped quotes and
|
|
1427
|
+
# nested objects. Never search for a delimiter inside content.
|
|
1428
|
+
content, end = json.JSONDecoder().raw_decode(text, len('{"content":'))
|
|
1429
|
+
suffix = text[end:].lstrip()
|
|
1430
|
+
envelope = json.loads("{" + suffix[1:]) if suffix.startswith(",") else {}
|
|
1431
|
+
if "content" in envelope:
|
|
1432
|
+
raise SnapshotIntegrityError(f"{source}: duplicate content member")
|
|
1433
|
+
envelope["content"] = content
|
|
1434
|
+
stored_digest = "sha256:" + hashlib.sha256(
|
|
1435
|
+
text[len('{"content":'):end].encode("utf-8")).hexdigest()
|
|
1436
|
+
else:
|
|
1437
|
+
envelope = json.loads(text)
|
|
1438
|
+
except (OSError, EOFError, ValueError) as exc:
|
|
1439
|
+
raise SnapshotIntegrityError(f"{source}: not a gzipped JSON snapshot: {exc}") from exc
|
|
1440
|
+
if not isinstance(envelope, dict) or envelope.get("format") != FORMAT:
|
|
1441
|
+
raise SnapshotIntegrityError(f"{source}: not a {FORMAT} file")
|
|
1442
|
+
if envelope.get("format_version") != FORMAT_VERSION:
|
|
1443
|
+
raise SnapshotIntegrityError(
|
|
1444
|
+
f"{source}: format version {envelope.get('format_version')!r}, "
|
|
1445
|
+
f"expected {FORMAT_VERSION}"
|
|
1446
|
+
)
|
|
1447
|
+
# Canonical writer output can be checked directly. Other JSON encodings
|
|
1448
|
+
# retain the original semantic hash check, including the signature check.
|
|
1449
|
+
digest = stored_digest
|
|
1450
|
+
if digest != envelope.get("content_hash") or digest is None:
|
|
1451
|
+
digest = content_hash(envelope["content"])
|
|
1452
|
+
if digest != envelope.get("content_hash"):
|
|
1453
|
+
raise SnapshotIntegrityError(f"{source}: content hash mismatch: the snapshot was altered")
|
|
1454
|
+
if envelope.get("snapshot_id") != snapshot_id_for(digest):
|
|
1455
|
+
raise SnapshotIntegrityError(f"{source}: snapshot ID does not match its content hash")
|
|
1456
|
+
key = env_key() if key is _FROM_ENV else _key_bytes(key)
|
|
1457
|
+
verified = False
|
|
1458
|
+
if key is not None:
|
|
1459
|
+
signature = envelope.get("signature")
|
|
1460
|
+
if not signature:
|
|
1461
|
+
raise SnapshotIntegrityError(f"{source}: unsigned snapshot, but a key was given")
|
|
1462
|
+
if (signature.get("alg") != SIGNATURE_ALG
|
|
1463
|
+
or not hmac.compare_digest(str(signature.get("value")), _sign(digest, key))):
|
|
1464
|
+
raise SnapshotIntegrityError(f"{source}: signature does not verify with this key")
|
|
1465
|
+
verified = True
|
|
1466
|
+
return LoadedSnapshot(envelope, include_archive=include_archive, signature_verified=verified)
|
|
1467
|
+
|
|
1468
|
+
|
|
1469
|
+
def load_snapshot(path: str | Path, *, key: bytes | str | None = _FROM_ENV,
|
|
1470
|
+
include_archive: bool = False) -> LoadedSnapshot:
|
|
1471
|
+
"""Read, check and index a snapshot.
|
|
1472
|
+
|
|
1473
|
+
The content hash is always checked. The key defaults to
|
|
1474
|
+
``MODELSPEC_SNAPSHOT_KEY``; with a key, the snapshot must be signed with it.
|
|
1475
|
+
Without one, the signature cannot be checked and ``signature_verified`` is
|
|
1476
|
+
false. Retired models are left out unless ``include_archive``.
|
|
1477
|
+
"""
|
|
1478
|
+
path = Path(path)
|
|
1479
|
+
try:
|
|
1480
|
+
data = path.read_bytes()
|
|
1481
|
+
except OSError as exc:
|
|
1482
|
+
raise SnapshotIntegrityError(f"{path}: not a gzipped JSON snapshot: {exc}") from exc
|
|
1483
|
+
return load_snapshot_bytes(data, key=key, include_archive=include_archive, source=str(path))
|