modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
decision/snapshot.py ADDED
@@ -0,0 +1,1483 @@
1
+ """The snapshot: verified facts and evidence, compiled, hashed and signed (MODEL-138).
2
+
3
+ A decision reads one snapshot and nothing else (design §4.2, §6). This module
4
+ builds it, gates it and loads it:
5
+
6
+ * ``build_snapshot`` compiles models, offerings and evidence into a columnar
7
+ snapshot. Only values whose latest verification is ``verified`` enter, and
8
+ only when every source they name resolves to a registered URL that is not an
9
+ excluded source. A ``verified`` from the collector's own model family is not
10
+ a second key (MODEL-159): it is skipped as if never logged, while a
11
+ same-family mismatch still counts. Retired models and their offerings go to
12
+ a separate ``archive`` section. With a premier list, the ``lineup`` holds only the
13
+ premier models and their offerings; other active models are counted in
14
+ ``out_of_lineup`` and left out (MODEL-157). The legacy flat
15
+ ``benchmarks.scores`` block is never read.
16
+ * The **completeness gate** fails the build when a guaranteed facet is unknown
17
+ or unverified for a premier model or one of its offerings, naming the
18
+ subject, the facet and the source. Computed facets (``computed_by`` in the
19
+ registry, such as ``estimate.capability``) are skipped: slice 1 does not
20
+ compute them.
21
+ * ``Snapshot.write`` serialises canonical JSON, gzips it with a fixed header,
22
+ and signs the content hash with HMAC-SHA256 when ``MODELSPEC_SNAPSHOT_KEY`` is
23
+ set. The same inputs give the same bytes.
24
+ * ``load_snapshot`` checks the hash and, when a key is available, the
25
+ signature, then builds the in-memory index: three-valued bitsets over the
26
+ candidates, per facet value.
27
+
28
+ Inputs are MODEL-134's records (``decision.model``) or their serialised dicts;
29
+ the builder reads them by field name, so either works.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import gzip
35
+ import hashlib
36
+ import hmac
37
+ import io
38
+ import json
39
+ import math
40
+ import os
41
+ from bisect import bisect_left, bisect_right
42
+ from collections import Counter
43
+ from collections.abc import Iterable, Mapping, Sequence
44
+ from dataclasses import dataclass, field
45
+ from datetime import date
46
+ from pathlib import Path
47
+ from typing import Any, Literal, Protocol, runtime_checkable
48
+
49
+ import yaml
50
+
51
+ from decision.excluded import ExcludedSources, excluded_sources
52
+ from decision.model import value_hash, verification_counts
53
+
54
+ FORMAT = "modelspec.decision-snapshot"
55
+ FORMAT_VERSION = 1
56
+ KEY_ENV = "MODELSPEC_SNAPSHOT_KEY"
57
+ SIGNATURE_ALG = "hmac-sha256"
58
+
59
+ FactState = Literal["known", "unknown", "not_disclosed", "requires_contract"]
60
+ Lifecycle = Literal["active", "deprecated", "retired"]
61
+ Directness = Literal["direct", "proxy"]
62
+ FACT_STATES = ("known", "unknown", "not_disclosed", "requires_contract")
63
+ LIFECYCLES = ("active", "deprecated", "retired")
64
+
65
+ #: A number facet may hold these literals instead of a number (MODEL-133).
66
+ UNBOUNDED = "unbounded"
67
+ NOT_OFFERED = "not_offered"
68
+
69
+
70
+ # ── errors ─────────────────────────────────────────────────────────────────
71
+
72
+
73
+ class SnapshotError(ValueError):
74
+ """A snapshot could not be built or read."""
75
+
76
+
77
+ class SnapshotBuildError(SnapshotError):
78
+ """The inputs cannot make a snapshot."""
79
+
80
+
81
+ class SnapshotIntegrityError(SnapshotError):
82
+ """A snapshot file failed its format, hash or signature check."""
83
+
84
+
85
+ @dataclass(frozen=True)
86
+ class Gap:
87
+ """One guaranteed facet a premier subject lacks."""
88
+
89
+ model: str
90
+ subject: str
91
+ facet: str
92
+ reason: str
93
+ sources: tuple[str, ...] = ()
94
+
95
+ def __str__(self) -> str:
96
+ where = ", ".join(self.sources) if self.sources else "no source recorded"
97
+ return f"{self.subject}: {self.facet} is {self.reason}; source: {where}"
98
+
99
+
100
+ class CompletenessError(SnapshotBuildError):
101
+ """The premier-set completeness gate failed (design §5)."""
102
+
103
+ def __init__(self, gaps: Sequence[Gap]):
104
+ self.gaps = tuple(gaps)
105
+ lines = "\n ".join(str(g) for g in self.gaps)
106
+ super().__init__(f"completeness gate: {len(self.gaps)} guaranteed fact(s) missing "
107
+ f"for the premier set:\n {lines}")
108
+
109
+
110
+ # ── the values an index returns ────────────────────────────────────────────
111
+
112
+
113
+ @dataclass(frozen=True)
114
+ class FactValue:
115
+ state: FactState
116
+ value: Any = None
117
+ sources: tuple[str, ...] = ()
118
+ record_id: str | None = field(default=None, compare=False)
119
+
120
+
121
+ UNKNOWN = FactValue("unknown")
122
+
123
+
124
+ @dataclass(frozen=True)
125
+ class EvidenceValue:
126
+ benchmark_id: str
127
+ version: str | None
128
+ subcategory: str | None
129
+ value: float
130
+ unit: str | None
131
+ measured_by: str | None
132
+ effort: str | None
133
+ harness: str | None
134
+ #: The evidence date; ``None`` when the source gave less than a full date.
135
+ date: date | None
136
+ source_ids: tuple[str, ...]
137
+ verified: bool = True
138
+ #: Set by ``evidence_for_domain`` only: how directly the benchmark measures
139
+ #: the domain asked about. An addition to the agreed field list.
140
+ directness: Directness | None = None
141
+ record_id: str | None = field(default=None, compare=False)
142
+ date_type: str | None = None
143
+ source_snapshot: str | None = None
144
+
145
+
146
+ @dataclass(frozen=True)
147
+ class CapabilityEstimateValue:
148
+ value: float
149
+ low: float
150
+ high: float
151
+ sd: float
152
+
153
+
154
+ @dataclass(frozen=True)
155
+ class CapabilityDriverValue:
156
+ record_id: str
157
+ benchmark_id: str
158
+ version: str | None
159
+ loading: float
160
+ weight: float
161
+ recency_weight: float
162
+
163
+
164
+ @dataclass(frozen=True)
165
+ class Bitset3:
166
+ """Three disjoint bitsets over ``candidates()``: bit ``i`` is candidate ``i``."""
167
+
168
+ passing: int
169
+ failing: int
170
+ unknown: int
171
+
172
+ def __post_init__(self) -> None:
173
+ if self.passing & self.failing or self.passing & self.unknown or self.failing & self.unknown:
174
+ raise ValueError("a Bitset3's three sets must be disjoint")
175
+
176
+
177
+ @runtime_checkable
178
+ class SnapshotIndex(Protocol):
179
+ snapshot_id: str
180
+
181
+ def candidates(self) -> Sequence[str]: ...
182
+
183
+ def lifecycle(self, cid: str) -> Lifecycle: ...
184
+
185
+ def fact(self, cid: str, facet_id: str) -> FactValue: ...
186
+
187
+ def ids_where(self, facet_id: str, op: str, arg: Any) -> Bitset3: ...
188
+
189
+ def evidence(self, cid: str, benchmark_id: str, *, measured_by: set[str] | None = None,
190
+ effort: str | None = None, harness: str | None = None,
191
+ after: date | None = None) -> Sequence[EvidenceValue]: ...
192
+
193
+ def evidence_where(
194
+ self,
195
+ benchmark_id: str,
196
+ op: str,
197
+ arg: Any,
198
+ *,
199
+ measured_by: set[str] | None = None,
200
+ effort: str | None = None,
201
+ harness: str | None = None,
202
+ after: date | None = None,
203
+ direct: bool = False,
204
+ domains: Iterable[str] = (),
205
+ ) -> Bitset3: ...
206
+
207
+ def direct_for(self, benchmark_id: str, domains: Iterable[str] = ()) -> bool: ...
208
+
209
+ def evidence_for_domain(self, cid: str, domain_id: str) -> Sequence[EvidenceValue]: ...
210
+
211
+ def capability_estimate(
212
+ self, cid: str, domain_id: str
213
+ ) -> CapabilityEstimateValue | None: ...
214
+
215
+ def capability_drivers(
216
+ self, cid: str, domain_id: str
217
+ ) -> Sequence[CapabilityDriverValue]: ...
218
+
219
+ def evidence_record(self, cid: str, record_id: str) -> EvidenceValue | None: ...
220
+
221
+ def kind(self, cid: str) -> Literal["model", "offering"]: ...
222
+
223
+ def model_of(self, cid: str) -> str: ...
224
+
225
+
226
+ @runtime_checkable
227
+ class ExplanationIndex(SnapshotIndex, Protocol):
228
+ """Snapshot metadata and retained records needed to transport a decision."""
229
+
230
+ def require_explanation_records(self) -> None: ...
231
+ def source_url(self, source_id: str) -> str: ...
232
+ def record(self, record_id: str) -> Mapping[str, Any]: ...
233
+ def facet_ids(self) -> tuple[str, ...]: ...
234
+ def domain_ids(self) -> tuple[str, ...]: ...
235
+ def benchmark_ids(self) -> tuple[str, ...]: ...
236
+ def benchmark_domain_tags(self) -> dict[str, tuple[tuple[str, str], ...]]: ...
237
+
238
+
239
+ # ── inputs ─────────────────────────────────────────────────────────────────
240
+
241
+
242
+ @dataclass(frozen=True)
243
+ class SnapshotInputs:
244
+ """What a snapshot is compiled from. Records may be MODEL-134 objects or dicts."""
245
+
246
+ models: Sequence[Any] = ()
247
+ offerings: Sequence[Any] = ()
248
+ evidence: Sequence[Any] = ()
249
+ #: Registered source ID -> URL.
250
+ sources: Mapping[str, str] = field(default_factory=dict)
251
+ #: Benchmark ID -> ((domain ID, directness), ...), from the benchmark pages.
252
+ benchmark_domains: Mapping[str, Sequence[Sequence[str]]] = field(default_factory=dict)
253
+ #: Benchmark measurement metadata used by the build-time capability fit.
254
+ benchmark_metadata: Mapping[str, Mapping[str, Any]] = field(default_factory=dict)
255
+ #: The verification log. The latest verification of a target wins, over an
256
+ #: inline one too.
257
+ verifications: Sequence[Any] = ()
258
+
259
+
260
+ def _as_dict(record: Any) -> dict[str, Any]:
261
+ if isinstance(record, Mapping):
262
+ return dict(record)
263
+ if hasattr(record, "model_dump"):
264
+ return record.model_dump(mode="json", by_alias=True)
265
+ raise SnapshotBuildError(f"not a record: {record!r}")
266
+
267
+
268
+ def _counts(v: Mapping[str, Any]) -> bool:
269
+ """A same-family ``verified`` is not a second key: it neither admits nor displaces."""
270
+ return verification_counts(str(v["outcome"]), str(v["collector"]["model_family"]),
271
+ str(v["verifier"]["model_family"]))
272
+
273
+
274
+ def _offering_id(o: Mapping[str, Any]) -> str:
275
+ return f"{o['provider']}/{o['model']}/{o['region']}/{o['tier']}"
276
+
277
+
278
+ # ── canonical form, hash and signature ─────────────────────────────────────
279
+
280
+
281
+ def canonical_json(value: Any) -> bytes:
282
+ return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False,
283
+ allow_nan=False).encode("utf-8")
284
+
285
+
286
+ def content_hash(content: Mapping[str, Any]) -> str:
287
+ return "sha256:" + hashlib.sha256(canonical_json(content)).hexdigest()
288
+
289
+
290
+ def snapshot_id_for(digest: str) -> str:
291
+ return "snap_" + digest.removeprefix("sha256:")[:16]
292
+
293
+
294
+ def _key_bytes(key: bytes | str | None) -> bytes | None:
295
+ if key is None or key == "" or key == b"":
296
+ return None
297
+ return key.encode("utf-8") if isinstance(key, str) else key
298
+
299
+
300
+ def _sign(digest: str, key: bytes) -> str:
301
+ return hmac.new(key, digest.encode("ascii"), hashlib.sha256).hexdigest()
302
+
303
+
304
+ _FROM_ENV: Any = object()
305
+
306
+
307
+ def env_key() -> bytes | None:
308
+ """The signing key from ``MODELSPEC_SNAPSHOT_KEY``, or ``None``."""
309
+ return _key_bytes(os.environ.get(KEY_ENV))
310
+
311
+
312
+ def _record_fields(
313
+ record: Mapping[str, Any], prefix: tuple[str, ...] = (),
314
+ ) -> Iterable[tuple[tuple[str, ...], Any]]:
315
+ for key, value in record.items():
316
+ path = (*prefix, key)
317
+ if isinstance(value, dict) and value:
318
+ yield from _record_fields(value, path)
319
+ else:
320
+ yield path, value
321
+
322
+
323
+ def _pack_records(records: Mapping[str, Mapping[str, Any]]) -> dict[str, Any]:
324
+ """Intern repeated provenance values, including verification and source data.
325
+
326
+ Nested dictionary paths preserve absent fields, explicit nulls and empty
327
+ dictionaries distinctly. Values stay JSON until a record is requested.
328
+ """
329
+ flattened = {rid: dict(_record_fields(record)) for rid, record in records.items()}
330
+ fields = sorted({path for record in flattened.values() for path in record})
331
+ values: list[str] = []
332
+ positions: dict[bytes, int] = {}
333
+ rows = {}
334
+ for rid, record in sorted(flattened.items()):
335
+ row = []
336
+ for path in fields:
337
+ if path not in record:
338
+ row.append(None)
339
+ continue
340
+ encoded = canonical_json(record[path])
341
+ if encoded not in positions:
342
+ positions[encoded] = len(values)
343
+ values.append(encoded.decode("utf-8"))
344
+ row.append(positions[encoded])
345
+ rows[rid] = row
346
+ return {"fields": fields, "values": values, "rows": rows}
347
+
348
+
349
+ def _unpack_record(table: Mapping[str, Any], rid: str) -> dict[str, Any]:
350
+ record: dict[str, Any] = {}
351
+ for path, position in zip(table["fields"], table["rows"][rid]):
352
+ if position is None:
353
+ continue
354
+ target = record
355
+ for key in path[:-1]:
356
+ target = target.setdefault(key, {})
357
+ target[path[-1]] = json.loads(table["values"][position])
358
+ return record
359
+
360
+
361
+ # ── the built snapshot ─────────────────────────────────────────────────────
362
+
363
+
364
+ @dataclass(frozen=True)
365
+ class Snapshot:
366
+ content: Mapping[str, Any]
367
+ content_hash: str
368
+ snapshot_id: str
369
+
370
+ def envelope(self, key: bytes | str | None = _FROM_ENV) -> dict[str, Any]:
371
+ key = env_key() if key is _FROM_ENV else _key_bytes(key)
372
+ signature = None if key is None else {"alg": SIGNATURE_ALG,
373
+ "value": _sign(self.content_hash, key)}
374
+ return {"format": FORMAT, "format_version": FORMAT_VERSION,
375
+ "snapshot_id": self.snapshot_id, "content_hash": self.content_hash,
376
+ "signature": signature, "content": self.content}
377
+
378
+ def to_bytes(self, key: bytes | str | None = _FROM_ENV) -> bytes:
379
+ buf = io.BytesIO()
380
+ # A fixed mtime and no file name keep the gzip header deterministic.
381
+ with gzip.GzipFile(filename="", mode="wb", fileobj=buf, compresslevel=9, mtime=0) as gz:
382
+ gz.write(canonical_json(self.envelope(key)))
383
+ return buf.getvalue()
384
+
385
+ def write(self, path: str | Path, *, key: bytes | str | None = _FROM_ENV) -> Path:
386
+ path = Path(path)
387
+ path.parent.mkdir(parents=True, exist_ok=True)
388
+ path.write_bytes(self.to_bytes(key))
389
+ return path
390
+
391
+
392
+ # ── building ───────────────────────────────────────────────────────────────
393
+
394
+
395
+ class _Compiler:
396
+ def __init__(self, inputs: SnapshotInputs, registry: Any, guard: ExcludedSources | None):
397
+ self.inputs = inputs
398
+ self.registry = registry
399
+ self.guard = guard
400
+ self.sources = {str(k): str(v) for k, v in inputs.sources.items()}
401
+ #: subject id (``None`` when a record names none) -> why its records stayed out.
402
+ self.excluded: dict[str | None, Counter[str]] = {}
403
+ #: (subject, facet) -> (reason, source URLs), for the gate's messages.
404
+ self.rejected: dict[tuple[str, str], tuple[str, tuple[str, ...]]] = {}
405
+ self.log = self._verification_log(inputs.verifications)
406
+ #: subject id -> {"kind", "model", "lifecycle"}
407
+ self.subjects: dict[str, dict[str, Any]] = {}
408
+ #: subject id -> facet -> [state, value, source ids]
409
+ self.facts: dict[str, dict[str, list[Any]]] = {}
410
+ self.facet_subject: dict[str, str] = {}
411
+ self.evidence: dict[str, list[list[Any]]] = {}
412
+ self.records: dict[str, dict[str, Any]] = {}
413
+ self.fact_records: dict[str, dict[str, str]] = {}
414
+
415
+ # verification ------------------------------------------------------------
416
+
417
+ @staticmethod
418
+ def _verification_log(
419
+ rows: Iterable[Any],
420
+ ) -> dict[tuple[str, str, str], tuple[str, int, dict]]:
421
+ latest: dict[tuple[str, str, str], tuple[str, int, dict]] = {}
422
+ for i, raw in enumerate(rows):
423
+ v = _as_dict(raw)
424
+ if not _counts(v):
425
+ continue
426
+ target = v["target"]
427
+ key = (target["kind"], target["id"], str(target.get("value_hash") or ""))
428
+ entry = (str(v["date"]), i + 1, v)
429
+ if key not in latest or entry[:2] >= latest[key][:2]:
430
+ latest[key] = entry
431
+ return latest
432
+
433
+ def _verification(self, kind: str, rid: Any, inline: Any, value: Any) -> dict | None:
434
+ """The winning verification record, or ``None``."""
435
+ expected = value_hash(value)
436
+ inline_record = None if inline is None else _as_dict(inline)
437
+ best = (
438
+ None
439
+ if inline_record is None
440
+ or not _counts(inline_record)
441
+ or inline_record.get("target", {}).get("value_hash") != expected
442
+ else (str(inline_record["date"]), 0, inline_record)
443
+ )
444
+ logged = self.log.get((kind, str(rid), expected)) if rid is not None else None
445
+ if logged is not None and (best is None or logged[:2] >= best[:2]):
446
+ best = logged
447
+ return None if best is None else best[2]
448
+
449
+ def _outcome(self, kind: str, rid: Any, inline: Any, value: Any) -> str:
450
+ record = self._verification(kind, rid, inline, value)
451
+ return "unverified" if record is None else str(record["outcome"])
452
+
453
+ def _retain(self, kind: str, record: dict) -> str:
454
+ rid = str(record.get("id") or content_hash(record))
455
+ value = record.get("value") if kind == "fact" else record.get("score")
456
+ retained = {**record, "verification": self._verification(
457
+ kind, record.get("id"), record.get("verification"), value)}
458
+ if rid in self.records and self.records[rid] != retained:
459
+ raise SnapshotBuildError(f"duplicate record ID {rid}")
460
+ self.records[rid] = retained
461
+ return rid
462
+
463
+ # admission ---------------------------------------------------------------
464
+
465
+ def _source_ids(self, refs: Iterable[Any]) -> list[str]:
466
+ return sorted({str(_as_dict(r)["source_id"]) for r in refs or ()})
467
+
468
+ def _admit(self, kind: str, rid: Any, inline: Any, value: Any, source_ids: list[str],
469
+ extra_urls: Iterable[Any] = (), benchmark: Any = None) -> str | None:
470
+ """Why a record stays out, or ``None`` when it enters."""
471
+ urls = [self.sources[s] for s in source_ids if s in self.sources]
472
+ if self.guard is not None and (any(self.guard.url(u) for u in [*urls, *extra_urls])
473
+ or (benchmark is not None
474
+ and self.guard.benchmark(benchmark))):
475
+ return "excluded_source"
476
+ outcome = self._outcome(kind, rid, inline, value)
477
+ if outcome != "verified":
478
+ return f"quarantined ({outcome})"
479
+ if not source_ids:
480
+ return "unsourced"
481
+ if len(urls) != len(source_ids):
482
+ return "unresolved_source"
483
+ return None
484
+
485
+ def _exclude(self, sid: str | None, reason: str) -> None:
486
+ kind = "quarantined" if reason.startswith("quarantined") else reason
487
+ self.excluded.setdefault(sid, Counter())[kind] += 1
488
+
489
+ def _reject(self, key: tuple[str, str], reason: str, source_ids: list[str]) -> None:
490
+ self._exclude(key[0], reason)
491
+ urls = tuple(self.sources.get(s, s) for s in source_ids)
492
+ self.rejected[key] = (reason, urls)
493
+
494
+ # subjects ----------------------------------------------------------------
495
+
496
+ def _check_facet(self, facet_id: str, kind: str) -> None:
497
+ if self.registry is not None:
498
+ try:
499
+ registered = self.registry.facet(facet_id)
500
+ except KeyError as exc:
501
+ raise SnapshotBuildError(f"facet {facet_id!r} is not registered") from exc
502
+ if getattr(registered, "computed_by", None):
503
+ raise SnapshotBuildError(
504
+ f"facet {facet_id!r} is computed ({registered.computed_by}), never authored")
505
+ seen = self.facet_subject.setdefault(facet_id, kind)
506
+ if seen != kind:
507
+ raise SnapshotBuildError(f"facet {facet_id!r} is used on both a {seen} and a {kind}")
508
+
509
+ def _add_facts(self, sid: str, kind: str, facts: Iterable[Any]) -> None:
510
+ row = self.facts.setdefault(sid, {})
511
+ for raw in facts or ():
512
+ f = _as_dict(raw)
513
+ facet_id, state = str(f["facet"]), str(f["state"])
514
+ self._check_facet(facet_id, kind)
515
+ if state not in FACT_STATES:
516
+ raise SnapshotBuildError(f"{sid}: {facet_id} has an unknown state {state!r}")
517
+ if facet_id in row or (sid, facet_id) in self.rejected:
518
+ raise SnapshotBuildError(f"{sid}: {facet_id} is stated twice")
519
+ source_ids = self._source_ids(f.get("sources"))
520
+ if state == "unknown":
521
+ self.rejected[(sid, facet_id)] = ("unknown", tuple(
522
+ self.sources.get(s, s) for s in source_ids))
523
+ continue
524
+ reason = self._admit(
525
+ "fact", f.get("id"), f.get("verification"), f.get("value"), source_ids
526
+ )
527
+ if reason is not None:
528
+ self._reject((sid, facet_id), reason, source_ids)
529
+ continue
530
+ self.fact_records.setdefault(sid, {})[facet_id] = self._retain("fact", f)
531
+ row[facet_id] = [state, f.get("value") if state == "known" else None, source_ids]
532
+
533
+ def add_model(self, raw: Any) -> None:
534
+ m = _as_dict(raw)
535
+ mid, lifecycle = str(m["id"]), str(m.get("lifecycle"))
536
+ if lifecycle not in LIFECYCLES:
537
+ raise SnapshotBuildError(f"{mid}: lifecycle {lifecycle!r} is not one of {LIFECYCLES}")
538
+ if mid in self.subjects:
539
+ raise SnapshotBuildError(f"model {mid} appears twice")
540
+ self.subjects[mid] = {"kind": "model", "model": mid, "lifecycle": lifecycle}
541
+ self._add_facts(mid, "model", m.get("facts"))
542
+
543
+ def add_offering(self, raw: Any) -> None:
544
+ o = _as_dict(raw)
545
+ oid, mid = _offering_id(o), str(o["model"])
546
+ if mid not in self.subjects:
547
+ raise SnapshotBuildError(f"offering {oid} names model {mid}, which is not in the catalogue")
548
+ if oid in self.subjects:
549
+ raise SnapshotBuildError(f"offering {oid} appears twice")
550
+ self.subjects[oid] = {"kind": "offering", "model": mid,
551
+ "lifecycle": self.subjects[mid]["lifecycle"]}
552
+ self._add_facts(oid, "offering", o.get("facts"))
553
+ # An offering's identity is its provider, region and tier: structural,
554
+ # not a sourced claim, so they carry no source.
555
+ row = self.facts[oid]
556
+ for part in ("provider", "region", "tier"):
557
+ facet_id = f"offering.{part}"
558
+ if facet_id not in row:
559
+ self._check_facet(facet_id, "offering")
560
+ row[facet_id] = ["known", str(o[part]), []]
561
+
562
+ def add_evidence(self, raw: Any) -> None:
563
+ e = _as_dict(raw)
564
+ subject = e.get("subject") or {}
565
+ sid = subject.get("id")
566
+ if sid is None:
567
+ self._exclude(None, "quarantined") # no subject: cannot be v2-verified
568
+ return
569
+ if sid not in self.subjects:
570
+ raise SnapshotBuildError(f"evidence {e.get('id')!r} names {sid}, which is not in the catalogue")
571
+ source_ids = self._source_ids(e.get("sources"))
572
+ reason = self._admit("evidence", e.get("id"), e.get("verification"), e.get("score"), source_ids,
573
+ extra_urls=[e.get("source_url")], benchmark=e.get("benchmark_id"))
574
+ if reason is not None:
575
+ self._exclude(sid, reason)
576
+ return
577
+ self.evidence.setdefault(sid, []).append([
578
+ str(e["benchmark_id"]), e.get("benchmark_version") or None, e.get("subcategory"),
579
+ float(e["score"]), e.get("unit"), e.get("measured_by"), e.get("effort"),
580
+ e.get("harness"), str(e.get("evidence_date") or "") or None, source_ids,
581
+ self._retain("evidence", e), e.get("date_type"),
582
+ next((r.get("snapshot_ref") for r in e.get("sources", [])
583
+ if r["source_id"] == source_ids[0]), None),
584
+ ])
585
+
586
+ # output ------------------------------------------------------------------
587
+
588
+ def _section(self, ids: list[str]) -> dict[str, Any]:
589
+ index = {sid: i for i, sid in enumerate(ids)}
590
+ columns: dict[str, dict[str, list[Any]]] = {}
591
+ for sid in ids:
592
+ for facet_id, (state, value, sources) in self.facts.get(sid, {}).items():
593
+ col = columns.setdefault(facet_id, {"row": [], "state": [], "value": [],
594
+ "sources": []})
595
+ col["row"].append(index[sid])
596
+ col["state"].append(state)
597
+ col["value"].append(value)
598
+ col["sources"].append(sources)
599
+ return {
600
+ "candidates": [{"id": sid, **self.subjects[sid]} for sid in ids],
601
+ "facets": columns,
602
+ "evidence": {sid: sorted(self.evidence[sid], key=lambda r: (
603
+ r[0], r[8] or "", r[3], canonical_json(r))) for sid in ids if sid in self.evidence},
604
+ }
605
+
606
+ def content(self, as_of: date | None, premier: Iterable[str] | None = None) -> dict[str, Any]:
607
+ """The snapshot content. With ``premier``, the lineup is the premier set.
608
+
609
+ Retired models always go to the archive. Active and deprecated models
610
+ outside the premier set leave the snapshot, with their offerings,
611
+ evidence and records; only their number is kept, as ``out_of_lineup``.
612
+ """
613
+ wanted = None if premier is None else set(premier)
614
+ archive = sorted(s for s, v in self.subjects.items() if v["lifecycle"] == "retired")
615
+ lineup = sorted(s for s, v in self.subjects.items() if v["lifecycle"] != "retired"
616
+ and (wanted is None or v["model"] in wanted))
617
+ kept = {*lineup, *archive}
618
+ out_of_lineup = sum(1 for s, v in self.subjects.items()
619
+ if v["kind"] == "model" and s not in kept)
620
+ sources: set[str] = set()
621
+ record_ids: set[str] = set()
622
+ for sid in kept:
623
+ for _state, _value, source_ids in self.facts.get(sid, {}).values():
624
+ sources.update(source_ids)
625
+ record_ids.update(self.fact_records.get(sid, {}).values())
626
+ for row in self.evidence.get(sid, ()):
627
+ sources.update(row[9])
628
+ record_ids.add(row[10])
629
+ excluded: Counter[str] = Counter()
630
+ for sid, counts in self.excluded.items():
631
+ if sid is None or sid in kept:
632
+ excluded.update(counts)
633
+ domains = {}
634
+ for bench, tags in sorted(self.inputs.benchmark_domains.items()):
635
+ if self.guard is not None and self.guard.benchmark(bench):
636
+ continue
637
+ domains[str(bench)] = sorted([str(d), str(k)] for d, k in tags)
638
+ capability: dict[str, Any] = {}
639
+ if self.inputs.benchmark_metadata and as_of is not None:
640
+ from decision.capability import (
641
+ BenchmarkSpec,
642
+ CapabilityObservation,
643
+ Directness,
644
+ fit_capabilities,
645
+ )
646
+
647
+ observations = []
648
+ fitted_models = {self.subjects[sid]["model"] for sid in kept}
649
+ for subject, evidence_rows in sorted(self.evidence.items()):
650
+ model_id = self.subjects[subject]["model"]
651
+ if model_id not in fitted_models:
652
+ continue
653
+ for row in evidence_rows:
654
+ evidence_date = _date(row[8])
655
+ tag_rows: list[tuple[str, Directness]] = []
656
+ for domain_id, raw_directness in self.inputs.benchmark_domains.get(row[0], ()):
657
+ directness = str(raw_directness)
658
+ if directness not in ("direct", "proxy"):
659
+ raise SnapshotBuildError(
660
+ f"{row[0]}: invalid capability directness {directness!r}"
661
+ )
662
+ tag_rows.append((str(domain_id), directness))
663
+ tags = tuple(tag_rows)
664
+ if evidence_date is None or not tags:
665
+ continue
666
+ observations.append(CapabilityObservation(
667
+ model_id=model_id,
668
+ benchmark_id=row[0],
669
+ value=float(row[3]),
670
+ unit=row[4],
671
+ measured_by=str(row[5] or ""),
672
+ date=evidence_date,
673
+ record_id=row[10],
674
+ version=row[1],
675
+ domains=tags,
676
+ ))
677
+ specs = {
678
+ benchmark: BenchmarkSpec(
679
+ random_baseline=metadata.get("random_baseline"),
680
+ sample_size=metadata.get("sample_size"),
681
+ direction=metadata.get("direction", "higher_is_better"),
682
+ )
683
+ for benchmark, metadata in self.inputs.benchmark_metadata.items()
684
+ }
685
+ fit = fit_capabilities(observations, specs, as_of=as_of)
686
+ capability = fit.to_payload(
687
+ self.subjects[sid]["model"] for sid in kept
688
+ if self.subjects[sid]["kind"] == "model"
689
+ )
690
+ return {
691
+ "format_version": FORMAT_VERSION,
692
+ "as_of": as_of.isoformat() if as_of else None,
693
+ "facet_subjects": dict(sorted(self.facet_subject.items())),
694
+ "lineup": self._section(lineup),
695
+ "archive": self._section(archive),
696
+ "out_of_lineup": out_of_lineup,
697
+ "benchmark_domains": domains,
698
+ "capability": capability,
699
+ "sources": {s: self.sources[s] for s in sorted(sources)},
700
+ "excluded": dict(sorted(excluded.items())),
701
+ "record_table": _pack_records({r: self.records[r] for r in record_ids}),
702
+ "fact_records": {sid: rows for sid, rows in self.fact_records.items() if sid in kept},
703
+ }
704
+
705
+ # the gate ----------------------------------------------------------------
706
+
707
+ def gaps(self, premier: Iterable[str]) -> list[Gap]:
708
+ if self.registry is None:
709
+ raise SnapshotBuildError("the completeness gate needs the facet registry")
710
+ guaranteed = [f for f in self.registry.facets()
711
+ if f.tier == "guaranteed" and not getattr(f, "computed_by", None)
712
+ and f.subject in ("model", "offering")]
713
+ out: list[Gap] = []
714
+ for mid in sorted(set(premier)):
715
+ subject = self.subjects.get(mid)
716
+ if subject is None:
717
+ out.append(Gap(mid, mid, "(model)", "not in the catalogue"))
718
+ continue
719
+ if subject["lifecycle"] == "retired":
720
+ continue # retired models leave the premier set (design §5)
721
+ offerings = sorted(s for s, v in self.subjects.items()
722
+ if v["kind"] == "offering" and v["model"] == mid)
723
+ for f in sorted(guaranteed, key=lambda f: f.id):
724
+ for sid in ([mid] if f.subject == "model" else offerings):
725
+ if f.id in self.facts.get(sid, {}):
726
+ continue
727
+ reason, urls = self.rejected.get((sid, f.id), ("unknown (no fact)", ()))
728
+ if reason == "unknown":
729
+ reason = "unknown (stated as unknown)"
730
+ out.append(Gap(mid, sid, f.id, reason, urls))
731
+ return out
732
+
733
+
734
+ def default_registry() -> Any:
735
+ try:
736
+ from decision import registry
737
+ except ImportError as exc: # MODEL-133 not present
738
+ raise SnapshotBuildError(f"the facet registry is not available: {exc}") from exc
739
+ return registry.default()
740
+
741
+
742
+ def build_snapshot(inputs: SnapshotInputs, *, registry: Any = None,
743
+ premier: Iterable[str] | None = None, as_of: date | None = None,
744
+ guard: ExcludedSources | None = None, gate: bool = True) -> Snapshot:
745
+ """Compile ``inputs``. With ``premier``, the lineup is the premier set.
746
+
747
+ With ``premier`` and ``gate`` (the default), the completeness gate runs
748
+ first. ``gate=False`` keeps the premier lineup but skips the gate, for an
749
+ audit that must run while facts are still missing (MODEL-146).
750
+ ``registry`` validates facet IDs and names the guaranteed facets; the gate
751
+ requires it. ``guard`` drops excluded sources and scans the output;
752
+ ``build_from_repo`` always passes it.
753
+ """
754
+ c = _Compiler(inputs, registry, guard)
755
+ for m in inputs.models:
756
+ c.add_model(m)
757
+ for o in inputs.offerings:
758
+ c.add_offering(o)
759
+ for e in inputs.evidence:
760
+ c.add_evidence(e)
761
+ premier = None if premier is None else tuple(premier)
762
+ if premier is not None and gate:
763
+ gaps = c.gaps(premier)
764
+ if gaps:
765
+ raise CompletenessError(gaps)
766
+ content = c.content(as_of, premier)
767
+ if guard is not None:
768
+ text = canonical_json(content).decode("utf-8")
769
+ hit = guard.text.search(text)
770
+ bad = [u for u in content["sources"].values() if guard.url(u)]
771
+ if hit or bad:
772
+ raise SnapshotBuildError(
773
+ f"excluded source in the snapshot output: {hit.group(0) if hit else bad[0]!r}")
774
+ digest = content_hash(content)
775
+ return Snapshot(content=content, content_hash=digest, snapshot_id=snapshot_id_for(digest))
776
+
777
+
778
+ # ── reading the repository ─────────────────────────────────────────────────
779
+
780
+
781
+ _LIFECYCLE_FROM_STATUS = {"deprecated": "deprecated", "sunset": "deprecated"}
782
+
783
+
784
+ def collect_repo(root: Path) -> SnapshotInputs:
785
+ """Read the snapshot's inputs from a repository checkout.
786
+
787
+ * Cards (``models/``): ``lifecycle`` (else the v1 ``status``: deprecated and
788
+ sunset map to ``deprecated``, everything else to ``active``), v2 ``facts``,
789
+ and ``benchmarks.evidence`` rows. ``benchmarks.scores`` is never read.
790
+ * Offerings: ``offerings/<provider>/<lab>/<model>.yaml``, each a list.
791
+ * Sources: canonical ``registry/sources.yaml``; see ``verification/README.md``.
792
+ * Domains: each benchmark page's ``domains`` tags.
793
+ * The verification log: ``verification/log.jsonl``.
794
+ """
795
+ from pipeline.load import load_benchmarks, load_models
796
+
797
+ root = Path(root)
798
+ models, evidence = [], []
799
+ for card in load_models(root):
800
+ mid = card.model_id
801
+ front = card.front
802
+ lifecycle = front.get("lifecycle") or _LIFECYCLE_FROM_STATUS.get(
803
+ str(front.get("status") or ""), "active")
804
+ facts = []
805
+ for f in front.get("facts") or []:
806
+ facts.append({"subject": {"kind": "model", "id": mid},
807
+ "id": f"{mid}#{f.get('facet')}", **f})
808
+ models.append({"id": mid, "lifecycle": lifecycle, "facts": facts})
809
+ for row in card.evidence:
810
+ evidence.append({"subject": {"kind": "model", "id": mid}, **row})
811
+ offerings = []
812
+ for path in sorted((root / "offerings").glob("*/*/*.yaml")):
813
+ rows = yaml.safe_load(path.read_text(encoding="utf-8")) or []
814
+ if not isinstance(rows, list):
815
+ raise SnapshotBuildError(f"{path}: an offering file is a list")
816
+ for o in rows:
817
+ oid = _offering_id(o)
818
+ o = dict(o)
819
+ o["facts"] = [{"subject": {"kind": "offering", "id": oid},
820
+ "id": f"{oid}#{f.get('facet')}", **f} for f in o.get("facts") or []]
821
+ offerings.append(o)
822
+ from decision.sources import load_sources
823
+
824
+ sources = {
825
+ source_id: str(source.url)
826
+ for source_id, source in load_sources(root / "registry" / "sources.yaml").items()
827
+ }
828
+ domains = {}
829
+ metadata = {}
830
+ for b in load_benchmarks(root):
831
+ tags = b.front.get("domains") or []
832
+ if tags:
833
+ domains[b.benchmark_id] = tuple((str(t["id"]), str(t["directness"])) for t in tags)
834
+ metric = b.front.get("metric") or {}
835
+ dataset = b.front.get("dataset") or {}
836
+ metadata[b.benchmark_id] = {
837
+ "random_baseline": metric.get("random_baseline"),
838
+ "sample_size": dataset.get("size"),
839
+ "direction": metric.get("direction", "higher_is_better"),
840
+ }
841
+ verifications = []
842
+ verification_log = root / "verification" / "log.jsonl"
843
+ if verification_log.is_file():
844
+ for line in verification_log.read_text(encoding="utf-8").splitlines():
845
+ if line.strip():
846
+ verifications.append(json.loads(line))
847
+ return SnapshotInputs(models=models, offerings=offerings, evidence=evidence, sources=sources,
848
+ benchmark_domains=domains, benchmark_metadata=metadata,
849
+ verifications=verifications)
850
+
851
+
852
+ def load_premier(path: str | Path) -> tuple[str, ...]:
853
+ """The premier model IDs from a YAML file: a list, or ``models:`` a list.
854
+
855
+ Each item is an ID or a mapping with ``id`` or ``model_id``.
856
+ """
857
+ path = Path(path)
858
+ try:
859
+ data = yaml.safe_load(path.read_text(encoding="utf-8"))
860
+ except (OSError, yaml.YAMLError) as exc:
861
+ raise SnapshotBuildError(f"cannot read the premier list {path}: {exc}") from exc
862
+ items = data.get("models") if isinstance(data, Mapping) else data
863
+ ids = set()
864
+ for item in items or []:
865
+ mid = item.get("id") or item.get("model_id") if isinstance(item, Mapping) else item
866
+ if not isinstance(mid, str) or not mid:
867
+ raise SnapshotBuildError(f"{path}: premier entry {item!r} has no model id")
868
+ ids.add(mid)
869
+ if not ids:
870
+ raise SnapshotBuildError(f"{path}: no premier models listed")
871
+ return tuple(sorted(ids))
872
+
873
+
874
+ def build_from_repo(root: Path, *, premier: str | Path | None, as_of: date | None,
875
+ registry: Any = None, gate: bool = True) -> Snapshot:
876
+ """The production build: collect, guard, gate and compile."""
877
+ root = Path(root)
878
+ return build_snapshot(
879
+ collect_repo(root),
880
+ registry=registry if registry is not None else default_registry(),
881
+ premier=load_premier(premier) if premier is not None else None,
882
+ as_of=as_of,
883
+ guard=excluded_sources(),
884
+ gate=gate,
885
+ )
886
+
887
+
888
+ # ── loading ────────────────────────────────────────────────────────────────
889
+
890
+
891
+ def _date(value: Any) -> date | None:
892
+ try:
893
+ return date.fromisoformat(str(value))
894
+ except (TypeError, ValueError):
895
+ return None
896
+
897
+
898
+ def _key(value: Any) -> tuple[str, Any]:
899
+ if isinstance(value, str):
900
+ return "str", value
901
+ if isinstance(value, bool):
902
+ return "bool", value
903
+ if isinstance(value, int | float):
904
+ return "number", value
905
+ return "json", json.dumps(value, sort_keys=True)
906
+
907
+
908
+ def _ordered(value: Any) -> Any:
909
+ if value == UNBOUNDED:
910
+ return math.inf
911
+ if isinstance(value, date):
912
+ return value.isoformat()
913
+ return value
914
+
915
+
916
+ def _holds(value: Any, op: str, arg: Any) -> bool:
917
+ """Whether a known value satisfies ``op arg``."""
918
+ if op in ("=", "=="):
919
+ if isinstance(value, list):
920
+ return isinstance(arg, (list, tuple, set, frozenset)) and sorted(value) == sorted(arg)
921
+ return _ordered(value) == _ordered(arg)
922
+ if op == "!=":
923
+ return not _holds(value, "=", arg)
924
+ if op in ("in", "not_in"):
925
+ members = {_key(_ordered(a)) for a in arg}
926
+ found = (any(_key(v) in members for v in value) if isinstance(value, list)
927
+ else _key(_ordered(value)) in members)
928
+ return found if op == "in" else not found
929
+ if op == "contains":
930
+ return isinstance(value, list) and arg in value
931
+ if op == "contains_all":
932
+ return isinstance(value, list) and set(arg) <= set(value)
933
+ if op == "contains_any":
934
+ return isinstance(value, list) and bool(set(arg) & set(value))
935
+ if op in ("<", "<=", ">", ">=", "between"):
936
+ if value == NOT_OFFERED or isinstance(value, (list, bool)):
937
+ return False
938
+ v = _ordered(value)
939
+ try:
940
+ if op == "between":
941
+ low, high = arg
942
+ return _ordered(low) <= v <= _ordered(high)
943
+ a = _ordered(arg)
944
+ return {"<": v < a, "<=": v <= a, ">": v > a, ">=": v >= a}[op]
945
+ except TypeError as exc:
946
+ raise SnapshotError(f"cannot compare {value!r} {op} {arg!r}") from exc
947
+ raise SnapshotError(f"unknown operator {op!r}")
948
+
949
+
950
+ def _equality_key(value: Any) -> tuple[str, Any]:
951
+ ordered = _ordered(value)
952
+ if isinstance(ordered, list):
953
+ return "list", tuple(sorted(_key(_ordered(member)) for member in ordered))
954
+ return "scalar", _key(ordered)
955
+
956
+
957
+ class _FacetBitsets:
958
+ """Bitsets for one facet, built once from its known values."""
959
+
960
+ def __init__(self, rows: Iterable[tuple[int, Any]], *, alternatives: bool = False):
961
+ self.alternatives = alternatives
962
+ self.known = 0
963
+ self.collections = 0
964
+ self.exact: dict[tuple[str, Any], int] = {}
965
+ self.members: dict[tuple[str, Any], int] = {}
966
+ self.contains: dict[tuple[str, Any], int] = {}
967
+ ordered: dict[Any, int] = {}
968
+ self._ordered_usable = True
969
+ for row, raw in rows:
970
+ bit = 1 << row
971
+ self.known |= bit
972
+ values = raw if alternatives else (raw,)
973
+ for value in values:
974
+ key = _equality_key(value)
975
+ self.exact[key] = self.exact.get(key, 0) | bit
976
+ members = value if isinstance(value, list) and not alternatives else (value,)
977
+ if isinstance(value, list) and not alternatives:
978
+ self.collections |= bit
979
+ for member in members:
980
+ member_key = _key(_ordered(member))
981
+ self.members[member_key] = self.members.get(member_key, 0) | bit
982
+ if isinstance(value, list) and not alternatives:
983
+ self.contains[member_key] = self.contains.get(member_key, 0) | bit
984
+ if value == NOT_OFFERED or isinstance(value, (list, bool)):
985
+ continue
986
+ try:
987
+ ordered_value = _ordered(value)
988
+ ordered[ordered_value] = ordered.get(ordered_value, 0) | bit
989
+ except (TypeError, ValueError):
990
+ self._ordered_usable = False
991
+ try:
992
+ self.ordered_values = tuple(sorted(ordered))
993
+ except TypeError:
994
+ self._ordered_usable = False
995
+ self.ordered_values = ()
996
+ prefixes = [0]
997
+ for value in self.ordered_values:
998
+ prefixes.append(prefixes[-1] | ordered[value])
999
+ self.prefixes = tuple(prefixes)
1000
+
1001
+ def _ordered(self, op: str, arg: Any) -> int | None:
1002
+ if not self._ordered_usable:
1003
+ return None
1004
+ try:
1005
+ if op == "between":
1006
+ low, high = (_ordered(value) for value in arg)
1007
+ left = bisect_left(self.ordered_values, low)
1008
+ right = bisect_right(self.ordered_values, high)
1009
+ return 0 if left > right else self.prefixes[right] & ~self.prefixes[left]
1010
+ value = _ordered(arg)
1011
+ if op == "<":
1012
+ return self.prefixes[bisect_left(self.ordered_values, value)]
1013
+ if op == "<=":
1014
+ return self.prefixes[bisect_right(self.ordered_values, value)]
1015
+ if op == ">":
1016
+ return self.prefixes[-1] & ~self.prefixes[bisect_right(self.ordered_values, value)]
1017
+ if op == ">=":
1018
+ return self.prefixes[-1] & ~self.prefixes[bisect_left(self.ordered_values, value)]
1019
+ except (TypeError, ValueError):
1020
+ return None
1021
+ return None
1022
+
1023
+ def passing(self, op: str, arg: Any) -> int | None:
1024
+ if op in ("=", "=="):
1025
+ return self.exact.get(_equality_key(arg), 0)
1026
+ if op == "!=":
1027
+ key = _equality_key(arg)
1028
+ if self.alternatives:
1029
+ passing = 0
1030
+ for value_key, bits in self.exact.items():
1031
+ if value_key != key:
1032
+ passing |= bits
1033
+ return passing
1034
+ return self.known & ~self.exact.get(key, 0)
1035
+ if op in ("in", "not_in"):
1036
+ hit = 0
1037
+ for value in arg:
1038
+ hit |= self.members.get(_key(_ordered(value)), 0)
1039
+ return self.known & (hit if op == "in" else ~hit)
1040
+ if op == "contains":
1041
+ return self.contains.get(_key(_ordered(arg)), 0)
1042
+ if op in ("contains_all", "contains_any"):
1043
+ values = tuple(arg)
1044
+ if op == "contains_all":
1045
+ if not values:
1046
+ return self.collections
1047
+ passing = self.known
1048
+ for value in values:
1049
+ passing &= self.contains.get(_key(_ordered(value)), 0)
1050
+ return passing
1051
+ passing = 0
1052
+ for value in values:
1053
+ passing |= self.contains.get(_key(_ordered(value)), 0)
1054
+ return passing
1055
+ if op in ("<", "<=", ">", ">=", "between"):
1056
+ return self._ordered(op, arg)
1057
+ return None
1058
+
1059
+
1060
+ class _Evidence(dict[str, tuple[EvidenceValue, ...]]):
1061
+ """Materialise one candidate's evidence on first access.
1062
+
1063
+ An offering answers its model's evidence: capability belongs to the model,
1064
+ and an offering is that model as one provider sells it. The offering's own
1065
+ measurements of a benchmark, when it has any, replace its model's for that
1066
+ benchmark. The stored snapshot keeps evidence under its subject only.
1067
+ """
1068
+
1069
+ def __init__(self, rows: Mapping[str, Sequence[Sequence[Any]]],
1070
+ model_of: Mapping[str, str]):
1071
+ super().__init__()
1072
+ self.rows = rows
1073
+ self.model_of = model_of
1074
+ self.records = {
1075
+ (model_of.get(cid, cid), row[10]): row
1076
+ for cid, candidate_rows in rows.items()
1077
+ for row in candidate_rows
1078
+ if len(row) > 10 and row[10] is not None
1079
+ }
1080
+
1081
+ @staticmethod
1082
+ def _value(row: Sequence[Any]) -> EvidenceValue:
1083
+ return EvidenceValue(
1084
+ benchmark_id=row[0], version=row[1], subcategory=row[2], value=row[3],
1085
+ unit=row[4], measured_by=row[5], effort=row[6], harness=row[7],
1086
+ date=_date(row[8]), source_ids=tuple(row[9]),
1087
+ record_id=row[10] if len(row) > 10 else None,
1088
+ date_type=row[11] if len(row) > 11 else None,
1089
+ source_snapshot=row[12] if len(row) > 12 else None,
1090
+ )
1091
+
1092
+ def record(self, cid: str, record_id: str) -> EvidenceValue | None:
1093
+ row = self.records.get((self.model_of.get(cid, cid), record_id))
1094
+ return None if row is None else self._value(row)
1095
+
1096
+ def _rows(self, cid: str) -> list[Sequence[Any]]:
1097
+ own = list(self.rows.get(cid, ()))
1098
+ model = self.model_of.get(cid, cid)
1099
+ if model == cid:
1100
+ return own
1101
+ measured = {r[0] for r in own}
1102
+ inherited = [r for r in self.rows.get(model, ()) if r[0] not in measured]
1103
+ return sorted(own + inherited, key=lambda r: (r[0], r[8] or "", r[3], canonical_json(r)))
1104
+
1105
+ def __missing__(self, cid: str) -> tuple[EvidenceValue, ...]:
1106
+ self[cid] = tuple(self._value(row) for row in self._rows(cid))
1107
+ return self[cid]
1108
+
1109
+
1110
+ class LoadedSnapshot:
1111
+ """The in-memory index over one snapshot. Implements ``SnapshotIndex``."""
1112
+
1113
+ def __init__(self, envelope: Mapping[str, Any], *, include_archive: bool,
1114
+ signature_verified: bool):
1115
+ content = envelope["content"]
1116
+ self.snapshot_id: str = envelope["snapshot_id"]
1117
+ self.content_hash: str = envelope["content_hash"]
1118
+ self.signature_verified = signature_verified
1119
+ self.as_of = _date(content.get("as_of"))
1120
+ self.excluded: dict[str, int] = dict(content.get("excluded") or {})
1121
+ #: Active models the build left out because they are not in the premier set.
1122
+ self.out_of_lineup: int = int(content.get("out_of_lineup") or 0)
1123
+ self._sources: dict[str, str] = dict(content["sources"])
1124
+ self._records = content.get("records", {})
1125
+ self._record_table = content.get("record_table")
1126
+ self.explanation_rebuild_required = (
1127
+ None
1128
+ if "fact_records" in content and ("record_table" in content or "records" in content)
1129
+ else "snapshot predates retained verification records; rebuild it before explaining"
1130
+ )
1131
+ sections = [content["lineup"]] + ([content["archive"]] if include_archive else [])
1132
+
1133
+ rows: list[tuple[dict[str, Any], dict[str, Any], int]] = []
1134
+ for section in sections:
1135
+ for i, cand in enumerate(section["candidates"]):
1136
+ rows.append((cand, section, i))
1137
+ rows.sort(key=lambda r: r[0]["id"])
1138
+ self._ids: tuple[str, ...] = tuple(r[0]["id"] for r in rows)
1139
+ self._row = {cid: i for i, cid in enumerate(self._ids)}
1140
+ self._meta = {r[0]["id"]: r[0] for r in rows}
1141
+ self._all = (1 << len(self._ids)) - 1
1142
+
1143
+ facts: dict[str, dict[int, FactValue]] = {}
1144
+ for section in sections:
1145
+ local = [c["id"] for c in section["candidates"]]
1146
+ for facet_id, col in section["facets"].items():
1147
+ target = facts.setdefault(facet_id, {})
1148
+ for r, state, value, srcs in zip(col["row"], col["state"], col["value"],
1149
+ col["sources"]):
1150
+ target[self._row[local[r]]] = FactValue(
1151
+ state, value, tuple(srcs),
1152
+ content.get("fact_records", {}).get(local[r], {}).get(facet_id))
1153
+ # An offering is its model as sold: it answers its model's facets.
1154
+ subjects = content.get("facet_subjects") or {}
1155
+ for facet_id, by_row in facts.items():
1156
+ if subjects.get(facet_id) != "model":
1157
+ continue
1158
+ for cid, meta in self._meta.items():
1159
+ if meta["kind"] == "offering":
1160
+ row, model_row = self._row[cid], self._row.get(meta["model"])
1161
+ if row not in by_row and model_row in by_row:
1162
+ by_row[row] = by_row[model_row]
1163
+ self._facts = facts
1164
+
1165
+ self._facet_bits = {
1166
+ facet_id: _FacetBitsets(
1167
+ (row, fv.value) for row, fv in by_row.items() if fv.state == "known"
1168
+ )
1169
+ for facet_id, by_row in facts.items()
1170
+ }
1171
+
1172
+ self._evidence = _Evidence({cid: rows for section in sections
1173
+ for cid, rows in section["evidence"].items()},
1174
+ {cid: meta["model"] for cid, meta in self._meta.items()})
1175
+ self._evidence_bits: dict[tuple[Any, ...], _FacetBitsets] = {}
1176
+ self._benchmarks = tuple(sorted(content["benchmark_domains"]))
1177
+ capability = content.get("capability") or {}
1178
+ # Parse the learned lookup once. Domain objectives are the page's
1179
+ # default, so reconstructing these values throughout filtering,
1180
+ # optimisation and explanation made the first Worker request pay the
1181
+ # same JSON-to-object cost repeatedly.
1182
+ self._capability_estimates = {
1183
+ (model_id, domain_id): CapabilityEstimateValue(*map(float, row))
1184
+ for model_id, domains in (capability.get("estimates") or {}).items()
1185
+ for domain_id, row in domains.items()
1186
+ }
1187
+ self._capability_drivers = {
1188
+ (model_id, domain_id): tuple(
1189
+ CapabilityDriverValue(
1190
+ record_id=row[0], benchmark_id=row[1], version=row[2],
1191
+ loading=float(row[3]), weight=float(row[4]),
1192
+ recency_weight=float(row[5]),
1193
+ )
1194
+ for row in rows
1195
+ )
1196
+ for model_id, domains in (capability.get("drivers") or {}).items()
1197
+ for domain_id, rows in domains.items()
1198
+ }
1199
+ self.capability_method = capability.get("method")
1200
+ self.capability_items = capability.get("items") or {}
1201
+ self.capability_source_offsets = capability.get("source_offsets") or {}
1202
+ self._domains: dict[str, list[tuple[str, str]]] = {}
1203
+ for bench, tags in content["benchmark_domains"].items():
1204
+ for domain_id, directness in tags:
1205
+ self._domains.setdefault(domain_id, []).append((bench, directness))
1206
+
1207
+ # SnapshotIndex -----------------------------------------------------------
1208
+
1209
+ def candidates(self) -> Sequence[str]:
1210
+ return self._ids
1211
+
1212
+ def _check(self, cid: str) -> int:
1213
+ try:
1214
+ return self._row[cid]
1215
+ except KeyError:
1216
+ raise KeyError(f"{cid!r} is not a candidate in snapshot {self.snapshot_id}") from None
1217
+
1218
+ def lifecycle(self, cid: str) -> Lifecycle:
1219
+ self._check(cid)
1220
+ return self._meta[cid]["lifecycle"]
1221
+
1222
+ def fact(self, cid: str, facet_id: str) -> FactValue:
1223
+ return self._facts.get(facet_id, {}).get(self._check(cid), UNKNOWN)
1224
+
1225
+ def ids_where(self, facet_id: str, op: str, arg: Any) -> Bitset3:
1226
+ """Candidates passing, failing, or unknown on ``facet op arg``.
1227
+
1228
+ Operators: ``=`` (or ``==``), ``!=``, ``<``, ``<=``, ``>``, ``>=``,
1229
+ ``between`` (a (low, high) pair, inclusive), ``in``, ``not_in``,
1230
+ ``contains``, ``contains_all``, ``contains_any`` (on set facets) and
1231
+ ``known``. Any state but ``known`` is unknown; ``known`` itself is
1232
+ never unknown. ``unbounded`` exceeds every number; ``not_offered``
1233
+ fails every ordered comparison.
1234
+ """
1235
+ column = self._facet_bits.get(facet_id)
1236
+ known = 0 if column is None else column.known
1237
+ if op == "known":
1238
+ return Bitset3(known, self._all & ~known, 0)
1239
+ passing = None if column is None else column.passing(op, arg)
1240
+ if passing is None:
1241
+ passing = 0
1242
+ by_row = self._facts.get(facet_id, {})
1243
+ for row in self._rows(known):
1244
+ if _holds(by_row[row].value, op, arg):
1245
+ passing |= 1 << row
1246
+ return Bitset3(passing, known & ~passing, self._all & ~known)
1247
+
1248
+ def evidence_where(
1249
+ self,
1250
+ benchmark_id: str,
1251
+ op: str,
1252
+ arg: Any,
1253
+ *,
1254
+ measured_by: set[str] | None = None,
1255
+ effort: str | None = None,
1256
+ harness: str | None = None,
1257
+ after: date | None = None,
1258
+ direct: bool = False,
1259
+ domains: Iterable[str] = (),
1260
+ ) -> Bitset3:
1261
+ """Three-valued evidence condition, indexed lazily per qualifier set.
1262
+
1263
+ ``direct`` admits the benchmark only when it is direct for one of
1264
+ ``domains``, the capabilities asked about (see ``direct_for``).
1265
+ """
1266
+ admits = not direct or self.direct_for(benchmark_id, domains)
1267
+ key = (
1268
+ benchmark_id,
1269
+ None if measured_by is None else frozenset(measured_by),
1270
+ effort,
1271
+ harness,
1272
+ after,
1273
+ admits,
1274
+ )
1275
+ column = self._evidence_bits.get(key)
1276
+ if column is None:
1277
+ admitted = []
1278
+ for row, cid in enumerate(self._ids if admits else ()):
1279
+ values = tuple(
1280
+ evidence.value
1281
+ for evidence in self.evidence(
1282
+ cid,
1283
+ benchmark_id,
1284
+ measured_by=measured_by,
1285
+ effort=effort,
1286
+ harness=harness,
1287
+ after=after,
1288
+ )
1289
+ )
1290
+ if values:
1291
+ admitted.append((row, values))
1292
+ column = _FacetBitsets(admitted, alternatives=True)
1293
+ self._evidence_bits[key] = column
1294
+ passing = column.passing(op, arg)
1295
+ if passing is None:
1296
+ passing = 0
1297
+ for row, cid in enumerate(self._ids if admits else ()):
1298
+ values = self.evidence(
1299
+ cid,
1300
+ benchmark_id,
1301
+ measured_by=measured_by,
1302
+ effort=effort,
1303
+ harness=harness,
1304
+ after=after,
1305
+ )
1306
+ if any(_holds(value.value, op, arg) for value in values):
1307
+ passing |= 1 << row
1308
+ return Bitset3(passing, column.known & ~passing, self._all & ~column.known)
1309
+
1310
+ def evidence(self, cid: str, benchmark_id: str, *, measured_by: set[str] | None = None,
1311
+ effort: str | None = None, harness: str | None = None,
1312
+ after: date | None = None) -> Sequence[EvidenceValue]:
1313
+ """Evidence for one benchmark; ``after`` is exclusive. Unknown dates never pass it."""
1314
+ self._check(cid)
1315
+ return tuple(
1316
+ e for e in self._evidence[cid]
1317
+ if e.benchmark_id == benchmark_id
1318
+ and (measured_by is None or e.measured_by in measured_by)
1319
+ and (effort is None or e.effort == effort)
1320
+ and (harness is None or e.harness == harness)
1321
+ and (after is None or (e.date is not None and e.date > after)))
1322
+
1323
+ def direct_for(self, benchmark_id: str, domains: Iterable[str] = ()) -> bool:
1324
+ """Whether ``benchmark_id`` directly measures a capability asked about.
1325
+
1326
+ Directness is relative to the request (design §4.1): the benchmark must
1327
+ be tagged ``direct`` for one of ``domains``. With no domain asked
1328
+ about, a ``direct`` tag for any domain is enough.
1329
+ """
1330
+ return any((benchmark_id, "direct") in self._domains.get(domain_id, ())
1331
+ for domain_id in (set(domains) or self._domains))
1332
+
1333
+ def evidence_for_domain(self, cid: str, domain_id: str) -> Sequence[EvidenceValue]:
1334
+ self._check(cid)
1335
+ directness = dict(self._domains.get(domain_id, ()))
1336
+ return tuple(
1337
+ EvidenceValue(**{**e.__dict__, "directness": directness[e.benchmark_id]})
1338
+ for e in self._evidence[cid] if e.benchmark_id in directness)
1339
+
1340
+ def capability_estimate(
1341
+ self, cid: str, domain_id: str
1342
+ ) -> CapabilityEstimateValue | None:
1343
+ self._check(cid)
1344
+ model_id = self._meta[cid]["model"]
1345
+ return self._capability_estimates.get((model_id, domain_id))
1346
+
1347
+ def capability_drivers(
1348
+ self, cid: str, domain_id: str
1349
+ ) -> Sequence[CapabilityDriverValue]:
1350
+ self._check(cid)
1351
+ model_id = self._meta[cid]["model"]
1352
+ return self._capability_drivers.get((model_id, domain_id), ())
1353
+
1354
+ def evidence_record(self, cid: str, record_id: str) -> EvidenceValue | None:
1355
+ """Return retained model evidence by record ID in constant time."""
1356
+ self._check(cid)
1357
+ return self._evidence.record(self._meta[cid]["model"], record_id)
1358
+
1359
+ # beyond the protocol -----------------------------------------------------
1360
+
1361
+ def require_explanation_records(self) -> None:
1362
+ """Refuse explanations from a snapshot built before provenance retention."""
1363
+ if self.explanation_rebuild_required is not None:
1364
+ raise SnapshotError(self.explanation_rebuild_required)
1365
+
1366
+ def kind(self, cid: str) -> Literal["model", "offering"]:
1367
+ self._check(cid)
1368
+ return self._meta[cid]["kind"]
1369
+
1370
+ def model_of(self, cid: str) -> str:
1371
+ self._check(cid)
1372
+ return self._meta[cid]["model"]
1373
+
1374
+ def record(self, record_id: str) -> Mapping[str, Any]:
1375
+ """The admitted record with its winning verification, retained verbatim."""
1376
+ if record_id not in self._records and self._record_table is not None:
1377
+ self._records[record_id] = _unpack_record(self._record_table, record_id)
1378
+ return self._records[record_id]
1379
+
1380
+ def facet_ids(self) -> tuple[str, ...]:
1381
+ return tuple(sorted(self._facts))
1382
+
1383
+ def domain_ids(self) -> tuple[str, ...]:
1384
+ return tuple(sorted(self._domains))
1385
+
1386
+ def benchmark_ids(self) -> tuple[str, ...]:
1387
+ return self._benchmarks
1388
+
1389
+ def benchmark_domain_tags(self) -> dict[str, tuple[tuple[str, str], ...]]:
1390
+ """Each benchmark's (domain, directness) tags, sorted by domain."""
1391
+ tags: dict[str, list[tuple[str, str]]] = {b: [] for b in self._benchmarks}
1392
+ for domain_id, rows in self._domains.items():
1393
+ for bench, directness in rows:
1394
+ tags.setdefault(bench, []).append((domain_id, directness))
1395
+ return {b: tuple(sorted(t)) for b, t in tags.items()}
1396
+
1397
+ def source_url(self, source_id: str) -> str:
1398
+ return self._sources[source_id]
1399
+
1400
+ def ids(self, bits: int) -> tuple[str, ...]:
1401
+ """The candidate IDs a bitset names."""
1402
+ return tuple(self._ids[r] for r in self._rows(bits))
1403
+
1404
+ @staticmethod
1405
+ def _rows(bits: int) -> Iterable[int]:
1406
+ row = 0
1407
+ while bits:
1408
+ if bits & 1:
1409
+ yield row
1410
+ bits >>= 1
1411
+ row += 1
1412
+
1413
+
1414
+ def load_snapshot_bytes(
1415
+ data: bytes,
1416
+ *,
1417
+ key: bytes | str | None = _FROM_ENV,
1418
+ include_archive: bool = False,
1419
+ source: str = "snapshot bytes",
1420
+ ) -> LoadedSnapshot:
1421
+ """Check and index a gzipped snapshot already held in memory."""
1422
+ try:
1423
+ text = gzip.decompress(data).decode("utf-8")
1424
+ stored_digest = None
1425
+ if text.startswith('{"content":{'):
1426
+ # raw_decode finds the JSON boundary, including escaped quotes and
1427
+ # nested objects. Never search for a delimiter inside content.
1428
+ content, end = json.JSONDecoder().raw_decode(text, len('{"content":'))
1429
+ suffix = text[end:].lstrip()
1430
+ envelope = json.loads("{" + suffix[1:]) if suffix.startswith(",") else {}
1431
+ if "content" in envelope:
1432
+ raise SnapshotIntegrityError(f"{source}: duplicate content member")
1433
+ envelope["content"] = content
1434
+ stored_digest = "sha256:" + hashlib.sha256(
1435
+ text[len('{"content":'):end].encode("utf-8")).hexdigest()
1436
+ else:
1437
+ envelope = json.loads(text)
1438
+ except (OSError, EOFError, ValueError) as exc:
1439
+ raise SnapshotIntegrityError(f"{source}: not a gzipped JSON snapshot: {exc}") from exc
1440
+ if not isinstance(envelope, dict) or envelope.get("format") != FORMAT:
1441
+ raise SnapshotIntegrityError(f"{source}: not a {FORMAT} file")
1442
+ if envelope.get("format_version") != FORMAT_VERSION:
1443
+ raise SnapshotIntegrityError(
1444
+ f"{source}: format version {envelope.get('format_version')!r}, "
1445
+ f"expected {FORMAT_VERSION}"
1446
+ )
1447
+ # Canonical writer output can be checked directly. Other JSON encodings
1448
+ # retain the original semantic hash check, including the signature check.
1449
+ digest = stored_digest
1450
+ if digest != envelope.get("content_hash") or digest is None:
1451
+ digest = content_hash(envelope["content"])
1452
+ if digest != envelope.get("content_hash"):
1453
+ raise SnapshotIntegrityError(f"{source}: content hash mismatch: the snapshot was altered")
1454
+ if envelope.get("snapshot_id") != snapshot_id_for(digest):
1455
+ raise SnapshotIntegrityError(f"{source}: snapshot ID does not match its content hash")
1456
+ key = env_key() if key is _FROM_ENV else _key_bytes(key)
1457
+ verified = False
1458
+ if key is not None:
1459
+ signature = envelope.get("signature")
1460
+ if not signature:
1461
+ raise SnapshotIntegrityError(f"{source}: unsigned snapshot, but a key was given")
1462
+ if (signature.get("alg") != SIGNATURE_ALG
1463
+ or not hmac.compare_digest(str(signature.get("value")), _sign(digest, key))):
1464
+ raise SnapshotIntegrityError(f"{source}: signature does not verify with this key")
1465
+ verified = True
1466
+ return LoadedSnapshot(envelope, include_archive=include_archive, signature_verified=verified)
1467
+
1468
+
1469
+ def load_snapshot(path: str | Path, *, key: bytes | str | None = _FROM_ENV,
1470
+ include_archive: bool = False) -> LoadedSnapshot:
1471
+ """Read, check and index a snapshot.
1472
+
1473
+ The content hash is always checked. The key defaults to
1474
+ ``MODELSPEC_SNAPSHOT_KEY``; with a key, the snapshot must be signed with it.
1475
+ Without one, the signature cannot be checked and ``signature_verified`` is
1476
+ false. Retired models are left out unless ``include_archive``.
1477
+ """
1478
+ path = Path(path)
1479
+ try:
1480
+ data = path.read_bytes()
1481
+ except OSError as exc:
1482
+ raise SnapshotIntegrityError(f"{path}: not a gzipped JSON snapshot: {exc}") from exc
1483
+ return load_snapshot_bytes(data, key=key, include_archive=include_archive, source=str(path))