modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
decision/model.py ADDED
@@ -0,0 +1,438 @@
1
+ """Sourced domain records for the decision engine (MODEL-134).
2
+
3
+ Registry lookups use decision.registry, or context={"registry": registry} for
4
+ an explicitly supplied registry with the same interface. No legacy flat scores
5
+ are read here. See docs/decision-model.md for the card evidence mapping.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import datetime
11
+ import hashlib
12
+ import json
13
+ import math
14
+ import re
15
+ from importlib import import_module
16
+ from pathlib import Path
17
+ from typing import Annotated, Literal, Self
18
+
19
+ import yaml
20
+ from pydantic import (
21
+ AwareDatetime,
22
+ BaseModel,
23
+ ConfigDict,
24
+ Field,
25
+ HttpUrl,
26
+ JsonValue,
27
+ TypeAdapter,
28
+ ValidationInfo,
29
+ model_validator,
30
+ )
31
+
32
+ from schema.card import BenchmarkEvidence
33
+
34
+ Text = Annotated[str, Field(min_length=1, pattern=r"\S", strict=True)]
35
+ ContentRef = Annotated[str, Field(pattern=r"^sha256:[0-9a-f]{64}$")]
36
+
37
+
38
+ class Record(BaseModel):
39
+ model_config = ConfigDict(extra="forbid")
40
+
41
+
42
+ class SubjectRef(Record):
43
+ kind: Literal["model", "offering", "provider"]
44
+ id: Text
45
+
46
+
47
+ class SourceRef(Record):
48
+ source_id: Text
49
+ snapshot_ref: ContentRef
50
+ cited_regions: list[Text] = Field(min_length=1)
51
+
52
+
53
+ class TargetRef(Record):
54
+ kind: Literal["fact", "evidence"]
55
+ id: Text
56
+
57
+
58
+ class VerificationTarget(TargetRef):
59
+ value_hash: ContentRef
60
+
61
+
62
+ class VerificationActor(Record):
63
+ agent: Text
64
+ model_family: Text
65
+ method: Text
66
+
67
+
68
+ #: The ``model_family`` of a reader that runs no model: always independent.
69
+ DETERMINISTIC = "deterministic"
70
+
71
+ #: A family name's first word -> the lab whose lineage it is.
72
+ _LINEAGES = {
73
+ "anthropic": "anthropic", "claude": "anthropic",
74
+ "openai": "openai", "gpt": "openai",
75
+ "google": "google", "gemini": "google", "gemma": "google",
76
+ "alibaba": "alibaba", "qwen": "alibaba",
77
+ "meta": "meta", "llama": "meta",
78
+ "xai": "xai", "grok": "xai",
79
+ "mistral": "mistral", "mixtral": "mistral", "codestral": "mistral",
80
+ "deepseek": "deepseek",
81
+ }
82
+
83
+
84
+ def model_family(name: str) -> str:
85
+ """The lineage a recorded ``model_family`` names: "claude" and "anthropic" are one.
86
+
87
+ Case and a trailing version are dropped ("gpt-5", "gemma4", "qwen3"); a
88
+ first word that names a known lineage ("claude-sonnet-5", "mistral-large")
89
+ maps to its lab. Gemma and Gemini are one family: a second key must not
90
+ share the first key's lineage. An unknown name is its own family.
91
+ """
92
+ base = re.sub(r"[-_ .]?v?\d[\w.\-]*$", "", name.strip().casefold()) or name.casefold()
93
+ return _LINEAGES.get(re.split(r"[-_ ]", base)[0], base)
94
+
95
+
96
+ def independent_families(collector: str, verifier: str) -> bool:
97
+ """Two keys: the verifier is deterministic, or of another model family (MODEL-140)."""
98
+ return verifier == DETERMINISTIC or model_family(collector) != model_family(verifier)
99
+
100
+
101
+ def verification_counts(outcome: str, collector: str, verifier: str) -> bool:
102
+ """Whether a logged verification decides its value's state.
103
+
104
+ A same-family ``verified`` is not a second key and does not count; any other
105
+ outcome counts, so a same-family mismatch still keeps a value out.
106
+ """
107
+ return outcome != "verified" or independent_families(collector, verifier)
108
+
109
+
110
+ class Verification(Record):
111
+ target: VerificationTarget
112
+ collector: VerificationActor
113
+ verifier: VerificationActor
114
+ method: Text
115
+ outcome: Literal["verified", "mismatch", "unreachable"]
116
+ date: datetime.date
117
+ diff: Text | None = None
118
+
119
+ @model_validator(mode="after")
120
+ def independent_check(self) -> Self:
121
+ if (
122
+ self.collector.agent == self.verifier.agent
123
+ and self.collector.model_family == self.verifier.model_family
124
+ ):
125
+ raise ValueError("verification must be independent in agent or model family")
126
+ if self.outcome == "mismatch" and self.diff is None:
127
+ raise ValueError("a mismatch requires a diff")
128
+ if self.outcome != "mismatch" and self.diff is not None:
129
+ raise ValueError("diff belongs only to a mismatch")
130
+ return self
131
+
132
+ @property
133
+ def independent(self) -> bool:
134
+ return independent_families(self.collector.model_family, self.verifier.model_family)
135
+
136
+ @property
137
+ def counts(self) -> bool:
138
+ """See ``verification_counts``: a same-family ``verified`` is ignored."""
139
+ return verification_counts(
140
+ self.outcome, self.collector.model_family, self.verifier.model_family)
141
+
142
+ @property
143
+ def quarantined(self) -> bool:
144
+ return self.outcome != "verified" or not self.independent
145
+
146
+
147
+ def _registry(info: ValidationInfo):
148
+ registry = (info.context or {}).get("registry")
149
+ if registry is None:
150
+ registry = import_module("decision.registry")
151
+ return registry
152
+
153
+
154
+ def _registered(info: ValidationInfo, collection: str, id: str):
155
+ registry = _registry(info)
156
+ try:
157
+ return getattr(registry, collection)(id)
158
+ except KeyError as exc:
159
+ raise ValueError(f"unknown {collection} ID: {id}") from exc
160
+
161
+
162
+ def _allowed_values(registry, facet) -> frozenset[str] | None:
163
+ if isinstance(facet.value_type, str):
164
+ return None
165
+ owner = registry.default() if hasattr(registry, "default") else registry
166
+ return owner.allowed_values(facet) if hasattr(owner, "allowed_values") else None
167
+
168
+
169
+ def _check_value(value: JsonValue, facet, registry) -> None:
170
+ value_type = facet.value_type
171
+ kind = value_type if isinstance(value_type, str) else value_type.kind
172
+ if not isinstance(value_type, str):
173
+ if value == "unbounded" and value_type.unbounded:
174
+ return
175
+ if value == "not_offered" and value_type.not_offered:
176
+ return
177
+ if kind == "date":
178
+ if not isinstance(value, str):
179
+ raise ValueError("date value must be an ISO date string")
180
+ parsed = datetime.date.fromisoformat(value)
181
+ valid = parsed.isoformat() == value
182
+ else:
183
+ checks = {
184
+ "integer": lambda: type(value) is int,
185
+ "number": lambda: type(value) in (int, float) and math.isfinite(value),
186
+ "bool": lambda: type(value) is bool,
187
+ "boolean": lambda: type(value) is bool,
188
+ "string": lambda: isinstance(value, str),
189
+ "string_set": lambda: (
190
+ isinstance(value, list)
191
+ and all(isinstance(item, str) for item in value)
192
+ and len(value) == len(set(value))
193
+ ),
194
+ "enum": lambda: isinstance(value, str),
195
+ "set": lambda: (
196
+ isinstance(value, list)
197
+ and all(isinstance(item, str) for item in value)
198
+ and len(value) == len(set(value))
199
+ ),
200
+ "range": lambda: (
201
+ isinstance(value, list)
202
+ and len(value) == 2
203
+ and all(type(item) in (int, float) and math.isfinite(item) for item in value)
204
+ and value[0] <= value[1]
205
+ ),
206
+ }
207
+ if kind not in checks:
208
+ raise ValueError(f"unsupported facet value_type: {kind}")
209
+ valid = checks[kind]()
210
+ if not valid:
211
+ raise ValueError(f"value must match facet value_type {kind}")
212
+ allowed = _allowed_values(registry, facet)
213
+ members = value if kind in ("set", "string_set") else [value]
214
+ if allowed is not None and any(member not in allowed for member in members):
215
+ raise ValueError(f"value must use the facet's registered values: {sorted(allowed)}")
216
+
217
+
218
+ def value_hash(value: JsonValue) -> str:
219
+ encoded = json.dumps(
220
+ value, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False
221
+ ).encode("utf-8")
222
+ return "sha256:" + hashlib.sha256(encoded).hexdigest()
223
+
224
+
225
+ def _check_target(verification: Verification | None, kind: str, id: str, value: JsonValue) -> None:
226
+ if verification is not None and (
227
+ verification.target.kind != kind
228
+ or verification.target.id != id
229
+ or verification.target.value_hash != value_hash(value)
230
+ ):
231
+ raise ValueError("verification target or value_hash does not match this record")
232
+
233
+
234
+ class Fact(Record):
235
+ id: Text
236
+ subject: SubjectRef
237
+ facet: Text
238
+ value: JsonValue = None
239
+ state: Literal["known", "unknown", "not_disclosed", "requires_contract"]
240
+ sources: list[SourceRef] = Field(default_factory=list)
241
+ checked_sources: list[Text] = Field(default_factory=list)
242
+ verification: Verification | None = None
243
+
244
+ @model_validator(mode="after")
245
+ def valid_fact(self, info: ValidationInfo) -> Self:
246
+ facet = _registered(info, "facet", self.facet)
247
+ facet_subject = getattr(facet, "subject", None)
248
+ if facet_subject is not None and facet_subject != self.subject.kind:
249
+ raise ValueError(
250
+ f"facet {self.facet} belongs to {facet_subject}, not {self.subject.kind}"
251
+ )
252
+ if self.subject.kind == "provider":
253
+ _registered(info, "provider", self.subject.id)
254
+ if self.state == "known":
255
+ if self.value is None or not self.sources:
256
+ raise ValueError("a known fact requires a value and sources")
257
+ _check_value(self.value, facet, _registry(info))
258
+ elif self.value is not None:
259
+ raise ValueError("only a known fact may have a value")
260
+ _check_target(self.verification, "fact", self.id, self.value)
261
+ return self
262
+
263
+ @property
264
+ def quarantined(self) -> bool:
265
+ return self.verification is None or self.verification.quarantined
266
+
267
+
268
+ class RegionLocator(Record):
269
+ kind: Literal["page", "css", "xpath", "heading", "heading_anchor", "table"]
270
+ value: str = ""
271
+
272
+
273
+ class CitedRegion(Record):
274
+ id: Text
275
+ locator: RegionLocator
276
+
277
+
278
+ class Source(Record):
279
+ id: Text
280
+ url: HttpUrl
281
+ #: Whether values read from this source age from the day it was observed.
282
+ #: Static is the safe default for papers, system cards, and launch posts.
283
+ volatility: Literal["static", "live"] = "static"
284
+ fetch: Literal["http", "conditional_http", "rendered"] = "conditional_http"
285
+ normaliser: Text = "html-default"
286
+ cited_regions: list[CitedRegion] = Field(default_factory=list)
287
+
288
+ @model_validator(mode="after")
289
+ def unique_regions(self) -> Self:
290
+ from decision.normalise import NORMALISERS, Locator
291
+
292
+ ids = [region.id for region in self.cited_regions]
293
+ if len(ids) != len(set(ids)):
294
+ raise ValueError("cited region IDs must be unique within a source")
295
+ if self.normaliser not in NORMALISERS:
296
+ raise ValueError(f"unknown normaliser {self.normaliser!r}")
297
+ for region in self.cited_regions:
298
+ locator = region.locator
299
+ if locator.kind == "xpath":
300
+ raise ValueError(
301
+ "xpath cited-region locators are not supported; register a css, "
302
+ "heading_anchor, table, or page locator"
303
+ )
304
+ kind = "heading" if locator.kind == "heading_anchor" else locator.kind
305
+ Locator(kind, locator.value)
306
+ if NORMALISERS[self.normaliser].content == "text" and kind != "page":
307
+ raise ValueError("text sources support only page locators")
308
+ return self
309
+
310
+
311
+ class SourceSnapshot(Record):
312
+ source_id: Text
313
+ retrieved_at: AwareDatetime
314
+ page_fingerprint: ContentRef
315
+ region_fingerprints: dict[Text, ContentRef | None]
316
+ copy_ref: ContentRef
317
+ etag: Text | None = None
318
+ last_modified: Text | None = None
319
+
320
+
321
+ class EvidenceSubjectRef(Record):
322
+ kind: Literal["model", "offering"]
323
+ id: Text
324
+
325
+
326
+ class Evidence(BenchmarkEvidence):
327
+ """Additive v2 evidence; legacy verified_at does not grant v2 verification."""
328
+
329
+ model_config = ConfigDict(extra="forbid")
330
+ id: Text | None = None
331
+ subject: EvidenceSubjectRef | None = None
332
+ harness: Text | None = None
333
+ effort: Text | None = None
334
+ tools: list[Text] | None = None
335
+ measured_by: (
336
+ Literal[
337
+ "benchmark_author",
338
+ "independent_evaluator",
339
+ "provider_self_report",
340
+ "modelspec",
341
+ "outcome_protocol",
342
+ ]
343
+ | None
344
+ ) = None
345
+ subcategory: Text | None = None
346
+ sources: list[SourceRef] = Field(default_factory=list)
347
+ verification: Verification | None = None
348
+
349
+ @model_validator(mode="after")
350
+ def qualified_evidence(self, info: ValidationInfo) -> Self:
351
+ if not math.isfinite(self.score):
352
+ raise ValueError("evidence score must be finite")
353
+ if self.harness is not None and self.harness != "unregistered":
354
+ _registered(info, "harness", self.harness)
355
+ if self.verification is not None:
356
+ if self.id is None or self.subject is None or not self.sources:
357
+ raise ValueError("verification requires an ID, subject and source snapshots")
358
+ _check_target(self.verification, "evidence", self.id, self.score)
359
+ return self
360
+
361
+ @property
362
+ def quarantined(self) -> bool:
363
+ return self.verification is None or self.verification.quarantined
364
+
365
+
366
+ Lifecycle = Literal["active", "deprecated", "retired"]
367
+ ModelId = Annotated[str, Field(pattern=r"^[^/\s]+/[^/\s]+$")]
368
+
369
+
370
+ class Model(Record):
371
+ id: ModelId
372
+ lifecycle: Lifecycle
373
+ facts: list[Fact] = Field(default_factory=list)
374
+
375
+ @model_validator(mode="after")
376
+ def own_facts(self) -> Self:
377
+ _check_facts(self.facts, "model", self.id)
378
+ return self
379
+
380
+ @property
381
+ def in_lineup(self) -> bool:
382
+ return self.lifecycle != "retired"
383
+
384
+ @property
385
+ def in_live_archive(self) -> bool:
386
+ return self.lifecycle == "retired"
387
+
388
+
389
+ def _check_facts(facts: list[Fact], kind: str, id: str) -> None:
390
+ ids = [fact.id for fact in facts]
391
+ facets = [fact.facet for fact in facts]
392
+ if len(ids) != len(set(ids)) or len(facets) != len(set(facets)):
393
+ raise ValueError("fact IDs and facets must be unique for a subject")
394
+ if any(fact.subject.kind != kind or fact.subject.id != id for fact in facts):
395
+ raise ValueError("fact subject does not match its owner")
396
+
397
+
398
+ PathPart = Annotated[str, Field(pattern=r"^[a-zA-Z0-9][a-zA-Z0-9._-]*$")]
399
+
400
+
401
+ class Offering(Record):
402
+ model: ModelId
403
+ provider: PathPart
404
+ region: PathPart
405
+ tier: PathPart
406
+ facts: list[Fact] = Field(default_factory=list)
407
+
408
+ @property
409
+ def id(self) -> str:
410
+ return f"{self.provider}/{self.model}/{self.region}/{self.tier}"
411
+
412
+ @model_validator(mode="after")
413
+ def valid_offering(self, info: ValidationInfo) -> Self:
414
+ _registered(info, "provider", self.provider)
415
+ _check_facts(self.facts, "offering", self.id)
416
+ return self
417
+
418
+
419
+ def load_offerings(path: str | Path, *, registry=None) -> list[Offering]:
420
+ """Read a list from offerings/<provider>/<lab>/<model>.yaml.
421
+
422
+ The last three path components must agree with every row's identity.
423
+ Source storage and cross-record source resolution belong to MODEL-137/138.
424
+ """
425
+ path = Path(path)
426
+ rows = TypeAdapter(list[Offering]).validate_python(
427
+ yaml.safe_load(path.read_text(encoding="utf-8")),
428
+ context={"registry": registry},
429
+ )
430
+ ids: set[str] = set()
431
+ for row in rows:
432
+ expected = Path(row.provider) / f"{row.model}.yaml"
433
+ if tuple(path.parts[-3:]) != expected.parts:
434
+ raise ValueError(f"offering {row.id} does not match path {expected}")
435
+ if row.id in ids:
436
+ raise ValueError(f"duplicate offering: {row.id}")
437
+ ids.add(row.id)
438
+ return rows