modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
pipeline/load.py
ADDED
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
"""Read the repository's sources of truth into plain Python objects.
|
|
2
|
+
|
|
3
|
+
Three sources, in dependency order:
|
|
4
|
+
|
|
5
|
+
* `models/` — one Markdown file per model card, YAML front matter.
|
|
6
|
+
* `benchmarks/` — one Markdown file per benchmark page, YAML front matter.
|
|
7
|
+
* the eligibility report — which benchmarks may be presented as current.
|
|
8
|
+
|
|
9
|
+
Nothing here reaches the network and nothing writes. The loader is deliberately
|
|
10
|
+
tolerant about *absent* data and strict about *malformed* data: a missing
|
|
11
|
+
optional field is normal in a 692-field schema where null means "not yet
|
|
12
|
+
researched", but a file that claims to be a card and will not parse is a defect
|
|
13
|
+
worth surfacing.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import json
|
|
19
|
+
import re
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
from datetime import date
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
import yaml
|
|
26
|
+
|
|
27
|
+
REPO_ROOT = Path(__file__).resolve().parent.parent
|
|
28
|
+
FRONT_MATTER = re.compile(r"^---\n(.*?)\n---\n?(.*)\Z", re.S)
|
|
29
|
+
|
|
30
|
+
#: Prose that lives beside the data and is not itself data.
|
|
31
|
+
NOT_CONTENT = {"LICENSE.md", "README.md", "AUTHORING.md", "CONTRIBUTING.md"}
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class LoadError(RuntimeError):
|
|
35
|
+
"""A file that should have parsed did not."""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def split_front_matter(text: str) -> tuple[dict[str, Any], str]:
|
|
39
|
+
match = FRONT_MATTER.match(text)
|
|
40
|
+
if not match:
|
|
41
|
+
raise LoadError("no YAML front matter")
|
|
42
|
+
try:
|
|
43
|
+
front = yaml.safe_load(match.group(1))
|
|
44
|
+
except yaml.YAMLError as exc:
|
|
45
|
+
raise LoadError(f"front matter is not valid YAML: {exc}") from exc
|
|
46
|
+
if not isinstance(front, dict):
|
|
47
|
+
raise LoadError("front matter is not a mapping")
|
|
48
|
+
return front, match.group(2)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass(frozen=True)
|
|
52
|
+
class Model:
|
|
53
|
+
model_id: str
|
|
54
|
+
path: Path
|
|
55
|
+
front: dict[str, Any]
|
|
56
|
+
body: str
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def slug(self) -> str:
|
|
60
|
+
return self.model_id
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def display_name(self) -> str:
|
|
64
|
+
return str(self.front.get("display_name") or self.model_id)
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def provider(self) -> str:
|
|
68
|
+
return str(self.front.get("provider") or self.path.parent.name)
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def provider_display(self) -> str:
|
|
72
|
+
return str(self.front.get("provider_display") or self.provider)
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def scores(self) -> dict[str, float]:
|
|
76
|
+
block = self.front.get("benchmarks") or {}
|
|
77
|
+
raw = block.get("scores") if isinstance(block, dict) else None
|
|
78
|
+
if not isinstance(raw, dict):
|
|
79
|
+
return {}
|
|
80
|
+
return {k: v for k, v in raw.items() if isinstance(v, (int, float))}
|
|
81
|
+
|
|
82
|
+
@property
|
|
83
|
+
def evidence(self) -> list[dict[str, Any]]:
|
|
84
|
+
"""Reviewed per-score records, each with its own source and date."""
|
|
85
|
+
block = self.front.get("benchmarks") or {}
|
|
86
|
+
raw = block.get("evidence") if isinstance(block, dict) else None
|
|
87
|
+
if not isinstance(raw, list):
|
|
88
|
+
return []
|
|
89
|
+
return [r for r in raw if isinstance(r, dict) and r.get("benchmark_id")
|
|
90
|
+
and isinstance(r.get("score"), (int, float))]
|
|
91
|
+
|
|
92
|
+
@property
|
|
93
|
+
def scores_as_of(self) -> str | None:
|
|
94
|
+
block = self.front.get("benchmarks") or {}
|
|
95
|
+
value = block.get("benchmark_as_of") if isinstance(block, dict) else None
|
|
96
|
+
return str(value) if value else None
|
|
97
|
+
|
|
98
|
+
@property
|
|
99
|
+
def scores_source(self) -> str | None:
|
|
100
|
+
block = self.front.get("benchmarks") or {}
|
|
101
|
+
value = block.get("benchmark_source") if isinstance(block, dict) else None
|
|
102
|
+
return str(value) if value else None
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
@dataclass(frozen=True)
|
|
106
|
+
class Benchmark:
|
|
107
|
+
benchmark_id: str
|
|
108
|
+
path: Path
|
|
109
|
+
front: dict[str, Any]
|
|
110
|
+
body: str
|
|
111
|
+
|
|
112
|
+
@property
|
|
113
|
+
def name(self) -> str:
|
|
114
|
+
return str(self.front.get("name") or self.benchmark_id)
|
|
115
|
+
|
|
116
|
+
@property
|
|
117
|
+
def category(self) -> str:
|
|
118
|
+
return str(self.front.get("category") or "uncategorised")
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def summary(self) -> str:
|
|
122
|
+
return str(self.front.get("summary") or "").strip()
|
|
123
|
+
|
|
124
|
+
@property
|
|
125
|
+
def aliases(self) -> list[str]:
|
|
126
|
+
raw = self.front.get("aliases")
|
|
127
|
+
return [str(a) for a in raw] if isinstance(raw, list) else []
|
|
128
|
+
|
|
129
|
+
@property
|
|
130
|
+
def lower_is_better(self) -> bool:
|
|
131
|
+
metric = self.front.get("metric")
|
|
132
|
+
return isinstance(metric, dict) and metric.get("direction") == "lower_is_better"
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
@dataclass(frozen=True)
|
|
136
|
+
class Disposition:
|
|
137
|
+
"""One benchmark's standing in the active catalogue.
|
|
138
|
+
|
|
139
|
+
`status` is the spec's vocabulary: active, historical, unverified, alias, or
|
|
140
|
+
unassessed for a discovery lead nobody has evaluated yet. `unassessed` is not
|
|
141
|
+
a judgement — the spec is explicit that missing evidence is not staleness and
|
|
142
|
+
not illegitimacy.
|
|
143
|
+
"""
|
|
144
|
+
|
|
145
|
+
status: str
|
|
146
|
+
canonical_id: str
|
|
147
|
+
reasons: tuple[str, ...] = ()
|
|
148
|
+
results: tuple[dict[str, Any], ...] = ()
|
|
149
|
+
|
|
150
|
+
@property
|
|
151
|
+
def is_active(self) -> bool:
|
|
152
|
+
return self.status == "active"
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
@dataclass
|
|
156
|
+
class Catalogue:
|
|
157
|
+
as_of: date
|
|
158
|
+
dispositions: dict[str, Disposition] = field(default_factory=dict)
|
|
159
|
+
|
|
160
|
+
def for_benchmark(self, benchmark_id: str) -> Disposition:
|
|
161
|
+
return self.dispositions.get(
|
|
162
|
+
benchmark_id, Disposition(status="unassessed", canonical_id=benchmark_id)
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
@property
|
|
166
|
+
def active_ids(self) -> list[str]:
|
|
167
|
+
return sorted(i for i, d in self.dispositions.items() if d.is_active)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def load_models(root: Path | None = None) -> list[Model]:
|
|
171
|
+
directory = (root or REPO_ROOT) / "models"
|
|
172
|
+
out: list[Model] = []
|
|
173
|
+
for path in sorted(directory.rglob("*.md")):
|
|
174
|
+
if path.name in NOT_CONTENT:
|
|
175
|
+
continue
|
|
176
|
+
front, body = split_front_matter(path.read_text(encoding="utf-8", errors="replace"))
|
|
177
|
+
model_id = front.get("model_id")
|
|
178
|
+
if not model_id:
|
|
179
|
+
# Prose without a model_id is not a card; skip rather than fail, the
|
|
180
|
+
# same rule the PR validator applies.
|
|
181
|
+
continue
|
|
182
|
+
out.append(Model(str(model_id), path, front, body))
|
|
183
|
+
return out
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def load_benchmarks(root: Path | None = None) -> list[Benchmark]:
|
|
187
|
+
directory = (root or REPO_ROOT) / "benchmarks"
|
|
188
|
+
out: list[Benchmark] = []
|
|
189
|
+
seen: dict[str, Path] = {}
|
|
190
|
+
for path in sorted(directory.glob("*.md")):
|
|
191
|
+
if path.name in NOT_CONTENT:
|
|
192
|
+
continue
|
|
193
|
+
front, body = split_front_matter(path.read_text(encoding="utf-8", errors="replace"))
|
|
194
|
+
benchmark_id = front.get("id")
|
|
195
|
+
if not benchmark_id:
|
|
196
|
+
raise LoadError(f"{path}: benchmark page has no id")
|
|
197
|
+
benchmark_id = str(benchmark_id)
|
|
198
|
+
if benchmark_id in seen:
|
|
199
|
+
raise LoadError(f"duplicate benchmark id {benchmark_id!r}: {seen[benchmark_id]} and {path}")
|
|
200
|
+
seen[benchmark_id] = path
|
|
201
|
+
out.append(Benchmark(benchmark_id, path, front, body))
|
|
202
|
+
return out
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def load_catalogue(root: Path | None = None) -> Catalogue:
|
|
206
|
+
"""Read the eligibility report.
|
|
207
|
+
|
|
208
|
+
Absent report means an empty active set, which the spec says is a valid
|
|
209
|
+
outcome. It must never fall back to treating the census queue as a catalogue.
|
|
210
|
+
"""
|
|
211
|
+
path = (root or REPO_ROOT) / "benchmarks/_census/eligibility/current-report.json"
|
|
212
|
+
if not path.is_file():
|
|
213
|
+
return Catalogue(as_of=date.today())
|
|
214
|
+
report = json.loads(path.read_text(encoding="utf-8"))
|
|
215
|
+
as_of = date.fromisoformat(str(report["as_of"]))
|
|
216
|
+
dispositions: dict[str, Disposition] = {}
|
|
217
|
+
for row in report.get("rows", []):
|
|
218
|
+
dispositions[str(row["candidate_id"])] = Disposition(
|
|
219
|
+
status=str(row["status"]),
|
|
220
|
+
canonical_id=str(row.get("canonical_id") or row["candidate_id"]),
|
|
221
|
+
reasons=tuple(str(r) for r in row.get("reasons", [])),
|
|
222
|
+
results=tuple(row.get("accepted_results", [])),
|
|
223
|
+
)
|
|
224
|
+
return Catalogue(as_of=as_of, dispositions=dispositions)
|