modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
|
@@ -0,0 +1,304 @@
|
|
|
1
|
+
"""Strict records and the mechanical eligibility gate for benchmark evidence.
|
|
2
|
+
|
|
3
|
+
This validates the shape, dates, and coverage of submitted evidence. It does
|
|
4
|
+
not establish that a source is truthful; that remains the reviewer's job.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from datetime import date, datetime
|
|
10
|
+
from math import isfinite
|
|
11
|
+
from typing import Literal
|
|
12
|
+
|
|
13
|
+
from pydantic import BaseModel, ConfigDict, Field, HttpUrl, field_validator
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _text(value: str) -> str:
|
|
17
|
+
if not isinstance(value, str) or not value.strip():
|
|
18
|
+
raise ValueError("must be a nonblank string")
|
|
19
|
+
return value.strip()
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _date_value(value: object) -> object:
|
|
23
|
+
if isinstance(value, date) and not isinstance(value, datetime):
|
|
24
|
+
return value
|
|
25
|
+
if isinstance(value, str):
|
|
26
|
+
try:
|
|
27
|
+
parsed = date.fromisoformat(value)
|
|
28
|
+
except ValueError as exc:
|
|
29
|
+
raise ValueError("must be an ISO date YYYY-MM-DD") from exc
|
|
30
|
+
if parsed.isoformat() != value:
|
|
31
|
+
raise ValueError("must be an exact ISO date YYYY-MM-DD")
|
|
32
|
+
return parsed
|
|
33
|
+
raise ValueError("must be an ISO date YYYY-MM-DD")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class StrictRecord(BaseModel):
|
|
37
|
+
model_config = ConfigDict(extra="forbid")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class Identity(StrictRecord):
|
|
41
|
+
url: HttpUrl
|
|
42
|
+
rationale: str = ""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Usefulness(StrictRecord):
|
|
46
|
+
verdict: Literal["useful", "saturated", "unknown"]
|
|
47
|
+
rationale: str
|
|
48
|
+
source_url: HttpUrl
|
|
49
|
+
|
|
50
|
+
_rationale = field_validator("rationale")(_text)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class Result(StrictRecord):
|
|
54
|
+
model_id: str
|
|
55
|
+
benchmark_version: str
|
|
56
|
+
configuration: str
|
|
57
|
+
score: float
|
|
58
|
+
unit: str
|
|
59
|
+
source_url: HttpUrl
|
|
60
|
+
evidence_date: date
|
|
61
|
+
date_type: Literal["evaluated", "published"]
|
|
62
|
+
verified_at: date
|
|
63
|
+
source_kind: Literal["benchmark_author", "independent_evaluator", "model_provider"]
|
|
64
|
+
|
|
65
|
+
_text_fields = field_validator("model_id", "benchmark_version", "configuration", "unit")(_text)
|
|
66
|
+
_dates = field_validator("evidence_date", "verified_at", mode="before")(_date_value)
|
|
67
|
+
|
|
68
|
+
@field_validator("score", mode="before")
|
|
69
|
+
@classmethod
|
|
70
|
+
def finite_score(cls, value: object) -> float:
|
|
71
|
+
if isinstance(value, (bool, str)) or not isinstance(value, (int, float)):
|
|
72
|
+
raise ValueError("score must be numeric")
|
|
73
|
+
value = float(value)
|
|
74
|
+
if not isfinite(value):
|
|
75
|
+
raise ValueError("score must be finite")
|
|
76
|
+
return value
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class Review(StrictRecord):
|
|
80
|
+
reviewer: str
|
|
81
|
+
reviewed_at: date
|
|
82
|
+
verdict: Literal["approved", "needs_work"]
|
|
83
|
+
rationale: str
|
|
84
|
+
|
|
85
|
+
_text_fields = field_validator("reviewer", "rationale")(_text)
|
|
86
|
+
_date = field_validator("reviewed_at", mode="before")(_date_value)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class BenchmarkEvidence(StrictRecord):
|
|
90
|
+
candidate_id: str
|
|
91
|
+
canonical_id: str
|
|
92
|
+
domain: str
|
|
93
|
+
identity: Identity
|
|
94
|
+
researcher: str
|
|
95
|
+
task: str = ""
|
|
96
|
+
metric: str = ""
|
|
97
|
+
protocol: str = ""
|
|
98
|
+
usefulness: Usefulness
|
|
99
|
+
results: list[Result] = Field(default_factory=list)
|
|
100
|
+
review: Review | None = None
|
|
101
|
+
|
|
102
|
+
_text_fields = field_validator("candidate_id", "canonical_id", "domain", "researcher")(_text)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class ReferenceModel(StrictRecord):
|
|
106
|
+
model_id: str
|
|
107
|
+
organization: str
|
|
108
|
+
domain: str
|
|
109
|
+
cohort: Literal["frontier", "open"]
|
|
110
|
+
openness: Literal["closed", "open_weight", "open_source"]
|
|
111
|
+
source_url: HttpUrl
|
|
112
|
+
rationale: str
|
|
113
|
+
|
|
114
|
+
_text_fields = field_validator("model_id", "organization", "domain", "rationale")(_text)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class ModelReferenceSet(StrictRecord):
|
|
118
|
+
as_of: date
|
|
119
|
+
#: May be empty. With no reference models no result can qualify, so the
|
|
120
|
+
#: active set is empty: that fails closed, which is the point.
|
|
121
|
+
models: list[ReferenceModel] = Field(default_factory=list)
|
|
122
|
+
|
|
123
|
+
_date = field_validator("as_of", mode="before")(_date_value)
|
|
124
|
+
|
|
125
|
+
@field_validator("models")
|
|
126
|
+
@classmethod
|
|
127
|
+
def unique_model_domains(cls, value: list[ReferenceModel]) -> list[ReferenceModel]:
|
|
128
|
+
keys = [(model.model_id, model.domain) for model in value]
|
|
129
|
+
if len(keys) != len(set(keys)):
|
|
130
|
+
raise ValueError("duplicate reference model_id/domain")
|
|
131
|
+
return value
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
class EligibilityRow(StrictRecord):
|
|
135
|
+
candidate_id: str
|
|
136
|
+
canonical_id: str
|
|
137
|
+
status: Literal["active", "historical", "unverified", "alias"]
|
|
138
|
+
reasons: list[str]
|
|
139
|
+
accepted_results: list[dict[str, object]]
|
|
140
|
+
source_evidence: list[str]
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
class EligibilityReport(StrictRecord):
|
|
144
|
+
as_of: date
|
|
145
|
+
active_ids: list[str]
|
|
146
|
+
rows: list[EligibilityRow]
|
|
147
|
+
mechanical_validation_note: str = (
|
|
148
|
+
"Mechanical validation does not prove source truth; reviewer approval is required."
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _age_days(as_of: date, evidence_date: date) -> int:
|
|
153
|
+
return (as_of - evidence_date).days
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def evaluate(
|
|
157
|
+
evidence: BenchmarkEvidence, references: ModelReferenceSet, as_of: date
|
|
158
|
+
) -> EligibilityRow:
|
|
159
|
+
"""Evaluate one already parsed candidate against the reference set."""
|
|
160
|
+
review_ok = evidence.review is not None and evidence.review.verdict == "approved"
|
|
161
|
+
identity_known = bool(evidence.identity.rationale.strip())
|
|
162
|
+
reasons: list[str] = []
|
|
163
|
+
if not review_ok:
|
|
164
|
+
reasons.append("review is not approved")
|
|
165
|
+
if evidence.review is not None and evidence.review.reviewer == evidence.researcher:
|
|
166
|
+
reasons.append("reviewer must be different from researcher")
|
|
167
|
+
if not identity_known:
|
|
168
|
+
reasons.append("identity is unknown")
|
|
169
|
+
if not all((evidence.task.strip(), evidence.metric.strip(), evidence.protocol.strip())):
|
|
170
|
+
reasons.append("task, metric, and protocol are incomplete")
|
|
171
|
+
if evidence.usefulness.verdict == "unknown":
|
|
172
|
+
reasons.append("usefulness is unknown")
|
|
173
|
+
if evidence.usefulness.verdict == "saturated":
|
|
174
|
+
reasons.append("benchmark is saturated")
|
|
175
|
+
if evidence.usefulness.verdict != "useful" or not evidence.usefulness.rationale.strip():
|
|
176
|
+
reasons.append("usefulness is not established")
|
|
177
|
+
|
|
178
|
+
reference_fresh = references.as_of <= as_of and _age_days(as_of, references.as_of) <= 30
|
|
179
|
+
if not reference_fresh:
|
|
180
|
+
reasons.append("reference set is older than 30 days or future-dated")
|
|
181
|
+
|
|
182
|
+
reference_by_id = {(model.model_id, model.domain): model for model in references.models}
|
|
183
|
+
valid: list[tuple[Result, ReferenceModel]] = []
|
|
184
|
+
stale: list[tuple[Result, ReferenceModel]] = []
|
|
185
|
+
invalid: list[str] = []
|
|
186
|
+
for result in evidence.results:
|
|
187
|
+
model = reference_by_id.get((result.model_id, evidence.domain))
|
|
188
|
+
if model is None:
|
|
189
|
+
invalid.append(
|
|
190
|
+
f"model {result.model_id} is absent from reference set for domain {evidence.domain}"
|
|
191
|
+
)
|
|
192
|
+
continue
|
|
193
|
+
if result.evidence_date > as_of:
|
|
194
|
+
invalid.append(f"result {result.model_id} has future evidence_date")
|
|
195
|
+
continue
|
|
196
|
+
if result.verified_at < result.evidence_date or result.verified_at > as_of:
|
|
197
|
+
invalid.append(f"result {result.model_id} has invalid verified_at")
|
|
198
|
+
continue
|
|
199
|
+
if (
|
|
200
|
+
evidence.review is None
|
|
201
|
+
or evidence.review.reviewed_at < result.verified_at
|
|
202
|
+
or evidence.review.reviewed_at > as_of
|
|
203
|
+
):
|
|
204
|
+
invalid.append(f"result {result.model_id} predates review or has future review")
|
|
205
|
+
continue
|
|
206
|
+
if model.cohort == "open" and model.openness not in ("open_weight", "open_source"):
|
|
207
|
+
invalid.append(f"open reference model {result.model_id} is not open")
|
|
208
|
+
continue
|
|
209
|
+
(stale if _age_days(as_of, result.evidence_date) > 60 else valid).append((result, model))
|
|
210
|
+
|
|
211
|
+
reasons.extend(invalid)
|
|
212
|
+
|
|
213
|
+
def groups(
|
|
214
|
+
items: list[tuple[Result, ReferenceModel]],
|
|
215
|
+
) -> dict[tuple[str, str, str], list[tuple[Result, ReferenceModel]]]:
|
|
216
|
+
grouped: dict[tuple[str, str, str], list[tuple[Result, ReferenceModel]]] = {}
|
|
217
|
+
for result, model in items:
|
|
218
|
+
key = (result.benchmark_version, result.configuration, result.unit)
|
|
219
|
+
grouped.setdefault(key, []).append((result, model))
|
|
220
|
+
return grouped
|
|
221
|
+
|
|
222
|
+
def qualifying(
|
|
223
|
+
items: list[tuple[Result, ReferenceModel]],
|
|
224
|
+
) -> list[tuple[Result, ReferenceModel]]:
|
|
225
|
+
for group in groups(items).values():
|
|
226
|
+
cohorts = {model.cohort for _, model in group}
|
|
227
|
+
orgs = {model.organization for _, model in group}
|
|
228
|
+
if {"frontier", "open"} <= cohorts and len(orgs) >= 2:
|
|
229
|
+
return group
|
|
230
|
+
return []
|
|
231
|
+
|
|
232
|
+
accepted = qualifying(valid)
|
|
233
|
+
all_coverage = qualifying(valid + stale)
|
|
234
|
+
historical = all_coverage if not accepted or evidence.usefulness.verdict == "saturated" else []
|
|
235
|
+
substantive = all(
|
|
236
|
+
value.strip() for value in (evidence.task, evidence.metric, evidence.protocol)
|
|
237
|
+
)
|
|
238
|
+
review_dates_ok = evidence.review is not None and evidence.review.reviewed_at <= as_of
|
|
239
|
+
full_identity_review = (
|
|
240
|
+
review_ok
|
|
241
|
+
and review_dates_ok
|
|
242
|
+
and identity_known
|
|
243
|
+
and (evidence.review is not None and evidence.review.reviewer != evidence.researcher)
|
|
244
|
+
)
|
|
245
|
+
if evidence.candidate_id != evidence.canonical_id:
|
|
246
|
+
status = "alias" if full_identity_review and reference_fresh else "unverified"
|
|
247
|
+
elif (
|
|
248
|
+
accepted
|
|
249
|
+
and full_identity_review
|
|
250
|
+
and substantive
|
|
251
|
+
and evidence.usefulness.verdict == "useful"
|
|
252
|
+
and reference_fresh
|
|
253
|
+
):
|
|
254
|
+
status = "active"
|
|
255
|
+
elif (
|
|
256
|
+
historical
|
|
257
|
+
and full_identity_review
|
|
258
|
+
and substantive
|
|
259
|
+
and evidence.usefulness.verdict in ("useful", "saturated")
|
|
260
|
+
and reference_fresh
|
|
261
|
+
):
|
|
262
|
+
status = "historical"
|
|
263
|
+
else:
|
|
264
|
+
status = "unverified"
|
|
265
|
+
if not accepted and status == "unverified":
|
|
266
|
+
reasons.append("no qualifying current frontier/open coverage from different organizations")
|
|
267
|
+
if historical and not accepted:
|
|
268
|
+
reasons.append("qualifying coverage requires evidence older than 60 days")
|
|
269
|
+
if status == "alias":
|
|
270
|
+
reasons = [
|
|
271
|
+
"independently reviewed alias; retain canonical benchmark and protocol distinctions"
|
|
272
|
+
]
|
|
273
|
+
if status == "active":
|
|
274
|
+
reasons.append("qualifying current coverage")
|
|
275
|
+
accepted_for_report = accepted or historical
|
|
276
|
+
if status == "unverified" or status == "alias":
|
|
277
|
+
accepted_for_report = []
|
|
278
|
+
attempted_sources = {
|
|
279
|
+
str(evidence.identity.url),
|
|
280
|
+
str(evidence.usefulness.source_url),
|
|
281
|
+
*(str(r.source_url) for r in evidence.results),
|
|
282
|
+
}
|
|
283
|
+
return EligibilityRow(
|
|
284
|
+
candidate_id=evidence.candidate_id,
|
|
285
|
+
canonical_id=evidence.canonical_id,
|
|
286
|
+
status=status,
|
|
287
|
+
reasons=sorted(set(reasons)),
|
|
288
|
+
accepted_results=[
|
|
289
|
+
{
|
|
290
|
+
"model_id": r.model_id,
|
|
291
|
+
"benchmark_version": r.benchmark_version,
|
|
292
|
+
"configuration": r.configuration,
|
|
293
|
+
"unit": r.unit,
|
|
294
|
+
"score": r.score,
|
|
295
|
+
"evidence_date": r.evidence_date.isoformat(),
|
|
296
|
+
"date_type": r.date_type,
|
|
297
|
+
"verified_at": r.verified_at.isoformat(),
|
|
298
|
+
"source_kind": r.source_kind,
|
|
299
|
+
"source_url": str(r.source_url),
|
|
300
|
+
}
|
|
301
|
+
for r, _ in accepted_for_report
|
|
302
|
+
],
|
|
303
|
+
source_evidence=sorted(attempted_sources),
|
|
304
|
+
)
|