modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
@@ -0,0 +1,304 @@
1
+ """Strict records and the mechanical eligibility gate for benchmark evidence.
2
+
3
+ This validates the shape, dates, and coverage of submitted evidence. It does
4
+ not establish that a source is truthful; that remains the reviewer's job.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from datetime import date, datetime
10
+ from math import isfinite
11
+ from typing import Literal
12
+
13
+ from pydantic import BaseModel, ConfigDict, Field, HttpUrl, field_validator
14
+
15
+
16
+ def _text(value: str) -> str:
17
+ if not isinstance(value, str) or not value.strip():
18
+ raise ValueError("must be a nonblank string")
19
+ return value.strip()
20
+
21
+
22
+ def _date_value(value: object) -> object:
23
+ if isinstance(value, date) and not isinstance(value, datetime):
24
+ return value
25
+ if isinstance(value, str):
26
+ try:
27
+ parsed = date.fromisoformat(value)
28
+ except ValueError as exc:
29
+ raise ValueError("must be an ISO date YYYY-MM-DD") from exc
30
+ if parsed.isoformat() != value:
31
+ raise ValueError("must be an exact ISO date YYYY-MM-DD")
32
+ return parsed
33
+ raise ValueError("must be an ISO date YYYY-MM-DD")
34
+
35
+
36
+ class StrictRecord(BaseModel):
37
+ model_config = ConfigDict(extra="forbid")
38
+
39
+
40
+ class Identity(StrictRecord):
41
+ url: HttpUrl
42
+ rationale: str = ""
43
+
44
+
45
+ class Usefulness(StrictRecord):
46
+ verdict: Literal["useful", "saturated", "unknown"]
47
+ rationale: str
48
+ source_url: HttpUrl
49
+
50
+ _rationale = field_validator("rationale")(_text)
51
+
52
+
53
+ class Result(StrictRecord):
54
+ model_id: str
55
+ benchmark_version: str
56
+ configuration: str
57
+ score: float
58
+ unit: str
59
+ source_url: HttpUrl
60
+ evidence_date: date
61
+ date_type: Literal["evaluated", "published"]
62
+ verified_at: date
63
+ source_kind: Literal["benchmark_author", "independent_evaluator", "model_provider"]
64
+
65
+ _text_fields = field_validator("model_id", "benchmark_version", "configuration", "unit")(_text)
66
+ _dates = field_validator("evidence_date", "verified_at", mode="before")(_date_value)
67
+
68
+ @field_validator("score", mode="before")
69
+ @classmethod
70
+ def finite_score(cls, value: object) -> float:
71
+ if isinstance(value, (bool, str)) or not isinstance(value, (int, float)):
72
+ raise ValueError("score must be numeric")
73
+ value = float(value)
74
+ if not isfinite(value):
75
+ raise ValueError("score must be finite")
76
+ return value
77
+
78
+
79
+ class Review(StrictRecord):
80
+ reviewer: str
81
+ reviewed_at: date
82
+ verdict: Literal["approved", "needs_work"]
83
+ rationale: str
84
+
85
+ _text_fields = field_validator("reviewer", "rationale")(_text)
86
+ _date = field_validator("reviewed_at", mode="before")(_date_value)
87
+
88
+
89
+ class BenchmarkEvidence(StrictRecord):
90
+ candidate_id: str
91
+ canonical_id: str
92
+ domain: str
93
+ identity: Identity
94
+ researcher: str
95
+ task: str = ""
96
+ metric: str = ""
97
+ protocol: str = ""
98
+ usefulness: Usefulness
99
+ results: list[Result] = Field(default_factory=list)
100
+ review: Review | None = None
101
+
102
+ _text_fields = field_validator("candidate_id", "canonical_id", "domain", "researcher")(_text)
103
+
104
+
105
+ class ReferenceModel(StrictRecord):
106
+ model_id: str
107
+ organization: str
108
+ domain: str
109
+ cohort: Literal["frontier", "open"]
110
+ openness: Literal["closed", "open_weight", "open_source"]
111
+ source_url: HttpUrl
112
+ rationale: str
113
+
114
+ _text_fields = field_validator("model_id", "organization", "domain", "rationale")(_text)
115
+
116
+
117
+ class ModelReferenceSet(StrictRecord):
118
+ as_of: date
119
+ #: May be empty. With no reference models no result can qualify, so the
120
+ #: active set is empty: that fails closed, which is the point.
121
+ models: list[ReferenceModel] = Field(default_factory=list)
122
+
123
+ _date = field_validator("as_of", mode="before")(_date_value)
124
+
125
+ @field_validator("models")
126
+ @classmethod
127
+ def unique_model_domains(cls, value: list[ReferenceModel]) -> list[ReferenceModel]:
128
+ keys = [(model.model_id, model.domain) for model in value]
129
+ if len(keys) != len(set(keys)):
130
+ raise ValueError("duplicate reference model_id/domain")
131
+ return value
132
+
133
+
134
+ class EligibilityRow(StrictRecord):
135
+ candidate_id: str
136
+ canonical_id: str
137
+ status: Literal["active", "historical", "unverified", "alias"]
138
+ reasons: list[str]
139
+ accepted_results: list[dict[str, object]]
140
+ source_evidence: list[str]
141
+
142
+
143
+ class EligibilityReport(StrictRecord):
144
+ as_of: date
145
+ active_ids: list[str]
146
+ rows: list[EligibilityRow]
147
+ mechanical_validation_note: str = (
148
+ "Mechanical validation does not prove source truth; reviewer approval is required."
149
+ )
150
+
151
+
152
+ def _age_days(as_of: date, evidence_date: date) -> int:
153
+ return (as_of - evidence_date).days
154
+
155
+
156
+ def evaluate(
157
+ evidence: BenchmarkEvidence, references: ModelReferenceSet, as_of: date
158
+ ) -> EligibilityRow:
159
+ """Evaluate one already parsed candidate against the reference set."""
160
+ review_ok = evidence.review is not None and evidence.review.verdict == "approved"
161
+ identity_known = bool(evidence.identity.rationale.strip())
162
+ reasons: list[str] = []
163
+ if not review_ok:
164
+ reasons.append("review is not approved")
165
+ if evidence.review is not None and evidence.review.reviewer == evidence.researcher:
166
+ reasons.append("reviewer must be different from researcher")
167
+ if not identity_known:
168
+ reasons.append("identity is unknown")
169
+ if not all((evidence.task.strip(), evidence.metric.strip(), evidence.protocol.strip())):
170
+ reasons.append("task, metric, and protocol are incomplete")
171
+ if evidence.usefulness.verdict == "unknown":
172
+ reasons.append("usefulness is unknown")
173
+ if evidence.usefulness.verdict == "saturated":
174
+ reasons.append("benchmark is saturated")
175
+ if evidence.usefulness.verdict != "useful" or not evidence.usefulness.rationale.strip():
176
+ reasons.append("usefulness is not established")
177
+
178
+ reference_fresh = references.as_of <= as_of and _age_days(as_of, references.as_of) <= 30
179
+ if not reference_fresh:
180
+ reasons.append("reference set is older than 30 days or future-dated")
181
+
182
+ reference_by_id = {(model.model_id, model.domain): model for model in references.models}
183
+ valid: list[tuple[Result, ReferenceModel]] = []
184
+ stale: list[tuple[Result, ReferenceModel]] = []
185
+ invalid: list[str] = []
186
+ for result in evidence.results:
187
+ model = reference_by_id.get((result.model_id, evidence.domain))
188
+ if model is None:
189
+ invalid.append(
190
+ f"model {result.model_id} is absent from reference set for domain {evidence.domain}"
191
+ )
192
+ continue
193
+ if result.evidence_date > as_of:
194
+ invalid.append(f"result {result.model_id} has future evidence_date")
195
+ continue
196
+ if result.verified_at < result.evidence_date or result.verified_at > as_of:
197
+ invalid.append(f"result {result.model_id} has invalid verified_at")
198
+ continue
199
+ if (
200
+ evidence.review is None
201
+ or evidence.review.reviewed_at < result.verified_at
202
+ or evidence.review.reviewed_at > as_of
203
+ ):
204
+ invalid.append(f"result {result.model_id} predates review or has future review")
205
+ continue
206
+ if model.cohort == "open" and model.openness not in ("open_weight", "open_source"):
207
+ invalid.append(f"open reference model {result.model_id} is not open")
208
+ continue
209
+ (stale if _age_days(as_of, result.evidence_date) > 60 else valid).append((result, model))
210
+
211
+ reasons.extend(invalid)
212
+
213
+ def groups(
214
+ items: list[tuple[Result, ReferenceModel]],
215
+ ) -> dict[tuple[str, str, str], list[tuple[Result, ReferenceModel]]]:
216
+ grouped: dict[tuple[str, str, str], list[tuple[Result, ReferenceModel]]] = {}
217
+ for result, model in items:
218
+ key = (result.benchmark_version, result.configuration, result.unit)
219
+ grouped.setdefault(key, []).append((result, model))
220
+ return grouped
221
+
222
+ def qualifying(
223
+ items: list[tuple[Result, ReferenceModel]],
224
+ ) -> list[tuple[Result, ReferenceModel]]:
225
+ for group in groups(items).values():
226
+ cohorts = {model.cohort for _, model in group}
227
+ orgs = {model.organization for _, model in group}
228
+ if {"frontier", "open"} <= cohorts and len(orgs) >= 2:
229
+ return group
230
+ return []
231
+
232
+ accepted = qualifying(valid)
233
+ all_coverage = qualifying(valid + stale)
234
+ historical = all_coverage if not accepted or evidence.usefulness.verdict == "saturated" else []
235
+ substantive = all(
236
+ value.strip() for value in (evidence.task, evidence.metric, evidence.protocol)
237
+ )
238
+ review_dates_ok = evidence.review is not None and evidence.review.reviewed_at <= as_of
239
+ full_identity_review = (
240
+ review_ok
241
+ and review_dates_ok
242
+ and identity_known
243
+ and (evidence.review is not None and evidence.review.reviewer != evidence.researcher)
244
+ )
245
+ if evidence.candidate_id != evidence.canonical_id:
246
+ status = "alias" if full_identity_review and reference_fresh else "unverified"
247
+ elif (
248
+ accepted
249
+ and full_identity_review
250
+ and substantive
251
+ and evidence.usefulness.verdict == "useful"
252
+ and reference_fresh
253
+ ):
254
+ status = "active"
255
+ elif (
256
+ historical
257
+ and full_identity_review
258
+ and substantive
259
+ and evidence.usefulness.verdict in ("useful", "saturated")
260
+ and reference_fresh
261
+ ):
262
+ status = "historical"
263
+ else:
264
+ status = "unverified"
265
+ if not accepted and status == "unverified":
266
+ reasons.append("no qualifying current frontier/open coverage from different organizations")
267
+ if historical and not accepted:
268
+ reasons.append("qualifying coverage requires evidence older than 60 days")
269
+ if status == "alias":
270
+ reasons = [
271
+ "independently reviewed alias; retain canonical benchmark and protocol distinctions"
272
+ ]
273
+ if status == "active":
274
+ reasons.append("qualifying current coverage")
275
+ accepted_for_report = accepted or historical
276
+ if status == "unverified" or status == "alias":
277
+ accepted_for_report = []
278
+ attempted_sources = {
279
+ str(evidence.identity.url),
280
+ str(evidence.usefulness.source_url),
281
+ *(str(r.source_url) for r in evidence.results),
282
+ }
283
+ return EligibilityRow(
284
+ candidate_id=evidence.candidate_id,
285
+ canonical_id=evidence.canonical_id,
286
+ status=status,
287
+ reasons=sorted(set(reasons)),
288
+ accepted_results=[
289
+ {
290
+ "model_id": r.model_id,
291
+ "benchmark_version": r.benchmark_version,
292
+ "configuration": r.configuration,
293
+ "unit": r.unit,
294
+ "score": r.score,
295
+ "evidence_date": r.evidence_date.isoformat(),
296
+ "date_type": r.date_type,
297
+ "verified_at": r.verified_at.isoformat(),
298
+ "source_kind": r.source_kind,
299
+ "source_url": str(r.source_url),
300
+ }
301
+ for r, _ in accepted_for_report
302
+ ],
303
+ source_evidence=sorted(attempted_sources),
304
+ )