modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
schema/card.py
ADDED
|
@@ -0,0 +1,1463 @@
|
|
|
1
|
+
"""ModelSpec Universal Model Card Schema — V3
|
|
2
|
+
|
|
3
|
+
Pydantic models for every section of the model intelligence card.
|
|
4
|
+
This is the source of truth. The YAML template, graph ingestion,
|
|
5
|
+
API responses, and CLI output all derive from these models.
|
|
6
|
+
|
|
7
|
+
Usage:
|
|
8
|
+
from schema.card import ModelCard
|
|
9
|
+
card = ModelCard.from_yaml("models/qwen/qwen3-30b-a3b.md")
|
|
10
|
+
card.validate()
|
|
11
|
+
print(card.applicable_field_coverage)
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from collections.abc import Iterable
|
|
17
|
+
from datetime import date, datetime
|
|
18
|
+
from functools import lru_cache
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any, ClassVar, Literal
|
|
21
|
+
|
|
22
|
+
import yaml
|
|
23
|
+
from pydantic import BaseModel, Field, computed_field, field_validator, model_validator
|
|
24
|
+
|
|
25
|
+
from .applicability import FIELD_RULES, model_types
|
|
26
|
+
from .enums import (
|
|
27
|
+
ArchitectureType,
|
|
28
|
+
AttentionType,
|
|
29
|
+
BaseModelRelation,
|
|
30
|
+
BenchmarkCategory,
|
|
31
|
+
ConfidenceLevel,
|
|
32
|
+
DisclosureState,
|
|
33
|
+
EUAIActRisk,
|
|
34
|
+
EvalStatus,
|
|
35
|
+
LicenseType,
|
|
36
|
+
ModelStatus,
|
|
37
|
+
ModelType,
|
|
38
|
+
Modality,
|
|
39
|
+
OrgType,
|
|
40
|
+
PlatformCategory,
|
|
41
|
+
PositionalEncoding,
|
|
42
|
+
QuantFormat,
|
|
43
|
+
ResistanceLevel,
|
|
44
|
+
RiskTier,
|
|
45
|
+
Tier,
|
|
46
|
+
TokenizerType,
|
|
47
|
+
TrainingMethod,
|
|
48
|
+
UsePermission,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
# ═══════════════════════════════════════════════════════════════
|
|
53
|
+
# Section 1: Identity
|
|
54
|
+
# ═══════════════════════════════════════════════════════════════
|
|
55
|
+
|
|
56
|
+
class Identity(BaseModel):
|
|
57
|
+
model_id: str = Field(..., description="Canonical ID: provider/model-name")
|
|
58
|
+
display_name: str
|
|
59
|
+
provider: str = Field(..., description="Provider slug (lowercase)")
|
|
60
|
+
provider_display: str = ""
|
|
61
|
+
family: str = ""
|
|
62
|
+
version: str = ""
|
|
63
|
+
release_date: str = ""
|
|
64
|
+
last_updated: str = ""
|
|
65
|
+
status: ModelStatus = ModelStatus.ACTIVE
|
|
66
|
+
model_type: ModelType | None = None
|
|
67
|
+
model_subtypes: list[ModelType] = []
|
|
68
|
+
tags: list[str] = []
|
|
69
|
+
pipeline_tag: str = ""
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# ═══════════════════════════════════════════════════════════════
|
|
73
|
+
# Section 2: Architecture
|
|
74
|
+
# ═══════════════════════════════════════════════════════════════
|
|
75
|
+
|
|
76
|
+
class Architecture(BaseModel):
|
|
77
|
+
type: ArchitectureType | None = None
|
|
78
|
+
total_parameters: int | None = None
|
|
79
|
+
#: How `total_parameters` was obtained. `safetensors` is an exact Hub
|
|
80
|
+
#: count; values starting `model_card_published:` are a README figure.
|
|
81
|
+
#: Empty means unknown — either still null, or a legacy fill.
|
|
82
|
+
total_parameters_source: str = ""
|
|
83
|
+
active_parameters: int | None = None
|
|
84
|
+
num_experts: int | None = None
|
|
85
|
+
experts_per_token: int | None = None
|
|
86
|
+
num_layers: int | None = None
|
|
87
|
+
hidden_size: int | None = None
|
|
88
|
+
intermediate_size: int | None = None
|
|
89
|
+
attention_type: AttentionType | None = None
|
|
90
|
+
num_attention_heads: int | None = None
|
|
91
|
+
num_kv_heads: int | None = None
|
|
92
|
+
positional_encoding: PositionalEncoding | None = None
|
|
93
|
+
rope_theta: float | None = None
|
|
94
|
+
vocab_size: int | None = None
|
|
95
|
+
tokenizer_type: TokenizerType | None = None
|
|
96
|
+
embedding_dimensions: int | None = None
|
|
97
|
+
activation_function: str = ""
|
|
98
|
+
precision_native: str = ""
|
|
99
|
+
flash_attention: bool | None = None
|
|
100
|
+
tie_word_embeddings: bool | None = None
|
|
101
|
+
sliding_window_size: int | None = None
|
|
102
|
+
# Vision encoder
|
|
103
|
+
vision_encoder: str = ""
|
|
104
|
+
vision_resolution_max: str = ""
|
|
105
|
+
vision_patch_size: int | None = None
|
|
106
|
+
# Diffusion
|
|
107
|
+
diffusion_scheduler: str = ""
|
|
108
|
+
diffusion_steps_default: int | None = None
|
|
109
|
+
vae_type: str = ""
|
|
110
|
+
|
|
111
|
+
@model_validator(mode="after")
|
|
112
|
+
def _active_cannot_exceed_total(self) -> Architecture:
|
|
113
|
+
"""A stored active count larger than the stored total is a catalogue bug.
|
|
114
|
+
|
|
115
|
+
That is how Mixtral-8x7B and GLM-4.5 were filed as 7B / 9B and told they
|
|
116
|
+
fitted on hardware that cannot hold the weights. Null on either side
|
|
117
|
+
stays legal — unknown is not a contradiction.
|
|
118
|
+
"""
|
|
119
|
+
if (
|
|
120
|
+
self.active_parameters is not None
|
|
121
|
+
and self.total_parameters is not None
|
|
122
|
+
and self.active_parameters > self.total_parameters
|
|
123
|
+
):
|
|
124
|
+
raise ValueError(
|
|
125
|
+
f"active_parameters ({self.active_parameters}) exceeds "
|
|
126
|
+
f"total_parameters ({self.total_parameters})"
|
|
127
|
+
)
|
|
128
|
+
return self
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
# ═══════════════════════════════════════════════════════════════
|
|
132
|
+
# Section 3: Lineage
|
|
133
|
+
# ═══════════════════════════════════════════════════════════════
|
|
134
|
+
|
|
135
|
+
class Lineage(BaseModel):
|
|
136
|
+
base_model: str = ""
|
|
137
|
+
base_model_relation: BaseModelRelation | None = None
|
|
138
|
+
merge_models: list[str] = []
|
|
139
|
+
adapter_type: str = ""
|
|
140
|
+
adapter_rank: int | None = None
|
|
141
|
+
training_datasets: list[str] = []
|
|
142
|
+
training_data_tokens: int | None = None
|
|
143
|
+
training_data_cutoff: str = ""
|
|
144
|
+
training_compute_flops: float | None = None
|
|
145
|
+
training_hardware: str = ""
|
|
146
|
+
training_time: str = ""
|
|
147
|
+
training_cost_estimate: str = ""
|
|
148
|
+
training_method: TrainingMethod | None = None
|
|
149
|
+
co2_emissions_kg: float | None = None
|
|
150
|
+
co2_source: str = ""
|
|
151
|
+
energy_kwh: float | None = None
|
|
152
|
+
library_name: str = ""
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
# ═══════════════════════════════════════════════════════════════
|
|
156
|
+
# Section 4: Licensing
|
|
157
|
+
# ═══════════════════════════════════════════════════════════════
|
|
158
|
+
|
|
159
|
+
#: Source kinds a policy determination may cite. `legacy-import` is the one
|
|
160
|
+
#: kind that is *not* evidence: see `PolicySource` below.
|
|
161
|
+
PolicySourceKind = Literal[
|
|
162
|
+
"license",
|
|
163
|
+
"terms_of_service",
|
|
164
|
+
"acceptable_use_policy",
|
|
165
|
+
"provider_documentation",
|
|
166
|
+
"provider_statement",
|
|
167
|
+
"legacy-import",
|
|
168
|
+
]
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
class PolicySource(BaseModel):
|
|
172
|
+
"""The document a policy determination was read from, and the day it was read.
|
|
173
|
+
|
|
174
|
+
A policy answer is only worth anything if the reader can go and check it,
|
|
175
|
+
and licences are rewritten without notice — so the date is as load-bearing
|
|
176
|
+
as the URL. This mirrors `BenchmarkEvidence`: everything needed to recheck
|
|
177
|
+
the claim, or it is not a claim.
|
|
178
|
+
|
|
179
|
+
`legacy-import` is the single exception and it is deliberately ugly. Eight
|
|
180
|
+
cards carried `commercial_use: true` with no citation from before this
|
|
181
|
+
shape existed. Discarding those values would lose information; dressing
|
|
182
|
+
them up with a plausible licence URL would manufacture evidence that was
|
|
183
|
+
never read. So they keep the value and carry a source that says, in the
|
|
184
|
+
published JSON, that nobody cited anything — the same thing
|
|
185
|
+
`evidence_basis: unverified-legacy` says about benchmark scores. It is
|
|
186
|
+
frozen to those eight by
|
|
187
|
+
`tests/test_policy_shape.py::test_legacy_import_is_frozen_to_the_migrated_eight`;
|
|
188
|
+
a ninth card cannot quietly use it.
|
|
189
|
+
"""
|
|
190
|
+
|
|
191
|
+
kind: PolicySourceKind
|
|
192
|
+
url: str = ""
|
|
193
|
+
#: The day the document at `url` was read. ISO `YYYY-MM-DD`.
|
|
194
|
+
read_on: str = ""
|
|
195
|
+
#: Optional. The operative clause, verbatim and short, so a reader can see
|
|
196
|
+
#: what the determination was made from without refetching the document.
|
|
197
|
+
quote: str = ""
|
|
198
|
+
|
|
199
|
+
@model_validator(mode="after")
|
|
200
|
+
def _evidence_or_an_admission(self) -> PolicySource:
|
|
201
|
+
if self.kind == "legacy-import":
|
|
202
|
+
if self.url or self.read_on or self.quote:
|
|
203
|
+
raise ValueError(
|
|
204
|
+
"a legacy-import source is the admission that nothing was "
|
|
205
|
+
"cited; it cannot carry a url, a read date or a quote. "
|
|
206
|
+
"If a document was actually read, cite it with a real kind."
|
|
207
|
+
)
|
|
208
|
+
return self
|
|
209
|
+
if not self.url.startswith(("http://", "https://")):
|
|
210
|
+
raise ValueError(
|
|
211
|
+
"url must be the document the determination was read from; "
|
|
212
|
+
"a policy answer without one cannot be rechecked"
|
|
213
|
+
)
|
|
214
|
+
try:
|
|
215
|
+
date.fromisoformat(self.read_on)
|
|
216
|
+
except ValueError as exc:
|
|
217
|
+
raise ValueError(
|
|
218
|
+
f"read_on must be the exact ISO date the document was read, "
|
|
219
|
+
f"got {self.read_on!r}. Licence terms change; an undated "
|
|
220
|
+
f"reading cannot be trusted later."
|
|
221
|
+
) from exc
|
|
222
|
+
return self
|
|
223
|
+
|
|
224
|
+
@property
|
|
225
|
+
def is_evidence(self) -> bool:
|
|
226
|
+
"""False for `legacy-import`, which is a value with no citation."""
|
|
227
|
+
return self.kind != "legacy-import"
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
#: The `commercial_use` values that assert something about the world, and so
|
|
231
|
+
#: require a source. `UNSPECIFIED` and `WITHHELD` assert nothing about the
|
|
232
|
+
#: licence — they describe this file's contents.
|
|
233
|
+
DETERMINED_PERMISSIONS = frozenset({
|
|
234
|
+
UsePermission.ALLOWED,
|
|
235
|
+
UsePermission.RESTRICTED,
|
|
236
|
+
UsePermission.PROHIBITED,
|
|
237
|
+
})
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class Licensing(BaseModel):
|
|
241
|
+
open_weights: bool = False
|
|
242
|
+
license_type: LicenseType | None = None
|
|
243
|
+
license_url: str = ""
|
|
244
|
+
tos_url: str = ""
|
|
245
|
+
acceptable_use_policy_url: str = ""
|
|
246
|
+
not_for_all_audiences: bool = False
|
|
247
|
+
#: Was `bool | None` until MODEL-77, which could not express the answer for
|
|
248
|
+
#: 169 cards whose licence grants commercial use *up to a threshold*
|
|
249
|
+
#: (llama-community, gemma, deepseek). That is neither true nor false; it
|
|
250
|
+
#: is `restricted`, with the threshold in `commercial_use_conditions`.
|
|
251
|
+
commercial_use: UsePermission = UsePermission.UNSPECIFIED
|
|
252
|
+
#: Required whenever `commercial_use` is a determination; forbidden when it
|
|
253
|
+
#: is not, so an empty value can never look sourced.
|
|
254
|
+
commercial_use_source: PolicySource | None = None
|
|
255
|
+
#: Where a `restricted` grant's condition lives: one short line stating the
|
|
256
|
+
#: threshold or carve-out ("free below 700M MAU; a licence is required
|
|
257
|
+
#: above it"), not prose buried in the card body where no consumer reads it.
|
|
258
|
+
commercial_use_conditions: str = ""
|
|
259
|
+
defense_use: UsePermission = UsePermission.UNSPECIFIED
|
|
260
|
+
government_use: UsePermission = UsePermission.UNSPECIFIED
|
|
261
|
+
medical_use: UsePermission = UsePermission.UNSPECIFIED
|
|
262
|
+
academic_use: UsePermission = UsePermission.UNSPECIFIED
|
|
263
|
+
geographic_restrictions: list[str] = []
|
|
264
|
+
export_control_notes: str = ""
|
|
265
|
+
origin_country: str = ""
|
|
266
|
+
origin_org_type: OrgType | None = None
|
|
267
|
+
|
|
268
|
+
@field_validator("commercial_use", mode="before")
|
|
269
|
+
@classmethod
|
|
270
|
+
def _reject_the_old_boolean(cls, value: Any) -> Any:
|
|
271
|
+
"""A pre-MODEL-77 card fails loudly rather than being guessed at.
|
|
272
|
+
|
|
273
|
+
`True`/`False` are not silently mapped: `False` in particular could
|
|
274
|
+
have meant "prohibited" or "restricted in a way the author could not
|
|
275
|
+
express", and picking one would invent a determination.
|
|
276
|
+
"""
|
|
277
|
+
if isinstance(value, bool):
|
|
278
|
+
raise ValueError(
|
|
279
|
+
"commercial_use is a UsePermission since MODEL-77, not a bool. "
|
|
280
|
+
"true became 'allowed'; false is ambiguous between 'prohibited' "
|
|
281
|
+
"and 'restricted' and must be re-read from the licence."
|
|
282
|
+
)
|
|
283
|
+
if value is None:
|
|
284
|
+
raise ValueError(
|
|
285
|
+
"commercial_use is no longer nullable: use 'unspecified' for "
|
|
286
|
+
"not-yet-researched, or 'withheld' for determined-not-published."
|
|
287
|
+
)
|
|
288
|
+
return value
|
|
289
|
+
|
|
290
|
+
@model_validator(mode="after")
|
|
291
|
+
def _a_determination_carries_its_source(self) -> Licensing:
|
|
292
|
+
determined = self.commercial_use in DETERMINED_PERMISSIONS
|
|
293
|
+
if determined and self.commercial_use_source is None:
|
|
294
|
+
raise ValueError(
|
|
295
|
+
f"commercial_use is {self.commercial_use.value!r} with no "
|
|
296
|
+
"commercial_use_source. Standing rule 1: an uncited policy "
|
|
297
|
+
"answer is not an answer."
|
|
298
|
+
)
|
|
299
|
+
if not determined and self.commercial_use_source is not None:
|
|
300
|
+
raise ValueError(
|
|
301
|
+
f"commercial_use is {self.commercial_use.value!r} but carries a "
|
|
302
|
+
"commercial_use_source. Nothing has been published to cite; a "
|
|
303
|
+
"withheld determination keeps its source in the enrichment "
|
|
304
|
+
"record, not on the public card."
|
|
305
|
+
)
|
|
306
|
+
if self.commercial_use is UsePermission.RESTRICTED and not self.commercial_use_conditions.strip():
|
|
307
|
+
raise ValueError(
|
|
308
|
+
"commercial_use is 'restricted' with no commercial_use_conditions. "
|
|
309
|
+
"'Restricted' without the restriction is unusable: it is the "
|
|
310
|
+
"condition that tells a reader whether they are inside it."
|
|
311
|
+
)
|
|
312
|
+
if self.commercial_use is not UsePermission.RESTRICTED and self.commercial_use_conditions.strip():
|
|
313
|
+
raise ValueError(
|
|
314
|
+
f"commercial_use is {self.commercial_use.value!r} but carries "
|
|
315
|
+
"commercial_use_conditions. Conditions belong to a restricted "
|
|
316
|
+
"grant; anywhere else they contradict the value."
|
|
317
|
+
)
|
|
318
|
+
return self
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
# ═══════════════════════════════════════════════════════════════
|
|
322
|
+
# Section 5: Modalities
|
|
323
|
+
# ═══════════════════════════════════════════════════════════════
|
|
324
|
+
#
|
|
325
|
+
# Nested modality and capability models declare ``__applicable_model_types__``.
|
|
326
|
+
# ``ModelCard.applicable_field_coverage`` counts a nested section only when the
|
|
327
|
+
# card's ``model_type`` (or a ``model_subtype``) is in that set. Sections with
|
|
328
|
+
# no declaration count for every type. The field *sets* are the nested models'
|
|
329
|
+
# own fields — not a hand list of paths.
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
#: Shared with the field-level table in `schema/applicability.py`, which is the
|
|
333
|
+
#: same idea one level down: sections here, named fields there.
|
|
334
|
+
_model_types = model_types
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
_LLM_TYPES = _model_types("llm-")
|
|
338
|
+
_EMBEDDING_TYPES = _model_types("embedding-")
|
|
339
|
+
_AUDIO_TYPES = _model_types("audio-")
|
|
340
|
+
_IMAGE_TYPES = _model_types("image-generation", "image-editing")
|
|
341
|
+
#: `decision-model` (MODEL-98) reads text state, so `max_input_tokens` and
|
|
342
|
+
#: `context_window` are real questions for it. It is deliberately absent from
|
|
343
|
+
#: `_GENERATIVE_TEXT_TYPES` below — it writes no text — and the output-shaped
|
|
344
|
+
#: fields inside this subtree are excluded field by field in
|
|
345
|
+
#: `schema/applicability.py`, which is finer than a whole-section gate can be.
|
|
346
|
+
_TEXT_TYPES = _LLM_TYPES | _model_types(
|
|
347
|
+
"vlm", "agent-model", "medical", "legal", "financial",
|
|
348
|
+
"router", "reward-model", "safety-classifier", "text-encoder", "document-ocr",
|
|
349
|
+
"decision-model",
|
|
350
|
+
)
|
|
351
|
+
_GENERATIVE_TEXT_TYPES = _LLM_TYPES | _model_types(
|
|
352
|
+
"vlm", "agent-model", "medical", "legal", "financial",
|
|
353
|
+
"router", "reward-model", "safety-classifier",
|
|
354
|
+
)
|
|
355
|
+
_VISION_TYPES = _IMAGE_TYPES | _model_types(
|
|
356
|
+
"vlm", "vision-encoder", "document-ocr", "embedding-multimodal",
|
|
357
|
+
)
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
class VisionDetail(BaseModel):
|
|
361
|
+
__applicable_model_types__: ClassVar[frozenset[ModelType]] = _VISION_TYPES
|
|
362
|
+
supported: bool = False
|
|
363
|
+
ocr: bool = False
|
|
364
|
+
chart_reading: bool = False
|
|
365
|
+
spatial_reasoning: bool = False
|
|
366
|
+
handwriting: bool = False
|
|
367
|
+
object_detection: bool = False
|
|
368
|
+
object_counting: bool = False
|
|
369
|
+
visual_grounding: bool = False
|
|
370
|
+
max_image_resolution: str = ""
|
|
371
|
+
max_images_per_request: int | None = None
|
|
372
|
+
video_frames: bool = False
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
class AudioDetail(BaseModel):
|
|
376
|
+
__applicable_model_types__: ClassVar[frozenset[ModelType]] = _AUDIO_TYPES
|
|
377
|
+
input_supported: bool = False
|
|
378
|
+
output_supported: bool = False
|
|
379
|
+
realtime_streaming: bool = False
|
|
380
|
+
asr_languages: list[str] = []
|
|
381
|
+
tts_languages: list[str] = []
|
|
382
|
+
tts_voices: int | None = None
|
|
383
|
+
voice_cloning: bool = False
|
|
384
|
+
speaker_diarization: bool = False
|
|
385
|
+
music_understanding: bool = False
|
|
386
|
+
music_generation: bool = False
|
|
387
|
+
max_audio_duration_sec: int | None = None
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
class VideoDetail(BaseModel):
|
|
391
|
+
__applicable_model_types__: ClassVar[frozenset[ModelType]] = _model_types("video-generation")
|
|
392
|
+
input_supported: bool = False
|
|
393
|
+
output_supported: bool = False
|
|
394
|
+
max_input_duration_sec: int | None = None
|
|
395
|
+
max_output_duration_sec: int | None = None
|
|
396
|
+
max_resolution: str = ""
|
|
397
|
+
max_fps: int | None = None
|
|
398
|
+
audio_sync: bool = False
|
|
399
|
+
temporal_reasoning: bool = False
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
class DocumentDetail(BaseModel):
|
|
403
|
+
__applicable_model_types__: ClassVar[frozenset[ModelType]] = _model_types(
|
|
404
|
+
"document-ocr", "vlm",
|
|
405
|
+
)
|
|
406
|
+
pdf_native: bool = False
|
|
407
|
+
table_extraction: bool = False
|
|
408
|
+
form_understanding: bool = False
|
|
409
|
+
max_pages: int | None = None
|
|
410
|
+
layout_analysis: bool = False
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
class ImageGenDetail(BaseModel):
|
|
414
|
+
__applicable_model_types__: ClassVar[frozenset[ModelType]] = _IMAGE_TYPES
|
|
415
|
+
supported: bool = False
|
|
416
|
+
max_resolution: str = ""
|
|
417
|
+
aspect_ratios: list[str] = []
|
|
418
|
+
inpainting: bool = False
|
|
419
|
+
outpainting: bool = False
|
|
420
|
+
img2img: bool = False
|
|
421
|
+
text_rendering_quality: str = ""
|
|
422
|
+
style_control: bool = False
|
|
423
|
+
controlnet_support: bool = False
|
|
424
|
+
lora_support: bool = False
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
class EmbeddingDetail(BaseModel):
|
|
428
|
+
__applicable_model_types__: ClassVar[frozenset[ModelType]] = _EMBEDDING_TYPES
|
|
429
|
+
supported: bool = False
|
|
430
|
+
dimensions: int | None = None
|
|
431
|
+
dimensions_configurable: bool = False
|
|
432
|
+
dimension_options: list[int] = []
|
|
433
|
+
max_input_tokens: int | None = None
|
|
434
|
+
similarity_metric: str = ""
|
|
435
|
+
normalized: bool | None = None
|
|
436
|
+
batch_size_max: int | None = None
|
|
437
|
+
instruction_aware: bool = False
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
class RerankingDetail(BaseModel):
|
|
441
|
+
__applicable_model_types__: ClassVar[frozenset[ModelType]] = _model_types("reranker")
|
|
442
|
+
supported: bool = False
|
|
443
|
+
max_input_pairs: int | None = None
|
|
444
|
+
max_input_length: int | None = None
|
|
445
|
+
cross_encoder: bool | None = None
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
class TextDetail(BaseModel):
|
|
449
|
+
__applicable_model_types__: ClassVar[frozenset[ModelType]] = _TEXT_TYPES
|
|
450
|
+
max_input_tokens: int | None = None
|
|
451
|
+
max_output_tokens: int | None = None
|
|
452
|
+
context_window: int | None = None
|
|
453
|
+
streaming: bool | None = None
|
|
454
|
+
fill_in_middle: bool | None = None
|
|
455
|
+
json_mode: bool | None = None
|
|
456
|
+
system_prompt: bool | None = None
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
class Modalities(BaseModel):
|
|
460
|
+
input: list[Modality] = []
|
|
461
|
+
output: list[Modality] = []
|
|
462
|
+
text: TextDetail = TextDetail()
|
|
463
|
+
vision: VisionDetail = VisionDetail()
|
|
464
|
+
audio: AudioDetail = AudioDetail()
|
|
465
|
+
video: VideoDetail = VideoDetail()
|
|
466
|
+
document: DocumentDetail = DocumentDetail()
|
|
467
|
+
image_generation: ImageGenDetail = ImageGenDetail()
|
|
468
|
+
embeddings: EmbeddingDetail = EmbeddingDetail()
|
|
469
|
+
reranking: RerankingDetail = RerankingDetail()
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
# ═══════════════════════════════════════════════════════════════
|
|
473
|
+
# Section 6: Capabilities
|
|
474
|
+
# ═══════════════════════════════════════════════════════════════
|
|
475
|
+
|
|
476
|
+
class CodingCapability(BaseModel):
|
|
477
|
+
overall: Tier | None = None
|
|
478
|
+
languages: list[str] = []
|
|
479
|
+
agentic_coding: bool = False
|
|
480
|
+
code_review: bool = False
|
|
481
|
+
refactoring: bool = False
|
|
482
|
+
debugging: bool = False
|
|
483
|
+
test_generation: bool = False
|
|
484
|
+
documentation: bool = False
|
|
485
|
+
code_completion: bool = False
|
|
486
|
+
multi_file_editing: bool = False
|
|
487
|
+
fill_in_middle: bool = False
|
|
488
|
+
lsp_integration: bool = False
|
|
489
|
+
repository_understanding: bool = False
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
class ReasoningCapability(BaseModel):
|
|
493
|
+
overall: Tier | None = None
|
|
494
|
+
mathematical: bool = False
|
|
495
|
+
logical: bool = False
|
|
496
|
+
scientific: bool = False
|
|
497
|
+
planning: bool = False
|
|
498
|
+
multi_step: bool = False
|
|
499
|
+
chain_of_thought: bool = False
|
|
500
|
+
self_correction: bool = False
|
|
501
|
+
spatial: bool = False
|
|
502
|
+
temporal: bool = False
|
|
503
|
+
causal: bool = False
|
|
504
|
+
think_budget_control: bool = False
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
class ToolUseCapability(BaseModel):
|
|
508
|
+
overall: Tier | None = None
|
|
509
|
+
function_calling: bool = False
|
|
510
|
+
mcp_compatible: bool = False
|
|
511
|
+
parallel_tool_calls: bool = False
|
|
512
|
+
tool_selection_accuracy: ConfidenceLevel | None = None
|
|
513
|
+
multi_turn_tool_use: bool = False
|
|
514
|
+
tool_error_recovery: bool = False
|
|
515
|
+
computer_use: bool = False
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
class LanguageCapability(BaseModel):
|
|
519
|
+
multilingual: bool = False
|
|
520
|
+
num_languages: int | None = None
|
|
521
|
+
strong_languages: list[str] = []
|
|
522
|
+
translation_quality: Tier | None = None
|
|
523
|
+
long_context_retrieval: Tier | None = None
|
|
524
|
+
|
|
525
|
+
|
|
526
|
+
class CreativeCapability(BaseModel):
|
|
527
|
+
writing: Tier | None = None
|
|
528
|
+
summarization: Tier | None = None
|
|
529
|
+
instruction_following: Tier | None = None
|
|
530
|
+
storytelling: Tier | None = None
|
|
531
|
+
technical_writing: Tier | None = None
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
class SafetyAlignment(BaseModel):
|
|
535
|
+
alignment_approach: str = ""
|
|
536
|
+
refusal_rate: ConfidenceLevel | None = None
|
|
537
|
+
jailbreak_resistance: ConfidenceLevel | None = None
|
|
538
|
+
content_safety_tier: Tier | None = None
|
|
539
|
+
guardrail_builtin: bool = False
|
|
540
|
+
|
|
541
|
+
|
|
542
|
+
class DomainCapability(BaseModel):
|
|
543
|
+
medical_knowledge: Tier | None = None
|
|
544
|
+
legal_knowledge: Tier | None = None
|
|
545
|
+
financial_knowledge: Tier | None = None
|
|
546
|
+
scientific_knowledge: Tier | None = None
|
|
547
|
+
|
|
548
|
+
|
|
549
|
+
class AgentCapability(BaseModel):
|
|
550
|
+
autonomous_execution: bool = False
|
|
551
|
+
web_browsing: bool = False
|
|
552
|
+
file_system_access: bool = False
|
|
553
|
+
code_execution: bool = False
|
|
554
|
+
long_running_tasks: bool = False
|
|
555
|
+
memory_management: bool = False
|
|
556
|
+
self_delegation: bool = False
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
class Capabilities(BaseModel):
|
|
560
|
+
__applicable_model_types__: ClassVar[frozenset[ModelType]] = _GENERATIVE_TEXT_TYPES
|
|
561
|
+
coding: CodingCapability = CodingCapability()
|
|
562
|
+
reasoning: ReasoningCapability = ReasoningCapability()
|
|
563
|
+
tool_use: ToolUseCapability = ToolUseCapability()
|
|
564
|
+
language: LanguageCapability = LanguageCapability()
|
|
565
|
+
creative: CreativeCapability = CreativeCapability()
|
|
566
|
+
safety_alignment: SafetyAlignment = SafetyAlignment()
|
|
567
|
+
domain_specific: DomainCapability = DomainCapability()
|
|
568
|
+
agent_capabilities: AgentCapability = AgentCapability()
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
# ═══════════════════════════════════════════════════════════════
|
|
572
|
+
# Section 7: Cost
|
|
573
|
+
# ═══════════════════════════════════════════════════════════════
|
|
574
|
+
|
|
575
|
+
class Cost(BaseModel):
|
|
576
|
+
input: float | None = None
|
|
577
|
+
output: float | None = None
|
|
578
|
+
reasoning: float | None = None
|
|
579
|
+
cache_read: float | None = None
|
|
580
|
+
cache_write: float | None = None
|
|
581
|
+
input_audio: float | None = None
|
|
582
|
+
output_audio: float | None = None
|
|
583
|
+
input_image: float | None = None
|
|
584
|
+
output_image: float | None = None
|
|
585
|
+
output_video_per_sec: float | None = None
|
|
586
|
+
batch_input: float | None = None
|
|
587
|
+
batch_output: float | None = None
|
|
588
|
+
embedding_per_million: float | None = None
|
|
589
|
+
reranking_per_million: float | None = None
|
|
590
|
+
finetune_per_million_tokens: float | None = None
|
|
591
|
+
finetune_hosting_per_hour: float | None = None
|
|
592
|
+
free_tier: bool = False
|
|
593
|
+
free_tier_limits: str = ""
|
|
594
|
+
note: str = ""
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
# ═══════════════════════════════════════════════════════════════
|
|
598
|
+
# Section 8: Availability
|
|
599
|
+
# ═══════════════════════════════════════════════════════════════
|
|
600
|
+
|
|
601
|
+
class PlatformEntry(BaseModel):
|
|
602
|
+
"""A single platform where a model may be available."""
|
|
603
|
+
available: bool = False
|
|
604
|
+
model_id: str = ""
|
|
605
|
+
url: str = ""
|
|
606
|
+
fine_tuning: bool = False
|
|
607
|
+
gated: bool = False
|
|
608
|
+
regions: list[str] = []
|
|
609
|
+
notes: str = ""
|
|
610
|
+
|
|
611
|
+
|
|
612
|
+
class PrimaryProvider(BaseModel):
|
|
613
|
+
name: str = ""
|
|
614
|
+
platform_url: str = ""
|
|
615
|
+
api_endpoint: str = ""
|
|
616
|
+
npm_package: str = ""
|
|
617
|
+
env_vars: list[str] = []
|
|
618
|
+
model_id_on_platform: str = ""
|
|
619
|
+
rate_limit_rpm: int | None = None
|
|
620
|
+
rate_limit_tpm: int | None = None
|
|
621
|
+
sla_uptime: str = ""
|
|
622
|
+
regions: list[str] = []
|
|
623
|
+
#: Where the provider commits to processing data. `null` until it is a
|
|
624
|
+
#: determination — see `data_residency_disclosure`. Was `list[str] = []`
|
|
625
|
+
#: until MODEL-77, where the default was indistinguishable from an answer
|
|
626
|
+
#: and sat on all 1,339 cards while being researched on none of them.
|
|
627
|
+
data_residency: list[str] | None = None
|
|
628
|
+
data_residency_disclosure: DisclosureState = DisclosureState.UNRESEARCHED
|
|
629
|
+
#: Required when the disclosure is `published`; forbidden otherwise, for
|
|
630
|
+
#: the same reason as `commercial_use_source`.
|
|
631
|
+
data_residency_source: PolicySource | None = None
|
|
632
|
+
hipaa_eligible: bool = False
|
|
633
|
+
fedramp_authorized: bool = False
|
|
634
|
+
soc2_compliant: bool = False
|
|
635
|
+
free_tier: bool = False
|
|
636
|
+
free_tier_details: str = ""
|
|
637
|
+
|
|
638
|
+
@model_validator(mode="after")
|
|
639
|
+
def _residency_states_are_not_interchangeable(self) -> PrimaryProvider:
|
|
640
|
+
state = self.data_residency_disclosure
|
|
641
|
+
if state is DisclosureState.PUBLISHED:
|
|
642
|
+
if self.data_residency is None:
|
|
643
|
+
raise ValueError(
|
|
644
|
+
"data_residency_disclosure is 'published' but data_residency "
|
|
645
|
+
"is null. An empty list is a legitimate published answer "
|
|
646
|
+
"('no residency commitment'); null is not an answer at all."
|
|
647
|
+
)
|
|
648
|
+
if self.data_residency_source is None:
|
|
649
|
+
raise ValueError(
|
|
650
|
+
"data_residency is published with no data_residency_source. "
|
|
651
|
+
"Standing rule 1: an uncited policy answer is not an answer."
|
|
652
|
+
)
|
|
653
|
+
return self
|
|
654
|
+
if self.data_residency is not None:
|
|
655
|
+
raise ValueError(
|
|
656
|
+
f"data_residency_disclosure is {state.value!r} but data_residency "
|
|
657
|
+
"carries a value. Only a published determination has one; set "
|
|
658
|
+
"the disclosure to 'published' and cite it, or clear the value."
|
|
659
|
+
)
|
|
660
|
+
if self.data_residency_source is not None:
|
|
661
|
+
raise ValueError(
|
|
662
|
+
f"data_residency_disclosure is {state.value!r} but carries a "
|
|
663
|
+
"data_residency_source. Nothing has been published to cite."
|
|
664
|
+
)
|
|
665
|
+
return self
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
class Availability(BaseModel):
|
|
669
|
+
primary_provider: PrimaryProvider = PrimaryProvider()
|
|
670
|
+
# Cloud platforms
|
|
671
|
+
aws_bedrock: PlatformEntry = PlatformEntry(url="https://aws.amazon.com/bedrock/")
|
|
672
|
+
azure_ai_foundry: PlatformEntry = PlatformEntry(url="https://ai.azure.com/")
|
|
673
|
+
google_vertex_ai: PlatformEntry = PlatformEntry(url="https://cloud.google.com/vertex-ai")
|
|
674
|
+
nvidia_nim: PlatformEntry = PlatformEntry(url="https://build.nvidia.com/")
|
|
675
|
+
ibm_watsonx: PlatformEntry = PlatformEntry(url="https://www.ibm.com/watsonx")
|
|
676
|
+
snowflake_cortex: PlatformEntry = PlatformEntry(url="https://www.snowflake.com/en/data-cloud/cortex/")
|
|
677
|
+
# Inference providers
|
|
678
|
+
groq: PlatformEntry = PlatformEntry(url="https://groq.com/")
|
|
679
|
+
together_ai: PlatformEntry = PlatformEntry(url="https://www.together.ai/")
|
|
680
|
+
fireworks_ai: PlatformEntry = PlatformEntry(url="https://fireworks.ai/")
|
|
681
|
+
replicate: PlatformEntry = PlatformEntry(url="https://replicate.com/")
|
|
682
|
+
deepinfra: PlatformEntry = PlatformEntry(url="https://deepinfra.com/")
|
|
683
|
+
cerebras: PlatformEntry = PlatformEntry(url="https://www.cerebras.ai/")
|
|
684
|
+
sambanova: PlatformEntry = PlatformEntry(url="https://sambanova.ai/")
|
|
685
|
+
# Aggregators
|
|
686
|
+
openrouter: PlatformEntry = PlatformEntry(url="https://openrouter.ai/")
|
|
687
|
+
# AI apps
|
|
688
|
+
cursor: PlatformEntry = PlatformEntry(url="https://cursor.com/")
|
|
689
|
+
github_copilot: PlatformEntry = PlatformEntry(url="https://github.com/features/copilot")
|
|
690
|
+
perplexity: PlatformEntry = PlatformEntry(url="https://www.perplexity.ai/")
|
|
691
|
+
raycast: PlatformEntry = PlatformEntry(url="https://www.raycast.com/")
|
|
692
|
+
poe: PlatformEntry = PlatformEntry(url="https://poe.com/")
|
|
693
|
+
# Consumer chat
|
|
694
|
+
chatgpt: PlatformEntry = PlatformEntry(url="https://chat.openai.com/")
|
|
695
|
+
claude_ai: PlatformEntry = PlatformEntry(url="https://claude.ai/")
|
|
696
|
+
gemini_app: PlatformEntry = PlatformEntry(url="https://gemini.google.com/")
|
|
697
|
+
grok_xai: PlatformEntry = PlatformEntry(url="https://x.ai/")
|
|
698
|
+
meta_ai: PlatformEntry = PlatformEntry(url="https://www.meta.ai/")
|
|
699
|
+
copilot_microsoft: PlatformEntry = PlatformEntry(url="https://copilot.microsoft.com/")
|
|
700
|
+
# Provider platforms
|
|
701
|
+
mistral_plateforme: PlatformEntry = PlatformEntry(url="https://console.mistral.ai/")
|
|
702
|
+
cohere: PlatformEntry = PlatformEntry(url="https://cohere.com/")
|
|
703
|
+
ai21_labs: PlatformEntry = PlatformEntry(url="https://www.ai21.com/")
|
|
704
|
+
stability_ai: PlatformEntry = PlatformEntry(url="https://stability.ai/")
|
|
705
|
+
# Chinese platforms
|
|
706
|
+
deepseek: PlatformEntry = PlatformEntry(url="https://platform.deepseek.com/")
|
|
707
|
+
qwen_alibaba: PlatformEntry = PlatformEntry(url="https://www.alibabacloud.com/en/solutions/generative-ai/qwen")
|
|
708
|
+
baidu_ernie: PlatformEntry = PlatformEntry(url="https://cloud.baidu.com/")
|
|
709
|
+
bytedance_doubao: PlatformEntry = PlatformEntry(url="https://www.volcengine.com/")
|
|
710
|
+
tencent_hunyuan: PlatformEntry = PlatformEntry(url="https://cloud.tencent.com/")
|
|
711
|
+
zhipu_glm: PlatformEntry = PlatformEntry(url="https://www.zhipuai.cn/")
|
|
712
|
+
moonshot_kimi: PlatformEntry = PlatformEntry(url="https://www.moonshot.cn/")
|
|
713
|
+
minimax: PlatformEntry = PlatformEntry(url="https://www.minimax.chat/")
|
|
714
|
+
zero_one_ai: PlatformEntry = PlatformEntry(url="https://www.01.ai/")
|
|
715
|
+
# Regional
|
|
716
|
+
tii_falcon: PlatformEntry = PlatformEntry(url="https://falconllm.tii.ae/")
|
|
717
|
+
samsung_gauss: PlatformEntry = PlatformEntry(url="https://www.samsung.com/")
|
|
718
|
+
upstage_solar: PlatformEntry = PlatformEntry(url="https://www.upstage.ai/")
|
|
719
|
+
# Local
|
|
720
|
+
ollama: PlatformEntry = PlatformEntry(url="https://ollama.com/")
|
|
721
|
+
lm_studio: PlatformEntry = PlatformEntry(url="https://lmstudio.ai/")
|
|
722
|
+
gpt4all: PlatformEntry = PlatformEntry(url="https://gpt4all.io/")
|
|
723
|
+
jan_ai: PlatformEntry = PlatformEntry(url="https://jan.ai/")
|
|
724
|
+
mlx_community: PlatformEntry = PlatformEntry(url="https://huggingface.co/mlx-community")
|
|
725
|
+
open_webui: PlatformEntry = PlatformEntry(url="https://openwebui.com/")
|
|
726
|
+
# Model hubs
|
|
727
|
+
huggingface: PlatformEntry = PlatformEntry(url="https://huggingface.co/")
|
|
728
|
+
modelscope: PlatformEntry = PlatformEntry(url="https://modelscope.cn/")
|
|
729
|
+
kaggle_models: PlatformEntry = PlatformEntry(url="https://www.kaggle.com/models")
|
|
730
|
+
# Overflow
|
|
731
|
+
other_platforms: list[PlatformEntry] = []
|
|
732
|
+
|
|
733
|
+
def platforms_available(self) -> list[str]:
|
|
734
|
+
"""Return names of all platforms where this model is available."""
|
|
735
|
+
available = []
|
|
736
|
+
for field_name, field_value in self:
|
|
737
|
+
if isinstance(field_value, PlatformEntry) and field_value.available:
|
|
738
|
+
available.append(field_name)
|
|
739
|
+
return available
|
|
740
|
+
|
|
741
|
+
|
|
742
|
+
# ═══════════════════════════════════════════════════════════════
|
|
743
|
+
# Section 9: Benchmarks
|
|
744
|
+
# ═══════════════════════════════════════════════════════════════
|
|
745
|
+
|
|
746
|
+
class BenchmarkEvidence(BaseModel):
|
|
747
|
+
"""One score, with everything needed to check it.
|
|
748
|
+
|
|
749
|
+
The flat `scores` dict below carries one collection date for a whole card
|
|
750
|
+
and a comma-joined source list, so no individual number can be attributed,
|
|
751
|
+
dated or rechecked. This record is the shape the benchmark catalogue's
|
|
752
|
+
evidence contract requires, and it mirrors the census evidence ledger so the
|
|
753
|
+
two can be reconciled rather than diverging.
|
|
754
|
+
|
|
755
|
+
Every field here is required. A record that cannot say where a number came
|
|
756
|
+
from or when is not evidence, and admitting a partial one would quietly
|
|
757
|
+
reintroduce exactly the problem this replaces.
|
|
758
|
+
"""
|
|
759
|
+
|
|
760
|
+
benchmark_id: str
|
|
761
|
+
model_id_as_evaluated: str
|
|
762
|
+
score: float
|
|
763
|
+
unit: str
|
|
764
|
+
source_url: str
|
|
765
|
+
#: benchmark author, independent evaluator, or the provider's own claim.
|
|
766
|
+
#: Provider self-report is legitimate and must be visibly distinguishable.
|
|
767
|
+
source_kind: Literal["benchmark_author", "independent_evaluator", "provider_self_report"]
|
|
768
|
+
evidence_date: str
|
|
769
|
+
#: `evaluated` when the run date is disclosed; `published` when only the
|
|
770
|
+
#: publication date is. Never infer a run date from a retrieval timestamp.
|
|
771
|
+
date_type: Literal["evaluated", "published"]
|
|
772
|
+
verified_at: str
|
|
773
|
+
benchmark_version: str = ""
|
|
774
|
+
configuration: str = ""
|
|
775
|
+
limitations: str = ""
|
|
776
|
+
|
|
777
|
+
@field_validator("source_url")
|
|
778
|
+
@classmethod
|
|
779
|
+
def _url_must_be_real(cls, value: str) -> str:
|
|
780
|
+
if not value.startswith(("http://", "https://")):
|
|
781
|
+
raise ValueError("source_url must be a URL; a score without one is not evidence")
|
|
782
|
+
return value
|
|
783
|
+
|
|
784
|
+
@field_validator("evidence_date", "verified_at")
|
|
785
|
+
@classmethod
|
|
786
|
+
def _dates_must_be_iso(cls, value: str) -> str:
|
|
787
|
+
from datetime import date as _date
|
|
788
|
+
try:
|
|
789
|
+
_date.fromisoformat(value)
|
|
790
|
+
except ValueError as exc:
|
|
791
|
+
raise ValueError(f"must be an exact ISO date YYYY-MM-DD, got {value!r}") from exc
|
|
792
|
+
return value
|
|
793
|
+
|
|
794
|
+
|
|
795
|
+
class Benchmarks(BaseModel):
|
|
796
|
+
# All benchmark scores in a single open-ended dictionary.
|
|
797
|
+
# Keys are benchmark identifiers (e.g. "humaneval", "mmlu_pro",
|
|
798
|
+
# "multipl_e_rust", "mmlu_chemistry", "pubmedqa", "flores_en_zh").
|
|
799
|
+
# No fixed schema — any benchmark can be added without code changes.
|
|
800
|
+
#: V2 quarantine: the decision engine never reads benchmarks.scores.
|
|
801
|
+
#: MODEL-118 re-sources these values; v1 retains its existing behavior.
|
|
802
|
+
scores: dict[str, float] = {}
|
|
803
|
+
|
|
804
|
+
#: Verified, per-score evidence. Everything in `scores` above that has no
|
|
805
|
+
#: matching record here is unverified-legacy and must be presented as such.
|
|
806
|
+
evidence: list[BenchmarkEvidence] = []
|
|
807
|
+
|
|
808
|
+
# Meta. These describe `scores` only, and are the reason it cannot be
|
|
809
|
+
# attributed: one date and one source list for the whole card.
|
|
810
|
+
benchmark_source: str = ""
|
|
811
|
+
benchmark_as_of: str = ""
|
|
812
|
+
benchmark_notes: str = ""
|
|
813
|
+
|
|
814
|
+
def filled_count(self) -> int:
|
|
815
|
+
return len(self.scores)
|
|
816
|
+
|
|
817
|
+
def verified_ids(self) -> set[str]:
|
|
818
|
+
return {e.benchmark_id for e in self.evidence}
|
|
819
|
+
|
|
820
|
+
def is_verified(self, benchmark_id: str) -> bool:
|
|
821
|
+
return benchmark_id in self.verified_ids()
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
# ═══════════════════════════════════════════════════════════════
|
|
825
|
+
# Section 10: Hardware & Deployment
|
|
826
|
+
# ═══════════════════════════════════════════════════════════════
|
|
827
|
+
|
|
828
|
+
class HardwareProfile(BaseModel):
|
|
829
|
+
fits: bool | None = None
|
|
830
|
+
best_quant: str = ""
|
|
831
|
+
vram_usage_gb: float | None = None
|
|
832
|
+
ram_usage_gb: float | None = None
|
|
833
|
+
tokens_per_sec: float | None = None
|
|
834
|
+
prompt_tps: float | None = None
|
|
835
|
+
ttft_ms: float | None = None
|
|
836
|
+
max_context_at_quant: int | None = None
|
|
837
|
+
inference_engine: str = ""
|
|
838
|
+
notes: str = ""
|
|
839
|
+
|
|
840
|
+
|
|
841
|
+
class Runtimes(BaseModel):
|
|
842
|
+
gguf: bool = False
|
|
843
|
+
ollama: bool = False
|
|
844
|
+
ollama_tag: str = ""
|
|
845
|
+
lm_studio: bool = False
|
|
846
|
+
vllm: bool = False
|
|
847
|
+
trt_llm: bool = False
|
|
848
|
+
mlx: bool = False
|
|
849
|
+
llama_cpp: bool = False
|
|
850
|
+
sglang: bool = False
|
|
851
|
+
transformers: bool = False
|
|
852
|
+
exllamav2: bool = False
|
|
853
|
+
core_ml: bool = False
|
|
854
|
+
onnx: bool = False
|
|
855
|
+
triton: bool = False
|
|
856
|
+
nim: bool = False
|
|
857
|
+
|
|
858
|
+
|
|
859
|
+
class Deployment(BaseModel):
|
|
860
|
+
api_only: bool = False
|
|
861
|
+
local_inference: bool = False
|
|
862
|
+
self_hostable: bool = False
|
|
863
|
+
fine_tuning_supported: bool = False
|
|
864
|
+
fine_tuning_methods: list[str] = []
|
|
865
|
+
quantizations_available: list[str] = []
|
|
866
|
+
hardware_profiles: dict[str, HardwareProfile] = Field(default_factory=lambda: {
|
|
867
|
+
"nvidia_5090_32gb": HardwareProfile(),
|
|
868
|
+
"dgx_spark_128gb": HardwareProfile(),
|
|
869
|
+
"macbook_m4_pro_64gb": HardwareProfile(),
|
|
870
|
+
"macbook_air_m4_24gb": HardwareProfile(),
|
|
871
|
+
})
|
|
872
|
+
custom_hardware: list[HardwareProfile] = []
|
|
873
|
+
runtimes: Runtimes = Runtimes()
|
|
874
|
+
|
|
875
|
+
|
|
876
|
+
# ═══════════════════════════════════════════════════════════════
|
|
877
|
+
# Section 11-14: Risk, Performance, Adoption, Downselect
|
|
878
|
+
# ═══════════════════════════════════════════════════════════════
|
|
879
|
+
|
|
880
|
+
class BiasEvaluation(BaseModel):
|
|
881
|
+
conducted: bool = False
|
|
882
|
+
methodology: str = ""
|
|
883
|
+
results_summary: str = ""
|
|
884
|
+
known_biases: list[str] = []
|
|
885
|
+
|
|
886
|
+
|
|
887
|
+
class PrivacyPosture(BaseModel):
|
|
888
|
+
data_retention_policy: str = ""
|
|
889
|
+
pii_handling: str = ""
|
|
890
|
+
training_data_pii_scrubbed: bool | None = None
|
|
891
|
+
data_processing_location: list[str] = []
|
|
892
|
+
gdpr_compliant: bool | None = None
|
|
893
|
+
hipaa_eligible: bool | None = None
|
|
894
|
+
ccpa_compliant: bool | None = None
|
|
895
|
+
|
|
896
|
+
|
|
897
|
+
class SupplyChain(BaseModel):
|
|
898
|
+
training_data_transparency: str = ""
|
|
899
|
+
model_provenance_documented: bool = False
|
|
900
|
+
third_party_dependencies: list[str] = []
|
|
901
|
+
ai_bom_available: bool = False
|
|
902
|
+
reproducible: bool = False
|
|
903
|
+
|
|
904
|
+
|
|
905
|
+
class RegulatoryAlignment(BaseModel):
|
|
906
|
+
eu_ai_act_risk_level: EUAIActRisk | None = None
|
|
907
|
+
nist_rmf_profile: str = ""
|
|
908
|
+
iso_42001_certified: bool | None = None
|
|
909
|
+
soc2_type2: bool | None = None
|
|
910
|
+
fedramp_level: str = ""
|
|
911
|
+
|
|
912
|
+
|
|
913
|
+
class RiskGovernance(BaseModel):
|
|
914
|
+
valid_and_reliable: str = ""
|
|
915
|
+
safe: str = ""
|
|
916
|
+
secure_and_resilient: str = ""
|
|
917
|
+
accountable_and_transparent: str = ""
|
|
918
|
+
explainable_and_interpretable: str = ""
|
|
919
|
+
privacy_enhanced: str = ""
|
|
920
|
+
fair_with_bias_managed: str = ""
|
|
921
|
+
bias_evaluation: BiasEvaluation = BiasEvaluation()
|
|
922
|
+
adversarial_robustness: ResistanceLevel = ResistanceLevel.UNTESTED
|
|
923
|
+
privacy: PrivacyPosture = PrivacyPosture()
|
|
924
|
+
supply_chain: SupplyChain = SupplyChain()
|
|
925
|
+
incident_history: list[str] = []
|
|
926
|
+
known_failure_modes: list[str] = []
|
|
927
|
+
regulatory: RegulatoryAlignment = RegulatoryAlignment()
|
|
928
|
+
|
|
929
|
+
|
|
930
|
+
class InferencePerformance(BaseModel):
|
|
931
|
+
api_latency_p50_ms: float | None = None
|
|
932
|
+
api_latency_p99_ms: float | None = None
|
|
933
|
+
api_ttft_ms: float | None = None
|
|
934
|
+
api_tps_output: float | None = None
|
|
935
|
+
api_tps_input: float | None = None
|
|
936
|
+
context_speed_degradation: str = ""
|
|
937
|
+
generation_time_sec: float | None = None
|
|
938
|
+
quality_per_dollar: float | None = None
|
|
939
|
+
quality_per_watt: float | None = None
|
|
940
|
+
|
|
941
|
+
|
|
942
|
+
class Adoption(BaseModel):
|
|
943
|
+
huggingface_downloads: int | None = None
|
|
944
|
+
huggingface_likes: int | None = None
|
|
945
|
+
ollama_pulls: int | None = None
|
|
946
|
+
community_forks: int | None = None
|
|
947
|
+
is_common_distillation_teacher: bool = False
|
|
948
|
+
is_common_finetune_base: bool = False
|
|
949
|
+
openrouter_ranking: int | None = None
|
|
950
|
+
open_webui_ranking: int | None = None
|
|
951
|
+
notable_users: list[str] = []
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
class Downselect(BaseModel):
|
|
955
|
+
compliance_tags: list[str] = []
|
|
956
|
+
clearance_tags: list[str] = []
|
|
957
|
+
defense_tags: list[str] = []
|
|
958
|
+
sovereignty_tags: list[str] = []
|
|
959
|
+
use_case_tags: list[str] = []
|
|
960
|
+
eval_status: EvalStatus | None = None
|
|
961
|
+
risk_tier: RiskTier | None = None
|
|
962
|
+
cost_tier: str = ""
|
|
963
|
+
custom_score: float | None = None
|
|
964
|
+
custom_notes: str = ""
|
|
965
|
+
reviewed_by: str = ""
|
|
966
|
+
review_date: str = ""
|
|
967
|
+
approval_authority: str = ""
|
|
968
|
+
next_review_date: str = ""
|
|
969
|
+
|
|
970
|
+
|
|
971
|
+
# ═══════════════════════════════════════════════════════════════
|
|
972
|
+
# Section 15: Sources & Metadata
|
|
973
|
+
# ═══════════════════════════════════════════════════════════════
|
|
974
|
+
|
|
975
|
+
class Sources(BaseModel):
|
|
976
|
+
models_dev_url: str = ""
|
|
977
|
+
provider_docs_url: str = ""
|
|
978
|
+
huggingface_url: str = ""
|
|
979
|
+
arxiv_url: str = ""
|
|
980
|
+
paper_url: str = ""
|
|
981
|
+
github_url: str = ""
|
|
982
|
+
ollama_url: str = ""
|
|
983
|
+
artificial_analysis_url: str = ""
|
|
984
|
+
arena_url: str = ""
|
|
985
|
+
# Freshness tracking
|
|
986
|
+
last_scraped_models_dev: str = ""
|
|
987
|
+
last_scraped_huggingface: str = ""
|
|
988
|
+
last_scraped_benchmarks: str = ""
|
|
989
|
+
last_scraped_pricing: str = ""
|
|
990
|
+
|
|
991
|
+
|
|
992
|
+
|
|
993
|
+
# ═══════════════════════════════════════════════════════════════
|
|
994
|
+
# Section 16: Authoring Guide (MODEL-8)
|
|
995
|
+
# ═══════════════════════════════════════════════════════════════
|
|
996
|
+
#
|
|
997
|
+
# How to write for one model: what helps, what wastes tokens. Every claim is a
|
|
998
|
+
# short paraphrase of the provider's own guidance, dated and sourced. A guide is
|
|
999
|
+
# pinned to the card's `version` (the provider's API model string); when that
|
|
1000
|
+
# changes, the guide must be re-reviewed or marked `stale`. Stale is never kept
|
|
1001
|
+
# silently.
|
|
1002
|
+
|
|
1003
|
+
GuideSourceKind = Literal["provider-guidance", "system-card", "release-notes", "model-docs"]
|
|
1004
|
+
GUIDE_SECTIONS = (
|
|
1005
|
+
"prompt_shape", "system_message", "reasoning_and_tools",
|
|
1006
|
+
"formatting", "failure_modes", "retry_advice",
|
|
1007
|
+
)
|
|
1008
|
+
|
|
1009
|
+
|
|
1010
|
+
def _iso_date(value: str, what: str) -> str:
|
|
1011
|
+
try:
|
|
1012
|
+
date.fromisoformat(str(value))
|
|
1013
|
+
except ValueError:
|
|
1014
|
+
raise ValueError(f"{what} must be an ISO date (YYYY-MM-DD), got {value!r}") from None
|
|
1015
|
+
return str(value)
|
|
1016
|
+
|
|
1017
|
+
|
|
1018
|
+
class GuideSource(BaseModel):
|
|
1019
|
+
url: str
|
|
1020
|
+
title: str = ""
|
|
1021
|
+
accessed: str
|
|
1022
|
+
kind: GuideSourceKind
|
|
1023
|
+
|
|
1024
|
+
@field_validator("url")
|
|
1025
|
+
@classmethod
|
|
1026
|
+
def _url_must_be_real(cls, value: str) -> str:
|
|
1027
|
+
if not str(value).startswith(("http://", "https://")):
|
|
1028
|
+
raise ValueError(f"guide source url must be http(s), got {value!r}")
|
|
1029
|
+
return value
|
|
1030
|
+
|
|
1031
|
+
@field_validator("accessed", mode="before")
|
|
1032
|
+
@classmethod
|
|
1033
|
+
def _accessed_iso(cls, value: Any) -> str:
|
|
1034
|
+
return _iso_date(value, "guide source accessed")
|
|
1035
|
+
|
|
1036
|
+
|
|
1037
|
+
class GuideClaim(BaseModel):
|
|
1038
|
+
text: str
|
|
1039
|
+
sources: list[GuideSource]
|
|
1040
|
+
|
|
1041
|
+
@model_validator(mode="after")
|
|
1042
|
+
def _needs_source(self) -> "GuideClaim":
|
|
1043
|
+
if not self.text.strip():
|
|
1044
|
+
raise ValueError("guide claim text is empty")
|
|
1045
|
+
if not self.sources:
|
|
1046
|
+
raise ValueError(f"guide claim has no source: {self.text[:60]!r}")
|
|
1047
|
+
return self
|
|
1048
|
+
|
|
1049
|
+
|
|
1050
|
+
class GuideAppliesTo(BaseModel):
|
|
1051
|
+
model_id: str
|
|
1052
|
+
version: str
|
|
1053
|
+
|
|
1054
|
+
|
|
1055
|
+
class GuideSections(BaseModel):
|
|
1056
|
+
prompt_shape: list[GuideClaim] = []
|
|
1057
|
+
system_message: list[GuideClaim] = []
|
|
1058
|
+
reasoning_and_tools: list[GuideClaim] = []
|
|
1059
|
+
formatting: list[GuideClaim] = []
|
|
1060
|
+
failure_modes: list[GuideClaim] = []
|
|
1061
|
+
retry_advice: list[GuideClaim] = []
|
|
1062
|
+
|
|
1063
|
+
|
|
1064
|
+
class AuthoringGuide(BaseModel):
|
|
1065
|
+
applies_to: GuideAppliesTo
|
|
1066
|
+
as_of: str
|
|
1067
|
+
status: Literal["current", "stale"]
|
|
1068
|
+
sections: GuideSections = GuideSections()
|
|
1069
|
+
|
|
1070
|
+
@field_validator("as_of", mode="before")
|
|
1071
|
+
@classmethod
|
|
1072
|
+
def _as_of_iso(cls, value: Any) -> str:
|
|
1073
|
+
return _iso_date(value, "authoring_guide.as_of")
|
|
1074
|
+
|
|
1075
|
+
|
|
1076
|
+
# ═══════════════════════════════════════════════════════════════
|
|
1077
|
+
# THE COMPLETE MODEL CARD
|
|
1078
|
+
# ═══════════════════════════════════════════════════════════════
|
|
1079
|
+
|
|
1080
|
+
class ModelCard(BaseModel):
|
|
1081
|
+
"""Universal Model Intelligence Card — V3.
|
|
1082
|
+
|
|
1083
|
+
This is the root object. It composes all sections into a single
|
|
1084
|
+
schema that covers every model type. Null fields = not yet researched.
|
|
1085
|
+
"""
|
|
1086
|
+
# Sections
|
|
1087
|
+
identity: Identity
|
|
1088
|
+
architecture: Architecture = Architecture()
|
|
1089
|
+
lineage: Lineage = Lineage()
|
|
1090
|
+
licensing: Licensing = Licensing()
|
|
1091
|
+
modalities: Modalities = Modalities()
|
|
1092
|
+
capabilities: Capabilities = Capabilities()
|
|
1093
|
+
cost: Cost = Cost()
|
|
1094
|
+
availability: Availability = Availability()
|
|
1095
|
+
benchmarks: Benchmarks = Benchmarks()
|
|
1096
|
+
deployment: Deployment = Deployment()
|
|
1097
|
+
risk_governance: RiskGovernance = RiskGovernance()
|
|
1098
|
+
inference_performance: InferencePerformance = InferencePerformance()
|
|
1099
|
+
adoption: Adoption = Adoption()
|
|
1100
|
+
downselect: Downselect = Downselect()
|
|
1101
|
+
sources: Sources = Sources()
|
|
1102
|
+
# Optional, additive (MODEL-8). Absent guides serialize and count as before.
|
|
1103
|
+
authoring_guide: AuthoringGuide | None = None
|
|
1104
|
+
|
|
1105
|
+
# Card metadata
|
|
1106
|
+
card_schema_version: str = "3.0"
|
|
1107
|
+
card_author: str = ""
|
|
1108
|
+
card_created: str = ""
|
|
1109
|
+
card_updated: str = ""
|
|
1110
|
+
prose_body: str = "" # The markdown content below the YAML frontmatter
|
|
1111
|
+
|
|
1112
|
+
@model_validator(mode="after")
|
|
1113
|
+
def _guide_matches_card(self) -> "ModelCard":
|
|
1114
|
+
guide = self.authoring_guide
|
|
1115
|
+
if guide is None:
|
|
1116
|
+
return self
|
|
1117
|
+
if guide.applies_to.model_id != self.identity.model_id:
|
|
1118
|
+
raise ValueError(
|
|
1119
|
+
f"authoring_guide.applies_to.model_id {guide.applies_to.model_id!r} "
|
|
1120
|
+
f"does not match card model_id {self.identity.model_id!r}")
|
|
1121
|
+
if guide.status != "stale" and guide.applies_to.version != self.identity.version:
|
|
1122
|
+
raise ValueError(
|
|
1123
|
+
f"authoring_guide was written for version {guide.applies_to.version!r} "
|
|
1124
|
+
f"but the card is now {self.identity.version!r}: re-review the guide "
|
|
1125
|
+
"against current provider guidance, or set status: stale")
|
|
1126
|
+
return self
|
|
1127
|
+
|
|
1128
|
+
def _coverage_types(self) -> frozenset[ModelType]:
|
|
1129
|
+
types: set[ModelType] = set()
|
|
1130
|
+
ident = self.identity
|
|
1131
|
+
if ident.model_type is not None:
|
|
1132
|
+
types.add(ident.model_type)
|
|
1133
|
+
types.update(ident.model_subtypes)
|
|
1134
|
+
return frozenset(types)
|
|
1135
|
+
|
|
1136
|
+
def _section_applies(self, obj: BaseModel) -> bool:
|
|
1137
|
+
applicable = getattr(type(obj), "__applicable_model_types__", None)
|
|
1138
|
+
if applicable is None:
|
|
1139
|
+
return True
|
|
1140
|
+
return bool(self._coverage_types() & applicable)
|
|
1141
|
+
|
|
1142
|
+
def warnings(self) -> list[str]:
|
|
1143
|
+
"""Non-fatal catalogue checks. CI reports these; they do not invalidate the card."""
|
|
1144
|
+
out: list[str] = []
|
|
1145
|
+
if (
|
|
1146
|
+
self.licensing.open_weights is True
|
|
1147
|
+
and self.licensing.license_type is LicenseType.PROPRIETARY
|
|
1148
|
+
):
|
|
1149
|
+
out.append(
|
|
1150
|
+
"open_weights is true with license_type proprietary: "
|
|
1151
|
+
"a proprietary licence does not distribute downloadable weights"
|
|
1152
|
+
)
|
|
1153
|
+
return out
|
|
1154
|
+
|
|
1155
|
+
@property
|
|
1156
|
+
def inapplicable_fields(self) -> tuple[str, ...]:
|
|
1157
|
+
"""Dotted paths this card's *class* cannot answer (MODEL-97).
|
|
1158
|
+
|
|
1159
|
+
Derived from `model_type` and `model_subtypes`, never stored, so it
|
|
1160
|
+
cannot drift from the card and cannot be lost in a YAML round-trip.
|
|
1161
|
+
A plain property rather than a `computed_field`: it is not card data
|
|
1162
|
+
and must not appear in `model_dump()` or `to_yaml()`.
|
|
1163
|
+
|
|
1164
|
+
An entry may be a field (`modalities.text.max_output_tokens`) or a
|
|
1165
|
+
subtree (`capabilities`), in which case every field beneath it is
|
|
1166
|
+
inapplicable. These are **not** the same as unresearched nulls: there
|
|
1167
|
+
is nothing here to research.
|
|
1168
|
+
|
|
1169
|
+
**The card wins.** A path where this card carries an actual value is
|
|
1170
|
+
never reported inapplicable, whatever the class table says: a
|
|
1171
|
+
`llm-reasoning` card with `modalities.vision.supported: true` is a
|
|
1172
|
+
model that sees, and publishing "vision does not apply" over its own
|
|
1173
|
+
data would be a false claim rather than a missing one. The pruning can
|
|
1174
|
+
only ever *remove* a claim, so it cannot invent applicability.
|
|
1175
|
+
"""
|
|
1176
|
+
return _answered_removed(
|
|
1177
|
+
inapplicable_paths(self.identity.model_type, self.identity.model_subtypes), self)
|
|
1178
|
+
|
|
1179
|
+
@computed_field
|
|
1180
|
+
@property
|
|
1181
|
+
def applicable_field_coverage(self) -> float:
|
|
1182
|
+
"""Internal statistic: percent of type-applicable schema fields filled.
|
|
1183
|
+
|
|
1184
|
+
Not published. It is not a Model node property, not in the graph
|
|
1185
|
+
export, and not shown by ``modelspec info`` or ``modelspec stats``.
|
|
1186
|
+
Nested modality details and the Capabilities block declare
|
|
1187
|
+
``__applicable_model_types__``; those subtrees count only when the
|
|
1188
|
+
card's ``model_type`` or a ``model_subtype`` is in the set. Untagged
|
|
1189
|
+
sections count for every type. ``card_*`` metadata, ``prose_body`` and
|
|
1190
|
+
``authoring_guide`` never count.
|
|
1191
|
+
"""
|
|
1192
|
+
filled, total = self._count_fields(self)
|
|
1193
|
+
return round((filled / total) * 100, 1) if total > 0 else 0.0
|
|
1194
|
+
|
|
1195
|
+
@property
|
|
1196
|
+
def card_completeness(self) -> float:
|
|
1197
|
+
"""Deprecated alias of ``applicable_field_coverage`` for scripts/**."""
|
|
1198
|
+
return self.applicable_field_coverage
|
|
1199
|
+
|
|
1200
|
+
def applicable_field_counts(self) -> tuple[int, int]:
|
|
1201
|
+
"""Filled and total fields that apply to this card's type."""
|
|
1202
|
+
return self._count_fields(self)
|
|
1203
|
+
|
|
1204
|
+
def _count_fields(self, obj: BaseModel, _depth: int = 0,
|
|
1205
|
+
_prefix: str = "",
|
|
1206
|
+
_inapplicable: frozenset[str] | None = None) -> tuple[int, int]:
|
|
1207
|
+
"""Recursively count filled vs type-applicable fields."""
|
|
1208
|
+
filled = 0
|
|
1209
|
+
total = 0
|
|
1210
|
+
inapplicable = (frozenset(self.inapplicable_fields)
|
|
1211
|
+
if _inapplicable is None else _inapplicable)
|
|
1212
|
+
for field_name, field_info in type(obj).model_fields.items():
|
|
1213
|
+
value = getattr(obj, field_name)
|
|
1214
|
+
if field_name == "authoring_guide":
|
|
1215
|
+
continue # guidance, not model facts: never moves coverage
|
|
1216
|
+
# A field this class cannot answer is outside the denominator: it
|
|
1217
|
+
# is not a gap, and counting it would be a permanent deduction for
|
|
1218
|
+
# a question that has no answer (MODEL-97).
|
|
1219
|
+
if f"{_prefix}{field_name}" in inapplicable:
|
|
1220
|
+
continue
|
|
1221
|
+
if isinstance(value, BaseModel):
|
|
1222
|
+
if not self._section_applies(value):
|
|
1223
|
+
continue
|
|
1224
|
+
f, t = self._count_fields(value, _depth + 1, f"{_prefix}{field_name}.",
|
|
1225
|
+
inapplicable)
|
|
1226
|
+
filled += f
|
|
1227
|
+
total += t
|
|
1228
|
+
elif isinstance(value, list):
|
|
1229
|
+
total += 1
|
|
1230
|
+
if len(value) > 0:
|
|
1231
|
+
filled += 1
|
|
1232
|
+
elif isinstance(value, dict):
|
|
1233
|
+
total += 1
|
|
1234
|
+
if len(value) > 0:
|
|
1235
|
+
filled += 1
|
|
1236
|
+
elif field_name.startswith("card_") or field_name in ("prose_body", "authoring_guide"):
|
|
1237
|
+
continue # Skip metadata fields
|
|
1238
|
+
else:
|
|
1239
|
+
total += 1
|
|
1240
|
+
if value is not None and value != "" and value is not False:
|
|
1241
|
+
filled += 1
|
|
1242
|
+
return filled, total
|
|
1243
|
+
|
|
1244
|
+
# ─── Serialization ─────────────────────────────────────────
|
|
1245
|
+
|
|
1246
|
+
@classmethod
|
|
1247
|
+
def from_yaml_file(cls, path: str | Path) -> "ModelCard":
|
|
1248
|
+
"""Load a model card from a YAML+Markdown file."""
|
|
1249
|
+
path = Path(path)
|
|
1250
|
+
content = path.read_text(encoding="utf-8")
|
|
1251
|
+
return cls.from_yaml_string(content)
|
|
1252
|
+
|
|
1253
|
+
@classmethod
|
|
1254
|
+
def from_yaml_string(cls, content: str) -> "ModelCard":
|
|
1255
|
+
"""Parse a model card from a string with YAML frontmatter."""
|
|
1256
|
+
parts = content.split("---", 2)
|
|
1257
|
+
if len(parts) >= 3:
|
|
1258
|
+
yaml_str = parts[1]
|
|
1259
|
+
prose = parts[2].strip()
|
|
1260
|
+
else:
|
|
1261
|
+
yaml_str = content
|
|
1262
|
+
prose = ""
|
|
1263
|
+
|
|
1264
|
+
data = yaml.safe_load(yaml_str) or {}
|
|
1265
|
+
|
|
1266
|
+
# Map flat YAML to nested Pydantic structure
|
|
1267
|
+
identity_data = {
|
|
1268
|
+
k: data.pop(k)
|
|
1269
|
+
for k in list(data.keys())
|
|
1270
|
+
if k in Identity.model_fields
|
|
1271
|
+
}
|
|
1272
|
+
|
|
1273
|
+
card_data = {
|
|
1274
|
+
"identity": identity_data,
|
|
1275
|
+
"prose_body": prose,
|
|
1276
|
+
}
|
|
1277
|
+
|
|
1278
|
+
# Map remaining top-level keys to sections
|
|
1279
|
+
section_map = {
|
|
1280
|
+
"architecture": Architecture,
|
|
1281
|
+
"lineage": Lineage,
|
|
1282
|
+
"licensing": Licensing,
|
|
1283
|
+
"modalities": Modalities,
|
|
1284
|
+
"capabilities": Capabilities,
|
|
1285
|
+
"cost": Cost,
|
|
1286
|
+
"availability": Availability,
|
|
1287
|
+
"benchmarks": Benchmarks,
|
|
1288
|
+
"deployment": Deployment,
|
|
1289
|
+
"risk_governance": RiskGovernance,
|
|
1290
|
+
"inference_performance": InferencePerformance,
|
|
1291
|
+
"adoption": Adoption,
|
|
1292
|
+
"downselect": Downselect,
|
|
1293
|
+
"sources": Sources,
|
|
1294
|
+
"authoring_guide": AuthoringGuide,
|
|
1295
|
+
}
|
|
1296
|
+
|
|
1297
|
+
for section_key, section_cls in section_map.items():
|
|
1298
|
+
if section_key in data:
|
|
1299
|
+
card_data[section_key] = data.pop(section_key)
|
|
1300
|
+
|
|
1301
|
+
# Remaining flat keys go to card metadata
|
|
1302
|
+
for k in ("card_schema_version", "card_author", "card_created", "card_updated"):
|
|
1303
|
+
if k in data:
|
|
1304
|
+
card_data[k] = data.pop(k)
|
|
1305
|
+
|
|
1306
|
+
return cls(**card_data)
|
|
1307
|
+
|
|
1308
|
+
def to_yaml(self) -> str:
|
|
1309
|
+
"""Serialize back to YAML frontmatter + Markdown.
|
|
1310
|
+
|
|
1311
|
+
The inverse of `from_yaml_string`: identity fields flat at the top
|
|
1312
|
+
level, every other section nested, enums as their string values.
|
|
1313
|
+
"""
|
|
1314
|
+
data = self.model_dump(
|
|
1315
|
+
mode="json",
|
|
1316
|
+
exclude_none=False,
|
|
1317
|
+
exclude={"prose_body", "applicable_field_coverage"},
|
|
1318
|
+
)
|
|
1319
|
+
if self.authoring_guide is None:
|
|
1320
|
+
data.pop("authoring_guide", None)
|
|
1321
|
+
out = data.pop("identity")
|
|
1322
|
+
for key in ModelCard.model_fields:
|
|
1323
|
+
if key in data:
|
|
1324
|
+
out[key] = data.pop(key)
|
|
1325
|
+
yaml_str = yaml.dump(out, default_flow_style=False, sort_keys=False, allow_unicode=True)
|
|
1326
|
+
return f"---\n{yaml_str}---\n\n{self.prose_body}"
|
|
1327
|
+
|
|
1328
|
+
|
|
1329
|
+
# ═══════════════════════════════════════════════════════════════
|
|
1330
|
+
# Applicability: the *other* kind of null (MODEL-97)
|
|
1331
|
+
# ═══════════════════════════════════════════════════════════════
|
|
1332
|
+
#
|
|
1333
|
+
# Two sources, one answer. `__applicable_model_types__` gates whole sections
|
|
1334
|
+
# and predates this; `schema.applicability.FIELD_RULES` gates named fields
|
|
1335
|
+
# inside sections that do apply. Both are pure functions of the card's class,
|
|
1336
|
+
# so this is cached per class rather than computed per card.
|
|
1337
|
+
|
|
1338
|
+
|
|
1339
|
+
def _coerce_types(model_type: Any, model_subtypes: Iterable[Any] = ()) -> frozenset[ModelType]:
|
|
1340
|
+
"""Card class plus subtypes, from enum members or from raw YAML strings.
|
|
1341
|
+
|
|
1342
|
+
An unrecognised string is dropped rather than raising: a card carrying a
|
|
1343
|
+
type this build does not know is a card whose class we do not know, and an
|
|
1344
|
+
unknown class must assert nothing about what does or does not apply.
|
|
1345
|
+
"""
|
|
1346
|
+
found: set[ModelType] = set()
|
|
1347
|
+
for raw in [model_type, *(model_subtypes or ())]:
|
|
1348
|
+
if raw is None or raw == "":
|
|
1349
|
+
continue
|
|
1350
|
+
if isinstance(raw, ModelType):
|
|
1351
|
+
found.add(raw)
|
|
1352
|
+
continue
|
|
1353
|
+
try:
|
|
1354
|
+
found.add(ModelType(str(raw)))
|
|
1355
|
+
except ValueError:
|
|
1356
|
+
continue
|
|
1357
|
+
return frozenset(found)
|
|
1358
|
+
|
|
1359
|
+
|
|
1360
|
+
def _gated_sections(model_cls: type[BaseModel], prefix: str,
|
|
1361
|
+
types: frozenset[ModelType], out: list[str]) -> None:
|
|
1362
|
+
for name, info in model_cls.model_fields.items():
|
|
1363
|
+
annotation = info.annotation
|
|
1364
|
+
if not (isinstance(annotation, type) and issubclass(annotation, BaseModel)):
|
|
1365
|
+
continue # a list, a dict, an optional union: not a gated section
|
|
1366
|
+
path = f"{prefix}{name}"
|
|
1367
|
+
gate = getattr(annotation, "__applicable_model_types__", None)
|
|
1368
|
+
if gate is not None and not (types & gate):
|
|
1369
|
+
out.append(path) # the whole subtree, named once
|
|
1370
|
+
continue
|
|
1371
|
+
_gated_sections(annotation, f"{path}.", types, out)
|
|
1372
|
+
|
|
1373
|
+
|
|
1374
|
+
@lru_cache(maxsize=None)
|
|
1375
|
+
def _inapplicable_paths(types: frozenset[ModelType]) -> tuple[str, ...]:
|
|
1376
|
+
if not types:
|
|
1377
|
+
return ()
|
|
1378
|
+
sections: list[str] = []
|
|
1379
|
+
_gated_sections(ModelCard, "", types, sections)
|
|
1380
|
+
fields = [rule.path for rule in FIELD_RULES if not (types & rule.applies_to)]
|
|
1381
|
+
# A field inside an already-named subtree is redundant: the subtree says it.
|
|
1382
|
+
# Naming both would make a consumer's `not_applicable` list disagree with
|
|
1383
|
+
# itself about how specific it is.
|
|
1384
|
+
covered = tuple(f"{path}." for path in sections)
|
|
1385
|
+
fields = [path for path in fields if not path.startswith(covered)]
|
|
1386
|
+
return tuple(sorted(set(sections) | set(fields)))
|
|
1387
|
+
|
|
1388
|
+
|
|
1389
|
+
def inapplicable_paths(model_type: Any,
|
|
1390
|
+
model_subtypes: Iterable[Any] = ()) -> tuple[str, ...]:
|
|
1391
|
+
"""Dotted paths a card of this class cannot answer, sorted.
|
|
1392
|
+
|
|
1393
|
+
Accepts `ModelType` members or the raw strings a published card carries, so
|
|
1394
|
+
the export and the site renderer can call it with a front-matter dict
|
|
1395
|
+
without paying to build a `ModelCard`.
|
|
1396
|
+
|
|
1397
|
+
A card with no `model_type` gets `()`. Unknown class is unknown: it must
|
|
1398
|
+
never be turned into "there is nothing to know".
|
|
1399
|
+
"""
|
|
1400
|
+
return _inapplicable_paths(_coerce_types(model_type, model_subtypes))
|
|
1401
|
+
|
|
1402
|
+
|
|
1403
|
+
def _value_at(obj: Any, path: str) -> Any:
|
|
1404
|
+
"""Follow a dotted path through a mapping or a model. None when absent."""
|
|
1405
|
+
current = obj
|
|
1406
|
+
for part in path.split("."):
|
|
1407
|
+
if isinstance(current, dict):
|
|
1408
|
+
current = current.get(part)
|
|
1409
|
+
elif isinstance(current, BaseModel):
|
|
1410
|
+
current = getattr(current, part, None)
|
|
1411
|
+
else:
|
|
1412
|
+
return None
|
|
1413
|
+
if current is None:
|
|
1414
|
+
return None
|
|
1415
|
+
return current
|
|
1416
|
+
|
|
1417
|
+
|
|
1418
|
+
def _is_answered(value: Any) -> bool:
|
|
1419
|
+
"""Whether a value — or anything beneath a section — is a real answer.
|
|
1420
|
+
|
|
1421
|
+
The same test coverage uses: `None`, `""`, `False` and an empty collection
|
|
1422
|
+
are all "nobody filled this in", not data.
|
|
1423
|
+
"""
|
|
1424
|
+
if value is None or value == "" or value is False:
|
|
1425
|
+
return False
|
|
1426
|
+
if isinstance(value, BaseModel):
|
|
1427
|
+
return any(_is_answered(getattr(value, name, None))
|
|
1428
|
+
for name in type(value).model_fields)
|
|
1429
|
+
if isinstance(value, dict):
|
|
1430
|
+
return any(_is_answered(item) for item in value.values())
|
|
1431
|
+
if isinstance(value, (list, tuple, set)):
|
|
1432
|
+
return len(value) > 0
|
|
1433
|
+
return True
|
|
1434
|
+
|
|
1435
|
+
|
|
1436
|
+
def _answered_removed(paths: Iterable[str], card: Any) -> tuple[str, ...]:
|
|
1437
|
+
"""Drop any path this card actually answers. The card outranks the table.
|
|
1438
|
+
|
|
1439
|
+
One-way: it can only remove an inapplicability claim, never add one, so a
|
|
1440
|
+
card can never talk the catalogue into asserting that a question has no
|
|
1441
|
+
answer.
|
|
1442
|
+
"""
|
|
1443
|
+
return tuple(path for path in paths if not _is_answered(_value_at(card, path)))
|
|
1444
|
+
|
|
1445
|
+
|
|
1446
|
+
def applicability_block(model_type: Any,
|
|
1447
|
+
model_subtypes: Iterable[Any] = (),
|
|
1448
|
+
card: Any = None) -> dict[str, Any]:
|
|
1449
|
+
"""The derived block published beside a card in `/api/models/<id>.json`.
|
|
1450
|
+
|
|
1451
|
+
Additive: it adds information rather than widening any card field, so it is
|
|
1452
|
+
not a contract break on its own (see `docs/cli-contract.md`). `card` is the
|
|
1453
|
+
card's frontmatter (or a `ModelCard`); pass it so a path the card answers
|
|
1454
|
+
is never published as one it cannot have.
|
|
1455
|
+
"""
|
|
1456
|
+
paths = inapplicable_paths(model_type, model_subtypes)
|
|
1457
|
+
if card is not None:
|
|
1458
|
+
paths = _answered_removed(paths, card)
|
|
1459
|
+
return {
|
|
1460
|
+
"basis": "model_type",
|
|
1461
|
+
"model_type": getattr(model_type, "value", model_type) or None,
|
|
1462
|
+
"not_applicable": list(paths),
|
|
1463
|
+
}
|