modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
schema/card.py ADDED
@@ -0,0 +1,1463 @@
1
+ """ModelSpec Universal Model Card Schema — V3
2
+
3
+ Pydantic models for every section of the model intelligence card.
4
+ This is the source of truth. The YAML template, graph ingestion,
5
+ API responses, and CLI output all derive from these models.
6
+
7
+ Usage:
8
+ from schema.card import ModelCard
9
+ card = ModelCard.from_yaml("models/qwen/qwen3-30b-a3b.md")
10
+ card.validate()
11
+ print(card.applicable_field_coverage)
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from collections.abc import Iterable
17
+ from datetime import date, datetime
18
+ from functools import lru_cache
19
+ from pathlib import Path
20
+ from typing import Any, ClassVar, Literal
21
+
22
+ import yaml
23
+ from pydantic import BaseModel, Field, computed_field, field_validator, model_validator
24
+
25
+ from .applicability import FIELD_RULES, model_types
26
+ from .enums import (
27
+ ArchitectureType,
28
+ AttentionType,
29
+ BaseModelRelation,
30
+ BenchmarkCategory,
31
+ ConfidenceLevel,
32
+ DisclosureState,
33
+ EUAIActRisk,
34
+ EvalStatus,
35
+ LicenseType,
36
+ ModelStatus,
37
+ ModelType,
38
+ Modality,
39
+ OrgType,
40
+ PlatformCategory,
41
+ PositionalEncoding,
42
+ QuantFormat,
43
+ ResistanceLevel,
44
+ RiskTier,
45
+ Tier,
46
+ TokenizerType,
47
+ TrainingMethod,
48
+ UsePermission,
49
+ )
50
+
51
+
52
+ # ═══════════════════════════════════════════════════════════════
53
+ # Section 1: Identity
54
+ # ═══════════════════════════════════════════════════════════════
55
+
56
+ class Identity(BaseModel):
57
+ model_id: str = Field(..., description="Canonical ID: provider/model-name")
58
+ display_name: str
59
+ provider: str = Field(..., description="Provider slug (lowercase)")
60
+ provider_display: str = ""
61
+ family: str = ""
62
+ version: str = ""
63
+ release_date: str = ""
64
+ last_updated: str = ""
65
+ status: ModelStatus = ModelStatus.ACTIVE
66
+ model_type: ModelType | None = None
67
+ model_subtypes: list[ModelType] = []
68
+ tags: list[str] = []
69
+ pipeline_tag: str = ""
70
+
71
+
72
+ # ═══════════════════════════════════════════════════════════════
73
+ # Section 2: Architecture
74
+ # ═══════════════════════════════════════════════════════════════
75
+
76
+ class Architecture(BaseModel):
77
+ type: ArchitectureType | None = None
78
+ total_parameters: int | None = None
79
+ #: How `total_parameters` was obtained. `safetensors` is an exact Hub
80
+ #: count; values starting `model_card_published:` are a README figure.
81
+ #: Empty means unknown — either still null, or a legacy fill.
82
+ total_parameters_source: str = ""
83
+ active_parameters: int | None = None
84
+ num_experts: int | None = None
85
+ experts_per_token: int | None = None
86
+ num_layers: int | None = None
87
+ hidden_size: int | None = None
88
+ intermediate_size: int | None = None
89
+ attention_type: AttentionType | None = None
90
+ num_attention_heads: int | None = None
91
+ num_kv_heads: int | None = None
92
+ positional_encoding: PositionalEncoding | None = None
93
+ rope_theta: float | None = None
94
+ vocab_size: int | None = None
95
+ tokenizer_type: TokenizerType | None = None
96
+ embedding_dimensions: int | None = None
97
+ activation_function: str = ""
98
+ precision_native: str = ""
99
+ flash_attention: bool | None = None
100
+ tie_word_embeddings: bool | None = None
101
+ sliding_window_size: int | None = None
102
+ # Vision encoder
103
+ vision_encoder: str = ""
104
+ vision_resolution_max: str = ""
105
+ vision_patch_size: int | None = None
106
+ # Diffusion
107
+ diffusion_scheduler: str = ""
108
+ diffusion_steps_default: int | None = None
109
+ vae_type: str = ""
110
+
111
+ @model_validator(mode="after")
112
+ def _active_cannot_exceed_total(self) -> Architecture:
113
+ """A stored active count larger than the stored total is a catalogue bug.
114
+
115
+ That is how Mixtral-8x7B and GLM-4.5 were filed as 7B / 9B and told they
116
+ fitted on hardware that cannot hold the weights. Null on either side
117
+ stays legal — unknown is not a contradiction.
118
+ """
119
+ if (
120
+ self.active_parameters is not None
121
+ and self.total_parameters is not None
122
+ and self.active_parameters > self.total_parameters
123
+ ):
124
+ raise ValueError(
125
+ f"active_parameters ({self.active_parameters}) exceeds "
126
+ f"total_parameters ({self.total_parameters})"
127
+ )
128
+ return self
129
+
130
+
131
+ # ═══════════════════════════════════════════════════════════════
132
+ # Section 3: Lineage
133
+ # ═══════════════════════════════════════════════════════════════
134
+
135
+ class Lineage(BaseModel):
136
+ base_model: str = ""
137
+ base_model_relation: BaseModelRelation | None = None
138
+ merge_models: list[str] = []
139
+ adapter_type: str = ""
140
+ adapter_rank: int | None = None
141
+ training_datasets: list[str] = []
142
+ training_data_tokens: int | None = None
143
+ training_data_cutoff: str = ""
144
+ training_compute_flops: float | None = None
145
+ training_hardware: str = ""
146
+ training_time: str = ""
147
+ training_cost_estimate: str = ""
148
+ training_method: TrainingMethod | None = None
149
+ co2_emissions_kg: float | None = None
150
+ co2_source: str = ""
151
+ energy_kwh: float | None = None
152
+ library_name: str = ""
153
+
154
+
155
+ # ═══════════════════════════════════════════════════════════════
156
+ # Section 4: Licensing
157
+ # ═══════════════════════════════════════════════════════════════
158
+
159
+ #: Source kinds a policy determination may cite. `legacy-import` is the one
160
+ #: kind that is *not* evidence: see `PolicySource` below.
161
+ PolicySourceKind = Literal[
162
+ "license",
163
+ "terms_of_service",
164
+ "acceptable_use_policy",
165
+ "provider_documentation",
166
+ "provider_statement",
167
+ "legacy-import",
168
+ ]
169
+
170
+
171
+ class PolicySource(BaseModel):
172
+ """The document a policy determination was read from, and the day it was read.
173
+
174
+ A policy answer is only worth anything if the reader can go and check it,
175
+ and licences are rewritten without notice — so the date is as load-bearing
176
+ as the URL. This mirrors `BenchmarkEvidence`: everything needed to recheck
177
+ the claim, or it is not a claim.
178
+
179
+ `legacy-import` is the single exception and it is deliberately ugly. Eight
180
+ cards carried `commercial_use: true` with no citation from before this
181
+ shape existed. Discarding those values would lose information; dressing
182
+ them up with a plausible licence URL would manufacture evidence that was
183
+ never read. So they keep the value and carry a source that says, in the
184
+ published JSON, that nobody cited anything — the same thing
185
+ `evidence_basis: unverified-legacy` says about benchmark scores. It is
186
+ frozen to those eight by
187
+ `tests/test_policy_shape.py::test_legacy_import_is_frozen_to_the_migrated_eight`;
188
+ a ninth card cannot quietly use it.
189
+ """
190
+
191
+ kind: PolicySourceKind
192
+ url: str = ""
193
+ #: The day the document at `url` was read. ISO `YYYY-MM-DD`.
194
+ read_on: str = ""
195
+ #: Optional. The operative clause, verbatim and short, so a reader can see
196
+ #: what the determination was made from without refetching the document.
197
+ quote: str = ""
198
+
199
+ @model_validator(mode="after")
200
+ def _evidence_or_an_admission(self) -> PolicySource:
201
+ if self.kind == "legacy-import":
202
+ if self.url or self.read_on or self.quote:
203
+ raise ValueError(
204
+ "a legacy-import source is the admission that nothing was "
205
+ "cited; it cannot carry a url, a read date or a quote. "
206
+ "If a document was actually read, cite it with a real kind."
207
+ )
208
+ return self
209
+ if not self.url.startswith(("http://", "https://")):
210
+ raise ValueError(
211
+ "url must be the document the determination was read from; "
212
+ "a policy answer without one cannot be rechecked"
213
+ )
214
+ try:
215
+ date.fromisoformat(self.read_on)
216
+ except ValueError as exc:
217
+ raise ValueError(
218
+ f"read_on must be the exact ISO date the document was read, "
219
+ f"got {self.read_on!r}. Licence terms change; an undated "
220
+ f"reading cannot be trusted later."
221
+ ) from exc
222
+ return self
223
+
224
+ @property
225
+ def is_evidence(self) -> bool:
226
+ """False for `legacy-import`, which is a value with no citation."""
227
+ return self.kind != "legacy-import"
228
+
229
+
230
+ #: The `commercial_use` values that assert something about the world, and so
231
+ #: require a source. `UNSPECIFIED` and `WITHHELD` assert nothing about the
232
+ #: licence — they describe this file's contents.
233
+ DETERMINED_PERMISSIONS = frozenset({
234
+ UsePermission.ALLOWED,
235
+ UsePermission.RESTRICTED,
236
+ UsePermission.PROHIBITED,
237
+ })
238
+
239
+
240
+ class Licensing(BaseModel):
241
+ open_weights: bool = False
242
+ license_type: LicenseType | None = None
243
+ license_url: str = ""
244
+ tos_url: str = ""
245
+ acceptable_use_policy_url: str = ""
246
+ not_for_all_audiences: bool = False
247
+ #: Was `bool | None` until MODEL-77, which could not express the answer for
248
+ #: 169 cards whose licence grants commercial use *up to a threshold*
249
+ #: (llama-community, gemma, deepseek). That is neither true nor false; it
250
+ #: is `restricted`, with the threshold in `commercial_use_conditions`.
251
+ commercial_use: UsePermission = UsePermission.UNSPECIFIED
252
+ #: Required whenever `commercial_use` is a determination; forbidden when it
253
+ #: is not, so an empty value can never look sourced.
254
+ commercial_use_source: PolicySource | None = None
255
+ #: Where a `restricted` grant's condition lives: one short line stating the
256
+ #: threshold or carve-out ("free below 700M MAU; a licence is required
257
+ #: above it"), not prose buried in the card body where no consumer reads it.
258
+ commercial_use_conditions: str = ""
259
+ defense_use: UsePermission = UsePermission.UNSPECIFIED
260
+ government_use: UsePermission = UsePermission.UNSPECIFIED
261
+ medical_use: UsePermission = UsePermission.UNSPECIFIED
262
+ academic_use: UsePermission = UsePermission.UNSPECIFIED
263
+ geographic_restrictions: list[str] = []
264
+ export_control_notes: str = ""
265
+ origin_country: str = ""
266
+ origin_org_type: OrgType | None = None
267
+
268
+ @field_validator("commercial_use", mode="before")
269
+ @classmethod
270
+ def _reject_the_old_boolean(cls, value: Any) -> Any:
271
+ """A pre-MODEL-77 card fails loudly rather than being guessed at.
272
+
273
+ `True`/`False` are not silently mapped: `False` in particular could
274
+ have meant "prohibited" or "restricted in a way the author could not
275
+ express", and picking one would invent a determination.
276
+ """
277
+ if isinstance(value, bool):
278
+ raise ValueError(
279
+ "commercial_use is a UsePermission since MODEL-77, not a bool. "
280
+ "true became 'allowed'; false is ambiguous between 'prohibited' "
281
+ "and 'restricted' and must be re-read from the licence."
282
+ )
283
+ if value is None:
284
+ raise ValueError(
285
+ "commercial_use is no longer nullable: use 'unspecified' for "
286
+ "not-yet-researched, or 'withheld' for determined-not-published."
287
+ )
288
+ return value
289
+
290
+ @model_validator(mode="after")
291
+ def _a_determination_carries_its_source(self) -> Licensing:
292
+ determined = self.commercial_use in DETERMINED_PERMISSIONS
293
+ if determined and self.commercial_use_source is None:
294
+ raise ValueError(
295
+ f"commercial_use is {self.commercial_use.value!r} with no "
296
+ "commercial_use_source. Standing rule 1: an uncited policy "
297
+ "answer is not an answer."
298
+ )
299
+ if not determined and self.commercial_use_source is not None:
300
+ raise ValueError(
301
+ f"commercial_use is {self.commercial_use.value!r} but carries a "
302
+ "commercial_use_source. Nothing has been published to cite; a "
303
+ "withheld determination keeps its source in the enrichment "
304
+ "record, not on the public card."
305
+ )
306
+ if self.commercial_use is UsePermission.RESTRICTED and not self.commercial_use_conditions.strip():
307
+ raise ValueError(
308
+ "commercial_use is 'restricted' with no commercial_use_conditions. "
309
+ "'Restricted' without the restriction is unusable: it is the "
310
+ "condition that tells a reader whether they are inside it."
311
+ )
312
+ if self.commercial_use is not UsePermission.RESTRICTED and self.commercial_use_conditions.strip():
313
+ raise ValueError(
314
+ f"commercial_use is {self.commercial_use.value!r} but carries "
315
+ "commercial_use_conditions. Conditions belong to a restricted "
316
+ "grant; anywhere else they contradict the value."
317
+ )
318
+ return self
319
+
320
+
321
+ # ═══════════════════════════════════════════════════════════════
322
+ # Section 5: Modalities
323
+ # ═══════════════════════════════════════════════════════════════
324
+ #
325
+ # Nested modality and capability models declare ``__applicable_model_types__``.
326
+ # ``ModelCard.applicable_field_coverage`` counts a nested section only when the
327
+ # card's ``model_type`` (or a ``model_subtype``) is in that set. Sections with
328
+ # no declaration count for every type. The field *sets* are the nested models'
329
+ # own fields — not a hand list of paths.
330
+
331
+
332
+ #: Shared with the field-level table in `schema/applicability.py`, which is the
333
+ #: same idea one level down: sections here, named fields there.
334
+ _model_types = model_types
335
+
336
+
337
+ _LLM_TYPES = _model_types("llm-")
338
+ _EMBEDDING_TYPES = _model_types("embedding-")
339
+ _AUDIO_TYPES = _model_types("audio-")
340
+ _IMAGE_TYPES = _model_types("image-generation", "image-editing")
341
+ #: `decision-model` (MODEL-98) reads text state, so `max_input_tokens` and
342
+ #: `context_window` are real questions for it. It is deliberately absent from
343
+ #: `_GENERATIVE_TEXT_TYPES` below — it writes no text — and the output-shaped
344
+ #: fields inside this subtree are excluded field by field in
345
+ #: `schema/applicability.py`, which is finer than a whole-section gate can be.
346
+ _TEXT_TYPES = _LLM_TYPES | _model_types(
347
+ "vlm", "agent-model", "medical", "legal", "financial",
348
+ "router", "reward-model", "safety-classifier", "text-encoder", "document-ocr",
349
+ "decision-model",
350
+ )
351
+ _GENERATIVE_TEXT_TYPES = _LLM_TYPES | _model_types(
352
+ "vlm", "agent-model", "medical", "legal", "financial",
353
+ "router", "reward-model", "safety-classifier",
354
+ )
355
+ _VISION_TYPES = _IMAGE_TYPES | _model_types(
356
+ "vlm", "vision-encoder", "document-ocr", "embedding-multimodal",
357
+ )
358
+
359
+
360
+ class VisionDetail(BaseModel):
361
+ __applicable_model_types__: ClassVar[frozenset[ModelType]] = _VISION_TYPES
362
+ supported: bool = False
363
+ ocr: bool = False
364
+ chart_reading: bool = False
365
+ spatial_reasoning: bool = False
366
+ handwriting: bool = False
367
+ object_detection: bool = False
368
+ object_counting: bool = False
369
+ visual_grounding: bool = False
370
+ max_image_resolution: str = ""
371
+ max_images_per_request: int | None = None
372
+ video_frames: bool = False
373
+
374
+
375
+ class AudioDetail(BaseModel):
376
+ __applicable_model_types__: ClassVar[frozenset[ModelType]] = _AUDIO_TYPES
377
+ input_supported: bool = False
378
+ output_supported: bool = False
379
+ realtime_streaming: bool = False
380
+ asr_languages: list[str] = []
381
+ tts_languages: list[str] = []
382
+ tts_voices: int | None = None
383
+ voice_cloning: bool = False
384
+ speaker_diarization: bool = False
385
+ music_understanding: bool = False
386
+ music_generation: bool = False
387
+ max_audio_duration_sec: int | None = None
388
+
389
+
390
+ class VideoDetail(BaseModel):
391
+ __applicable_model_types__: ClassVar[frozenset[ModelType]] = _model_types("video-generation")
392
+ input_supported: bool = False
393
+ output_supported: bool = False
394
+ max_input_duration_sec: int | None = None
395
+ max_output_duration_sec: int | None = None
396
+ max_resolution: str = ""
397
+ max_fps: int | None = None
398
+ audio_sync: bool = False
399
+ temporal_reasoning: bool = False
400
+
401
+
402
+ class DocumentDetail(BaseModel):
403
+ __applicable_model_types__: ClassVar[frozenset[ModelType]] = _model_types(
404
+ "document-ocr", "vlm",
405
+ )
406
+ pdf_native: bool = False
407
+ table_extraction: bool = False
408
+ form_understanding: bool = False
409
+ max_pages: int | None = None
410
+ layout_analysis: bool = False
411
+
412
+
413
+ class ImageGenDetail(BaseModel):
414
+ __applicable_model_types__: ClassVar[frozenset[ModelType]] = _IMAGE_TYPES
415
+ supported: bool = False
416
+ max_resolution: str = ""
417
+ aspect_ratios: list[str] = []
418
+ inpainting: bool = False
419
+ outpainting: bool = False
420
+ img2img: bool = False
421
+ text_rendering_quality: str = ""
422
+ style_control: bool = False
423
+ controlnet_support: bool = False
424
+ lora_support: bool = False
425
+
426
+
427
+ class EmbeddingDetail(BaseModel):
428
+ __applicable_model_types__: ClassVar[frozenset[ModelType]] = _EMBEDDING_TYPES
429
+ supported: bool = False
430
+ dimensions: int | None = None
431
+ dimensions_configurable: bool = False
432
+ dimension_options: list[int] = []
433
+ max_input_tokens: int | None = None
434
+ similarity_metric: str = ""
435
+ normalized: bool | None = None
436
+ batch_size_max: int | None = None
437
+ instruction_aware: bool = False
438
+
439
+
440
+ class RerankingDetail(BaseModel):
441
+ __applicable_model_types__: ClassVar[frozenset[ModelType]] = _model_types("reranker")
442
+ supported: bool = False
443
+ max_input_pairs: int | None = None
444
+ max_input_length: int | None = None
445
+ cross_encoder: bool | None = None
446
+
447
+
448
+ class TextDetail(BaseModel):
449
+ __applicable_model_types__: ClassVar[frozenset[ModelType]] = _TEXT_TYPES
450
+ max_input_tokens: int | None = None
451
+ max_output_tokens: int | None = None
452
+ context_window: int | None = None
453
+ streaming: bool | None = None
454
+ fill_in_middle: bool | None = None
455
+ json_mode: bool | None = None
456
+ system_prompt: bool | None = None
457
+
458
+
459
+ class Modalities(BaseModel):
460
+ input: list[Modality] = []
461
+ output: list[Modality] = []
462
+ text: TextDetail = TextDetail()
463
+ vision: VisionDetail = VisionDetail()
464
+ audio: AudioDetail = AudioDetail()
465
+ video: VideoDetail = VideoDetail()
466
+ document: DocumentDetail = DocumentDetail()
467
+ image_generation: ImageGenDetail = ImageGenDetail()
468
+ embeddings: EmbeddingDetail = EmbeddingDetail()
469
+ reranking: RerankingDetail = RerankingDetail()
470
+
471
+
472
+ # ═══════════════════════════════════════════════════════════════
473
+ # Section 6: Capabilities
474
+ # ═══════════════════════════════════════════════════════════════
475
+
476
+ class CodingCapability(BaseModel):
477
+ overall: Tier | None = None
478
+ languages: list[str] = []
479
+ agentic_coding: bool = False
480
+ code_review: bool = False
481
+ refactoring: bool = False
482
+ debugging: bool = False
483
+ test_generation: bool = False
484
+ documentation: bool = False
485
+ code_completion: bool = False
486
+ multi_file_editing: bool = False
487
+ fill_in_middle: bool = False
488
+ lsp_integration: bool = False
489
+ repository_understanding: bool = False
490
+
491
+
492
+ class ReasoningCapability(BaseModel):
493
+ overall: Tier | None = None
494
+ mathematical: bool = False
495
+ logical: bool = False
496
+ scientific: bool = False
497
+ planning: bool = False
498
+ multi_step: bool = False
499
+ chain_of_thought: bool = False
500
+ self_correction: bool = False
501
+ spatial: bool = False
502
+ temporal: bool = False
503
+ causal: bool = False
504
+ think_budget_control: bool = False
505
+
506
+
507
+ class ToolUseCapability(BaseModel):
508
+ overall: Tier | None = None
509
+ function_calling: bool = False
510
+ mcp_compatible: bool = False
511
+ parallel_tool_calls: bool = False
512
+ tool_selection_accuracy: ConfidenceLevel | None = None
513
+ multi_turn_tool_use: bool = False
514
+ tool_error_recovery: bool = False
515
+ computer_use: bool = False
516
+
517
+
518
+ class LanguageCapability(BaseModel):
519
+ multilingual: bool = False
520
+ num_languages: int | None = None
521
+ strong_languages: list[str] = []
522
+ translation_quality: Tier | None = None
523
+ long_context_retrieval: Tier | None = None
524
+
525
+
526
+ class CreativeCapability(BaseModel):
527
+ writing: Tier | None = None
528
+ summarization: Tier | None = None
529
+ instruction_following: Tier | None = None
530
+ storytelling: Tier | None = None
531
+ technical_writing: Tier | None = None
532
+
533
+
534
+ class SafetyAlignment(BaseModel):
535
+ alignment_approach: str = ""
536
+ refusal_rate: ConfidenceLevel | None = None
537
+ jailbreak_resistance: ConfidenceLevel | None = None
538
+ content_safety_tier: Tier | None = None
539
+ guardrail_builtin: bool = False
540
+
541
+
542
+ class DomainCapability(BaseModel):
543
+ medical_knowledge: Tier | None = None
544
+ legal_knowledge: Tier | None = None
545
+ financial_knowledge: Tier | None = None
546
+ scientific_knowledge: Tier | None = None
547
+
548
+
549
+ class AgentCapability(BaseModel):
550
+ autonomous_execution: bool = False
551
+ web_browsing: bool = False
552
+ file_system_access: bool = False
553
+ code_execution: bool = False
554
+ long_running_tasks: bool = False
555
+ memory_management: bool = False
556
+ self_delegation: bool = False
557
+
558
+
559
+ class Capabilities(BaseModel):
560
+ __applicable_model_types__: ClassVar[frozenset[ModelType]] = _GENERATIVE_TEXT_TYPES
561
+ coding: CodingCapability = CodingCapability()
562
+ reasoning: ReasoningCapability = ReasoningCapability()
563
+ tool_use: ToolUseCapability = ToolUseCapability()
564
+ language: LanguageCapability = LanguageCapability()
565
+ creative: CreativeCapability = CreativeCapability()
566
+ safety_alignment: SafetyAlignment = SafetyAlignment()
567
+ domain_specific: DomainCapability = DomainCapability()
568
+ agent_capabilities: AgentCapability = AgentCapability()
569
+
570
+
571
+ # ═══════════════════════════════════════════════════════════════
572
+ # Section 7: Cost
573
+ # ═══════════════════════════════════════════════════════════════
574
+
575
+ class Cost(BaseModel):
576
+ input: float | None = None
577
+ output: float | None = None
578
+ reasoning: float | None = None
579
+ cache_read: float | None = None
580
+ cache_write: float | None = None
581
+ input_audio: float | None = None
582
+ output_audio: float | None = None
583
+ input_image: float | None = None
584
+ output_image: float | None = None
585
+ output_video_per_sec: float | None = None
586
+ batch_input: float | None = None
587
+ batch_output: float | None = None
588
+ embedding_per_million: float | None = None
589
+ reranking_per_million: float | None = None
590
+ finetune_per_million_tokens: float | None = None
591
+ finetune_hosting_per_hour: float | None = None
592
+ free_tier: bool = False
593
+ free_tier_limits: str = ""
594
+ note: str = ""
595
+
596
+
597
+ # ═══════════════════════════════════════════════════════════════
598
+ # Section 8: Availability
599
+ # ═══════════════════════════════════════════════════════════════
600
+
601
+ class PlatformEntry(BaseModel):
602
+ """A single platform where a model may be available."""
603
+ available: bool = False
604
+ model_id: str = ""
605
+ url: str = ""
606
+ fine_tuning: bool = False
607
+ gated: bool = False
608
+ regions: list[str] = []
609
+ notes: str = ""
610
+
611
+
612
+ class PrimaryProvider(BaseModel):
613
+ name: str = ""
614
+ platform_url: str = ""
615
+ api_endpoint: str = ""
616
+ npm_package: str = ""
617
+ env_vars: list[str] = []
618
+ model_id_on_platform: str = ""
619
+ rate_limit_rpm: int | None = None
620
+ rate_limit_tpm: int | None = None
621
+ sla_uptime: str = ""
622
+ regions: list[str] = []
623
+ #: Where the provider commits to processing data. `null` until it is a
624
+ #: determination — see `data_residency_disclosure`. Was `list[str] = []`
625
+ #: until MODEL-77, where the default was indistinguishable from an answer
626
+ #: and sat on all 1,339 cards while being researched on none of them.
627
+ data_residency: list[str] | None = None
628
+ data_residency_disclosure: DisclosureState = DisclosureState.UNRESEARCHED
629
+ #: Required when the disclosure is `published`; forbidden otherwise, for
630
+ #: the same reason as `commercial_use_source`.
631
+ data_residency_source: PolicySource | None = None
632
+ hipaa_eligible: bool = False
633
+ fedramp_authorized: bool = False
634
+ soc2_compliant: bool = False
635
+ free_tier: bool = False
636
+ free_tier_details: str = ""
637
+
638
+ @model_validator(mode="after")
639
+ def _residency_states_are_not_interchangeable(self) -> PrimaryProvider:
640
+ state = self.data_residency_disclosure
641
+ if state is DisclosureState.PUBLISHED:
642
+ if self.data_residency is None:
643
+ raise ValueError(
644
+ "data_residency_disclosure is 'published' but data_residency "
645
+ "is null. An empty list is a legitimate published answer "
646
+ "('no residency commitment'); null is not an answer at all."
647
+ )
648
+ if self.data_residency_source is None:
649
+ raise ValueError(
650
+ "data_residency is published with no data_residency_source. "
651
+ "Standing rule 1: an uncited policy answer is not an answer."
652
+ )
653
+ return self
654
+ if self.data_residency is not None:
655
+ raise ValueError(
656
+ f"data_residency_disclosure is {state.value!r} but data_residency "
657
+ "carries a value. Only a published determination has one; set "
658
+ "the disclosure to 'published' and cite it, or clear the value."
659
+ )
660
+ if self.data_residency_source is not None:
661
+ raise ValueError(
662
+ f"data_residency_disclosure is {state.value!r} but carries a "
663
+ "data_residency_source. Nothing has been published to cite."
664
+ )
665
+ return self
666
+
667
+
668
+ class Availability(BaseModel):
669
+ primary_provider: PrimaryProvider = PrimaryProvider()
670
+ # Cloud platforms
671
+ aws_bedrock: PlatformEntry = PlatformEntry(url="https://aws.amazon.com/bedrock/")
672
+ azure_ai_foundry: PlatformEntry = PlatformEntry(url="https://ai.azure.com/")
673
+ google_vertex_ai: PlatformEntry = PlatformEntry(url="https://cloud.google.com/vertex-ai")
674
+ nvidia_nim: PlatformEntry = PlatformEntry(url="https://build.nvidia.com/")
675
+ ibm_watsonx: PlatformEntry = PlatformEntry(url="https://www.ibm.com/watsonx")
676
+ snowflake_cortex: PlatformEntry = PlatformEntry(url="https://www.snowflake.com/en/data-cloud/cortex/")
677
+ # Inference providers
678
+ groq: PlatformEntry = PlatformEntry(url="https://groq.com/")
679
+ together_ai: PlatformEntry = PlatformEntry(url="https://www.together.ai/")
680
+ fireworks_ai: PlatformEntry = PlatformEntry(url="https://fireworks.ai/")
681
+ replicate: PlatformEntry = PlatformEntry(url="https://replicate.com/")
682
+ deepinfra: PlatformEntry = PlatformEntry(url="https://deepinfra.com/")
683
+ cerebras: PlatformEntry = PlatformEntry(url="https://www.cerebras.ai/")
684
+ sambanova: PlatformEntry = PlatformEntry(url="https://sambanova.ai/")
685
+ # Aggregators
686
+ openrouter: PlatformEntry = PlatformEntry(url="https://openrouter.ai/")
687
+ # AI apps
688
+ cursor: PlatformEntry = PlatformEntry(url="https://cursor.com/")
689
+ github_copilot: PlatformEntry = PlatformEntry(url="https://github.com/features/copilot")
690
+ perplexity: PlatformEntry = PlatformEntry(url="https://www.perplexity.ai/")
691
+ raycast: PlatformEntry = PlatformEntry(url="https://www.raycast.com/")
692
+ poe: PlatformEntry = PlatformEntry(url="https://poe.com/")
693
+ # Consumer chat
694
+ chatgpt: PlatformEntry = PlatformEntry(url="https://chat.openai.com/")
695
+ claude_ai: PlatformEntry = PlatformEntry(url="https://claude.ai/")
696
+ gemini_app: PlatformEntry = PlatformEntry(url="https://gemini.google.com/")
697
+ grok_xai: PlatformEntry = PlatformEntry(url="https://x.ai/")
698
+ meta_ai: PlatformEntry = PlatformEntry(url="https://www.meta.ai/")
699
+ copilot_microsoft: PlatformEntry = PlatformEntry(url="https://copilot.microsoft.com/")
700
+ # Provider platforms
701
+ mistral_plateforme: PlatformEntry = PlatformEntry(url="https://console.mistral.ai/")
702
+ cohere: PlatformEntry = PlatformEntry(url="https://cohere.com/")
703
+ ai21_labs: PlatformEntry = PlatformEntry(url="https://www.ai21.com/")
704
+ stability_ai: PlatformEntry = PlatformEntry(url="https://stability.ai/")
705
+ # Chinese platforms
706
+ deepseek: PlatformEntry = PlatformEntry(url="https://platform.deepseek.com/")
707
+ qwen_alibaba: PlatformEntry = PlatformEntry(url="https://www.alibabacloud.com/en/solutions/generative-ai/qwen")
708
+ baidu_ernie: PlatformEntry = PlatformEntry(url="https://cloud.baidu.com/")
709
+ bytedance_doubao: PlatformEntry = PlatformEntry(url="https://www.volcengine.com/")
710
+ tencent_hunyuan: PlatformEntry = PlatformEntry(url="https://cloud.tencent.com/")
711
+ zhipu_glm: PlatformEntry = PlatformEntry(url="https://www.zhipuai.cn/")
712
+ moonshot_kimi: PlatformEntry = PlatformEntry(url="https://www.moonshot.cn/")
713
+ minimax: PlatformEntry = PlatformEntry(url="https://www.minimax.chat/")
714
+ zero_one_ai: PlatformEntry = PlatformEntry(url="https://www.01.ai/")
715
+ # Regional
716
+ tii_falcon: PlatformEntry = PlatformEntry(url="https://falconllm.tii.ae/")
717
+ samsung_gauss: PlatformEntry = PlatformEntry(url="https://www.samsung.com/")
718
+ upstage_solar: PlatformEntry = PlatformEntry(url="https://www.upstage.ai/")
719
+ # Local
720
+ ollama: PlatformEntry = PlatformEntry(url="https://ollama.com/")
721
+ lm_studio: PlatformEntry = PlatformEntry(url="https://lmstudio.ai/")
722
+ gpt4all: PlatformEntry = PlatformEntry(url="https://gpt4all.io/")
723
+ jan_ai: PlatformEntry = PlatformEntry(url="https://jan.ai/")
724
+ mlx_community: PlatformEntry = PlatformEntry(url="https://huggingface.co/mlx-community")
725
+ open_webui: PlatformEntry = PlatformEntry(url="https://openwebui.com/")
726
+ # Model hubs
727
+ huggingface: PlatformEntry = PlatformEntry(url="https://huggingface.co/")
728
+ modelscope: PlatformEntry = PlatformEntry(url="https://modelscope.cn/")
729
+ kaggle_models: PlatformEntry = PlatformEntry(url="https://www.kaggle.com/models")
730
+ # Overflow
731
+ other_platforms: list[PlatformEntry] = []
732
+
733
+ def platforms_available(self) -> list[str]:
734
+ """Return names of all platforms where this model is available."""
735
+ available = []
736
+ for field_name, field_value in self:
737
+ if isinstance(field_value, PlatformEntry) and field_value.available:
738
+ available.append(field_name)
739
+ return available
740
+
741
+
742
+ # ═══════════════════════════════════════════════════════════════
743
+ # Section 9: Benchmarks
744
+ # ═══════════════════════════════════════════════════════════════
745
+
746
+ class BenchmarkEvidence(BaseModel):
747
+ """One score, with everything needed to check it.
748
+
749
+ The flat `scores` dict below carries one collection date for a whole card
750
+ and a comma-joined source list, so no individual number can be attributed,
751
+ dated or rechecked. This record is the shape the benchmark catalogue's
752
+ evidence contract requires, and it mirrors the census evidence ledger so the
753
+ two can be reconciled rather than diverging.
754
+
755
+ Every field here is required. A record that cannot say where a number came
756
+ from or when is not evidence, and admitting a partial one would quietly
757
+ reintroduce exactly the problem this replaces.
758
+ """
759
+
760
+ benchmark_id: str
761
+ model_id_as_evaluated: str
762
+ score: float
763
+ unit: str
764
+ source_url: str
765
+ #: benchmark author, independent evaluator, or the provider's own claim.
766
+ #: Provider self-report is legitimate and must be visibly distinguishable.
767
+ source_kind: Literal["benchmark_author", "independent_evaluator", "provider_self_report"]
768
+ evidence_date: str
769
+ #: `evaluated` when the run date is disclosed; `published` when only the
770
+ #: publication date is. Never infer a run date from a retrieval timestamp.
771
+ date_type: Literal["evaluated", "published"]
772
+ verified_at: str
773
+ benchmark_version: str = ""
774
+ configuration: str = ""
775
+ limitations: str = ""
776
+
777
+ @field_validator("source_url")
778
+ @classmethod
779
+ def _url_must_be_real(cls, value: str) -> str:
780
+ if not value.startswith(("http://", "https://")):
781
+ raise ValueError("source_url must be a URL; a score without one is not evidence")
782
+ return value
783
+
784
+ @field_validator("evidence_date", "verified_at")
785
+ @classmethod
786
+ def _dates_must_be_iso(cls, value: str) -> str:
787
+ from datetime import date as _date
788
+ try:
789
+ _date.fromisoformat(value)
790
+ except ValueError as exc:
791
+ raise ValueError(f"must be an exact ISO date YYYY-MM-DD, got {value!r}") from exc
792
+ return value
793
+
794
+
795
+ class Benchmarks(BaseModel):
796
+ # All benchmark scores in a single open-ended dictionary.
797
+ # Keys are benchmark identifiers (e.g. "humaneval", "mmlu_pro",
798
+ # "multipl_e_rust", "mmlu_chemistry", "pubmedqa", "flores_en_zh").
799
+ # No fixed schema — any benchmark can be added without code changes.
800
+ #: V2 quarantine: the decision engine never reads benchmarks.scores.
801
+ #: MODEL-118 re-sources these values; v1 retains its existing behavior.
802
+ scores: dict[str, float] = {}
803
+
804
+ #: Verified, per-score evidence. Everything in `scores` above that has no
805
+ #: matching record here is unverified-legacy and must be presented as such.
806
+ evidence: list[BenchmarkEvidence] = []
807
+
808
+ # Meta. These describe `scores` only, and are the reason it cannot be
809
+ # attributed: one date and one source list for the whole card.
810
+ benchmark_source: str = ""
811
+ benchmark_as_of: str = ""
812
+ benchmark_notes: str = ""
813
+
814
+ def filled_count(self) -> int:
815
+ return len(self.scores)
816
+
817
+ def verified_ids(self) -> set[str]:
818
+ return {e.benchmark_id for e in self.evidence}
819
+
820
+ def is_verified(self, benchmark_id: str) -> bool:
821
+ return benchmark_id in self.verified_ids()
822
+
823
+
824
+ # ═══════════════════════════════════════════════════════════════
825
+ # Section 10: Hardware & Deployment
826
+ # ═══════════════════════════════════════════════════════════════
827
+
828
+ class HardwareProfile(BaseModel):
829
+ fits: bool | None = None
830
+ best_quant: str = ""
831
+ vram_usage_gb: float | None = None
832
+ ram_usage_gb: float | None = None
833
+ tokens_per_sec: float | None = None
834
+ prompt_tps: float | None = None
835
+ ttft_ms: float | None = None
836
+ max_context_at_quant: int | None = None
837
+ inference_engine: str = ""
838
+ notes: str = ""
839
+
840
+
841
+ class Runtimes(BaseModel):
842
+ gguf: bool = False
843
+ ollama: bool = False
844
+ ollama_tag: str = ""
845
+ lm_studio: bool = False
846
+ vllm: bool = False
847
+ trt_llm: bool = False
848
+ mlx: bool = False
849
+ llama_cpp: bool = False
850
+ sglang: bool = False
851
+ transformers: bool = False
852
+ exllamav2: bool = False
853
+ core_ml: bool = False
854
+ onnx: bool = False
855
+ triton: bool = False
856
+ nim: bool = False
857
+
858
+
859
+ class Deployment(BaseModel):
860
+ api_only: bool = False
861
+ local_inference: bool = False
862
+ self_hostable: bool = False
863
+ fine_tuning_supported: bool = False
864
+ fine_tuning_methods: list[str] = []
865
+ quantizations_available: list[str] = []
866
+ hardware_profiles: dict[str, HardwareProfile] = Field(default_factory=lambda: {
867
+ "nvidia_5090_32gb": HardwareProfile(),
868
+ "dgx_spark_128gb": HardwareProfile(),
869
+ "macbook_m4_pro_64gb": HardwareProfile(),
870
+ "macbook_air_m4_24gb": HardwareProfile(),
871
+ })
872
+ custom_hardware: list[HardwareProfile] = []
873
+ runtimes: Runtimes = Runtimes()
874
+
875
+
876
+ # ═══════════════════════════════════════════════════════════════
877
+ # Section 11-14: Risk, Performance, Adoption, Downselect
878
+ # ═══════════════════════════════════════════════════════════════
879
+
880
+ class BiasEvaluation(BaseModel):
881
+ conducted: bool = False
882
+ methodology: str = ""
883
+ results_summary: str = ""
884
+ known_biases: list[str] = []
885
+
886
+
887
+ class PrivacyPosture(BaseModel):
888
+ data_retention_policy: str = ""
889
+ pii_handling: str = ""
890
+ training_data_pii_scrubbed: bool | None = None
891
+ data_processing_location: list[str] = []
892
+ gdpr_compliant: bool | None = None
893
+ hipaa_eligible: bool | None = None
894
+ ccpa_compliant: bool | None = None
895
+
896
+
897
+ class SupplyChain(BaseModel):
898
+ training_data_transparency: str = ""
899
+ model_provenance_documented: bool = False
900
+ third_party_dependencies: list[str] = []
901
+ ai_bom_available: bool = False
902
+ reproducible: bool = False
903
+
904
+
905
+ class RegulatoryAlignment(BaseModel):
906
+ eu_ai_act_risk_level: EUAIActRisk | None = None
907
+ nist_rmf_profile: str = ""
908
+ iso_42001_certified: bool | None = None
909
+ soc2_type2: bool | None = None
910
+ fedramp_level: str = ""
911
+
912
+
913
+ class RiskGovernance(BaseModel):
914
+ valid_and_reliable: str = ""
915
+ safe: str = ""
916
+ secure_and_resilient: str = ""
917
+ accountable_and_transparent: str = ""
918
+ explainable_and_interpretable: str = ""
919
+ privacy_enhanced: str = ""
920
+ fair_with_bias_managed: str = ""
921
+ bias_evaluation: BiasEvaluation = BiasEvaluation()
922
+ adversarial_robustness: ResistanceLevel = ResistanceLevel.UNTESTED
923
+ privacy: PrivacyPosture = PrivacyPosture()
924
+ supply_chain: SupplyChain = SupplyChain()
925
+ incident_history: list[str] = []
926
+ known_failure_modes: list[str] = []
927
+ regulatory: RegulatoryAlignment = RegulatoryAlignment()
928
+
929
+
930
+ class InferencePerformance(BaseModel):
931
+ api_latency_p50_ms: float | None = None
932
+ api_latency_p99_ms: float | None = None
933
+ api_ttft_ms: float | None = None
934
+ api_tps_output: float | None = None
935
+ api_tps_input: float | None = None
936
+ context_speed_degradation: str = ""
937
+ generation_time_sec: float | None = None
938
+ quality_per_dollar: float | None = None
939
+ quality_per_watt: float | None = None
940
+
941
+
942
+ class Adoption(BaseModel):
943
+ huggingface_downloads: int | None = None
944
+ huggingface_likes: int | None = None
945
+ ollama_pulls: int | None = None
946
+ community_forks: int | None = None
947
+ is_common_distillation_teacher: bool = False
948
+ is_common_finetune_base: bool = False
949
+ openrouter_ranking: int | None = None
950
+ open_webui_ranking: int | None = None
951
+ notable_users: list[str] = []
952
+
953
+
954
+ class Downselect(BaseModel):
955
+ compliance_tags: list[str] = []
956
+ clearance_tags: list[str] = []
957
+ defense_tags: list[str] = []
958
+ sovereignty_tags: list[str] = []
959
+ use_case_tags: list[str] = []
960
+ eval_status: EvalStatus | None = None
961
+ risk_tier: RiskTier | None = None
962
+ cost_tier: str = ""
963
+ custom_score: float | None = None
964
+ custom_notes: str = ""
965
+ reviewed_by: str = ""
966
+ review_date: str = ""
967
+ approval_authority: str = ""
968
+ next_review_date: str = ""
969
+
970
+
971
+ # ═══════════════════════════════════════════════════════════════
972
+ # Section 15: Sources & Metadata
973
+ # ═══════════════════════════════════════════════════════════════
974
+
975
+ class Sources(BaseModel):
976
+ models_dev_url: str = ""
977
+ provider_docs_url: str = ""
978
+ huggingface_url: str = ""
979
+ arxiv_url: str = ""
980
+ paper_url: str = ""
981
+ github_url: str = ""
982
+ ollama_url: str = ""
983
+ artificial_analysis_url: str = ""
984
+ arena_url: str = ""
985
+ # Freshness tracking
986
+ last_scraped_models_dev: str = ""
987
+ last_scraped_huggingface: str = ""
988
+ last_scraped_benchmarks: str = ""
989
+ last_scraped_pricing: str = ""
990
+
991
+
992
+
993
+ # ═══════════════════════════════════════════════════════════════
994
+ # Section 16: Authoring Guide (MODEL-8)
995
+ # ═══════════════════════════════════════════════════════════════
996
+ #
997
+ # How to write for one model: what helps, what wastes tokens. Every claim is a
998
+ # short paraphrase of the provider's own guidance, dated and sourced. A guide is
999
+ # pinned to the card's `version` (the provider's API model string); when that
1000
+ # changes, the guide must be re-reviewed or marked `stale`. Stale is never kept
1001
+ # silently.
1002
+
1003
+ GuideSourceKind = Literal["provider-guidance", "system-card", "release-notes", "model-docs"]
1004
+ GUIDE_SECTIONS = (
1005
+ "prompt_shape", "system_message", "reasoning_and_tools",
1006
+ "formatting", "failure_modes", "retry_advice",
1007
+ )
1008
+
1009
+
1010
+ def _iso_date(value: str, what: str) -> str:
1011
+ try:
1012
+ date.fromisoformat(str(value))
1013
+ except ValueError:
1014
+ raise ValueError(f"{what} must be an ISO date (YYYY-MM-DD), got {value!r}") from None
1015
+ return str(value)
1016
+
1017
+
1018
+ class GuideSource(BaseModel):
1019
+ url: str
1020
+ title: str = ""
1021
+ accessed: str
1022
+ kind: GuideSourceKind
1023
+
1024
+ @field_validator("url")
1025
+ @classmethod
1026
+ def _url_must_be_real(cls, value: str) -> str:
1027
+ if not str(value).startswith(("http://", "https://")):
1028
+ raise ValueError(f"guide source url must be http(s), got {value!r}")
1029
+ return value
1030
+
1031
+ @field_validator("accessed", mode="before")
1032
+ @classmethod
1033
+ def _accessed_iso(cls, value: Any) -> str:
1034
+ return _iso_date(value, "guide source accessed")
1035
+
1036
+
1037
+ class GuideClaim(BaseModel):
1038
+ text: str
1039
+ sources: list[GuideSource]
1040
+
1041
+ @model_validator(mode="after")
1042
+ def _needs_source(self) -> "GuideClaim":
1043
+ if not self.text.strip():
1044
+ raise ValueError("guide claim text is empty")
1045
+ if not self.sources:
1046
+ raise ValueError(f"guide claim has no source: {self.text[:60]!r}")
1047
+ return self
1048
+
1049
+
1050
+ class GuideAppliesTo(BaseModel):
1051
+ model_id: str
1052
+ version: str
1053
+
1054
+
1055
+ class GuideSections(BaseModel):
1056
+ prompt_shape: list[GuideClaim] = []
1057
+ system_message: list[GuideClaim] = []
1058
+ reasoning_and_tools: list[GuideClaim] = []
1059
+ formatting: list[GuideClaim] = []
1060
+ failure_modes: list[GuideClaim] = []
1061
+ retry_advice: list[GuideClaim] = []
1062
+
1063
+
1064
+ class AuthoringGuide(BaseModel):
1065
+ applies_to: GuideAppliesTo
1066
+ as_of: str
1067
+ status: Literal["current", "stale"]
1068
+ sections: GuideSections = GuideSections()
1069
+
1070
+ @field_validator("as_of", mode="before")
1071
+ @classmethod
1072
+ def _as_of_iso(cls, value: Any) -> str:
1073
+ return _iso_date(value, "authoring_guide.as_of")
1074
+
1075
+
1076
+ # ═══════════════════════════════════════════════════════════════
1077
+ # THE COMPLETE MODEL CARD
1078
+ # ═══════════════════════════════════════════════════════════════
1079
+
1080
+ class ModelCard(BaseModel):
1081
+ """Universal Model Intelligence Card — V3.
1082
+
1083
+ This is the root object. It composes all sections into a single
1084
+ schema that covers every model type. Null fields = not yet researched.
1085
+ """
1086
+ # Sections
1087
+ identity: Identity
1088
+ architecture: Architecture = Architecture()
1089
+ lineage: Lineage = Lineage()
1090
+ licensing: Licensing = Licensing()
1091
+ modalities: Modalities = Modalities()
1092
+ capabilities: Capabilities = Capabilities()
1093
+ cost: Cost = Cost()
1094
+ availability: Availability = Availability()
1095
+ benchmarks: Benchmarks = Benchmarks()
1096
+ deployment: Deployment = Deployment()
1097
+ risk_governance: RiskGovernance = RiskGovernance()
1098
+ inference_performance: InferencePerformance = InferencePerformance()
1099
+ adoption: Adoption = Adoption()
1100
+ downselect: Downselect = Downselect()
1101
+ sources: Sources = Sources()
1102
+ # Optional, additive (MODEL-8). Absent guides serialize and count as before.
1103
+ authoring_guide: AuthoringGuide | None = None
1104
+
1105
+ # Card metadata
1106
+ card_schema_version: str = "3.0"
1107
+ card_author: str = ""
1108
+ card_created: str = ""
1109
+ card_updated: str = ""
1110
+ prose_body: str = "" # The markdown content below the YAML frontmatter
1111
+
1112
+ @model_validator(mode="after")
1113
+ def _guide_matches_card(self) -> "ModelCard":
1114
+ guide = self.authoring_guide
1115
+ if guide is None:
1116
+ return self
1117
+ if guide.applies_to.model_id != self.identity.model_id:
1118
+ raise ValueError(
1119
+ f"authoring_guide.applies_to.model_id {guide.applies_to.model_id!r} "
1120
+ f"does not match card model_id {self.identity.model_id!r}")
1121
+ if guide.status != "stale" and guide.applies_to.version != self.identity.version:
1122
+ raise ValueError(
1123
+ f"authoring_guide was written for version {guide.applies_to.version!r} "
1124
+ f"but the card is now {self.identity.version!r}: re-review the guide "
1125
+ "against current provider guidance, or set status: stale")
1126
+ return self
1127
+
1128
+ def _coverage_types(self) -> frozenset[ModelType]:
1129
+ types: set[ModelType] = set()
1130
+ ident = self.identity
1131
+ if ident.model_type is not None:
1132
+ types.add(ident.model_type)
1133
+ types.update(ident.model_subtypes)
1134
+ return frozenset(types)
1135
+
1136
+ def _section_applies(self, obj: BaseModel) -> bool:
1137
+ applicable = getattr(type(obj), "__applicable_model_types__", None)
1138
+ if applicable is None:
1139
+ return True
1140
+ return bool(self._coverage_types() & applicable)
1141
+
1142
+ def warnings(self) -> list[str]:
1143
+ """Non-fatal catalogue checks. CI reports these; they do not invalidate the card."""
1144
+ out: list[str] = []
1145
+ if (
1146
+ self.licensing.open_weights is True
1147
+ and self.licensing.license_type is LicenseType.PROPRIETARY
1148
+ ):
1149
+ out.append(
1150
+ "open_weights is true with license_type proprietary: "
1151
+ "a proprietary licence does not distribute downloadable weights"
1152
+ )
1153
+ return out
1154
+
1155
+ @property
1156
+ def inapplicable_fields(self) -> tuple[str, ...]:
1157
+ """Dotted paths this card's *class* cannot answer (MODEL-97).
1158
+
1159
+ Derived from `model_type` and `model_subtypes`, never stored, so it
1160
+ cannot drift from the card and cannot be lost in a YAML round-trip.
1161
+ A plain property rather than a `computed_field`: it is not card data
1162
+ and must not appear in `model_dump()` or `to_yaml()`.
1163
+
1164
+ An entry may be a field (`modalities.text.max_output_tokens`) or a
1165
+ subtree (`capabilities`), in which case every field beneath it is
1166
+ inapplicable. These are **not** the same as unresearched nulls: there
1167
+ is nothing here to research.
1168
+
1169
+ **The card wins.** A path where this card carries an actual value is
1170
+ never reported inapplicable, whatever the class table says: a
1171
+ `llm-reasoning` card with `modalities.vision.supported: true` is a
1172
+ model that sees, and publishing "vision does not apply" over its own
1173
+ data would be a false claim rather than a missing one. The pruning can
1174
+ only ever *remove* a claim, so it cannot invent applicability.
1175
+ """
1176
+ return _answered_removed(
1177
+ inapplicable_paths(self.identity.model_type, self.identity.model_subtypes), self)
1178
+
1179
+ @computed_field
1180
+ @property
1181
+ def applicable_field_coverage(self) -> float:
1182
+ """Internal statistic: percent of type-applicable schema fields filled.
1183
+
1184
+ Not published. It is not a Model node property, not in the graph
1185
+ export, and not shown by ``modelspec info`` or ``modelspec stats``.
1186
+ Nested modality details and the Capabilities block declare
1187
+ ``__applicable_model_types__``; those subtrees count only when the
1188
+ card's ``model_type`` or a ``model_subtype`` is in the set. Untagged
1189
+ sections count for every type. ``card_*`` metadata, ``prose_body`` and
1190
+ ``authoring_guide`` never count.
1191
+ """
1192
+ filled, total = self._count_fields(self)
1193
+ return round((filled / total) * 100, 1) if total > 0 else 0.0
1194
+
1195
+ @property
1196
+ def card_completeness(self) -> float:
1197
+ """Deprecated alias of ``applicable_field_coverage`` for scripts/**."""
1198
+ return self.applicable_field_coverage
1199
+
1200
+ def applicable_field_counts(self) -> tuple[int, int]:
1201
+ """Filled and total fields that apply to this card's type."""
1202
+ return self._count_fields(self)
1203
+
1204
+ def _count_fields(self, obj: BaseModel, _depth: int = 0,
1205
+ _prefix: str = "",
1206
+ _inapplicable: frozenset[str] | None = None) -> tuple[int, int]:
1207
+ """Recursively count filled vs type-applicable fields."""
1208
+ filled = 0
1209
+ total = 0
1210
+ inapplicable = (frozenset(self.inapplicable_fields)
1211
+ if _inapplicable is None else _inapplicable)
1212
+ for field_name, field_info in type(obj).model_fields.items():
1213
+ value = getattr(obj, field_name)
1214
+ if field_name == "authoring_guide":
1215
+ continue # guidance, not model facts: never moves coverage
1216
+ # A field this class cannot answer is outside the denominator: it
1217
+ # is not a gap, and counting it would be a permanent deduction for
1218
+ # a question that has no answer (MODEL-97).
1219
+ if f"{_prefix}{field_name}" in inapplicable:
1220
+ continue
1221
+ if isinstance(value, BaseModel):
1222
+ if not self._section_applies(value):
1223
+ continue
1224
+ f, t = self._count_fields(value, _depth + 1, f"{_prefix}{field_name}.",
1225
+ inapplicable)
1226
+ filled += f
1227
+ total += t
1228
+ elif isinstance(value, list):
1229
+ total += 1
1230
+ if len(value) > 0:
1231
+ filled += 1
1232
+ elif isinstance(value, dict):
1233
+ total += 1
1234
+ if len(value) > 0:
1235
+ filled += 1
1236
+ elif field_name.startswith("card_") or field_name in ("prose_body", "authoring_guide"):
1237
+ continue # Skip metadata fields
1238
+ else:
1239
+ total += 1
1240
+ if value is not None and value != "" and value is not False:
1241
+ filled += 1
1242
+ return filled, total
1243
+
1244
+ # ─── Serialization ─────────────────────────────────────────
1245
+
1246
+ @classmethod
1247
+ def from_yaml_file(cls, path: str | Path) -> "ModelCard":
1248
+ """Load a model card from a YAML+Markdown file."""
1249
+ path = Path(path)
1250
+ content = path.read_text(encoding="utf-8")
1251
+ return cls.from_yaml_string(content)
1252
+
1253
+ @classmethod
1254
+ def from_yaml_string(cls, content: str) -> "ModelCard":
1255
+ """Parse a model card from a string with YAML frontmatter."""
1256
+ parts = content.split("---", 2)
1257
+ if len(parts) >= 3:
1258
+ yaml_str = parts[1]
1259
+ prose = parts[2].strip()
1260
+ else:
1261
+ yaml_str = content
1262
+ prose = ""
1263
+
1264
+ data = yaml.safe_load(yaml_str) or {}
1265
+
1266
+ # Map flat YAML to nested Pydantic structure
1267
+ identity_data = {
1268
+ k: data.pop(k)
1269
+ for k in list(data.keys())
1270
+ if k in Identity.model_fields
1271
+ }
1272
+
1273
+ card_data = {
1274
+ "identity": identity_data,
1275
+ "prose_body": prose,
1276
+ }
1277
+
1278
+ # Map remaining top-level keys to sections
1279
+ section_map = {
1280
+ "architecture": Architecture,
1281
+ "lineage": Lineage,
1282
+ "licensing": Licensing,
1283
+ "modalities": Modalities,
1284
+ "capabilities": Capabilities,
1285
+ "cost": Cost,
1286
+ "availability": Availability,
1287
+ "benchmarks": Benchmarks,
1288
+ "deployment": Deployment,
1289
+ "risk_governance": RiskGovernance,
1290
+ "inference_performance": InferencePerformance,
1291
+ "adoption": Adoption,
1292
+ "downselect": Downselect,
1293
+ "sources": Sources,
1294
+ "authoring_guide": AuthoringGuide,
1295
+ }
1296
+
1297
+ for section_key, section_cls in section_map.items():
1298
+ if section_key in data:
1299
+ card_data[section_key] = data.pop(section_key)
1300
+
1301
+ # Remaining flat keys go to card metadata
1302
+ for k in ("card_schema_version", "card_author", "card_created", "card_updated"):
1303
+ if k in data:
1304
+ card_data[k] = data.pop(k)
1305
+
1306
+ return cls(**card_data)
1307
+
1308
+ def to_yaml(self) -> str:
1309
+ """Serialize back to YAML frontmatter + Markdown.
1310
+
1311
+ The inverse of `from_yaml_string`: identity fields flat at the top
1312
+ level, every other section nested, enums as their string values.
1313
+ """
1314
+ data = self.model_dump(
1315
+ mode="json",
1316
+ exclude_none=False,
1317
+ exclude={"prose_body", "applicable_field_coverage"},
1318
+ )
1319
+ if self.authoring_guide is None:
1320
+ data.pop("authoring_guide", None)
1321
+ out = data.pop("identity")
1322
+ for key in ModelCard.model_fields:
1323
+ if key in data:
1324
+ out[key] = data.pop(key)
1325
+ yaml_str = yaml.dump(out, default_flow_style=False, sort_keys=False, allow_unicode=True)
1326
+ return f"---\n{yaml_str}---\n\n{self.prose_body}"
1327
+
1328
+
1329
+ # ═══════════════════════════════════════════════════════════════
1330
+ # Applicability: the *other* kind of null (MODEL-97)
1331
+ # ═══════════════════════════════════════════════════════════════
1332
+ #
1333
+ # Two sources, one answer. `__applicable_model_types__` gates whole sections
1334
+ # and predates this; `schema.applicability.FIELD_RULES` gates named fields
1335
+ # inside sections that do apply. Both are pure functions of the card's class,
1336
+ # so this is cached per class rather than computed per card.
1337
+
1338
+
1339
+ def _coerce_types(model_type: Any, model_subtypes: Iterable[Any] = ()) -> frozenset[ModelType]:
1340
+ """Card class plus subtypes, from enum members or from raw YAML strings.
1341
+
1342
+ An unrecognised string is dropped rather than raising: a card carrying a
1343
+ type this build does not know is a card whose class we do not know, and an
1344
+ unknown class must assert nothing about what does or does not apply.
1345
+ """
1346
+ found: set[ModelType] = set()
1347
+ for raw in [model_type, *(model_subtypes or ())]:
1348
+ if raw is None or raw == "":
1349
+ continue
1350
+ if isinstance(raw, ModelType):
1351
+ found.add(raw)
1352
+ continue
1353
+ try:
1354
+ found.add(ModelType(str(raw)))
1355
+ except ValueError:
1356
+ continue
1357
+ return frozenset(found)
1358
+
1359
+
1360
+ def _gated_sections(model_cls: type[BaseModel], prefix: str,
1361
+ types: frozenset[ModelType], out: list[str]) -> None:
1362
+ for name, info in model_cls.model_fields.items():
1363
+ annotation = info.annotation
1364
+ if not (isinstance(annotation, type) and issubclass(annotation, BaseModel)):
1365
+ continue # a list, a dict, an optional union: not a gated section
1366
+ path = f"{prefix}{name}"
1367
+ gate = getattr(annotation, "__applicable_model_types__", None)
1368
+ if gate is not None and not (types & gate):
1369
+ out.append(path) # the whole subtree, named once
1370
+ continue
1371
+ _gated_sections(annotation, f"{path}.", types, out)
1372
+
1373
+
1374
+ @lru_cache(maxsize=None)
1375
+ def _inapplicable_paths(types: frozenset[ModelType]) -> tuple[str, ...]:
1376
+ if not types:
1377
+ return ()
1378
+ sections: list[str] = []
1379
+ _gated_sections(ModelCard, "", types, sections)
1380
+ fields = [rule.path for rule in FIELD_RULES if not (types & rule.applies_to)]
1381
+ # A field inside an already-named subtree is redundant: the subtree says it.
1382
+ # Naming both would make a consumer's `not_applicable` list disagree with
1383
+ # itself about how specific it is.
1384
+ covered = tuple(f"{path}." for path in sections)
1385
+ fields = [path for path in fields if not path.startswith(covered)]
1386
+ return tuple(sorted(set(sections) | set(fields)))
1387
+
1388
+
1389
+ def inapplicable_paths(model_type: Any,
1390
+ model_subtypes: Iterable[Any] = ()) -> tuple[str, ...]:
1391
+ """Dotted paths a card of this class cannot answer, sorted.
1392
+
1393
+ Accepts `ModelType` members or the raw strings a published card carries, so
1394
+ the export and the site renderer can call it with a front-matter dict
1395
+ without paying to build a `ModelCard`.
1396
+
1397
+ A card with no `model_type` gets `()`. Unknown class is unknown: it must
1398
+ never be turned into "there is nothing to know".
1399
+ """
1400
+ return _inapplicable_paths(_coerce_types(model_type, model_subtypes))
1401
+
1402
+
1403
+ def _value_at(obj: Any, path: str) -> Any:
1404
+ """Follow a dotted path through a mapping or a model. None when absent."""
1405
+ current = obj
1406
+ for part in path.split("."):
1407
+ if isinstance(current, dict):
1408
+ current = current.get(part)
1409
+ elif isinstance(current, BaseModel):
1410
+ current = getattr(current, part, None)
1411
+ else:
1412
+ return None
1413
+ if current is None:
1414
+ return None
1415
+ return current
1416
+
1417
+
1418
+ def _is_answered(value: Any) -> bool:
1419
+ """Whether a value — or anything beneath a section — is a real answer.
1420
+
1421
+ The same test coverage uses: `None`, `""`, `False` and an empty collection
1422
+ are all "nobody filled this in", not data.
1423
+ """
1424
+ if value is None or value == "" or value is False:
1425
+ return False
1426
+ if isinstance(value, BaseModel):
1427
+ return any(_is_answered(getattr(value, name, None))
1428
+ for name in type(value).model_fields)
1429
+ if isinstance(value, dict):
1430
+ return any(_is_answered(item) for item in value.values())
1431
+ if isinstance(value, (list, tuple, set)):
1432
+ return len(value) > 0
1433
+ return True
1434
+
1435
+
1436
+ def _answered_removed(paths: Iterable[str], card: Any) -> tuple[str, ...]:
1437
+ """Drop any path this card actually answers. The card outranks the table.
1438
+
1439
+ One-way: it can only remove an inapplicability claim, never add one, so a
1440
+ card can never talk the catalogue into asserting that a question has no
1441
+ answer.
1442
+ """
1443
+ return tuple(path for path in paths if not _is_answered(_value_at(card, path)))
1444
+
1445
+
1446
+ def applicability_block(model_type: Any,
1447
+ model_subtypes: Iterable[Any] = (),
1448
+ card: Any = None) -> dict[str, Any]:
1449
+ """The derived block published beside a card in `/api/models/<id>.json`.
1450
+
1451
+ Additive: it adds information rather than widening any card field, so it is
1452
+ not a contract break on its own (see `docs/cli-contract.md`). `card` is the
1453
+ card's frontmatter (or a `ModelCard`); pass it so a path the card answers
1454
+ is never published as one it cannot have.
1455
+ """
1456
+ paths = inapplicable_paths(model_type, model_subtypes)
1457
+ if card is not None:
1458
+ paths = _answered_removed(paths, card)
1459
+ return {
1460
+ "basis": "model_type",
1461
+ "model_type": getattr(model_type, "value", model_type) or None,
1462
+ "not_applicable": list(paths),
1463
+ }