modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
schema/enrichment.py ADDED
@@ -0,0 +1,162 @@
1
+ """The private enrichment record — the shape a policy determination is kept in.
2
+
3
+ MODEL-77. A determination has two halves and they are one decision:
4
+
5
+ * what the **public** card says (`schema/card.py`: `commercial_use`,
6
+ `commercial_use_source`, `commercial_use_conditions`, and the
7
+ `data_residency` trio), and
8
+ * the determination itself, which is the product being sold and therefore is
9
+ **not** in this repository.
10
+
11
+ Defining only the first half is how the two drift: the public marker starts
12
+ meaning something the private store does not carry, or the store gains a field
13
+ the card cannot express, and nobody notices until a customer is told something
14
+ untrue. So the record is defined here, next to the card, and
15
+ `EnrichmentRecord.public_fields()` is the *only* mapping from one to the other.
16
+ `tests/test_policy_shape.py` feeds that mapping straight into the card models,
17
+ so a divergence is a test failure rather than a support ticket.
18
+
19
+ **Where the records live.** Not here. This repository is public and the
20
+ determinations are the paid product (the private business decision record, §3
21
+ and §5). The store is a JSON Lines file outside the repo — one
22
+ `EnrichmentRecord` per line, serialised with `model_dump_json()`. This module
23
+ is the schema for those lines and nothing else: it reads no file, names no
24
+ path, and holds no data. Serving them is MODEL-80.
25
+
26
+ **What a public card shows.** A determination that is withheld publishes
27
+ `commercial_use: withheld` — not `unspecified`, which would claim nobody had
28
+ looked, and not `null`, which the field no longer allows. See
29
+ `UsePermission.WITHHELD` and `docs/cli-contract.md`.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ from datetime import date
35
+ from typing import Any, Literal
36
+
37
+ from pydantic import BaseModel, model_validator
38
+
39
+ from .card import PolicySource
40
+ from .enums import DisclosureState, UsePermission
41
+
42
+ #: The fields a determination can be made about. Both are policy answers a
43
+ #: buyer pays for; both are `unspecified`/`unresearched` across the corpus
44
+ #: until the determinations are made (MODEL-78, MODEL-79).
45
+ EnrichedField = Literal["commercial_use", "data_residency"]
46
+
47
+
48
+ class EnrichmentRecord(BaseModel):
49
+ """One determination about one model, with everything needed to defend it.
50
+
51
+ Every field except `conditions` is required. A record that cannot say who
52
+ decided, from what document, and on what day is not a determination — it is
53
+ an opinion, and the catalogue's whole value is that it does not sell those.
54
+ """
55
+
56
+ model_id: str
57
+ field: EnrichedField
58
+
59
+ #: Set exactly one, matching `field`.
60
+ commercial_use: UsePermission | None = None
61
+ data_residency: list[str] | None = None
62
+
63
+ #: Short and structured, the same line the public card would carry. Required
64
+ #: for a `restricted` commercial grant: "restricted" without the restriction
65
+ #: tells a buyer nothing they can act on.
66
+ conditions: str = ""
67
+
68
+ #: The document read, and the day it was read. Never `legacy-import`: that
69
+ #: kind exists only to mark the eight uncited values MODEL-77 inherited on
70
+ #: public cards, and a new determination has no excuse for it.
71
+ source: PolicySource
72
+ #: Who made the call. A person or an agent id — not "the pipeline".
73
+ determined_by: str
74
+ #: The day the determination was made, which is not necessarily the day the
75
+ #: document was read (`source.read_on`).
76
+ determined_on: str
77
+
78
+ #: False (the default) means the public card shows the withheld marker.
79
+ #: True means the determination is mirrored onto the public card in full.
80
+ published: bool = False
81
+
82
+ @model_validator(mode="after")
83
+ def _one_determination_fully_defended(self) -> EnrichmentRecord:
84
+ values = {
85
+ "commercial_use": self.commercial_use,
86
+ "data_residency": self.data_residency,
87
+ }
88
+ for name, value in values.items():
89
+ if name == self.field and value is None:
90
+ raise ValueError(f"field is {self.field!r} but {name} is not set")
91
+ if name != self.field and value is not None:
92
+ raise ValueError(
93
+ f"field is {self.field!r} but {name} is also set; one record "
94
+ "carries one determination, so that it can be withheld, "
95
+ "published or corrected on its own"
96
+ )
97
+
98
+ if self.commercial_use is not None and self.commercial_use not in (
99
+ UsePermission.ALLOWED,
100
+ UsePermission.RESTRICTED,
101
+ UsePermission.PROHIBITED,
102
+ ):
103
+ raise ValueError(
104
+ f"commercial_use {self.commercial_use.value!r} is not a "
105
+ "determination. 'unspecified' and 'withheld' describe a public "
106
+ "card's contents; a record exists because something was decided."
107
+ )
108
+
109
+ if self.source.kind == "legacy-import":
110
+ raise ValueError(
111
+ "a determination cites the document it was read from. "
112
+ "'legacy-import' marks the uncited values inherited on public "
113
+ "cards, and may not be used to record new work."
114
+ )
115
+
116
+ if not self.determined_by.strip():
117
+ raise ValueError("determined_by is required: a determination has an author")
118
+ try:
119
+ date.fromisoformat(self.determined_on)
120
+ except ValueError as exc:
121
+ raise ValueError(
122
+ f"determined_on must be an exact ISO date, got {self.determined_on!r}"
123
+ ) from exc
124
+
125
+ if self.commercial_use is UsePermission.RESTRICTED and not self.conditions.strip():
126
+ raise ValueError(
127
+ "a 'restricted' grant must state its restriction in `conditions`"
128
+ )
129
+ return self
130
+
131
+ def public_fields(self) -> dict[str, Any]:
132
+ """Exactly what the public card must carry for this determination.
133
+
134
+ The single source of truth for the public/private relationship. Callers
135
+ splat it into `Licensing(...)` or `PrimaryProvider(...)`; the card
136
+ validators then decide whether it is legal, so the two halves cannot
137
+ drift apart without a test going red.
138
+ """
139
+ if self.field == "commercial_use":
140
+ if not self.published:
141
+ return {
142
+ "commercial_use": UsePermission.WITHHELD,
143
+ "commercial_use_source": None,
144
+ "commercial_use_conditions": "",
145
+ }
146
+ return {
147
+ "commercial_use": self.commercial_use,
148
+ "commercial_use_source": self.source,
149
+ "commercial_use_conditions": self.conditions,
150
+ }
151
+
152
+ if not self.published:
153
+ return {
154
+ "data_residency": None,
155
+ "data_residency_disclosure": DisclosureState.WITHHELD,
156
+ "data_residency_source": None,
157
+ }
158
+ return {
159
+ "data_residency": list(self.data_residency or []),
160
+ "data_residency_disclosure": DisclosureState.PUBLISHED,
161
+ "data_residency_source": self.source,
162
+ }
schema/enums.py ADDED
@@ -0,0 +1,327 @@
1
+ """Controlled vocabularies for the ModelRank schema.
2
+
3
+ Every constrained field in the model card maps to an enum here.
4
+ This is the single source of truth for valid values.
5
+ """
6
+
7
+ from enum import Enum
8
+
9
+
10
+ class ModelType(str, Enum):
11
+ """Primary model classification."""
12
+ LLM_CHAT = "llm-chat"
13
+ LLM_REASONING = "llm-reasoning"
14
+ LLM_CODE = "llm-code"
15
+ LLM_BASE = "llm-base"
16
+ VLM = "vlm"
17
+ EMBEDDING_TEXT = "embedding-text"
18
+ EMBEDDING_MULTIMODAL = "embedding-multimodal"
19
+ EMBEDDING_CODE = "embedding-code"
20
+ RERANKER = "reranker"
21
+ SAFETY_CLASSIFIER = "safety-classifier"
22
+ IMAGE_GENERATION = "image-generation"
23
+ IMAGE_EDITING = "image-editing"
24
+ VIDEO_GENERATION = "video-generation"
25
+ AUDIO_ASR = "audio-asr"
26
+ AUDIO_TTS = "audio-tts"
27
+ AUDIO_MUSIC = "audio-music"
28
+ AUDIO_REALTIME = "audio-realtime"
29
+ DOCUMENT_OCR = "document-ocr"
30
+ MEDICAL = "medical"
31
+ LEGAL = "legal"
32
+ FINANCIAL = "financial"
33
+ ROBOTICS = "robotics"
34
+ WORLD_MODEL = "world-model"
35
+ REWARD_MODEL = "reward-model"
36
+ ROUTER = "router"
37
+ AGENT_MODEL = "agent-model"
38
+ #: Evaluates supplied state against caller-defined typed questions,
39
+ #: returning calibrated choices, scores or probabilities; generates no text.
40
+ #:
41
+ #: MODEL-98. Every near miss fails for a specific reason: `llm-reasoning`
42
+ #: emits text and a chain of thought; `agent-model` takes actions, while
43
+ #: here the calling code owns the loop; `router` and `reranker` are
44
+ #: *applications* of this class rather than the class; `reward-model`
45
+ #: scores model outputs for training or ranking, not arbitrary questions
46
+ #: over arbitrary state; `safety-classifier` has a fixed harm taxonomy,
47
+ #: while these labels are defined per request.
48
+ #:
49
+ #: Adding this value widened a published enum and cost the export contract
50
+ #: a major version (2.0 -> 3.0). See `docs/cli-contract.md` and
51
+ #: `docs/design/class-and-null-semantics.md`.
52
+ DECISION_MODEL = "decision-model"
53
+ ADAPTER = "adapter"
54
+ QUANTIZED_VARIANT = "quantized-variant"
55
+ DISTILLED = "distilled"
56
+ MERGED = "merged"
57
+ #: Forecasts a numeric sequence forward (e.g. PatchTST, Moirai). No tokens in or out.
58
+ TIME_SERIES = "time-series"
59
+ #: Vision backbone for perception tasks — classification, segmentation,
60
+ #: detection, depth — that predict labels/masks/boxes, not text tokens.
61
+ VISION_ENCODER = "vision-encoder"
62
+ #: Masked/token-level text encoder (e.g. BERT-style fill-mask, token
63
+ #: classification). Reads text but does not autoregressively generate it.
64
+ TEXT_ENCODER = "text-encoder"
65
+ #: For models no specific type describes honestly, such as model components (VAEs), speaker verification and speech tokenizers.
66
+ MISCELLANEOUS = "miscellaneous"
67
+
68
+
69
+ class ModelStatus(str, Enum):
70
+ ACTIVE = "active"
71
+ BETA = "beta"
72
+ ALPHA = "alpha"
73
+ DEPRECATED = "deprecated"
74
+ SUNSET = "sunset"
75
+ PREVIEW = "preview"
76
+
77
+
78
+ class ArchitectureType(str, Enum):
79
+ DENSE_TRANSFORMER = "dense-transformer"
80
+ MOE = "MoE"
81
+ SSM = "SSM"
82
+ HYBRID_SSM_TRANSFORMER = "hybrid-SSM-transformer"
83
+ DIFFUSION = "diffusion"
84
+ GAN = "GAN"
85
+ FLOW_MATCHING = "flow-matching"
86
+ ENCODER_DECODER = "encoder-decoder"
87
+ ENCODER_ONLY = "encoder-only"
88
+ DECODER_ONLY = "decoder-only"
89
+ OTHER = "other"
90
+
91
+
92
+ class AttentionType(str, Enum):
93
+ MHA = "MHA"
94
+ GQA = "GQA"
95
+ MQA = "MQA"
96
+ MLA = "MLA"
97
+ LINEAR = "linear"
98
+ SLIDING_WINDOW = "sliding-window"
99
+ NONE = "none"
100
+
101
+
102
+ class PositionalEncoding(str, Enum):
103
+ ROPE = "RoPE"
104
+ ALIBI = "ALiBi"
105
+ NTK_AWARE_ROPE = "NTK-aware-RoPE"
106
+ YARN = "YaRN"
107
+ ABSOLUTE = "absolute"
108
+ RELATIVE = "relative"
109
+ NONE = "none"
110
+
111
+
112
+ class TokenizerType(str, Enum):
113
+ BPE = "BPE"
114
+ SENTENCEPIECE = "SentencePiece"
115
+ TIKTOKEN = "tiktoken"
116
+ UNIGRAM = "Unigram"
117
+ OTHER = "other"
118
+
119
+
120
+ class BaseModelRelation(str, Enum):
121
+ ORIGINAL = "original"
122
+ FINETUNE = "finetune"
123
+ ADAPTER = "adapter"
124
+ QUANTIZED = "quantized"
125
+ MERGE = "merge"
126
+ DISTILLATION = "distillation"
127
+ CONTINUATION = "continuation"
128
+ #: Same weights re-uploaded under another repo (a mirror or a re-save).
129
+ #: Byte-identical for fit purposes, so excluded from default fit pools.
130
+ REPACKAGED = "repackaged"
131
+
132
+
133
+ class TrainingMethod(str, Enum):
134
+ PRETRAINING = "pretraining"
135
+ SFT = "SFT"
136
+ RLHF = "RLHF"
137
+ DPO = "DPO"
138
+ GRPO = "GRPO"
139
+ RLAIF = "RLAIF"
140
+ CONTRASTIVE = "contrastive"
141
+ OTHER = "other"
142
+
143
+
144
+ class LicenseType(str, Enum):
145
+ PROPRIETARY = "proprietary"
146
+ APACHE_2_0 = "apache-2.0"
147
+ MIT = "mit"
148
+ LLAMA_COMMUNITY = "llama-community"
149
+ QWEN = "qwen"
150
+ DEEPSEEK = "deepseek"
151
+ GEMMA = "gemma"
152
+ CC_BY_4_0 = "cc-by-4.0"
153
+ CC_BY_NC_4_0 = "cc-by-nc-4.0"
154
+ OPENRAIL = "openrail"
155
+ GPL_3_0 = "gpl-3.0"
156
+ OTHER = "other"
157
+
158
+
159
+ class UsePermission(str, Enum):
160
+ ALLOWED = "allowed"
161
+ RESTRICTED = "restricted"
162
+ PROHIBITED = "prohibited"
163
+ #: Nothing has been determined. The honest empty state.
164
+ UNSPECIFIED = "unspecified"
165
+ #: Determined, and deliberately not published here (MODEL-77).
166
+ #:
167
+ #: This value exists because the alternative lies. Once a determination is
168
+ #: made and held back as enrichment, publishing `unspecified` would assert
169
+ #: "not yet researched" on cards where it is false — at corpus scale, in
170
+ #: the one place the catalogue's reputation lives. `withheld` says what is
171
+ #: true: the answer exists, it is not in this file.
172
+ #:
173
+ #: It carries no `*_source` and no conditions. The determination itself
174
+ #: lives in a private `schema.enrichment.EnrichmentRecord`.
175
+ WITHHELD = "withheld"
176
+
177
+
178
+ class DisclosureState(str, Enum):
179
+ """Why a *collection-valued* policy field is empty, when it is empty.
180
+
181
+ `UsePermission` carries its own empty states in-band, so a consumer that
182
+ reads only the value cannot misread it. A list cannot do that: `[]` means
183
+ "no regions" and "nobody looked" equally well, which is how
184
+ `data_residency` came to be `[]` on all 1,339 cards while being researched
185
+ on none of them. So the state moves to a companion field, and the list is
186
+ `null` whenever it is not a published determination.
187
+ """
188
+
189
+ #: Nobody has looked. The value must be `null`.
190
+ UNRESEARCHED = "unresearched"
191
+ #: Determined and published here. The value is the determination — an
192
+ #: empty list means "the provider commits to no residency", which is an
193
+ #: answer, not an absence.
194
+ PUBLISHED = "published"
195
+ #: Determined, not published here. The value must be `null`.
196
+ WITHHELD = "withheld"
197
+
198
+
199
+ class OrgType(str, Enum):
200
+ PRIVATE = "private"
201
+ STATE_BACKED = "state-backed"
202
+ ACADEMIC = "academic"
203
+ OPEN_COLLECTIVE = "open-collective"
204
+ GOVERNMENT = "government"
205
+ NONPROFIT = "nonprofit"
206
+
207
+
208
+ class Tier(str, Enum):
209
+ TIER_1 = "tier-1"
210
+ TIER_2 = "tier-2"
211
+ TIER_3 = "tier-3"
212
+ NA = "n/a"
213
+
214
+
215
+ class ConfidenceLevel(str, Enum):
216
+ HIGH = "high"
217
+ MEDIUM = "medium"
218
+ LOW = "low"
219
+
220
+
221
+ class ResistanceLevel(str, Enum):
222
+ RESISTANT = "resistant"
223
+ MODERATE = "moderate"
224
+ VULNERABLE = "vulnerable"
225
+ UNTESTED = "untested"
226
+
227
+
228
+ class EvalStatus(str, Enum):
229
+ EVAL_COMPLETE = "eval-complete"
230
+ EVAL_PENDING = "eval-pending"
231
+ EVAL_FAILED = "eval-failed"
232
+ NOT_STARTED = "not-started"
233
+
234
+
235
+ class RiskTier(str, Enum):
236
+ LOW = "low"
237
+ MEDIUM = "medium"
238
+ HIGH = "high"
239
+ CRITICAL = "critical"
240
+
241
+
242
+ class PlatformCategory(str, Enum):
243
+ CLOUD = "cloud"
244
+ INFERENCE = "inference"
245
+ AGGREGATOR = "aggregator"
246
+ AI_APP = "ai-app"
247
+ PROVIDER = "provider"
248
+ CN_PLATFORM = "cn-platform"
249
+ REGIONAL = "regional"
250
+ LOCAL = "local"
251
+ MODEL_HUB = "model-hub"
252
+
253
+
254
+ class BenchmarkCategory(str, Enum):
255
+ KNOWLEDGE = "knowledge"
256
+ MATH = "math"
257
+ CODING = "coding"
258
+ MULTIMODAL = "multimodal"
259
+ SAFETY = "safety"
260
+ HUMAN_PREFERENCE = "human-preference"
261
+ EMBEDDING = "embedding"
262
+ GENERATION = "generation"
263
+ DOMAIN = "domain"
264
+ AGENTIC = "agentic"
265
+ COMPOSITE = "composite"
266
+ REASONING = "reasoning"
267
+ INSTRUCTION_FOLLOWING = "instruction-following"
268
+ LONG_CONTEXT = "long-context"
269
+ TRANSLATION = "translation"
270
+
271
+
272
+ class QuantFormat(str, Enum):
273
+ GGUF = "gguf"
274
+ AWQ = "awq"
275
+ GPTQ = "gptq"
276
+ EXL2 = "exl2"
277
+ FP16 = "fp16"
278
+ BF16 = "bf16"
279
+ FP8 = "fp8"
280
+ INT8 = "int8"
281
+ INT4 = "int4"
282
+
283
+
284
+ class Modality(str, Enum):
285
+ TEXT = "text"
286
+ IMAGE = "image"
287
+ AUDIO = "audio"
288
+ VIDEO = "video"
289
+ PDF = "pdf"
290
+ THREE_D = "3d"
291
+ TABULAR = "tabular"
292
+ CODE = "code"
293
+ EMBEDDINGS = "embeddings"
294
+ ACTIONS = "actions"
295
+ CLASSIFICATIONS = "classifications"
296
+ SCORES = "scores"
297
+
298
+
299
+ class EUAIActRisk(str, Enum):
300
+ UNACCEPTABLE = "unacceptable"
301
+ HIGH = "high"
302
+ LIMITED = "limited"
303
+ MINIMAL = "minimal"
304
+ NA = "n/a"
305
+
306
+
307
+ class DeviceClass(str, Enum):
308
+ """What kind of machine a `hardware/*.yaml` device is (MODEL-76).
309
+
310
+ The vocabulary the hardware records have always used, promoted to an enum
311
+ so the graph can group and filter Hardware nodes by it. A typo in a record
312
+ would otherwise flow into the export as a sixth class nobody queries for.
313
+
314
+ This is a property of the *part*, not of who bought it: a datacentre GPU
315
+ under a desk is still `datacentre`.
316
+ """
317
+
318
+ #: Sold to individuals; gaming and prosumer boards.
319
+ CONSUMER = "consumer"
320
+ #: Professional desk-side parts (RTX A/Ada workstation lines).
321
+ WORKSTATION = "workstation"
322
+ #: Rack parts and accelerators — H100/H200, MI300X, TPUs, Gaudi.
323
+ DATACENTRE = "datacentre"
324
+ #: Embedded and robotics modules (Jetson).
325
+ EDGE = "edge"
326
+ #: Memory unified with the host SoC — Apple silicon, Ryzen AI Max, X Elite.
327
+ INTEGRATED = "integrated"