modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
registry/facets.yaml
ADDED
|
@@ -0,0 +1,888 @@
|
|
|
1
|
+
# The facet registry (MODEL-133; design §4.1 and §8).
|
|
2
|
+
#
|
|
3
|
+
# A facet is something a spec can constrain or optimise. Adding one is an entry
|
|
4
|
+
# here plus data; the engine reads this file and has no facet list of its own.
|
|
5
|
+
# Loader and validation: decision/registry.py.
|
|
6
|
+
#
|
|
7
|
+
# Entry fields
|
|
8
|
+
# id dotted snake_case, unique
|
|
9
|
+
# subject model | offering | evidence
|
|
10
|
+
# value_type.kind number | enum | boolean | date | set | range
|
|
11
|
+
# number, range need `unit` (one of `units` below). `unbounded: true`
|
|
12
|
+
# lets a known value be the literal `unbounded`
|
|
13
|
+
# (for example a licence with no user cap), and
|
|
14
|
+
# `not_offered: true` the literal `not_offered` (for
|
|
15
|
+
# example a provider with no batch interface).
|
|
16
|
+
# enum, set need `values` or `values_from` (a named list the
|
|
17
|
+
# loader knows, such as `registry:providers`).
|
|
18
|
+
# parameter optional: the facet is a family, one per value of a
|
|
19
|
+
# named list (for example one estimate per domain).
|
|
20
|
+
# definition exact. Says what a value means and what it does not.
|
|
21
|
+
# tier guaranteed | best_effort (design §8, G and B)
|
|
22
|
+
# risk capability | governance. Sets the default unknown
|
|
23
|
+
# policy: capability -> may_qualify,
|
|
24
|
+
# governance -> not_satisfied.
|
|
25
|
+
# permitted_source_kinds ids from `source_kinds` below
|
|
26
|
+
# unit id from `units` below, where applicable
|
|
27
|
+
# required_qualifiers optional: qualifiers every fact must carry
|
|
28
|
+
# computed_by optional: the ticket whose build step produces the
|
|
29
|
+
# value. Such facets are never authored on a card.
|
|
30
|
+
# label optional: a short name for people ("Input price").
|
|
31
|
+
# The id stays the identifier; the label is display.
|
|
32
|
+
#
|
|
33
|
+
# A fact's state (known, unknown, not_disclosed, requires_contract) is separate
|
|
34
|
+
# from its value and is not declared here.
|
|
35
|
+
|
|
36
|
+
schema_version: 1
|
|
37
|
+
|
|
38
|
+
units:
|
|
39
|
+
- id: usd_per_1m_tokens
|
|
40
|
+
definition: >-
|
|
41
|
+
United States dollars per 1,000,000 tokens, at the provider's list price
|
|
42
|
+
for the offering, before discounts, credits or committed-use pricing.
|
|
43
|
+
Tokens are counted by the provider's own tokenizer for that model.
|
|
44
|
+
- id: tokens
|
|
45
|
+
definition: >-
|
|
46
|
+
Tokens as counted by the model's own tokenizer, as documented by its lab.
|
|
47
|
+
Not words, characters or another model's tokens.
|
|
48
|
+
- id: tokens_per_second
|
|
49
|
+
definition: >-
|
|
50
|
+
Output tokens generated per second of wall-clock time, measured from the
|
|
51
|
+
first output token to the last, for a single request.
|
|
52
|
+
- id: milliseconds
|
|
53
|
+
definition: >-
|
|
54
|
+
Milliseconds of wall-clock time.
|
|
55
|
+
- id: parameters
|
|
56
|
+
definition: >-
|
|
57
|
+
A count of trainable weights (not billions; 7e9 is written 7000000000).
|
|
58
|
+
- id: days
|
|
59
|
+
definition: >-
|
|
60
|
+
Calendar days.
|
|
61
|
+
- id: requests_per_minute
|
|
62
|
+
definition: >-
|
|
63
|
+
Requests accepted per 60-second window for one account at the tier the
|
|
64
|
+
offering names, before any limit increase negotiated by contract.
|
|
65
|
+
- id: tokens_per_minute
|
|
66
|
+
definition: >-
|
|
67
|
+
Tokens (input plus output unless the provider states otherwise, in which
|
|
68
|
+
case the fact's qualifiers say so) accepted per 60-second window for one
|
|
69
|
+
account at the tier the offering names.
|
|
70
|
+
- id: percent
|
|
71
|
+
definition: >-
|
|
72
|
+
A percentage from 0 to 100.
|
|
73
|
+
- id: monthly_active_users
|
|
74
|
+
definition: >-
|
|
75
|
+
Distinct end users in a calendar month, as the licence itself defines the
|
|
76
|
+
count.
|
|
77
|
+
- id: benchmark_metric
|
|
78
|
+
definition: >-
|
|
79
|
+
The unit declared by the benchmark page's `metric.unit` for the benchmark
|
|
80
|
+
version the evidence names. Evidence on different benchmarks is never
|
|
81
|
+
compared in this unit.
|
|
82
|
+
- id: usd_per_task
|
|
83
|
+
definition: >-
|
|
84
|
+
United States dollars for one task, at list prices: the offering's input
|
|
85
|
+
price times the spec's input tokens per task, plus its output price times
|
|
86
|
+
the spec's output tokens per task, divided by 1,000,000. Excludes cached,
|
|
87
|
+
batch and committed-use pricing.
|
|
88
|
+
- id: capability_scale
|
|
89
|
+
definition: >-
|
|
90
|
+
The latent scale of the capability model (MODEL-129). Values are
|
|
91
|
+
comparable within one domain and one snapshot only.
|
|
92
|
+
|
|
93
|
+
source_kinds:
|
|
94
|
+
- id: lab_documentation
|
|
95
|
+
definition: >-
|
|
96
|
+
A model card, system card, technical report or API reference published by
|
|
97
|
+
the lab that trained the model.
|
|
98
|
+
- id: lab_announcement
|
|
99
|
+
definition: >-
|
|
100
|
+
A release post or changelog entry published by the lab that trained the
|
|
101
|
+
model.
|
|
102
|
+
- id: licence_text
|
|
103
|
+
definition: >-
|
|
104
|
+
The licence or terms of use under which the model's weights or outputs are
|
|
105
|
+
made available, as published by the licensor.
|
|
106
|
+
- id: weights_repository
|
|
107
|
+
definition: >-
|
|
108
|
+
Configuration and metadata files published alongside the weights by the
|
|
109
|
+
lab or its designated repository (for example a model repository's config).
|
|
110
|
+
- id: provider_documentation
|
|
111
|
+
definition: >-
|
|
112
|
+
A model page, pricing page, rate-limit page or API reference published by
|
|
113
|
+
the provider that sells the offering.
|
|
114
|
+
- id: provider_terms
|
|
115
|
+
definition: >-
|
|
116
|
+
The provider's service terms, data processing addendum or product-specific
|
|
117
|
+
terms that govern the offering.
|
|
118
|
+
- id: provider_trust_center
|
|
119
|
+
definition: >-
|
|
120
|
+
The provider's own page listing its certifications, attestations and
|
|
121
|
+
audit reports.
|
|
122
|
+
- id: official_registry
|
|
123
|
+
definition: >-
|
|
124
|
+
A register kept by the body that grants a status, such as an
|
|
125
|
+
authorization marketplace or a company register.
|
|
126
|
+
- id: benchmark_author
|
|
127
|
+
definition: >-
|
|
128
|
+
Results published by the authors or maintainers of the benchmark.
|
|
129
|
+
- id: independent_evaluator
|
|
130
|
+
definition: >-
|
|
131
|
+
Results published by an evaluator that is neither the benchmark's authors
|
|
132
|
+
nor the lab or provider being measured.
|
|
133
|
+
- id: provider_self_report
|
|
134
|
+
definition: >-
|
|
135
|
+
Results published by the lab or provider about its own model or offering.
|
|
136
|
+
Legitimate, and always distinguishable from independent evidence.
|
|
137
|
+
- id: modelspec_measurement
|
|
138
|
+
definition: >-
|
|
139
|
+
A measurement ModelSpec ran itself under a published method (ADR 0004).
|
|
140
|
+
- id: outcome_protocol
|
|
141
|
+
definition: >-
|
|
142
|
+
Aggregated outcome records reported through the outcome protocol (design
|
|
143
|
+
§9).
|
|
144
|
+
- id: modelspec_estimate
|
|
145
|
+
definition: >-
|
|
146
|
+
A value ModelSpec computes from other verified facts and evidence, such as
|
|
147
|
+
a hardware fit or a capability estimate. Labelled as an estimate.
|
|
148
|
+
|
|
149
|
+
facets:
|
|
150
|
+
# ── Model: class, modalities, context ─────────────────────────────────────
|
|
151
|
+
- id: model.class
|
|
152
|
+
label: Model class
|
|
153
|
+
subject: model
|
|
154
|
+
value_type: {kind: enum, values_from: model_classes}
|
|
155
|
+
value_labels:
|
|
156
|
+
actor: Agent (acts on tools or environments)
|
|
157
|
+
analyser: Analyser (labels parts of its input)
|
|
158
|
+
decider: Decision model
|
|
159
|
+
forecaster: Forecaster
|
|
160
|
+
labeller: Classifier
|
|
161
|
+
"media-generator": Media generator
|
|
162
|
+
orderer: Reranker
|
|
163
|
+
scorer: Scorer
|
|
164
|
+
simulator: Simulator
|
|
165
|
+
"text-generator": Text generator
|
|
166
|
+
transcriber: Speech recognition
|
|
167
|
+
vectoriser: Embedding model
|
|
168
|
+
definition: >-
|
|
169
|
+
The model's class as derived from its model type by `api/classes.py`:
|
|
170
|
+
what it consumes, what it emits and what decision it makes. One value per
|
|
171
|
+
model. Class is derived, not authored, and says nothing about quality.
|
|
172
|
+
tier: guaranteed
|
|
173
|
+
risk: capability
|
|
174
|
+
permitted_source_kinds: [lab_documentation, lab_announcement, weights_repository]
|
|
175
|
+
- id: model.input_modalities
|
|
176
|
+
label: Input modalities
|
|
177
|
+
subject: model
|
|
178
|
+
value_type: {kind: set, values: [text, image, audio, video, document, structured]}
|
|
179
|
+
value_labels:
|
|
180
|
+
audio: Audio
|
|
181
|
+
document: Documents
|
|
182
|
+
image: Images
|
|
183
|
+
structured: Structured data
|
|
184
|
+
text: Text
|
|
185
|
+
video: Video
|
|
186
|
+
definition: >-
|
|
187
|
+
Every modality the model accepts as input natively, without a separate
|
|
188
|
+
model transcribing or captioning it first. `document` means files such as
|
|
189
|
+
PDF read as documents, not text extracted by the caller.
|
|
190
|
+
tier: guaranteed
|
|
191
|
+
risk: capability
|
|
192
|
+
permitted_source_kinds: [lab_documentation, lab_announcement, weights_repository]
|
|
193
|
+
- id: model.output_modalities
|
|
194
|
+
label: Output modalities
|
|
195
|
+
subject: model
|
|
196
|
+
value_type: {kind: set, values: [text, image, audio, video, embedding, score, label, action]}
|
|
197
|
+
value_labels:
|
|
198
|
+
action: Actions
|
|
199
|
+
audio: Audio
|
|
200
|
+
embedding: Embeddings
|
|
201
|
+
image: Images
|
|
202
|
+
label: Labels
|
|
203
|
+
score: Scores
|
|
204
|
+
text: Text
|
|
205
|
+
video: Video
|
|
206
|
+
definition: >-
|
|
207
|
+
Every modality the model emits natively. `embedding` is a vector,
|
|
208
|
+
`score` a relevance or reward number, `label` a class from a fixed or
|
|
209
|
+
caller-defined set, and `action` a tool or environment action.
|
|
210
|
+
tier: guaranteed
|
|
211
|
+
risk: capability
|
|
212
|
+
permitted_source_kinds: [lab_documentation, lab_announcement, weights_repository]
|
|
213
|
+
- id: model.context_window
|
|
214
|
+
label: Context window
|
|
215
|
+
subject: model
|
|
216
|
+
value_type: {kind: number}
|
|
217
|
+
unit: tokens
|
|
218
|
+
definition: >-
|
|
219
|
+
The maximum number of tokens the model accepts in one request, input and
|
|
220
|
+
output together, as documented by its lab for its largest supported
|
|
221
|
+
configuration. A provider that serves a smaller window records that on the
|
|
222
|
+
offering, not here.
|
|
223
|
+
tier: guaranteed
|
|
224
|
+
risk: capability
|
|
225
|
+
permitted_source_kinds: [lab_documentation, lab_announcement, weights_repository]
|
|
226
|
+
- id: model.max_output_tokens
|
|
227
|
+
label: Maximum output tokens
|
|
228
|
+
subject: model
|
|
229
|
+
value_type: {kind: number}
|
|
230
|
+
unit: tokens
|
|
231
|
+
definition: >-
|
|
232
|
+
The maximum number of tokens the model can emit in one response,
|
|
233
|
+
including any reasoning tokens the lab counts against the limit, as
|
|
234
|
+
documented by its lab.
|
|
235
|
+
tier: guaranteed
|
|
236
|
+
risk: capability
|
|
237
|
+
permitted_source_kinds: [lab_documentation, lab_announcement]
|
|
238
|
+
|
|
239
|
+
# ── Model: weights, parameters, architecture ──────────────────────────────
|
|
240
|
+
- id: model.weights_openness
|
|
241
|
+
label: Open weights
|
|
242
|
+
subject: model
|
|
243
|
+
value_type: {kind: enum, values: [open_weights, closed_weights]}
|
|
244
|
+
value_labels:
|
|
245
|
+
closed_weights: Closed weights
|
|
246
|
+
open_weights: Open weights
|
|
247
|
+
definition: >-
|
|
248
|
+
`open_weights` when the lab publishes the trained weights for download
|
|
249
|
+
under any licence, including a gated or restrictive one; `closed_weights`
|
|
250
|
+
otherwise. Says nothing about the licence terms, which are the `licence.*`
|
|
251
|
+
facets.
|
|
252
|
+
tier: guaranteed
|
|
253
|
+
risk: capability
|
|
254
|
+
permitted_source_kinds: [lab_documentation, lab_announcement, weights_repository, licence_text]
|
|
255
|
+
- id: model.parameters_total
|
|
256
|
+
label: Total parameters
|
|
257
|
+
subject: model
|
|
258
|
+
value_type: {kind: number}
|
|
259
|
+
unit: parameters
|
|
260
|
+
definition: >-
|
|
261
|
+
The total number of parameters in the released model, counting every
|
|
262
|
+
expert of a mixture-of-experts model. `not_disclosed` is a state of the
|
|
263
|
+
fact, never an estimate.
|
|
264
|
+
tier: best_effort
|
|
265
|
+
risk: capability
|
|
266
|
+
permitted_source_kinds: [lab_documentation, weights_repository]
|
|
267
|
+
- id: model.parameters_active
|
|
268
|
+
label: Active parameters
|
|
269
|
+
subject: model
|
|
270
|
+
value_type: {kind: number}
|
|
271
|
+
unit: parameters
|
|
272
|
+
definition: >-
|
|
273
|
+
The number of parameters used to process one token. Equal to the total
|
|
274
|
+
for a dense model; for a mixture-of-experts model, as the lab documents
|
|
275
|
+
it.
|
|
276
|
+
tier: best_effort
|
|
277
|
+
risk: capability
|
|
278
|
+
permitted_source_kinds: [lab_documentation, weights_repository]
|
|
279
|
+
- id: model.architecture
|
|
280
|
+
label: Architecture
|
|
281
|
+
subject: model
|
|
282
|
+
value_type: {kind: enum, values_from: architecture_types}
|
|
283
|
+
value_labels:
|
|
284
|
+
GAN: GAN
|
|
285
|
+
MoE: Mixture of experts
|
|
286
|
+
SSM: State-space model
|
|
287
|
+
"decoder-only": Decoder-only transformer
|
|
288
|
+
"dense-transformer": Dense transformer
|
|
289
|
+
diffusion: Diffusion
|
|
290
|
+
"encoder-decoder": Encoder-decoder transformer
|
|
291
|
+
"encoder-only": Encoder-only transformer
|
|
292
|
+
"flow-matching": Flow matching
|
|
293
|
+
"hybrid-SSM-transformer": Hybrid state-space and transformer
|
|
294
|
+
other: Other
|
|
295
|
+
definition: >-
|
|
296
|
+
The model's architecture family as `ArchitectureType` in `schema/enums.py`
|
|
297
|
+
names it. An undisclosed architecture is unknown, never inapplicable.
|
|
298
|
+
tier: best_effort
|
|
299
|
+
risk: capability
|
|
300
|
+
permitted_source_kinds: [lab_documentation, weights_repository]
|
|
301
|
+
|
|
302
|
+
# ── Model: licence rights (governance) ────────────────────────────────────
|
|
303
|
+
- id: licence.commercial_use
|
|
304
|
+
label: Commercial use
|
|
305
|
+
subject: model
|
|
306
|
+
value_type: {kind: enum, values: [permitted, permitted_with_conditions, prohibited]}
|
|
307
|
+
value_labels:
|
|
308
|
+
permitted: Permitted
|
|
309
|
+
permitted_with_conditions: Permitted with conditions
|
|
310
|
+
prohibited: Prohibited
|
|
311
|
+
definition: >-
|
|
312
|
+
Whether the licence governing the model lets a customer use the model or
|
|
313
|
+
its outputs in a commercial product. `permitted_with_conditions` covers
|
|
314
|
+
any condition (user caps, attribution, field-of-use limits); the other
|
|
315
|
+
`licence.*` facets say which.
|
|
316
|
+
tier: guaranteed
|
|
317
|
+
risk: governance
|
|
318
|
+
permitted_source_kinds: [licence_text, provider_terms]
|
|
319
|
+
- id: licence.user_cap
|
|
320
|
+
label: Licence user cap
|
|
321
|
+
subject: model
|
|
322
|
+
value_type: {kind: number, unbounded: true}
|
|
323
|
+
unit: monthly_active_users
|
|
324
|
+
definition: >-
|
|
325
|
+
The largest number of monthly active users a licensee may serve before the
|
|
326
|
+
licence requires a separate agreement. `unbounded` when the licence sets
|
|
327
|
+
no such cap.
|
|
328
|
+
tier: guaranteed
|
|
329
|
+
risk: governance
|
|
330
|
+
permitted_source_kinds: [licence_text]
|
|
331
|
+
- id: licence.output_training
|
|
332
|
+
label: Training on outputs
|
|
333
|
+
subject: model
|
|
334
|
+
value_type: {kind: enum, values: [permitted, restricted, prohibited]}
|
|
335
|
+
value_labels:
|
|
336
|
+
permitted: Permitted
|
|
337
|
+
prohibited: Prohibited
|
|
338
|
+
restricted: Restricted
|
|
339
|
+
definition: >-
|
|
340
|
+
Whether the licence or terms let a customer use the model's outputs to
|
|
341
|
+
train or improve another model. `restricted` when permitted only for some
|
|
342
|
+
purposes or models (for example not for a competing model).
|
|
343
|
+
tier: guaranteed
|
|
344
|
+
risk: governance
|
|
345
|
+
permitted_source_kinds: [licence_text, provider_terms]
|
|
346
|
+
- id: licence.fine_tuning
|
|
347
|
+
label: Fine-tuning rights
|
|
348
|
+
subject: model
|
|
349
|
+
value_type: {kind: enum, values: [permitted, permitted_with_conditions, prohibited]}
|
|
350
|
+
value_labels:
|
|
351
|
+
permitted: Permitted
|
|
352
|
+
permitted_with_conditions: Permitted with conditions
|
|
353
|
+
prohibited: Prohibited
|
|
354
|
+
definition: >-
|
|
355
|
+
Whether the licence lets a customer modify the model's weights by further
|
|
356
|
+
training and use the result. Whether a provider offers fine-tuning as a
|
|
357
|
+
service is `offering.fine_tuning`, a different fact.
|
|
358
|
+
tier: guaranteed
|
|
359
|
+
risk: governance
|
|
360
|
+
permitted_source_kinds: [licence_text, provider_terms]
|
|
361
|
+
|
|
362
|
+
# ── Model: origin, three separate facets (governance) ─────────────────────
|
|
363
|
+
- id: origin.lab_jurisdiction
|
|
364
|
+
label: Lab jurisdiction
|
|
365
|
+
subject: model
|
|
366
|
+
value_type: {kind: set, values_from: iso_3166_1_alpha_2}
|
|
367
|
+
definition: >-
|
|
368
|
+
The country or countries where the lab that trained this model is
|
|
369
|
+
incorporated, as ISO 3166-1 alpha-2 codes. Where the lab has a parent
|
|
370
|
+
company, the parent's country is included too. Says nothing about where
|
|
371
|
+
the model was based on or where it runs.
|
|
372
|
+
tier: guaranteed
|
|
373
|
+
risk: governance
|
|
374
|
+
permitted_source_kinds: [lab_documentation, official_registry, provider_terms]
|
|
375
|
+
- id: origin.base_lineage
|
|
376
|
+
label: Base model lineage
|
|
377
|
+
subject: model
|
|
378
|
+
value_type: {kind: set, values_from: iso_3166_1_alpha_2}
|
|
379
|
+
definition: >-
|
|
380
|
+
The countries of incorporation of the labs that trained every model this
|
|
381
|
+
model's weights were derived from (fine-tuned, distilled into, merged or
|
|
382
|
+
continued from), as ISO 3166-1 alpha-2 codes. The empty set when the model
|
|
383
|
+
was trained from random initialisation. Excludes this model's own lab
|
|
384
|
+
unless it also trained an ancestor.
|
|
385
|
+
tier: guaranteed
|
|
386
|
+
risk: governance
|
|
387
|
+
permitted_source_kinds: [lab_documentation, weights_repository, licence_text]
|
|
388
|
+
- id: origin.weights_hosting
|
|
389
|
+
label: Weights hosted in
|
|
390
|
+
subject: model
|
|
391
|
+
value_type: {kind: set, values_from: iso_3166_1_alpha_2}
|
|
392
|
+
definition: >-
|
|
393
|
+
The countries in which the lab itself stores and serves the weights for
|
|
394
|
+
the hosted inference it sells first-party, as ISO 3166-1 alpha-2 codes.
|
|
395
|
+
The empty set when the lab sells no hosted inference of this model.
|
|
396
|
+
Where a third-party provider runs the model is `offering.region`, not this.
|
|
397
|
+
tier: guaranteed
|
|
398
|
+
risk: governance
|
|
399
|
+
permitted_source_kinds: [lab_documentation, provider_terms, provider_documentation]
|
|
400
|
+
- id: origin.base_models
|
|
401
|
+
label: Base models
|
|
402
|
+
subject: model
|
|
403
|
+
value_type: {kind: set, values_from: model_ids}
|
|
404
|
+
definition: >-
|
|
405
|
+
The `lab/model-id` of every model this model's weights were directly
|
|
406
|
+
derived from. The empty set when trained from random initialisation.
|
|
407
|
+
Backs `origin.base_lineage` with the models themselves.
|
|
408
|
+
tier: best_effort
|
|
409
|
+
risk: governance
|
|
410
|
+
permitted_source_kinds: [lab_documentation, weights_repository, licence_text]
|
|
411
|
+
|
|
412
|
+
# ── Model: lifecycle ──────────────────────────────────────────────────────
|
|
413
|
+
- id: model.release_date
|
|
414
|
+
label: Release date
|
|
415
|
+
subject: model
|
|
416
|
+
value_type: {kind: date}
|
|
417
|
+
definition: >-
|
|
418
|
+
The date the lab first made the model generally available to the public,
|
|
419
|
+
by API or download. A preview or waitlist release counts only if anyone
|
|
420
|
+
could sign up.
|
|
421
|
+
tier: guaranteed
|
|
422
|
+
risk: capability
|
|
423
|
+
permitted_source_kinds: [lab_announcement, lab_documentation]
|
|
424
|
+
- id: model.lifecycle
|
|
425
|
+
label: Lifecycle
|
|
426
|
+
subject: model
|
|
427
|
+
value_type: {kind: enum, values: [active, deprecated, retired]}
|
|
428
|
+
value_labels:
|
|
429
|
+
active: Active
|
|
430
|
+
deprecated: Deprecated
|
|
431
|
+
retired: Retired
|
|
432
|
+
definition: >-
|
|
433
|
+
`deprecated` when the lab has announced a retirement date; `retired` when
|
|
434
|
+
the lab no longer serves or supports the model; `active` otherwise.
|
|
435
|
+
Retired models are excluded from decisions unless a spec asks for them.
|
|
436
|
+
tier: guaranteed
|
|
437
|
+
risk: capability
|
|
438
|
+
permitted_source_kinds: [lab_documentation, lab_announcement]
|
|
439
|
+
- id: model.knowledge_cutoff
|
|
440
|
+
label: Knowledge cutoff
|
|
441
|
+
subject: model
|
|
442
|
+
value_type: {kind: date}
|
|
443
|
+
definition: >-
|
|
444
|
+
The latest date of training data the lab documents for the model. A month
|
|
445
|
+
is recorded as its first day, and the fact's qualifiers say so.
|
|
446
|
+
tier: best_effort
|
|
447
|
+
risk: capability
|
|
448
|
+
permitted_source_kinds: [lab_documentation]
|
|
449
|
+
- id: model.deprecation_date
|
|
450
|
+
label: Deprecation date
|
|
451
|
+
subject: model
|
|
452
|
+
value_type: {kind: date}
|
|
453
|
+
definition: >-
|
|
454
|
+
The date the lab has announced the model will stop being served by its own
|
|
455
|
+
API. Unknown until announced; not the date of the announcement.
|
|
456
|
+
tier: best_effort
|
|
457
|
+
risk: capability
|
|
458
|
+
permitted_source_kinds: [lab_documentation, lab_announcement]
|
|
459
|
+
|
|
460
|
+
# ── Model: features ───────────────────────────────────────────────────────
|
|
461
|
+
- id: feature.tool_calling
|
|
462
|
+
label: Tool calling
|
|
463
|
+
subject: model
|
|
464
|
+
value_type: {kind: boolean}
|
|
465
|
+
definition: >-
|
|
466
|
+
True when the lab documents that the model emits structured function or
|
|
467
|
+
tool calls against caller-supplied tool definitions through at least one
|
|
468
|
+
first-party interface.
|
|
469
|
+
tier: guaranteed
|
|
470
|
+
risk: capability
|
|
471
|
+
permitted_source_kinds: [lab_documentation]
|
|
472
|
+
- id: feature.structured_output
|
|
473
|
+
label: Structured output
|
|
474
|
+
subject: model
|
|
475
|
+
value_type: {kind: boolean}
|
|
476
|
+
definition: >-
|
|
477
|
+
True when the lab documents constrained output that conforms to a
|
|
478
|
+
caller-supplied JSON schema. A prompt-only "JSON mode" without a schema
|
|
479
|
+
guarantee is false.
|
|
480
|
+
tier: guaranteed
|
|
481
|
+
risk: capability
|
|
482
|
+
permitted_source_kinds: [lab_documentation]
|
|
483
|
+
- id: feature.effort_controls
|
|
484
|
+
label: Effort controls
|
|
485
|
+
subject: model
|
|
486
|
+
value_type: {kind: boolean}
|
|
487
|
+
definition: >-
|
|
488
|
+
True when the lab documents a request parameter that changes how much
|
|
489
|
+
reasoning the model does (a reasoning effort level or thinking budget).
|
|
490
|
+
tier: guaranteed
|
|
491
|
+
risk: capability
|
|
492
|
+
permitted_source_kinds: [lab_documentation]
|
|
493
|
+
- id: feature.batch
|
|
494
|
+
label: Batch interface
|
|
495
|
+
subject: model
|
|
496
|
+
value_type: {kind: boolean}
|
|
497
|
+
definition: >-
|
|
498
|
+
True when the lab's first-party API offers an asynchronous batch interface
|
|
499
|
+
for this model. Batch pricing is `offering.price.batch_*`.
|
|
500
|
+
tier: guaranteed
|
|
501
|
+
risk: capability
|
|
502
|
+
permitted_source_kinds: [lab_documentation, provider_documentation]
|
|
503
|
+
- id: feature.streaming
|
|
504
|
+
label: Streaming
|
|
505
|
+
subject: model
|
|
506
|
+
value_type: {kind: boolean}
|
|
507
|
+
definition: >-
|
|
508
|
+
True when the lab's first-party API can return the response incrementally
|
|
509
|
+
as it is generated.
|
|
510
|
+
tier: guaranteed
|
|
511
|
+
risk: capability
|
|
512
|
+
permitted_source_kinds: [lab_documentation, provider_documentation]
|
|
513
|
+
|
|
514
|
+
# ── Model: best-effort extras ─────────────────────────────────────────────
|
|
515
|
+
- id: model.languages
|
|
516
|
+
label: Languages
|
|
517
|
+
subject: model
|
|
518
|
+
value_type: {kind: set, values_from: bcp_47}
|
|
519
|
+
definition: >-
|
|
520
|
+
The natural languages the lab documents the model as supporting, as BCP 47
|
|
521
|
+
tags. Absence from the set means undocumented, not unsupported.
|
|
522
|
+
tier: best_effort
|
|
523
|
+
risk: capability
|
|
524
|
+
permitted_source_kinds: [lab_documentation, weights_repository]
|
|
525
|
+
- id: model.fits_hardware
|
|
526
|
+
label: Fits hardware
|
|
527
|
+
subject: model
|
|
528
|
+
value_type: {kind: set, values_from: hardware}
|
|
529
|
+
definition: >-
|
|
530
|
+
The hardware SKUs in `hardware/` on which some published quantisation of
|
|
531
|
+
the model's weights is estimated to load and run. An estimate, always
|
|
532
|
+
labelled as one.
|
|
533
|
+
tier: best_effort
|
|
534
|
+
risk: capability
|
|
535
|
+
permitted_source_kinds: [modelspec_estimate]
|
|
536
|
+
|
|
537
|
+
# ── Offering: identity ────────────────────────────────────────────────────
|
|
538
|
+
- id: offering.provider
|
|
539
|
+
label: Provider
|
|
540
|
+
subject: offering
|
|
541
|
+
value_type: {kind: enum, values_from: "registry:providers"}
|
|
542
|
+
definition: >-
|
|
543
|
+
The provider in `registry/providers.yaml` that sells the offering and
|
|
544
|
+
bills the customer for it.
|
|
545
|
+
tier: guaranteed
|
|
546
|
+
risk: capability
|
|
547
|
+
permitted_source_kinds: [provider_documentation]
|
|
548
|
+
- id: offering.region
|
|
549
|
+
label: Inference region
|
|
550
|
+
subject: offering
|
|
551
|
+
value_type: {kind: set, values_from: iso_3166_1_alpha_2}
|
|
552
|
+
value_labels:
|
|
553
|
+
global: Global
|
|
554
|
+
"global-cross-region": Global, cross-region routing
|
|
555
|
+
"global-short-context": Global, short context
|
|
556
|
+
definition: >-
|
|
557
|
+
The countries in which the provider documents that inference for this
|
|
558
|
+
offering runs, as ISO 3166-1 alpha-2 codes. A global or routed offering
|
|
559
|
+
lists every country it may run in. The provider's own region name is a
|
|
560
|
+
qualifier on the fact.
|
|
561
|
+
tier: guaranteed
|
|
562
|
+
risk: governance
|
|
563
|
+
permitted_source_kinds: [provider_documentation, provider_terms]
|
|
564
|
+
- id: offering.tier
|
|
565
|
+
label: Account tier
|
|
566
|
+
subject: offering
|
|
567
|
+
value_type: {kind: enum, values: [standard, enterprise, zero_retention, government]}
|
|
568
|
+
value_labels:
|
|
569
|
+
enterprise: Enterprise
|
|
570
|
+
government: Government
|
|
571
|
+
standard: Standard
|
|
572
|
+
zero_retention: Zero retention
|
|
573
|
+
definition: >-
|
|
574
|
+
The account tier the offering is sold under. A tier is its own offering
|
|
575
|
+
only when a guaranteed fact differs from the standard tier: price, data
|
|
576
|
+
handling or an attestation (design §4.1).
|
|
577
|
+
tier: guaranteed
|
|
578
|
+
risk: capability
|
|
579
|
+
permitted_source_kinds: [provider_documentation, provider_terms]
|
|
580
|
+
|
|
581
|
+
# ── Offering: price ───────────────────────────────────────────────────────
|
|
582
|
+
- id: offering.price.input
|
|
583
|
+
label: Input price
|
|
584
|
+
subject: offering
|
|
585
|
+
value_type: {kind: number}
|
|
586
|
+
unit: usd_per_1m_tokens
|
|
587
|
+
definition: >-
|
|
588
|
+
The list price of uncached input tokens for a synchronous request. Where
|
|
589
|
+
the price varies with prompt length, the lowest band, with the band as a
|
|
590
|
+
qualifier.
|
|
591
|
+
tier: guaranteed
|
|
592
|
+
risk: capability
|
|
593
|
+
permitted_source_kinds: [provider_documentation]
|
|
594
|
+
- id: offering.price.output
|
|
595
|
+
label: Output price
|
|
596
|
+
subject: offering
|
|
597
|
+
value_type: {kind: number}
|
|
598
|
+
unit: usd_per_1m_tokens
|
|
599
|
+
definition: >-
|
|
600
|
+
The list price of output tokens, including reasoning tokens the provider
|
|
601
|
+
bills as output, for a synchronous request, lowest band as for input.
|
|
602
|
+
tier: guaranteed
|
|
603
|
+
risk: capability
|
|
604
|
+
permitted_source_kinds: [provider_documentation]
|
|
605
|
+
- id: offering.price.cached_input
|
|
606
|
+
label: Cached input price
|
|
607
|
+
subject: offering
|
|
608
|
+
value_type: {kind: number, not_offered: true}
|
|
609
|
+
unit: usd_per_1m_tokens
|
|
610
|
+
definition: >-
|
|
611
|
+
The list price of input tokens read from the provider's prompt cache.
|
|
612
|
+
Cache-write surcharges are a qualifier. `not_offered` when the provider
|
|
613
|
+
has no prompt cache for this offering, which is a known value, not an
|
|
614
|
+
unknown one.
|
|
615
|
+
tier: guaranteed
|
|
616
|
+
risk: capability
|
|
617
|
+
permitted_source_kinds: [provider_documentation]
|
|
618
|
+
- id: offering.price.batch_input
|
|
619
|
+
label: Batch input price
|
|
620
|
+
subject: offering
|
|
621
|
+
value_type: {kind: number, not_offered: true}
|
|
622
|
+
unit: usd_per_1m_tokens
|
|
623
|
+
definition: >-
|
|
624
|
+
The list price of input tokens submitted through the provider's
|
|
625
|
+
asynchronous batch interface. `not_offered` when the provider has no
|
|
626
|
+
batch interface for this offering.
|
|
627
|
+
tier: guaranteed
|
|
628
|
+
risk: capability
|
|
629
|
+
permitted_source_kinds: [provider_documentation]
|
|
630
|
+
- id: offering.price.batch_output
|
|
631
|
+
label: Batch output price
|
|
632
|
+
subject: offering
|
|
633
|
+
value_type: {kind: number, not_offered: true}
|
|
634
|
+
unit: usd_per_1m_tokens
|
|
635
|
+
definition: >-
|
|
636
|
+
The list price of output tokens returned through the provider's
|
|
637
|
+
asynchronous batch interface. `not_offered` when the provider has no
|
|
638
|
+
batch interface for this offering.
|
|
639
|
+
tier: guaranteed
|
|
640
|
+
risk: capability
|
|
641
|
+
permitted_source_kinds: [provider_documentation]
|
|
642
|
+
|
|
643
|
+
- id: offering.cost_per_task
|
|
644
|
+
label: Cost per task
|
|
645
|
+
subject: offering
|
|
646
|
+
value_type: {kind: number}
|
|
647
|
+
unit: usd_per_task
|
|
648
|
+
definition: >-
|
|
649
|
+
What one task costs on this offering at list prices, computed per
|
|
650
|
+
decision from the spec's `task_tokens`: (offering.price.input × input
|
|
651
|
+
tokens + offering.price.output × output tokens) / 1,000,000. Unknown when
|
|
652
|
+
either price is unknown. Never authored on a card.
|
|
653
|
+
tier: guaranteed
|
|
654
|
+
risk: capability
|
|
655
|
+
permitted_source_kinds: [modelspec_estimate]
|
|
656
|
+
computed_by: MODEL-153
|
|
657
|
+
|
|
658
|
+
# ── Offering: speed, limits, SLA ──────────────────────────────────────────
|
|
659
|
+
- id: offering.speed.time_to_first_token
|
|
660
|
+
label: Time to first token
|
|
661
|
+
subject: offering
|
|
662
|
+
value_type: {kind: number}
|
|
663
|
+
unit: milliseconds
|
|
664
|
+
definition: >-
|
|
665
|
+
The median time from sending a request to receiving the first output
|
|
666
|
+
token. Always carries the method: prompt length, output length, effort,
|
|
667
|
+
region of the client, sample size and date.
|
|
668
|
+
tier: best_effort
|
|
669
|
+
risk: capability
|
|
670
|
+
permitted_source_kinds: [modelspec_measurement, independent_evaluator, provider_self_report, outcome_protocol]
|
|
671
|
+
required_qualifiers: [method]
|
|
672
|
+
- id: offering.speed.throughput
|
|
673
|
+
label: Output throughput
|
|
674
|
+
subject: offering
|
|
675
|
+
value_type: {kind: number}
|
|
676
|
+
unit: tokens_per_second
|
|
677
|
+
definition: >-
|
|
678
|
+
The median output throughput of one request after its first token. Always
|
|
679
|
+
carries the method, as for time to first token. Self-hosted throughput is
|
|
680
|
+
a labelled estimate only.
|
|
681
|
+
tier: best_effort
|
|
682
|
+
risk: capability
|
|
683
|
+
permitted_source_kinds: [modelspec_measurement, independent_evaluator, provider_self_report, outcome_protocol]
|
|
684
|
+
required_qualifiers: [method]
|
|
685
|
+
- id: offering.rate_limit.requests
|
|
686
|
+
label: Request rate limit
|
|
687
|
+
subject: offering
|
|
688
|
+
value_type: {kind: number}
|
|
689
|
+
unit: requests_per_minute
|
|
690
|
+
definition: >-
|
|
691
|
+
The documented default request rate limit for a new account at this tier,
|
|
692
|
+
at the lowest paid usage level the provider documents.
|
|
693
|
+
tier: best_effort
|
|
694
|
+
risk: capability
|
|
695
|
+
permitted_source_kinds: [provider_documentation]
|
|
696
|
+
- id: offering.rate_limit.tokens
|
|
697
|
+
label: Token rate limit
|
|
698
|
+
subject: offering
|
|
699
|
+
value_type: {kind: number}
|
|
700
|
+
unit: tokens_per_minute
|
|
701
|
+
definition: >-
|
|
702
|
+
The documented default token rate limit for a new account at this tier,
|
|
703
|
+
at the lowest paid usage level the provider documents.
|
|
704
|
+
tier: best_effort
|
|
705
|
+
risk: capability
|
|
706
|
+
permitted_source_kinds: [provider_documentation]
|
|
707
|
+
- id: offering.sla_uptime
|
|
708
|
+
label: SLA uptime
|
|
709
|
+
subject: offering
|
|
710
|
+
value_type: {kind: number}
|
|
711
|
+
unit: percent
|
|
712
|
+
definition: >-
|
|
713
|
+
The monthly uptime the provider commits to in a published service level
|
|
714
|
+
agreement for this offering, with service credits. A status-page history
|
|
715
|
+
is not an SLA.
|
|
716
|
+
tier: best_effort
|
|
717
|
+
risk: capability
|
|
718
|
+
permitted_source_kinds: [provider_terms]
|
|
719
|
+
|
|
720
|
+
# ── Offering: data handling (governance) ──────────────────────────────────
|
|
721
|
+
- id: offering.data.retention
|
|
722
|
+
label: Data retention
|
|
723
|
+
subject: offering
|
|
724
|
+
value_type: {kind: number, unbounded: true}
|
|
725
|
+
unit: days
|
|
726
|
+
definition: >-
|
|
727
|
+
The longest period the provider's terms say it keeps prompts and outputs
|
|
728
|
+
for this offering by default, for any purpose including abuse monitoring.
|
|
729
|
+
0 means not stored after the response. `unbounded` when the terms set no
|
|
730
|
+
limit.
|
|
731
|
+
tier: guaranteed
|
|
732
|
+
risk: governance
|
|
733
|
+
permitted_source_kinds: [provider_terms, provider_documentation]
|
|
734
|
+
- id: offering.data.trains_on_customer_data
|
|
735
|
+
label: Trains on customer data
|
|
736
|
+
subject: offering
|
|
737
|
+
value_type: {kind: boolean}
|
|
738
|
+
definition: >-
|
|
739
|
+
True when the provider's default terms for this offering let it use
|
|
740
|
+
customer prompts or outputs to train or improve models. An opt-out the
|
|
741
|
+
customer must take still makes this true.
|
|
742
|
+
tier: guaranteed
|
|
743
|
+
risk: governance
|
|
744
|
+
permitted_source_kinds: [provider_terms]
|
|
745
|
+
- id: offering.data.zero_retention
|
|
746
|
+
label: Zero retention available
|
|
747
|
+
subject: offering
|
|
748
|
+
value_type: {kind: boolean}
|
|
749
|
+
definition: >-
|
|
750
|
+
True when the provider documents that a customer can have prompts and
|
|
751
|
+
outputs for this offering not stored at all, including for abuse
|
|
752
|
+
monitoring, whether by default or on request.
|
|
753
|
+
tier: guaranteed
|
|
754
|
+
risk: governance
|
|
755
|
+
permitted_source_kinds: [provider_terms, provider_documentation]
|
|
756
|
+
|
|
757
|
+
# ── Offering: attestations (governance; documented, never certified) ──────
|
|
758
|
+
- id: offering.attestation.soc2
|
|
759
|
+
label: SOC 2 report
|
|
760
|
+
subject: offering
|
|
761
|
+
value_type: {kind: enum, values: [type_1, type_2, none]}
|
|
762
|
+
value_labels:
|
|
763
|
+
"none": None
|
|
764
|
+
type_1: SOC 2 Type I
|
|
765
|
+
type_2: SOC 2 Type II
|
|
766
|
+
definition: >-
|
|
767
|
+
The SOC 2 report type the provider documents for the service this offering
|
|
768
|
+
runs on, when that report's scope covers the service. `none` when the
|
|
769
|
+
provider documents no such report. Documented, never "compliant".
|
|
770
|
+
tier: guaranteed
|
|
771
|
+
risk: governance
|
|
772
|
+
permitted_source_kinds: [provider_trust_center, provider_documentation]
|
|
773
|
+
- id: offering.attestation.baa
|
|
774
|
+
label: HIPAA BAA
|
|
775
|
+
subject: offering
|
|
776
|
+
value_type: {kind: boolean}
|
|
777
|
+
definition: >-
|
|
778
|
+
True when the provider documents that it will sign a HIPAA business
|
|
779
|
+
associate agreement that covers this offering. Says nothing about whether
|
|
780
|
+
a customer's use is compliant.
|
|
781
|
+
tier: guaranteed
|
|
782
|
+
risk: governance
|
|
783
|
+
permitted_source_kinds: [provider_trust_center, provider_documentation, provider_terms]
|
|
784
|
+
- id: offering.attestation.fedramp
|
|
785
|
+
label: FedRAMP level
|
|
786
|
+
subject: offering
|
|
787
|
+
value_type: {kind: enum, values: [high, moderate, low, li_saas, none]}
|
|
788
|
+
value_labels:
|
|
789
|
+
high: FedRAMP High
|
|
790
|
+
li_saas: FedRAMP Low Impact SaaS
|
|
791
|
+
low: FedRAMP Low
|
|
792
|
+
moderate: FedRAMP Moderate
|
|
793
|
+
"none": None
|
|
794
|
+
definition: >-
|
|
795
|
+
The FedRAMP authorization level of the cloud service offering this
|
|
796
|
+
offering runs in, as the FedRAMP marketplace lists it for this model.
|
|
797
|
+
`none` when not listed.
|
|
798
|
+
tier: best_effort
|
|
799
|
+
risk: governance
|
|
800
|
+
permitted_source_kinds: [official_registry, provider_documentation]
|
|
801
|
+
- id: offering.attestation.iso_27001
|
|
802
|
+
label: ISO/IEC 27001
|
|
803
|
+
subject: offering
|
|
804
|
+
value_type: {kind: boolean}
|
|
805
|
+
definition: >-
|
|
806
|
+
True when the provider documents an ISO/IEC 27001 certificate whose scope
|
|
807
|
+
covers the service this offering runs on.
|
|
808
|
+
tier: best_effort
|
|
809
|
+
risk: governance
|
|
810
|
+
permitted_source_kinds: [provider_trust_center, provider_documentation]
|
|
811
|
+
|
|
812
|
+
# ── Offering: best-effort extras ──────────────────────────────────────────
|
|
813
|
+
- id: offering.fine_tuning
|
|
814
|
+
label: Fine-tuning service
|
|
815
|
+
subject: offering
|
|
816
|
+
value_type: {kind: boolean}
|
|
817
|
+
definition: >-
|
|
818
|
+
True when the provider sells fine-tuning of this model as a service and
|
|
819
|
+
serves the tuned model under this offering's terms.
|
|
820
|
+
tier: best_effort
|
|
821
|
+
risk: capability
|
|
822
|
+
permitted_source_kinds: [provider_documentation]
|
|
823
|
+
- id: offering.private_deployment
|
|
824
|
+
label: Private deployment
|
|
825
|
+
subject: offering
|
|
826
|
+
value_type: {kind: boolean}
|
|
827
|
+
definition: >-
|
|
828
|
+
True when the provider sells dedicated capacity for this model that is
|
|
829
|
+
not shared with other customers (provisioned throughput, a dedicated
|
|
830
|
+
endpoint or deployment in the customer's own cloud account).
|
|
831
|
+
tier: best_effort
|
|
832
|
+
risk: capability
|
|
833
|
+
permitted_source_kinds: [provider_documentation]
|
|
834
|
+
- id: offering.harness_compatibility
|
|
835
|
+
label: Harness compatibility
|
|
836
|
+
subject: offering
|
|
837
|
+
value_type: {kind: set, values_from: "registry:harnesses"}
|
|
838
|
+
definition: >-
|
|
839
|
+
The registered harnesses whose own documentation names this provider as a
|
|
840
|
+
supported backend for this model. Absence means undocumented.
|
|
841
|
+
tier: best_effort
|
|
842
|
+
risk: capability
|
|
843
|
+
permitted_source_kinds: [provider_documentation, lab_documentation]
|
|
844
|
+
|
|
845
|
+
# ── Evidence and estimates ────────────────────────────────────────────────
|
|
846
|
+
- id: evidence.benchmark
|
|
847
|
+
label: Benchmark result
|
|
848
|
+
subject: evidence
|
|
849
|
+
value_type: {kind: number}
|
|
850
|
+
unit: benchmark_metric
|
|
851
|
+
parameter: {name: benchmark, values_from: benchmarks}
|
|
852
|
+
definition: >-
|
|
853
|
+
One measured result on a benchmark version and sub-category, with its
|
|
854
|
+
qualifiers (effort, harness, tools, configuration, measured_by) and
|
|
855
|
+
evidence date. Never copied between models; a missing value is missing.
|
|
856
|
+
tier: best_effort
|
|
857
|
+
risk: capability
|
|
858
|
+
permitted_source_kinds: [benchmark_author, independent_evaluator, provider_self_report, modelspec_measurement]
|
|
859
|
+
required_qualifiers: [measured_by, evidence_date, date_type]
|
|
860
|
+
- id: evidence.outcome
|
|
861
|
+
label: Outcome success rate
|
|
862
|
+
subject: evidence
|
|
863
|
+
value_type: {kind: number}
|
|
864
|
+
unit: percent
|
|
865
|
+
parameter: {name: task_type, values_from: outcome_task_types}
|
|
866
|
+
definition: >-
|
|
867
|
+
The share of outcome records for a task type, offering and harness whose
|
|
868
|
+
objective result was success, with the record count. Published only above
|
|
869
|
+
the protocol's minimum count.
|
|
870
|
+
tier: best_effort
|
|
871
|
+
risk: capability
|
|
872
|
+
permitted_source_kinds: [outcome_protocol]
|
|
873
|
+
required_qualifiers: [harness, record_count]
|
|
874
|
+
- id: estimate.capability
|
|
875
|
+
label: Capability estimate
|
|
876
|
+
subject: model
|
|
877
|
+
value_type: {kind: range}
|
|
878
|
+
unit: capability_scale
|
|
879
|
+
parameter: {name: domain, values_from: "registry:domains"}
|
|
880
|
+
definition: >-
|
|
881
|
+
The capability model's estimate of the model's capability in one domain,
|
|
882
|
+
with its interval, computed from verified evidence only. Constraints never
|
|
883
|
+
add to it. Slice 1 ships no estimates: stage 3 shows evidence per domain,
|
|
884
|
+
unblended.
|
|
885
|
+
tier: guaranteed
|
|
886
|
+
risk: capability
|
|
887
|
+
permitted_source_kinds: [modelspec_estimate]
|
|
888
|
+
computed_by: MODEL-129
|