modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
registry/facets.yaml ADDED
@@ -0,0 +1,888 @@
1
+ # The facet registry (MODEL-133; design §4.1 and §8).
2
+ #
3
+ # A facet is something a spec can constrain or optimise. Adding one is an entry
4
+ # here plus data; the engine reads this file and has no facet list of its own.
5
+ # Loader and validation: decision/registry.py.
6
+ #
7
+ # Entry fields
8
+ # id dotted snake_case, unique
9
+ # subject model | offering | evidence
10
+ # value_type.kind number | enum | boolean | date | set | range
11
+ # number, range need `unit` (one of `units` below). `unbounded: true`
12
+ # lets a known value be the literal `unbounded`
13
+ # (for example a licence with no user cap), and
14
+ # `not_offered: true` the literal `not_offered` (for
15
+ # example a provider with no batch interface).
16
+ # enum, set need `values` or `values_from` (a named list the
17
+ # loader knows, such as `registry:providers`).
18
+ # parameter optional: the facet is a family, one per value of a
19
+ # named list (for example one estimate per domain).
20
+ # definition exact. Says what a value means and what it does not.
21
+ # tier guaranteed | best_effort (design §8, G and B)
22
+ # risk capability | governance. Sets the default unknown
23
+ # policy: capability -> may_qualify,
24
+ # governance -> not_satisfied.
25
+ # permitted_source_kinds ids from `source_kinds` below
26
+ # unit id from `units` below, where applicable
27
+ # required_qualifiers optional: qualifiers every fact must carry
28
+ # computed_by optional: the ticket whose build step produces the
29
+ # value. Such facets are never authored on a card.
30
+ # label optional: a short name for people ("Input price").
31
+ # The id stays the identifier; the label is display.
32
+ #
33
+ # A fact's state (known, unknown, not_disclosed, requires_contract) is separate
34
+ # from its value and is not declared here.
35
+
36
+ schema_version: 1
37
+
38
+ units:
39
+ - id: usd_per_1m_tokens
40
+ definition: >-
41
+ United States dollars per 1,000,000 tokens, at the provider's list price
42
+ for the offering, before discounts, credits or committed-use pricing.
43
+ Tokens are counted by the provider's own tokenizer for that model.
44
+ - id: tokens
45
+ definition: >-
46
+ Tokens as counted by the model's own tokenizer, as documented by its lab.
47
+ Not words, characters or another model's tokens.
48
+ - id: tokens_per_second
49
+ definition: >-
50
+ Output tokens generated per second of wall-clock time, measured from the
51
+ first output token to the last, for a single request.
52
+ - id: milliseconds
53
+ definition: >-
54
+ Milliseconds of wall-clock time.
55
+ - id: parameters
56
+ definition: >-
57
+ A count of trainable weights (not billions; 7e9 is written 7000000000).
58
+ - id: days
59
+ definition: >-
60
+ Calendar days.
61
+ - id: requests_per_minute
62
+ definition: >-
63
+ Requests accepted per 60-second window for one account at the tier the
64
+ offering names, before any limit increase negotiated by contract.
65
+ - id: tokens_per_minute
66
+ definition: >-
67
+ Tokens (input plus output unless the provider states otherwise, in which
68
+ case the fact's qualifiers say so) accepted per 60-second window for one
69
+ account at the tier the offering names.
70
+ - id: percent
71
+ definition: >-
72
+ A percentage from 0 to 100.
73
+ - id: monthly_active_users
74
+ definition: >-
75
+ Distinct end users in a calendar month, as the licence itself defines the
76
+ count.
77
+ - id: benchmark_metric
78
+ definition: >-
79
+ The unit declared by the benchmark page's `metric.unit` for the benchmark
80
+ version the evidence names. Evidence on different benchmarks is never
81
+ compared in this unit.
82
+ - id: usd_per_task
83
+ definition: >-
84
+ United States dollars for one task, at list prices: the offering's input
85
+ price times the spec's input tokens per task, plus its output price times
86
+ the spec's output tokens per task, divided by 1,000,000. Excludes cached,
87
+ batch and committed-use pricing.
88
+ - id: capability_scale
89
+ definition: >-
90
+ The latent scale of the capability model (MODEL-129). Values are
91
+ comparable within one domain and one snapshot only.
92
+
93
+ source_kinds:
94
+ - id: lab_documentation
95
+ definition: >-
96
+ A model card, system card, technical report or API reference published by
97
+ the lab that trained the model.
98
+ - id: lab_announcement
99
+ definition: >-
100
+ A release post or changelog entry published by the lab that trained the
101
+ model.
102
+ - id: licence_text
103
+ definition: >-
104
+ The licence or terms of use under which the model's weights or outputs are
105
+ made available, as published by the licensor.
106
+ - id: weights_repository
107
+ definition: >-
108
+ Configuration and metadata files published alongside the weights by the
109
+ lab or its designated repository (for example a model repository's config).
110
+ - id: provider_documentation
111
+ definition: >-
112
+ A model page, pricing page, rate-limit page or API reference published by
113
+ the provider that sells the offering.
114
+ - id: provider_terms
115
+ definition: >-
116
+ The provider's service terms, data processing addendum or product-specific
117
+ terms that govern the offering.
118
+ - id: provider_trust_center
119
+ definition: >-
120
+ The provider's own page listing its certifications, attestations and
121
+ audit reports.
122
+ - id: official_registry
123
+ definition: >-
124
+ A register kept by the body that grants a status, such as an
125
+ authorization marketplace or a company register.
126
+ - id: benchmark_author
127
+ definition: >-
128
+ Results published by the authors or maintainers of the benchmark.
129
+ - id: independent_evaluator
130
+ definition: >-
131
+ Results published by an evaluator that is neither the benchmark's authors
132
+ nor the lab or provider being measured.
133
+ - id: provider_self_report
134
+ definition: >-
135
+ Results published by the lab or provider about its own model or offering.
136
+ Legitimate, and always distinguishable from independent evidence.
137
+ - id: modelspec_measurement
138
+ definition: >-
139
+ A measurement ModelSpec ran itself under a published method (ADR 0004).
140
+ - id: outcome_protocol
141
+ definition: >-
142
+ Aggregated outcome records reported through the outcome protocol (design
143
+ §9).
144
+ - id: modelspec_estimate
145
+ definition: >-
146
+ A value ModelSpec computes from other verified facts and evidence, such as
147
+ a hardware fit or a capability estimate. Labelled as an estimate.
148
+
149
+ facets:
150
+ # ── Model: class, modalities, context ─────────────────────────────────────
151
+ - id: model.class
152
+ label: Model class
153
+ subject: model
154
+ value_type: {kind: enum, values_from: model_classes}
155
+ value_labels:
156
+ actor: Agent (acts on tools or environments)
157
+ analyser: Analyser (labels parts of its input)
158
+ decider: Decision model
159
+ forecaster: Forecaster
160
+ labeller: Classifier
161
+ "media-generator": Media generator
162
+ orderer: Reranker
163
+ scorer: Scorer
164
+ simulator: Simulator
165
+ "text-generator": Text generator
166
+ transcriber: Speech recognition
167
+ vectoriser: Embedding model
168
+ definition: >-
169
+ The model's class as derived from its model type by `api/classes.py`:
170
+ what it consumes, what it emits and what decision it makes. One value per
171
+ model. Class is derived, not authored, and says nothing about quality.
172
+ tier: guaranteed
173
+ risk: capability
174
+ permitted_source_kinds: [lab_documentation, lab_announcement, weights_repository]
175
+ - id: model.input_modalities
176
+ label: Input modalities
177
+ subject: model
178
+ value_type: {kind: set, values: [text, image, audio, video, document, structured]}
179
+ value_labels:
180
+ audio: Audio
181
+ document: Documents
182
+ image: Images
183
+ structured: Structured data
184
+ text: Text
185
+ video: Video
186
+ definition: >-
187
+ Every modality the model accepts as input natively, without a separate
188
+ model transcribing or captioning it first. `document` means files such as
189
+ PDF read as documents, not text extracted by the caller.
190
+ tier: guaranteed
191
+ risk: capability
192
+ permitted_source_kinds: [lab_documentation, lab_announcement, weights_repository]
193
+ - id: model.output_modalities
194
+ label: Output modalities
195
+ subject: model
196
+ value_type: {kind: set, values: [text, image, audio, video, embedding, score, label, action]}
197
+ value_labels:
198
+ action: Actions
199
+ audio: Audio
200
+ embedding: Embeddings
201
+ image: Images
202
+ label: Labels
203
+ score: Scores
204
+ text: Text
205
+ video: Video
206
+ definition: >-
207
+ Every modality the model emits natively. `embedding` is a vector,
208
+ `score` a relevance or reward number, `label` a class from a fixed or
209
+ caller-defined set, and `action` a tool or environment action.
210
+ tier: guaranteed
211
+ risk: capability
212
+ permitted_source_kinds: [lab_documentation, lab_announcement, weights_repository]
213
+ - id: model.context_window
214
+ label: Context window
215
+ subject: model
216
+ value_type: {kind: number}
217
+ unit: tokens
218
+ definition: >-
219
+ The maximum number of tokens the model accepts in one request, input and
220
+ output together, as documented by its lab for its largest supported
221
+ configuration. A provider that serves a smaller window records that on the
222
+ offering, not here.
223
+ tier: guaranteed
224
+ risk: capability
225
+ permitted_source_kinds: [lab_documentation, lab_announcement, weights_repository]
226
+ - id: model.max_output_tokens
227
+ label: Maximum output tokens
228
+ subject: model
229
+ value_type: {kind: number}
230
+ unit: tokens
231
+ definition: >-
232
+ The maximum number of tokens the model can emit in one response,
233
+ including any reasoning tokens the lab counts against the limit, as
234
+ documented by its lab.
235
+ tier: guaranteed
236
+ risk: capability
237
+ permitted_source_kinds: [lab_documentation, lab_announcement]
238
+
239
+ # ── Model: weights, parameters, architecture ──────────────────────────────
240
+ - id: model.weights_openness
241
+ label: Open weights
242
+ subject: model
243
+ value_type: {kind: enum, values: [open_weights, closed_weights]}
244
+ value_labels:
245
+ closed_weights: Closed weights
246
+ open_weights: Open weights
247
+ definition: >-
248
+ `open_weights` when the lab publishes the trained weights for download
249
+ under any licence, including a gated or restrictive one; `closed_weights`
250
+ otherwise. Says nothing about the licence terms, which are the `licence.*`
251
+ facets.
252
+ tier: guaranteed
253
+ risk: capability
254
+ permitted_source_kinds: [lab_documentation, lab_announcement, weights_repository, licence_text]
255
+ - id: model.parameters_total
256
+ label: Total parameters
257
+ subject: model
258
+ value_type: {kind: number}
259
+ unit: parameters
260
+ definition: >-
261
+ The total number of parameters in the released model, counting every
262
+ expert of a mixture-of-experts model. `not_disclosed` is a state of the
263
+ fact, never an estimate.
264
+ tier: best_effort
265
+ risk: capability
266
+ permitted_source_kinds: [lab_documentation, weights_repository]
267
+ - id: model.parameters_active
268
+ label: Active parameters
269
+ subject: model
270
+ value_type: {kind: number}
271
+ unit: parameters
272
+ definition: >-
273
+ The number of parameters used to process one token. Equal to the total
274
+ for a dense model; for a mixture-of-experts model, as the lab documents
275
+ it.
276
+ tier: best_effort
277
+ risk: capability
278
+ permitted_source_kinds: [lab_documentation, weights_repository]
279
+ - id: model.architecture
280
+ label: Architecture
281
+ subject: model
282
+ value_type: {kind: enum, values_from: architecture_types}
283
+ value_labels:
284
+ GAN: GAN
285
+ MoE: Mixture of experts
286
+ SSM: State-space model
287
+ "decoder-only": Decoder-only transformer
288
+ "dense-transformer": Dense transformer
289
+ diffusion: Diffusion
290
+ "encoder-decoder": Encoder-decoder transformer
291
+ "encoder-only": Encoder-only transformer
292
+ "flow-matching": Flow matching
293
+ "hybrid-SSM-transformer": Hybrid state-space and transformer
294
+ other: Other
295
+ definition: >-
296
+ The model's architecture family as `ArchitectureType` in `schema/enums.py`
297
+ names it. An undisclosed architecture is unknown, never inapplicable.
298
+ tier: best_effort
299
+ risk: capability
300
+ permitted_source_kinds: [lab_documentation, weights_repository]
301
+
302
+ # ── Model: licence rights (governance) ────────────────────────────────────
303
+ - id: licence.commercial_use
304
+ label: Commercial use
305
+ subject: model
306
+ value_type: {kind: enum, values: [permitted, permitted_with_conditions, prohibited]}
307
+ value_labels:
308
+ permitted: Permitted
309
+ permitted_with_conditions: Permitted with conditions
310
+ prohibited: Prohibited
311
+ definition: >-
312
+ Whether the licence governing the model lets a customer use the model or
313
+ its outputs in a commercial product. `permitted_with_conditions` covers
314
+ any condition (user caps, attribution, field-of-use limits); the other
315
+ `licence.*` facets say which.
316
+ tier: guaranteed
317
+ risk: governance
318
+ permitted_source_kinds: [licence_text, provider_terms]
319
+ - id: licence.user_cap
320
+ label: Licence user cap
321
+ subject: model
322
+ value_type: {kind: number, unbounded: true}
323
+ unit: monthly_active_users
324
+ definition: >-
325
+ The largest number of monthly active users a licensee may serve before the
326
+ licence requires a separate agreement. `unbounded` when the licence sets
327
+ no such cap.
328
+ tier: guaranteed
329
+ risk: governance
330
+ permitted_source_kinds: [licence_text]
331
+ - id: licence.output_training
332
+ label: Training on outputs
333
+ subject: model
334
+ value_type: {kind: enum, values: [permitted, restricted, prohibited]}
335
+ value_labels:
336
+ permitted: Permitted
337
+ prohibited: Prohibited
338
+ restricted: Restricted
339
+ definition: >-
340
+ Whether the licence or terms let a customer use the model's outputs to
341
+ train or improve another model. `restricted` when permitted only for some
342
+ purposes or models (for example not for a competing model).
343
+ tier: guaranteed
344
+ risk: governance
345
+ permitted_source_kinds: [licence_text, provider_terms]
346
+ - id: licence.fine_tuning
347
+ label: Fine-tuning rights
348
+ subject: model
349
+ value_type: {kind: enum, values: [permitted, permitted_with_conditions, prohibited]}
350
+ value_labels:
351
+ permitted: Permitted
352
+ permitted_with_conditions: Permitted with conditions
353
+ prohibited: Prohibited
354
+ definition: >-
355
+ Whether the licence lets a customer modify the model's weights by further
356
+ training and use the result. Whether a provider offers fine-tuning as a
357
+ service is `offering.fine_tuning`, a different fact.
358
+ tier: guaranteed
359
+ risk: governance
360
+ permitted_source_kinds: [licence_text, provider_terms]
361
+
362
+ # ── Model: origin, three separate facets (governance) ─────────────────────
363
+ - id: origin.lab_jurisdiction
364
+ label: Lab jurisdiction
365
+ subject: model
366
+ value_type: {kind: set, values_from: iso_3166_1_alpha_2}
367
+ definition: >-
368
+ The country or countries where the lab that trained this model is
369
+ incorporated, as ISO 3166-1 alpha-2 codes. Where the lab has a parent
370
+ company, the parent's country is included too. Says nothing about where
371
+ the model was based on or where it runs.
372
+ tier: guaranteed
373
+ risk: governance
374
+ permitted_source_kinds: [lab_documentation, official_registry, provider_terms]
375
+ - id: origin.base_lineage
376
+ label: Base model lineage
377
+ subject: model
378
+ value_type: {kind: set, values_from: iso_3166_1_alpha_2}
379
+ definition: >-
380
+ The countries of incorporation of the labs that trained every model this
381
+ model's weights were derived from (fine-tuned, distilled into, merged or
382
+ continued from), as ISO 3166-1 alpha-2 codes. The empty set when the model
383
+ was trained from random initialisation. Excludes this model's own lab
384
+ unless it also trained an ancestor.
385
+ tier: guaranteed
386
+ risk: governance
387
+ permitted_source_kinds: [lab_documentation, weights_repository, licence_text]
388
+ - id: origin.weights_hosting
389
+ label: Weights hosted in
390
+ subject: model
391
+ value_type: {kind: set, values_from: iso_3166_1_alpha_2}
392
+ definition: >-
393
+ The countries in which the lab itself stores and serves the weights for
394
+ the hosted inference it sells first-party, as ISO 3166-1 alpha-2 codes.
395
+ The empty set when the lab sells no hosted inference of this model.
396
+ Where a third-party provider runs the model is `offering.region`, not this.
397
+ tier: guaranteed
398
+ risk: governance
399
+ permitted_source_kinds: [lab_documentation, provider_terms, provider_documentation]
400
+ - id: origin.base_models
401
+ label: Base models
402
+ subject: model
403
+ value_type: {kind: set, values_from: model_ids}
404
+ definition: >-
405
+ The `lab/model-id` of every model this model's weights were directly
406
+ derived from. The empty set when trained from random initialisation.
407
+ Backs `origin.base_lineage` with the models themselves.
408
+ tier: best_effort
409
+ risk: governance
410
+ permitted_source_kinds: [lab_documentation, weights_repository, licence_text]
411
+
412
+ # ── Model: lifecycle ──────────────────────────────────────────────────────
413
+ - id: model.release_date
414
+ label: Release date
415
+ subject: model
416
+ value_type: {kind: date}
417
+ definition: >-
418
+ The date the lab first made the model generally available to the public,
419
+ by API or download. A preview or waitlist release counts only if anyone
420
+ could sign up.
421
+ tier: guaranteed
422
+ risk: capability
423
+ permitted_source_kinds: [lab_announcement, lab_documentation]
424
+ - id: model.lifecycle
425
+ label: Lifecycle
426
+ subject: model
427
+ value_type: {kind: enum, values: [active, deprecated, retired]}
428
+ value_labels:
429
+ active: Active
430
+ deprecated: Deprecated
431
+ retired: Retired
432
+ definition: >-
433
+ `deprecated` when the lab has announced a retirement date; `retired` when
434
+ the lab no longer serves or supports the model; `active` otherwise.
435
+ Retired models are excluded from decisions unless a spec asks for them.
436
+ tier: guaranteed
437
+ risk: capability
438
+ permitted_source_kinds: [lab_documentation, lab_announcement]
439
+ - id: model.knowledge_cutoff
440
+ label: Knowledge cutoff
441
+ subject: model
442
+ value_type: {kind: date}
443
+ definition: >-
444
+ The latest date of training data the lab documents for the model. A month
445
+ is recorded as its first day, and the fact's qualifiers say so.
446
+ tier: best_effort
447
+ risk: capability
448
+ permitted_source_kinds: [lab_documentation]
449
+ - id: model.deprecation_date
450
+ label: Deprecation date
451
+ subject: model
452
+ value_type: {kind: date}
453
+ definition: >-
454
+ The date the lab has announced the model will stop being served by its own
455
+ API. Unknown until announced; not the date of the announcement.
456
+ tier: best_effort
457
+ risk: capability
458
+ permitted_source_kinds: [lab_documentation, lab_announcement]
459
+
460
+ # ── Model: features ───────────────────────────────────────────────────────
461
+ - id: feature.tool_calling
462
+ label: Tool calling
463
+ subject: model
464
+ value_type: {kind: boolean}
465
+ definition: >-
466
+ True when the lab documents that the model emits structured function or
467
+ tool calls against caller-supplied tool definitions through at least one
468
+ first-party interface.
469
+ tier: guaranteed
470
+ risk: capability
471
+ permitted_source_kinds: [lab_documentation]
472
+ - id: feature.structured_output
473
+ label: Structured output
474
+ subject: model
475
+ value_type: {kind: boolean}
476
+ definition: >-
477
+ True when the lab documents constrained output that conforms to a
478
+ caller-supplied JSON schema. A prompt-only "JSON mode" without a schema
479
+ guarantee is false.
480
+ tier: guaranteed
481
+ risk: capability
482
+ permitted_source_kinds: [lab_documentation]
483
+ - id: feature.effort_controls
484
+ label: Effort controls
485
+ subject: model
486
+ value_type: {kind: boolean}
487
+ definition: >-
488
+ True when the lab documents a request parameter that changes how much
489
+ reasoning the model does (a reasoning effort level or thinking budget).
490
+ tier: guaranteed
491
+ risk: capability
492
+ permitted_source_kinds: [lab_documentation]
493
+ - id: feature.batch
494
+ label: Batch interface
495
+ subject: model
496
+ value_type: {kind: boolean}
497
+ definition: >-
498
+ True when the lab's first-party API offers an asynchronous batch interface
499
+ for this model. Batch pricing is `offering.price.batch_*`.
500
+ tier: guaranteed
501
+ risk: capability
502
+ permitted_source_kinds: [lab_documentation, provider_documentation]
503
+ - id: feature.streaming
504
+ label: Streaming
505
+ subject: model
506
+ value_type: {kind: boolean}
507
+ definition: >-
508
+ True when the lab's first-party API can return the response incrementally
509
+ as it is generated.
510
+ tier: guaranteed
511
+ risk: capability
512
+ permitted_source_kinds: [lab_documentation, provider_documentation]
513
+
514
+ # ── Model: best-effort extras ─────────────────────────────────────────────
515
+ - id: model.languages
516
+ label: Languages
517
+ subject: model
518
+ value_type: {kind: set, values_from: bcp_47}
519
+ definition: >-
520
+ The natural languages the lab documents the model as supporting, as BCP 47
521
+ tags. Absence from the set means undocumented, not unsupported.
522
+ tier: best_effort
523
+ risk: capability
524
+ permitted_source_kinds: [lab_documentation, weights_repository]
525
+ - id: model.fits_hardware
526
+ label: Fits hardware
527
+ subject: model
528
+ value_type: {kind: set, values_from: hardware}
529
+ definition: >-
530
+ The hardware SKUs in `hardware/` on which some published quantisation of
531
+ the model's weights is estimated to load and run. An estimate, always
532
+ labelled as one.
533
+ tier: best_effort
534
+ risk: capability
535
+ permitted_source_kinds: [modelspec_estimate]
536
+
537
+ # ── Offering: identity ────────────────────────────────────────────────────
538
+ - id: offering.provider
539
+ label: Provider
540
+ subject: offering
541
+ value_type: {kind: enum, values_from: "registry:providers"}
542
+ definition: >-
543
+ The provider in `registry/providers.yaml` that sells the offering and
544
+ bills the customer for it.
545
+ tier: guaranteed
546
+ risk: capability
547
+ permitted_source_kinds: [provider_documentation]
548
+ - id: offering.region
549
+ label: Inference region
550
+ subject: offering
551
+ value_type: {kind: set, values_from: iso_3166_1_alpha_2}
552
+ value_labels:
553
+ global: Global
554
+ "global-cross-region": Global, cross-region routing
555
+ "global-short-context": Global, short context
556
+ definition: >-
557
+ The countries in which the provider documents that inference for this
558
+ offering runs, as ISO 3166-1 alpha-2 codes. A global or routed offering
559
+ lists every country it may run in. The provider's own region name is a
560
+ qualifier on the fact.
561
+ tier: guaranteed
562
+ risk: governance
563
+ permitted_source_kinds: [provider_documentation, provider_terms]
564
+ - id: offering.tier
565
+ label: Account tier
566
+ subject: offering
567
+ value_type: {kind: enum, values: [standard, enterprise, zero_retention, government]}
568
+ value_labels:
569
+ enterprise: Enterprise
570
+ government: Government
571
+ standard: Standard
572
+ zero_retention: Zero retention
573
+ definition: >-
574
+ The account tier the offering is sold under. A tier is its own offering
575
+ only when a guaranteed fact differs from the standard tier: price, data
576
+ handling or an attestation (design §4.1).
577
+ tier: guaranteed
578
+ risk: capability
579
+ permitted_source_kinds: [provider_documentation, provider_terms]
580
+
581
+ # ── Offering: price ───────────────────────────────────────────────────────
582
+ - id: offering.price.input
583
+ label: Input price
584
+ subject: offering
585
+ value_type: {kind: number}
586
+ unit: usd_per_1m_tokens
587
+ definition: >-
588
+ The list price of uncached input tokens for a synchronous request. Where
589
+ the price varies with prompt length, the lowest band, with the band as a
590
+ qualifier.
591
+ tier: guaranteed
592
+ risk: capability
593
+ permitted_source_kinds: [provider_documentation]
594
+ - id: offering.price.output
595
+ label: Output price
596
+ subject: offering
597
+ value_type: {kind: number}
598
+ unit: usd_per_1m_tokens
599
+ definition: >-
600
+ The list price of output tokens, including reasoning tokens the provider
601
+ bills as output, for a synchronous request, lowest band as for input.
602
+ tier: guaranteed
603
+ risk: capability
604
+ permitted_source_kinds: [provider_documentation]
605
+ - id: offering.price.cached_input
606
+ label: Cached input price
607
+ subject: offering
608
+ value_type: {kind: number, not_offered: true}
609
+ unit: usd_per_1m_tokens
610
+ definition: >-
611
+ The list price of input tokens read from the provider's prompt cache.
612
+ Cache-write surcharges are a qualifier. `not_offered` when the provider
613
+ has no prompt cache for this offering, which is a known value, not an
614
+ unknown one.
615
+ tier: guaranteed
616
+ risk: capability
617
+ permitted_source_kinds: [provider_documentation]
618
+ - id: offering.price.batch_input
619
+ label: Batch input price
620
+ subject: offering
621
+ value_type: {kind: number, not_offered: true}
622
+ unit: usd_per_1m_tokens
623
+ definition: >-
624
+ The list price of input tokens submitted through the provider's
625
+ asynchronous batch interface. `not_offered` when the provider has no
626
+ batch interface for this offering.
627
+ tier: guaranteed
628
+ risk: capability
629
+ permitted_source_kinds: [provider_documentation]
630
+ - id: offering.price.batch_output
631
+ label: Batch output price
632
+ subject: offering
633
+ value_type: {kind: number, not_offered: true}
634
+ unit: usd_per_1m_tokens
635
+ definition: >-
636
+ The list price of output tokens returned through the provider's
637
+ asynchronous batch interface. `not_offered` when the provider has no
638
+ batch interface for this offering.
639
+ tier: guaranteed
640
+ risk: capability
641
+ permitted_source_kinds: [provider_documentation]
642
+
643
+ - id: offering.cost_per_task
644
+ label: Cost per task
645
+ subject: offering
646
+ value_type: {kind: number}
647
+ unit: usd_per_task
648
+ definition: >-
649
+ What one task costs on this offering at list prices, computed per
650
+ decision from the spec's `task_tokens`: (offering.price.input × input
651
+ tokens + offering.price.output × output tokens) / 1,000,000. Unknown when
652
+ either price is unknown. Never authored on a card.
653
+ tier: guaranteed
654
+ risk: capability
655
+ permitted_source_kinds: [modelspec_estimate]
656
+ computed_by: MODEL-153
657
+
658
+ # ── Offering: speed, limits, SLA ──────────────────────────────────────────
659
+ - id: offering.speed.time_to_first_token
660
+ label: Time to first token
661
+ subject: offering
662
+ value_type: {kind: number}
663
+ unit: milliseconds
664
+ definition: >-
665
+ The median time from sending a request to receiving the first output
666
+ token. Always carries the method: prompt length, output length, effort,
667
+ region of the client, sample size and date.
668
+ tier: best_effort
669
+ risk: capability
670
+ permitted_source_kinds: [modelspec_measurement, independent_evaluator, provider_self_report, outcome_protocol]
671
+ required_qualifiers: [method]
672
+ - id: offering.speed.throughput
673
+ label: Output throughput
674
+ subject: offering
675
+ value_type: {kind: number}
676
+ unit: tokens_per_second
677
+ definition: >-
678
+ The median output throughput of one request after its first token. Always
679
+ carries the method, as for time to first token. Self-hosted throughput is
680
+ a labelled estimate only.
681
+ tier: best_effort
682
+ risk: capability
683
+ permitted_source_kinds: [modelspec_measurement, independent_evaluator, provider_self_report, outcome_protocol]
684
+ required_qualifiers: [method]
685
+ - id: offering.rate_limit.requests
686
+ label: Request rate limit
687
+ subject: offering
688
+ value_type: {kind: number}
689
+ unit: requests_per_minute
690
+ definition: >-
691
+ The documented default request rate limit for a new account at this tier,
692
+ at the lowest paid usage level the provider documents.
693
+ tier: best_effort
694
+ risk: capability
695
+ permitted_source_kinds: [provider_documentation]
696
+ - id: offering.rate_limit.tokens
697
+ label: Token rate limit
698
+ subject: offering
699
+ value_type: {kind: number}
700
+ unit: tokens_per_minute
701
+ definition: >-
702
+ The documented default token rate limit for a new account at this tier,
703
+ at the lowest paid usage level the provider documents.
704
+ tier: best_effort
705
+ risk: capability
706
+ permitted_source_kinds: [provider_documentation]
707
+ - id: offering.sla_uptime
708
+ label: SLA uptime
709
+ subject: offering
710
+ value_type: {kind: number}
711
+ unit: percent
712
+ definition: >-
713
+ The monthly uptime the provider commits to in a published service level
714
+ agreement for this offering, with service credits. A status-page history
715
+ is not an SLA.
716
+ tier: best_effort
717
+ risk: capability
718
+ permitted_source_kinds: [provider_terms]
719
+
720
+ # ── Offering: data handling (governance) ──────────────────────────────────
721
+ - id: offering.data.retention
722
+ label: Data retention
723
+ subject: offering
724
+ value_type: {kind: number, unbounded: true}
725
+ unit: days
726
+ definition: >-
727
+ The longest period the provider's terms say it keeps prompts and outputs
728
+ for this offering by default, for any purpose including abuse monitoring.
729
+ 0 means not stored after the response. `unbounded` when the terms set no
730
+ limit.
731
+ tier: guaranteed
732
+ risk: governance
733
+ permitted_source_kinds: [provider_terms, provider_documentation]
734
+ - id: offering.data.trains_on_customer_data
735
+ label: Trains on customer data
736
+ subject: offering
737
+ value_type: {kind: boolean}
738
+ definition: >-
739
+ True when the provider's default terms for this offering let it use
740
+ customer prompts or outputs to train or improve models. An opt-out the
741
+ customer must take still makes this true.
742
+ tier: guaranteed
743
+ risk: governance
744
+ permitted_source_kinds: [provider_terms]
745
+ - id: offering.data.zero_retention
746
+ label: Zero retention available
747
+ subject: offering
748
+ value_type: {kind: boolean}
749
+ definition: >-
750
+ True when the provider documents that a customer can have prompts and
751
+ outputs for this offering not stored at all, including for abuse
752
+ monitoring, whether by default or on request.
753
+ tier: guaranteed
754
+ risk: governance
755
+ permitted_source_kinds: [provider_terms, provider_documentation]
756
+
757
+ # ── Offering: attestations (governance; documented, never certified) ──────
758
+ - id: offering.attestation.soc2
759
+ label: SOC 2 report
760
+ subject: offering
761
+ value_type: {kind: enum, values: [type_1, type_2, none]}
762
+ value_labels:
763
+ "none": None
764
+ type_1: SOC 2 Type I
765
+ type_2: SOC 2 Type II
766
+ definition: >-
767
+ The SOC 2 report type the provider documents for the service this offering
768
+ runs on, when that report's scope covers the service. `none` when the
769
+ provider documents no such report. Documented, never "compliant".
770
+ tier: guaranteed
771
+ risk: governance
772
+ permitted_source_kinds: [provider_trust_center, provider_documentation]
773
+ - id: offering.attestation.baa
774
+ label: HIPAA BAA
775
+ subject: offering
776
+ value_type: {kind: boolean}
777
+ definition: >-
778
+ True when the provider documents that it will sign a HIPAA business
779
+ associate agreement that covers this offering. Says nothing about whether
780
+ a customer's use is compliant.
781
+ tier: guaranteed
782
+ risk: governance
783
+ permitted_source_kinds: [provider_trust_center, provider_documentation, provider_terms]
784
+ - id: offering.attestation.fedramp
785
+ label: FedRAMP level
786
+ subject: offering
787
+ value_type: {kind: enum, values: [high, moderate, low, li_saas, none]}
788
+ value_labels:
789
+ high: FedRAMP High
790
+ li_saas: FedRAMP Low Impact SaaS
791
+ low: FedRAMP Low
792
+ moderate: FedRAMP Moderate
793
+ "none": None
794
+ definition: >-
795
+ The FedRAMP authorization level of the cloud service offering this
796
+ offering runs in, as the FedRAMP marketplace lists it for this model.
797
+ `none` when not listed.
798
+ tier: best_effort
799
+ risk: governance
800
+ permitted_source_kinds: [official_registry, provider_documentation]
801
+ - id: offering.attestation.iso_27001
802
+ label: ISO/IEC 27001
803
+ subject: offering
804
+ value_type: {kind: boolean}
805
+ definition: >-
806
+ True when the provider documents an ISO/IEC 27001 certificate whose scope
807
+ covers the service this offering runs on.
808
+ tier: best_effort
809
+ risk: governance
810
+ permitted_source_kinds: [provider_trust_center, provider_documentation]
811
+
812
+ # ── Offering: best-effort extras ──────────────────────────────────────────
813
+ - id: offering.fine_tuning
814
+ label: Fine-tuning service
815
+ subject: offering
816
+ value_type: {kind: boolean}
817
+ definition: >-
818
+ True when the provider sells fine-tuning of this model as a service and
819
+ serves the tuned model under this offering's terms.
820
+ tier: best_effort
821
+ risk: capability
822
+ permitted_source_kinds: [provider_documentation]
823
+ - id: offering.private_deployment
824
+ label: Private deployment
825
+ subject: offering
826
+ value_type: {kind: boolean}
827
+ definition: >-
828
+ True when the provider sells dedicated capacity for this model that is
829
+ not shared with other customers (provisioned throughput, a dedicated
830
+ endpoint or deployment in the customer's own cloud account).
831
+ tier: best_effort
832
+ risk: capability
833
+ permitted_source_kinds: [provider_documentation]
834
+ - id: offering.harness_compatibility
835
+ label: Harness compatibility
836
+ subject: offering
837
+ value_type: {kind: set, values_from: "registry:harnesses"}
838
+ definition: >-
839
+ The registered harnesses whose own documentation names this provider as a
840
+ supported backend for this model. Absence means undocumented.
841
+ tier: best_effort
842
+ risk: capability
843
+ permitted_source_kinds: [provider_documentation, lab_documentation]
844
+
845
+ # ── Evidence and estimates ────────────────────────────────────────────────
846
+ - id: evidence.benchmark
847
+ label: Benchmark result
848
+ subject: evidence
849
+ value_type: {kind: number}
850
+ unit: benchmark_metric
851
+ parameter: {name: benchmark, values_from: benchmarks}
852
+ definition: >-
853
+ One measured result on a benchmark version and sub-category, with its
854
+ qualifiers (effort, harness, tools, configuration, measured_by) and
855
+ evidence date. Never copied between models; a missing value is missing.
856
+ tier: best_effort
857
+ risk: capability
858
+ permitted_source_kinds: [benchmark_author, independent_evaluator, provider_self_report, modelspec_measurement]
859
+ required_qualifiers: [measured_by, evidence_date, date_type]
860
+ - id: evidence.outcome
861
+ label: Outcome success rate
862
+ subject: evidence
863
+ value_type: {kind: number}
864
+ unit: percent
865
+ parameter: {name: task_type, values_from: outcome_task_types}
866
+ definition: >-
867
+ The share of outcome records for a task type, offering and harness whose
868
+ objective result was success, with the record count. Published only above
869
+ the protocol's minimum count.
870
+ tier: best_effort
871
+ risk: capability
872
+ permitted_source_kinds: [outcome_protocol]
873
+ required_qualifiers: [harness, record_count]
874
+ - id: estimate.capability
875
+ label: Capability estimate
876
+ subject: model
877
+ value_type: {kind: range}
878
+ unit: capability_scale
879
+ parameter: {name: domain, values_from: "registry:domains"}
880
+ definition: >-
881
+ The capability model's estimate of the model's capability in one domain,
882
+ with its interval, computed from verified evidence only. Constraints never
883
+ add to it. Slice 1 ships no estimates: stage 3 shows evidence per domain,
884
+ unblended.
885
+ tier: guaranteed
886
+ risk: capability
887
+ permitted_source_kinds: [modelspec_estimate]
888
+ computed_by: MODEL-129