modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
registry/templates.yaml
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# Decision templates (MODEL-178). These are partial decision specs: `where`
|
|
2
|
+
# contains Musts, while `weights` contains Prefers. The loader validates every
|
|
3
|
+
# condition, objective, class, and domain against the decision registries.
|
|
4
|
+
|
|
5
|
+
schema_version: 1
|
|
6
|
+
|
|
7
|
+
templates:
|
|
8
|
+
- id: budget-coding
|
|
9
|
+
name: Coding agent on a budget
|
|
10
|
+
purpose: Find an active coding model with enough context while keeping one task under $0.25.
|
|
11
|
+
where:
|
|
12
|
+
- condition: model.class = text-generator
|
|
13
|
+
reason: Coding work needs a model that generates text and code.
|
|
14
|
+
- condition: model.lifecycle = active
|
|
15
|
+
reason: An active model is still supported by its lab or provider.
|
|
16
|
+
- condition: model.context_window >= 200000
|
|
17
|
+
reason: A large repository needs room for code, tests, and instructions.
|
|
18
|
+
- condition: offering.cost_per_task <= 0.25
|
|
19
|
+
reason: The offering must stay within the per-task budget.
|
|
20
|
+
weights:
|
|
21
|
+
software_engineering:
|
|
22
|
+
weight: 0.6
|
|
23
|
+
reason: Prefer stronger evidence on real software-engineering tasks.
|
|
24
|
+
-offering.cost_per_task:
|
|
25
|
+
weight: 0.4
|
|
26
|
+
reason: Among qualifying offerings, prefer the cheaper task.
|
|
27
|
+
needs:
|
|
28
|
+
classes: [text-generator]
|
|
29
|
+
domains: [software_engineering]
|
|
30
|
+
teaches: Use Must for non-negotiable context and budget limits, then Prefer to trade coding strength against cost.
|
|
31
|
+
|
|
32
|
+
- id: private-self-host
|
|
33
|
+
name: Private assistant you host yourself
|
|
34
|
+
purpose: Find a commercially usable open-weights assistant that you can operate yourself.
|
|
35
|
+
where:
|
|
36
|
+
- condition: model.class = text-generator
|
|
37
|
+
reason: A conversational assistant needs a text-generating model.
|
|
38
|
+
- condition: model.weights_openness = open_weights
|
|
39
|
+
reason: Self-hosting requires downloadable weights; fits-your-hardware is coming in MODEL-174.
|
|
40
|
+
- condition: licence.commercial_use in {permitted, permitted_with_conditions}
|
|
41
|
+
reason: The licence must allow commercial use, including use with stated conditions.
|
|
42
|
+
weights:
|
|
43
|
+
chat_preference:
|
|
44
|
+
weight: 0.7
|
|
45
|
+
reason: Prefer assistants whose open-ended responses people rate more highly.
|
|
46
|
+
-offering.cost_per_task:
|
|
47
|
+
weight: 0.3
|
|
48
|
+
reason: Prefer lower task cost when the snapshot has an offering price.
|
|
49
|
+
needs:
|
|
50
|
+
classes: [text-generator]
|
|
51
|
+
domains: [chat_preference]
|
|
52
|
+
teaches: Use Must for deployment and licence requirements, then Prefer to balance assistant quality and cost.
|
|
53
|
+
|
|
54
|
+
- id: regulated-data
|
|
55
|
+
name: Regulated data
|
|
56
|
+
purpose: Find an offering with the data-handling commitments needed for regulated workloads.
|
|
57
|
+
where:
|
|
58
|
+
- condition: offering.data.trains_on_customer_data = false
|
|
59
|
+
reason: Customer prompts and outputs must not be available for model training.
|
|
60
|
+
- condition: offering.data.zero_retention = true
|
|
61
|
+
reason: The provider must offer a mode that stores neither prompts nor outputs.
|
|
62
|
+
- condition: offering.attestation.baa = true
|
|
63
|
+
reason: The provider must offer a HIPAA business associate agreement for the service.
|
|
64
|
+
weights:
|
|
65
|
+
chat_preference:
|
|
66
|
+
weight: 0.6
|
|
67
|
+
reason: Prefer the stronger general assistant among compliant offerings.
|
|
68
|
+
-offering.cost_per_task:
|
|
69
|
+
weight: 0.4
|
|
70
|
+
reason: Prefer the lower task cost after the compliance gates pass.
|
|
71
|
+
needs:
|
|
72
|
+
classes: []
|
|
73
|
+
domains: [chat_preference]
|
|
74
|
+
teaches: Governance promises are Musts because a high score cannot compensate for a missing data commitment.
|
|
75
|
+
|
|
76
|
+
- id: maths
|
|
77
|
+
name: Maths and proofs
|
|
78
|
+
purpose: Find a text model with the strongest evidence on mathematical problems and proofs.
|
|
79
|
+
where:
|
|
80
|
+
- condition: model.class = text-generator
|
|
81
|
+
reason: The answer must be generated as mathematical text or a proof.
|
|
82
|
+
weights:
|
|
83
|
+
maths:
|
|
84
|
+
weight: 1.0
|
|
85
|
+
reason: Rank only by the maths capability estimate.
|
|
86
|
+
needs:
|
|
87
|
+
classes: [text-generator]
|
|
88
|
+
domains: [maths]
|
|
89
|
+
teaches: Use one Must to select the right model class and one Prefer to state what best means.
|
|
90
|
+
|
|
91
|
+
- id: retrieval-embeddings
|
|
92
|
+
name: Retrieval embeddings
|
|
93
|
+
purpose: Find an embedding model that ranks relevant documents well for retrieval.
|
|
94
|
+
where:
|
|
95
|
+
- condition: model.class = vectoriser
|
|
96
|
+
reason: Retrieval embeddings require the model class that emits vectors.
|
|
97
|
+
weights:
|
|
98
|
+
retrieval:
|
|
99
|
+
weight: 1.0
|
|
100
|
+
reason: Rank only by the retrieval capability estimate.
|
|
101
|
+
needs:
|
|
102
|
+
classes: [vectoriser]
|
|
103
|
+
domains: [retrieval]
|
|
104
|
+
teaches: Model class is a Must, while measured retrieval quality is a Prefer.
|
|
105
|
+
|
|
106
|
+
- id: high-volume
|
|
107
|
+
name: High volume, good enough
|
|
108
|
+
purpose: Find a low-cost text model for many short assistant requests without ignoring response quality.
|
|
109
|
+
where:
|
|
110
|
+
- condition: model.class = text-generator
|
|
111
|
+
reason: The workload needs generated text responses.
|
|
112
|
+
weights:
|
|
113
|
+
-offering.cost_per_task:
|
|
114
|
+
weight: 0.7
|
|
115
|
+
reason: At high volume, small differences in task cost dominate total spend.
|
|
116
|
+
chat_preference:
|
|
117
|
+
weight: 0.3
|
|
118
|
+
reason: Keep a quality signal so cheap but poor responses do not win by cost alone.
|
|
119
|
+
task_tokens: {input: 2000, output: 500}
|
|
120
|
+
needs:
|
|
121
|
+
classes: [text-generator]
|
|
122
|
+
domains: [chat_preference]
|
|
123
|
+
teaches: Keep the model class as a Must, then make cost the stronger Prefer without turning quality into a gate.
|
|
124
|
+
|
|
125
|
+
- id: long-documents
|
|
126
|
+
name: Long documents
|
|
127
|
+
purpose: Find a text model that can accept million-token documents and produce strong written analysis.
|
|
128
|
+
where:
|
|
129
|
+
- condition: model.class = text-generator
|
|
130
|
+
reason: The result must be generated as text.
|
|
131
|
+
- condition: model.context_window >= 1000000
|
|
132
|
+
reason: The model must accept a million-token document in one request.
|
|
133
|
+
weights:
|
|
134
|
+
writing:
|
|
135
|
+
weight: 0.5
|
|
136
|
+
reason: Prefer clear, high-quality long-form writing.
|
|
137
|
+
reasoning:
|
|
138
|
+
weight: 0.3
|
|
139
|
+
reason: Prefer stronger multi-step analysis of the document.
|
|
140
|
+
-offering.cost_per_task:
|
|
141
|
+
weight: 0.2
|
|
142
|
+
reason: Use cost as a smaller tie-breaker after writing and reasoning quality.
|
|
143
|
+
needs:
|
|
144
|
+
classes: [text-generator]
|
|
145
|
+
domains: [writing, reasoning]
|
|
146
|
+
teaches: Context capacity is a Must; writing, reasoning, and cost remain trade-offs expressed as Prefers.
|
|
147
|
+
|
|
148
|
+
- id: eu-data
|
|
149
|
+
name: EU-only data handling
|
|
150
|
+
purpose: Find an offering that runs inference in an EU member state and does not train on customer data.
|
|
151
|
+
where:
|
|
152
|
+
- condition: offering.region in {AT, BE, BG, HR, CY, CZ, DE, DK, EE, ES, FI, FR, GR, HU, IE, IT, LT, LU, LV, MT, NL, PL, PT, RO, SE, SI, SK}
|
|
153
|
+
reason: Inference must run in one of the EU's 27 member-state ISO country codes (EU Publications Office Interinstitutional Style Guide, https://style-guide.europa.eu/en/content/-/isg/topic?identifier=annex-a5-list-countries-territories-currencies, checked 2026-09-27).
|
|
154
|
+
- condition: offering.data.trains_on_customer_data = false
|
|
155
|
+
reason: Customer prompts and outputs must not be available for model training.
|
|
156
|
+
weights:
|
|
157
|
+
chat_preference:
|
|
158
|
+
weight: 0.6
|
|
159
|
+
reason: Prefer the stronger general assistant among offerings that meet the data rules.
|
|
160
|
+
-offering.cost_per_task:
|
|
161
|
+
weight: 0.4
|
|
162
|
+
reason: Prefer lower task cost after the data-handling gates pass.
|
|
163
|
+
needs:
|
|
164
|
+
classes: []
|
|
165
|
+
domains: [chat_preference]
|
|
166
|
+
teaches: Residency and training policy are Musts; quality and cost only rank offerings that pass them.
|
schema/__init__.py
ADDED
|
File without changes
|
schema/applicability.py
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""Which questions a model's *class* can answer at all (MODEL-97).
|
|
2
|
+
|
|
3
|
+
A `null` in this schema has always meant "not yet researched" — somebody should
|
|
4
|
+
go and look. For a catalogue of one class that reading is nearly always right.
|
|
5
|
+
For a catalogue of several it is sometimes a lie: a model that emits no text has
|
|
6
|
+
no maximum output-token count to research, and publishing that `null` with the
|
|
7
|
+
same meaning as an unresearched one asserts a gap that can never be closed.
|
|
8
|
+
|
|
9
|
+
**Applicability is derived from `model_type`, never written on a card.** There
|
|
10
|
+
is no per-card list to fall out of date, nothing to round-trip through
|
|
11
|
+
`ModelCard.to_yaml()`, and no way for one card to disagree with its own class.
|
|
12
|
+
The table here is reviewed once instead of 1,339 times, and
|
|
13
|
+
`tests/test_class_and_null_semantics.py` fails if a path in it stops naming a
|
|
14
|
+
real schema field.
|
|
15
|
+
|
|
16
|
+
Two rules the table obeys, and the second is the important one:
|
|
17
|
+
|
|
18
|
+
1. A rule names the classes for which a field **is** applicable. A card whose
|
|
19
|
+
class is not among them has nothing to research there.
|
|
20
|
+
2. **A rule may depend only on the class.** Not on another card field. The
|
|
21
|
+
tempting rule — "hardware profiles are inapplicable when `open_weights` is
|
|
22
|
+
false" — is refused, because `licensing.open_weights` is `bool = False` with
|
|
23
|
+
no null state: an unresearched card is indistinguishable from a genuinely
|
|
24
|
+
closed one, so the rule would manufacture "there is nothing to know" across
|
|
25
|
+
the catalogue. That is MODEL-77's `data_residency: []` mistake repainted.
|
|
26
|
+
|
|
27
|
+
What this file deliberately does **not** call inapplicable: architecture.
|
|
28
|
+
A closed model still *has* layers and parameters; nobody has published them.
|
|
29
|
+
That is unknown, possibly unobtainable, and much nearer `withheld` than
|
|
30
|
+
"meaningless". Calling it inapplicable would assert there is nothing to know
|
|
31
|
+
and quietly excuse the catalogue from ever asking.
|
|
32
|
+
|
|
33
|
+
The section-level counterpart of this table is `__applicable_model_types__` on
|
|
34
|
+
the nested models in `schema/card.py`, which predates it. `schema.card
|
|
35
|
+
.inapplicable_paths` evaluates both and is the single entry point consumers use.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
from __future__ import annotations
|
|
39
|
+
|
|
40
|
+
from dataclasses import dataclass
|
|
41
|
+
|
|
42
|
+
from .enums import ModelType
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def model_types(*keys: str) -> frozenset[ModelType]:
|
|
46
|
+
"""Resolve ModelType members by exact value or hyphen-prefix (``'llm-'``)."""
|
|
47
|
+
found: set[ModelType] = set()
|
|
48
|
+
for key in keys:
|
|
49
|
+
if key.endswith("-"):
|
|
50
|
+
matched = [t for t in ModelType if t.value.startswith(key)]
|
|
51
|
+
if not matched:
|
|
52
|
+
raise ValueError(f"no ModelType starts with {key!r}")
|
|
53
|
+
found.update(matched)
|
|
54
|
+
else:
|
|
55
|
+
found.add(ModelType(key))
|
|
56
|
+
return frozenset(found)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
#: Classes that emit text tokens, so an output-token limit, a token stream and
|
|
60
|
+
#: an output tok/s are questions they can answer.
|
|
61
|
+
#:
|
|
62
|
+
#: Same membership as `pipeline.hardware.TOKEN_GENERATING_MODEL_TYPES`, which
|
|
63
|
+
#: answers the neighbouring question "does it decode tokens, so is a tok/s
|
|
64
|
+
#: prediction honest" (MODEL-53). Two names because they are two questions;
|
|
65
|
+
#: a test holds them equal so neither drifts, and that test is where a future
|
|
66
|
+
#: divergence gets argued rather than discovered.
|
|
67
|
+
TEXT_WRITING_MODEL_TYPES: frozenset[ModelType] = model_types(
|
|
68
|
+
"llm-", "vlm", "medical", "legal", "financial", "agent-model", "router",
|
|
69
|
+
"adapter", "quantized-variant", "distilled", "merged",
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
#: Classes that take text *in*, so an input-token limit and a context window
|
|
73
|
+
#: are questions they can answer. A decision model reads supplied state, so it
|
|
74
|
+
#: is here; it writes no text, so it is not above.
|
|
75
|
+
TEXT_READING_MODEL_TYPES: frozenset[ModelType] = TEXT_WRITING_MODEL_TYPES | model_types(
|
|
76
|
+
"text-encoder", "document-ocr", "reranker", "embedding-", "decision-model",
|
|
77
|
+
"reward-model", "safety-classifier",
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass(frozen=True)
|
|
82
|
+
class FieldRule:
|
|
83
|
+
"""One field, the classes it applies to, and why it applies to no others."""
|
|
84
|
+
|
|
85
|
+
#: Dotted path from the `ModelCard` root. Held to a real field by a test.
|
|
86
|
+
path: str
|
|
87
|
+
#: What a reader is told the page cannot show.
|
|
88
|
+
label: str
|
|
89
|
+
#: The classes for which this field is a real question.
|
|
90
|
+
applies_to: frozenset[ModelType]
|
|
91
|
+
#: The sentence a page or a reviewer needs. Not published per field; it is
|
|
92
|
+
#: the reason the rule survives review.
|
|
93
|
+
because: str
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
FIELD_RULES: tuple[FieldRule, ...] = (
|
|
97
|
+
FieldRule("modalities.text.max_output_tokens", "Max output tokens",
|
|
98
|
+
TEXT_WRITING_MODEL_TYPES, "this class emits no text"),
|
|
99
|
+
FieldRule("modalities.text.streaming", "Token streaming",
|
|
100
|
+
TEXT_WRITING_MODEL_TYPES, "this class emits no token stream"),
|
|
101
|
+
FieldRule("modalities.text.fill_in_middle", "Fill-in-the-middle",
|
|
102
|
+
TEXT_WRITING_MODEL_TYPES, "this class emits no text"),
|
|
103
|
+
# Added writing the first `decision-model` card (MODEL-101). "JSON mode" is
|
|
104
|
+
# a constraint on *generated text*: a toggle that makes a model's prose come
|
|
105
|
+
# back as valid JSON. A class that emits no prose has no such mode — and for
|
|
106
|
+
# a decision model the wrong reading is the dangerous one, since its answers
|
|
107
|
+
# are always typed, so a blank here invites "it cannot do structured output"
|
|
108
|
+
# exactly backwards. Not a claim that it lacks structure: a claim that the
|
|
109
|
+
# question is about text this class never writes.
|
|
110
|
+
FieldRule("modalities.text.json_mode", "JSON output mode",
|
|
111
|
+
TEXT_WRITING_MODEL_TYPES, "this class emits no text to constrain"),
|
|
112
|
+
FieldRule("inference_performance.api_tps_output", "API output tok/s",
|
|
113
|
+
TEXT_WRITING_MODEL_TYPES, "this class has no output token stream"),
|
|
114
|
+
FieldRule("modalities.text.max_input_tokens", "Max input tokens",
|
|
115
|
+
TEXT_READING_MODEL_TYPES, "this class takes no text input"),
|
|
116
|
+
FieldRule("modalities.text.context_window", "Context window",
|
|
117
|
+
TEXT_READING_MODEL_TYPES, "this class takes no text input"),
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
#: Human labels for the *sections* `__applicable_model_types__` gates. Used
|
|
121
|
+
#: when a page or a report has to name a subtree rather than a field.
|
|
122
|
+
SECTION_LABELS: dict[str, str] = {
|
|
123
|
+
"capabilities": "Generative capabilities",
|
|
124
|
+
"modalities.text": "Text modality",
|
|
125
|
+
"modalities.vision": "Vision",
|
|
126
|
+
"modalities.audio": "Audio",
|
|
127
|
+
"modalities.video": "Video",
|
|
128
|
+
"modalities.document": "Documents",
|
|
129
|
+
"modalities.image_generation": "Image generation",
|
|
130
|
+
"modalities.embeddings": "Embeddings",
|
|
131
|
+
"modalities.reranking": "Reranking",
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
_FIELD_LABELS: dict[str, str] = {rule.path: rule.label for rule in FIELD_RULES}
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def label_for(path: str) -> str:
|
|
138
|
+
"""A reader's name for a dotted path; the path itself when none is set."""
|
|
139
|
+
return _FIELD_LABELS.get(path) or SECTION_LABELS.get(path) or path
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def reason_for(path: str) -> str | None:
|
|
143
|
+
"""Why this path does not apply, for the rules that state one."""
|
|
144
|
+
for rule in FIELD_RULES:
|
|
145
|
+
if rule.path == path:
|
|
146
|
+
return rule.because
|
|
147
|
+
return None
|
schema/benchmark.py
ADDED
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
"""BenchmarkCard: the schema for one benchmark wiki page (benchmarks/<id>.md front matter).
|
|
2
|
+
|
|
3
|
+
The template IS the schema, as with model cards: every field here is a front-matter key.
|
|
4
|
+
Unknown = empty/None, never a guess. `models_covered` is derived from the model cards at build time
|
|
5
|
+
and must not be authored.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
from typing import Literal
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
|
13
|
+
|
|
14
|
+
from schema.enums import BenchmarkCategory
|
|
15
|
+
|
|
16
|
+
ID_RE = re.compile(r"^[a-z0-9][a-z0-9_]{1,80}$")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class Metric(BaseModel):
|
|
20
|
+
name: str = "" # e.g. "accuracy", "pass@1", "% resolved", "Elo"
|
|
21
|
+
direction: Literal["higher_is_better", "lower_is_better"] = "higher_is_better"
|
|
22
|
+
unit: str = "" # "%", "points", "Elo"
|
|
23
|
+
max_score: float | None = None
|
|
24
|
+
random_baseline: float | None = None
|
|
25
|
+
human_baseline: float | None = None
|
|
26
|
+
baseline_note: str = ""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Dataset(BaseModel):
|
|
30
|
+
size: int | None = None # number of items/tasks/questions
|
|
31
|
+
size_note: str = "" # "500 tasks", "57 subjects x ~14k questions"
|
|
32
|
+
url: str = ""
|
|
33
|
+
license: str = "" # SPDX id or the licence name as published
|
|
34
|
+
languages: list[str] = Field(default_factory=list)
|
|
35
|
+
modalities: list[str] = Field(default_factory=list) # text, code, image, video, audio
|
|
36
|
+
splits: str = ""
|
|
37
|
+
public_test_set: bool | None = None # False when answers are held out
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class Publisher(BaseModel):
|
|
41
|
+
org: str = ""
|
|
42
|
+
authors: list[str] = Field(default_factory=list)
|
|
43
|
+
url: str = ""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class Paper(BaseModel):
|
|
47
|
+
title: str = ""
|
|
48
|
+
arxiv: str = "" # e.g. "2310.06770"
|
|
49
|
+
url: str = ""
|
|
50
|
+
year: int | None = None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class Lineage(BaseModel):
|
|
54
|
+
family: str = "" # id of the family page, e.g. "mmlu", "swe_bench"
|
|
55
|
+
predecessor: str = "" # id
|
|
56
|
+
successors: list[str] = Field(default_factory=list) # ids
|
|
57
|
+
variants: list[str] = Field(default_factory=list) # ids of subsets / language / category variants
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class Saturation(BaseModel):
|
|
61
|
+
status: Literal["open", "watch", "saturated", "unknown"] = "unknown"
|
|
62
|
+
top_score: float | None = None
|
|
63
|
+
as_of: str = "" # YYYY-MM
|
|
64
|
+
note: str = ""
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class Contamination(BaseModel):
|
|
68
|
+
risk: Literal["low", "medium", "high", "unknown"] = "unknown"
|
|
69
|
+
note: str = ""
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class Harness(BaseModel):
|
|
73
|
+
lm_eval: str = "" # lm-evaluation-harness task name
|
|
74
|
+
inspect_evals: str = "" # inspect_evals id
|
|
75
|
+
helm: str = "" # HELM scenario
|
|
76
|
+
opencompass: str = "" # OpenCompass dataset
|
|
77
|
+
bigbench: str = "" # BIG-bench task
|
|
78
|
+
other: str = ""
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class Source(BaseModel):
|
|
82
|
+
url: str
|
|
83
|
+
title: str = ""
|
|
84
|
+
accessed: str = "" # YYYY-MM-DD
|
|
85
|
+
|
|
86
|
+
@field_validator("url")
|
|
87
|
+
@classmethod
|
|
88
|
+
def _http(cls, v: str) -> str:
|
|
89
|
+
if not v.startswith(("http://", "https://")):
|
|
90
|
+
raise ValueError("source url must be http(s)")
|
|
91
|
+
return v
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class DomainTag(BaseModel):
|
|
95
|
+
"""One domain a benchmark measures, and how directly (MODEL-133).
|
|
96
|
+
|
|
97
|
+
`id` names a domain in `registry/domains.yaml`; `tests/test_decision_registry.py`
|
|
98
|
+
checks every tag against it. `direct`: the benchmark's tasks are instances of
|
|
99
|
+
the domain. `proxy`: correlated with it, a slice of it, or preference rather
|
|
100
|
+
than correctness. An absent tag means "not tagged", never "no domain".
|
|
101
|
+
"""
|
|
102
|
+
|
|
103
|
+
model_config = ConfigDict(extra="forbid")
|
|
104
|
+
|
|
105
|
+
id: str
|
|
106
|
+
directness: Literal["direct", "proxy"]
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class Freshness(BaseModel):
|
|
110
|
+
researched: str = "" # YYYY-MM-DD
|
|
111
|
+
researched_by: str = "" # "sonnet-5 agent, batch 1"
|
|
112
|
+
reviewed: str = ""
|
|
113
|
+
reviewed_by: str = ""
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class BenchmarkCard(BaseModel):
|
|
117
|
+
id: str
|
|
118
|
+
name: str
|
|
119
|
+
aliases: list[str] = Field(default_factory=list)
|
|
120
|
+
page_kind: Literal["benchmark", "family", "subset"] = "benchmark"
|
|
121
|
+
category: BenchmarkCategory
|
|
122
|
+
subcategory: str = ""
|
|
123
|
+
status: Literal["active", "saturated", "deprecated", "superseded", "proposed", "unknown"] = "unknown"
|
|
124
|
+
summary: str = "" # one sentence, <= 200 chars
|
|
125
|
+
measures: str = "" # one paragraph
|
|
126
|
+
task_format: str = ""
|
|
127
|
+
metric: Metric = Metric()
|
|
128
|
+
dataset: Dataset = Dataset()
|
|
129
|
+
publisher: Publisher = Publisher()
|
|
130
|
+
paper: Paper = Paper()
|
|
131
|
+
leaderboard_url: str = ""
|
|
132
|
+
repo_url: str = ""
|
|
133
|
+
released: str = "" # YYYY or YYYY-MM
|
|
134
|
+
last_updated: str = ""
|
|
135
|
+
lineage: Lineage = Lineage()
|
|
136
|
+
saturation: Saturation = Saturation()
|
|
137
|
+
contamination: Contamination = Contamination()
|
|
138
|
+
harness: Harness = Harness()
|
|
139
|
+
tags: list[str] = Field(default_factory=list)
|
|
140
|
+
#: Optional and additive: pages without it still load (MODEL-133).
|
|
141
|
+
domains: list[DomainTag] = Field(default_factory=list)
|
|
142
|
+
sources: list[Source] = Field(default_factory=list)
|
|
143
|
+
freshness: Freshness = Freshness()
|
|
144
|
+
|
|
145
|
+
@field_validator("id")
|
|
146
|
+
@classmethod
|
|
147
|
+
def _id(cls, v: str) -> str:
|
|
148
|
+
if not ID_RE.match(v):
|
|
149
|
+
raise ValueError("id must be snake_case: lowercase letters, digits, underscores")
|
|
150
|
+
return v
|
|
151
|
+
|
|
152
|
+
@field_validator("domains")
|
|
153
|
+
@classmethod
|
|
154
|
+
def _one_tag_per_domain(cls, v: list[DomainTag]) -> list[DomainTag]:
|
|
155
|
+
ids = [tag.id for tag in v]
|
|
156
|
+
repeated = sorted({i for i in ids if ids.count(i) > 1})
|
|
157
|
+
if repeated:
|
|
158
|
+
raise ValueError(f"a domain may be tagged once per page: {', '.join(repeated)}")
|
|
159
|
+
return v
|
|
160
|
+
|
|
161
|
+
@field_validator("summary")
|
|
162
|
+
@classmethod
|
|
163
|
+
def _summary(cls, v: str) -> str:
|
|
164
|
+
if len(v) > 220:
|
|
165
|
+
raise ValueError("summary must be one sentence, at most 220 characters")
|
|
166
|
+
return v
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
REQUIRED_SECTIONS = {
|
|
170
|
+
"benchmark": ["What it measures", "How it is scored", "Dataset and licence", "Who publishes it", "Lineage",
|
|
171
|
+
"Saturation and contamination", "How to run it", "Reading the numbers"],
|
|
172
|
+
"family": ["What it measures", "How it is scored", "Dataset and licence", "Who publishes it", "Lineage",
|
|
173
|
+
"Saturation and contamination", "How to run it", "Reading the numbers"],
|
|
174
|
+
"subset": ["What it measures", "Reading the numbers"],
|
|
175
|
+
}
|