modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
@@ -0,0 +1,166 @@
1
+ # Decision templates (MODEL-178). These are partial decision specs: `where`
2
+ # contains Musts, while `weights` contains Prefers. The loader validates every
3
+ # condition, objective, class, and domain against the decision registries.
4
+
5
+ schema_version: 1
6
+
7
+ templates:
8
+ - id: budget-coding
9
+ name: Coding agent on a budget
10
+ purpose: Find an active coding model with enough context while keeping one task under $0.25.
11
+ where:
12
+ - condition: model.class = text-generator
13
+ reason: Coding work needs a model that generates text and code.
14
+ - condition: model.lifecycle = active
15
+ reason: An active model is still supported by its lab or provider.
16
+ - condition: model.context_window >= 200000
17
+ reason: A large repository needs room for code, tests, and instructions.
18
+ - condition: offering.cost_per_task <= 0.25
19
+ reason: The offering must stay within the per-task budget.
20
+ weights:
21
+ software_engineering:
22
+ weight: 0.6
23
+ reason: Prefer stronger evidence on real software-engineering tasks.
24
+ -offering.cost_per_task:
25
+ weight: 0.4
26
+ reason: Among qualifying offerings, prefer the cheaper task.
27
+ needs:
28
+ classes: [text-generator]
29
+ domains: [software_engineering]
30
+ teaches: Use Must for non-negotiable context and budget limits, then Prefer to trade coding strength against cost.
31
+
32
+ - id: private-self-host
33
+ name: Private assistant you host yourself
34
+ purpose: Find a commercially usable open-weights assistant that you can operate yourself.
35
+ where:
36
+ - condition: model.class = text-generator
37
+ reason: A conversational assistant needs a text-generating model.
38
+ - condition: model.weights_openness = open_weights
39
+ reason: Self-hosting requires downloadable weights; fits-your-hardware is coming in MODEL-174.
40
+ - condition: licence.commercial_use in {permitted, permitted_with_conditions}
41
+ reason: The licence must allow commercial use, including use with stated conditions.
42
+ weights:
43
+ chat_preference:
44
+ weight: 0.7
45
+ reason: Prefer assistants whose open-ended responses people rate more highly.
46
+ -offering.cost_per_task:
47
+ weight: 0.3
48
+ reason: Prefer lower task cost when the snapshot has an offering price.
49
+ needs:
50
+ classes: [text-generator]
51
+ domains: [chat_preference]
52
+ teaches: Use Must for deployment and licence requirements, then Prefer to balance assistant quality and cost.
53
+
54
+ - id: regulated-data
55
+ name: Regulated data
56
+ purpose: Find an offering with the data-handling commitments needed for regulated workloads.
57
+ where:
58
+ - condition: offering.data.trains_on_customer_data = false
59
+ reason: Customer prompts and outputs must not be available for model training.
60
+ - condition: offering.data.zero_retention = true
61
+ reason: The provider must offer a mode that stores neither prompts nor outputs.
62
+ - condition: offering.attestation.baa = true
63
+ reason: The provider must offer a HIPAA business associate agreement for the service.
64
+ weights:
65
+ chat_preference:
66
+ weight: 0.6
67
+ reason: Prefer the stronger general assistant among compliant offerings.
68
+ -offering.cost_per_task:
69
+ weight: 0.4
70
+ reason: Prefer the lower task cost after the compliance gates pass.
71
+ needs:
72
+ classes: []
73
+ domains: [chat_preference]
74
+ teaches: Governance promises are Musts because a high score cannot compensate for a missing data commitment.
75
+
76
+ - id: maths
77
+ name: Maths and proofs
78
+ purpose: Find a text model with the strongest evidence on mathematical problems and proofs.
79
+ where:
80
+ - condition: model.class = text-generator
81
+ reason: The answer must be generated as mathematical text or a proof.
82
+ weights:
83
+ maths:
84
+ weight: 1.0
85
+ reason: Rank only by the maths capability estimate.
86
+ needs:
87
+ classes: [text-generator]
88
+ domains: [maths]
89
+ teaches: Use one Must to select the right model class and one Prefer to state what best means.
90
+
91
+ - id: retrieval-embeddings
92
+ name: Retrieval embeddings
93
+ purpose: Find an embedding model that ranks relevant documents well for retrieval.
94
+ where:
95
+ - condition: model.class = vectoriser
96
+ reason: Retrieval embeddings require the model class that emits vectors.
97
+ weights:
98
+ retrieval:
99
+ weight: 1.0
100
+ reason: Rank only by the retrieval capability estimate.
101
+ needs:
102
+ classes: [vectoriser]
103
+ domains: [retrieval]
104
+ teaches: Model class is a Must, while measured retrieval quality is a Prefer.
105
+
106
+ - id: high-volume
107
+ name: High volume, good enough
108
+ purpose: Find a low-cost text model for many short assistant requests without ignoring response quality.
109
+ where:
110
+ - condition: model.class = text-generator
111
+ reason: The workload needs generated text responses.
112
+ weights:
113
+ -offering.cost_per_task:
114
+ weight: 0.7
115
+ reason: At high volume, small differences in task cost dominate total spend.
116
+ chat_preference:
117
+ weight: 0.3
118
+ reason: Keep a quality signal so cheap but poor responses do not win by cost alone.
119
+ task_tokens: {input: 2000, output: 500}
120
+ needs:
121
+ classes: [text-generator]
122
+ domains: [chat_preference]
123
+ teaches: Keep the model class as a Must, then make cost the stronger Prefer without turning quality into a gate.
124
+
125
+ - id: long-documents
126
+ name: Long documents
127
+ purpose: Find a text model that can accept million-token documents and produce strong written analysis.
128
+ where:
129
+ - condition: model.class = text-generator
130
+ reason: The result must be generated as text.
131
+ - condition: model.context_window >= 1000000
132
+ reason: The model must accept a million-token document in one request.
133
+ weights:
134
+ writing:
135
+ weight: 0.5
136
+ reason: Prefer clear, high-quality long-form writing.
137
+ reasoning:
138
+ weight: 0.3
139
+ reason: Prefer stronger multi-step analysis of the document.
140
+ -offering.cost_per_task:
141
+ weight: 0.2
142
+ reason: Use cost as a smaller tie-breaker after writing and reasoning quality.
143
+ needs:
144
+ classes: [text-generator]
145
+ domains: [writing, reasoning]
146
+ teaches: Context capacity is a Must; writing, reasoning, and cost remain trade-offs expressed as Prefers.
147
+
148
+ - id: eu-data
149
+ name: EU-only data handling
150
+ purpose: Find an offering that runs inference in an EU member state and does not train on customer data.
151
+ where:
152
+ - condition: offering.region in {AT, BE, BG, HR, CY, CZ, DE, DK, EE, ES, FI, FR, GR, HU, IE, IT, LT, LU, LV, MT, NL, PL, PT, RO, SE, SI, SK}
153
+ reason: Inference must run in one of the EU's 27 member-state ISO country codes (EU Publications Office Interinstitutional Style Guide, https://style-guide.europa.eu/en/content/-/isg/topic?identifier=annex-a5-list-countries-territories-currencies, checked 2026-09-27).
154
+ - condition: offering.data.trains_on_customer_data = false
155
+ reason: Customer prompts and outputs must not be available for model training.
156
+ weights:
157
+ chat_preference:
158
+ weight: 0.6
159
+ reason: Prefer the stronger general assistant among offerings that meet the data rules.
160
+ -offering.cost_per_task:
161
+ weight: 0.4
162
+ reason: Prefer lower task cost after the data-handling gates pass.
163
+ needs:
164
+ classes: []
165
+ domains: [chat_preference]
166
+ teaches: Residency and training policy are Musts; quality and cost only rank offerings that pass them.
schema/__init__.py ADDED
File without changes
@@ -0,0 +1,147 @@
1
+ """Which questions a model's *class* can answer at all (MODEL-97).
2
+
3
+ A `null` in this schema has always meant "not yet researched" — somebody should
4
+ go and look. For a catalogue of one class that reading is nearly always right.
5
+ For a catalogue of several it is sometimes a lie: a model that emits no text has
6
+ no maximum output-token count to research, and publishing that `null` with the
7
+ same meaning as an unresearched one asserts a gap that can never be closed.
8
+
9
+ **Applicability is derived from `model_type`, never written on a card.** There
10
+ is no per-card list to fall out of date, nothing to round-trip through
11
+ `ModelCard.to_yaml()`, and no way for one card to disagree with its own class.
12
+ The table here is reviewed once instead of 1,339 times, and
13
+ `tests/test_class_and_null_semantics.py` fails if a path in it stops naming a
14
+ real schema field.
15
+
16
+ Two rules the table obeys, and the second is the important one:
17
+
18
+ 1. A rule names the classes for which a field **is** applicable. A card whose
19
+ class is not among them has nothing to research there.
20
+ 2. **A rule may depend only on the class.** Not on another card field. The
21
+ tempting rule — "hardware profiles are inapplicable when `open_weights` is
22
+ false" — is refused, because `licensing.open_weights` is `bool = False` with
23
+ no null state: an unresearched card is indistinguishable from a genuinely
24
+ closed one, so the rule would manufacture "there is nothing to know" across
25
+ the catalogue. That is MODEL-77's `data_residency: []` mistake repainted.
26
+
27
+ What this file deliberately does **not** call inapplicable: architecture.
28
+ A closed model still *has* layers and parameters; nobody has published them.
29
+ That is unknown, possibly unobtainable, and much nearer `withheld` than
30
+ "meaningless". Calling it inapplicable would assert there is nothing to know
31
+ and quietly excuse the catalogue from ever asking.
32
+
33
+ The section-level counterpart of this table is `__applicable_model_types__` on
34
+ the nested models in `schema/card.py`, which predates it. `schema.card
35
+ .inapplicable_paths` evaluates both and is the single entry point consumers use.
36
+ """
37
+
38
+ from __future__ import annotations
39
+
40
+ from dataclasses import dataclass
41
+
42
+ from .enums import ModelType
43
+
44
+
45
+ def model_types(*keys: str) -> frozenset[ModelType]:
46
+ """Resolve ModelType members by exact value or hyphen-prefix (``'llm-'``)."""
47
+ found: set[ModelType] = set()
48
+ for key in keys:
49
+ if key.endswith("-"):
50
+ matched = [t for t in ModelType if t.value.startswith(key)]
51
+ if not matched:
52
+ raise ValueError(f"no ModelType starts with {key!r}")
53
+ found.update(matched)
54
+ else:
55
+ found.add(ModelType(key))
56
+ return frozenset(found)
57
+
58
+
59
+ #: Classes that emit text tokens, so an output-token limit, a token stream and
60
+ #: an output tok/s are questions they can answer.
61
+ #:
62
+ #: Same membership as `pipeline.hardware.TOKEN_GENERATING_MODEL_TYPES`, which
63
+ #: answers the neighbouring question "does it decode tokens, so is a tok/s
64
+ #: prediction honest" (MODEL-53). Two names because they are two questions;
65
+ #: a test holds them equal so neither drifts, and that test is where a future
66
+ #: divergence gets argued rather than discovered.
67
+ TEXT_WRITING_MODEL_TYPES: frozenset[ModelType] = model_types(
68
+ "llm-", "vlm", "medical", "legal", "financial", "agent-model", "router",
69
+ "adapter", "quantized-variant", "distilled", "merged",
70
+ )
71
+
72
+ #: Classes that take text *in*, so an input-token limit and a context window
73
+ #: are questions they can answer. A decision model reads supplied state, so it
74
+ #: is here; it writes no text, so it is not above.
75
+ TEXT_READING_MODEL_TYPES: frozenset[ModelType] = TEXT_WRITING_MODEL_TYPES | model_types(
76
+ "text-encoder", "document-ocr", "reranker", "embedding-", "decision-model",
77
+ "reward-model", "safety-classifier",
78
+ )
79
+
80
+
81
+ @dataclass(frozen=True)
82
+ class FieldRule:
83
+ """One field, the classes it applies to, and why it applies to no others."""
84
+
85
+ #: Dotted path from the `ModelCard` root. Held to a real field by a test.
86
+ path: str
87
+ #: What a reader is told the page cannot show.
88
+ label: str
89
+ #: The classes for which this field is a real question.
90
+ applies_to: frozenset[ModelType]
91
+ #: The sentence a page or a reviewer needs. Not published per field; it is
92
+ #: the reason the rule survives review.
93
+ because: str
94
+
95
+
96
+ FIELD_RULES: tuple[FieldRule, ...] = (
97
+ FieldRule("modalities.text.max_output_tokens", "Max output tokens",
98
+ TEXT_WRITING_MODEL_TYPES, "this class emits no text"),
99
+ FieldRule("modalities.text.streaming", "Token streaming",
100
+ TEXT_WRITING_MODEL_TYPES, "this class emits no token stream"),
101
+ FieldRule("modalities.text.fill_in_middle", "Fill-in-the-middle",
102
+ TEXT_WRITING_MODEL_TYPES, "this class emits no text"),
103
+ # Added writing the first `decision-model` card (MODEL-101). "JSON mode" is
104
+ # a constraint on *generated text*: a toggle that makes a model's prose come
105
+ # back as valid JSON. A class that emits no prose has no such mode — and for
106
+ # a decision model the wrong reading is the dangerous one, since its answers
107
+ # are always typed, so a blank here invites "it cannot do structured output"
108
+ # exactly backwards. Not a claim that it lacks structure: a claim that the
109
+ # question is about text this class never writes.
110
+ FieldRule("modalities.text.json_mode", "JSON output mode",
111
+ TEXT_WRITING_MODEL_TYPES, "this class emits no text to constrain"),
112
+ FieldRule("inference_performance.api_tps_output", "API output tok/s",
113
+ TEXT_WRITING_MODEL_TYPES, "this class has no output token stream"),
114
+ FieldRule("modalities.text.max_input_tokens", "Max input tokens",
115
+ TEXT_READING_MODEL_TYPES, "this class takes no text input"),
116
+ FieldRule("modalities.text.context_window", "Context window",
117
+ TEXT_READING_MODEL_TYPES, "this class takes no text input"),
118
+ )
119
+
120
+ #: Human labels for the *sections* `__applicable_model_types__` gates. Used
121
+ #: when a page or a report has to name a subtree rather than a field.
122
+ SECTION_LABELS: dict[str, str] = {
123
+ "capabilities": "Generative capabilities",
124
+ "modalities.text": "Text modality",
125
+ "modalities.vision": "Vision",
126
+ "modalities.audio": "Audio",
127
+ "modalities.video": "Video",
128
+ "modalities.document": "Documents",
129
+ "modalities.image_generation": "Image generation",
130
+ "modalities.embeddings": "Embeddings",
131
+ "modalities.reranking": "Reranking",
132
+ }
133
+
134
+ _FIELD_LABELS: dict[str, str] = {rule.path: rule.label for rule in FIELD_RULES}
135
+
136
+
137
+ def label_for(path: str) -> str:
138
+ """A reader's name for a dotted path; the path itself when none is set."""
139
+ return _FIELD_LABELS.get(path) or SECTION_LABELS.get(path) or path
140
+
141
+
142
+ def reason_for(path: str) -> str | None:
143
+ """Why this path does not apply, for the rules that state one."""
144
+ for rule in FIELD_RULES:
145
+ if rule.path == path:
146
+ return rule.because
147
+ return None
schema/benchmark.py ADDED
@@ -0,0 +1,175 @@
1
+ """BenchmarkCard: the schema for one benchmark wiki page (benchmarks/<id>.md front matter).
2
+
3
+ The template IS the schema, as with model cards: every field here is a front-matter key.
4
+ Unknown = empty/None, never a guess. `models_covered` is derived from the model cards at build time
5
+ and must not be authored.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import re
10
+ from typing import Literal
11
+
12
+ from pydantic import BaseModel, ConfigDict, Field, field_validator
13
+
14
+ from schema.enums import BenchmarkCategory
15
+
16
+ ID_RE = re.compile(r"^[a-z0-9][a-z0-9_]{1,80}$")
17
+
18
+
19
+ class Metric(BaseModel):
20
+ name: str = "" # e.g. "accuracy", "pass@1", "% resolved", "Elo"
21
+ direction: Literal["higher_is_better", "lower_is_better"] = "higher_is_better"
22
+ unit: str = "" # "%", "points", "Elo"
23
+ max_score: float | None = None
24
+ random_baseline: float | None = None
25
+ human_baseline: float | None = None
26
+ baseline_note: str = ""
27
+
28
+
29
+ class Dataset(BaseModel):
30
+ size: int | None = None # number of items/tasks/questions
31
+ size_note: str = "" # "500 tasks", "57 subjects x ~14k questions"
32
+ url: str = ""
33
+ license: str = "" # SPDX id or the licence name as published
34
+ languages: list[str] = Field(default_factory=list)
35
+ modalities: list[str] = Field(default_factory=list) # text, code, image, video, audio
36
+ splits: str = ""
37
+ public_test_set: bool | None = None # False when answers are held out
38
+
39
+
40
+ class Publisher(BaseModel):
41
+ org: str = ""
42
+ authors: list[str] = Field(default_factory=list)
43
+ url: str = ""
44
+
45
+
46
+ class Paper(BaseModel):
47
+ title: str = ""
48
+ arxiv: str = "" # e.g. "2310.06770"
49
+ url: str = ""
50
+ year: int | None = None
51
+
52
+
53
+ class Lineage(BaseModel):
54
+ family: str = "" # id of the family page, e.g. "mmlu", "swe_bench"
55
+ predecessor: str = "" # id
56
+ successors: list[str] = Field(default_factory=list) # ids
57
+ variants: list[str] = Field(default_factory=list) # ids of subsets / language / category variants
58
+
59
+
60
+ class Saturation(BaseModel):
61
+ status: Literal["open", "watch", "saturated", "unknown"] = "unknown"
62
+ top_score: float | None = None
63
+ as_of: str = "" # YYYY-MM
64
+ note: str = ""
65
+
66
+
67
+ class Contamination(BaseModel):
68
+ risk: Literal["low", "medium", "high", "unknown"] = "unknown"
69
+ note: str = ""
70
+
71
+
72
+ class Harness(BaseModel):
73
+ lm_eval: str = "" # lm-evaluation-harness task name
74
+ inspect_evals: str = "" # inspect_evals id
75
+ helm: str = "" # HELM scenario
76
+ opencompass: str = "" # OpenCompass dataset
77
+ bigbench: str = "" # BIG-bench task
78
+ other: str = ""
79
+
80
+
81
+ class Source(BaseModel):
82
+ url: str
83
+ title: str = ""
84
+ accessed: str = "" # YYYY-MM-DD
85
+
86
+ @field_validator("url")
87
+ @classmethod
88
+ def _http(cls, v: str) -> str:
89
+ if not v.startswith(("http://", "https://")):
90
+ raise ValueError("source url must be http(s)")
91
+ return v
92
+
93
+
94
+ class DomainTag(BaseModel):
95
+ """One domain a benchmark measures, and how directly (MODEL-133).
96
+
97
+ `id` names a domain in `registry/domains.yaml`; `tests/test_decision_registry.py`
98
+ checks every tag against it. `direct`: the benchmark's tasks are instances of
99
+ the domain. `proxy`: correlated with it, a slice of it, or preference rather
100
+ than correctness. An absent tag means "not tagged", never "no domain".
101
+ """
102
+
103
+ model_config = ConfigDict(extra="forbid")
104
+
105
+ id: str
106
+ directness: Literal["direct", "proxy"]
107
+
108
+
109
+ class Freshness(BaseModel):
110
+ researched: str = "" # YYYY-MM-DD
111
+ researched_by: str = "" # "sonnet-5 agent, batch 1"
112
+ reviewed: str = ""
113
+ reviewed_by: str = ""
114
+
115
+
116
+ class BenchmarkCard(BaseModel):
117
+ id: str
118
+ name: str
119
+ aliases: list[str] = Field(default_factory=list)
120
+ page_kind: Literal["benchmark", "family", "subset"] = "benchmark"
121
+ category: BenchmarkCategory
122
+ subcategory: str = ""
123
+ status: Literal["active", "saturated", "deprecated", "superseded", "proposed", "unknown"] = "unknown"
124
+ summary: str = "" # one sentence, <= 200 chars
125
+ measures: str = "" # one paragraph
126
+ task_format: str = ""
127
+ metric: Metric = Metric()
128
+ dataset: Dataset = Dataset()
129
+ publisher: Publisher = Publisher()
130
+ paper: Paper = Paper()
131
+ leaderboard_url: str = ""
132
+ repo_url: str = ""
133
+ released: str = "" # YYYY or YYYY-MM
134
+ last_updated: str = ""
135
+ lineage: Lineage = Lineage()
136
+ saturation: Saturation = Saturation()
137
+ contamination: Contamination = Contamination()
138
+ harness: Harness = Harness()
139
+ tags: list[str] = Field(default_factory=list)
140
+ #: Optional and additive: pages without it still load (MODEL-133).
141
+ domains: list[DomainTag] = Field(default_factory=list)
142
+ sources: list[Source] = Field(default_factory=list)
143
+ freshness: Freshness = Freshness()
144
+
145
+ @field_validator("id")
146
+ @classmethod
147
+ def _id(cls, v: str) -> str:
148
+ if not ID_RE.match(v):
149
+ raise ValueError("id must be snake_case: lowercase letters, digits, underscores")
150
+ return v
151
+
152
+ @field_validator("domains")
153
+ @classmethod
154
+ def _one_tag_per_domain(cls, v: list[DomainTag]) -> list[DomainTag]:
155
+ ids = [tag.id for tag in v]
156
+ repeated = sorted({i for i in ids if ids.count(i) > 1})
157
+ if repeated:
158
+ raise ValueError(f"a domain may be tagged once per page: {', '.join(repeated)}")
159
+ return v
160
+
161
+ @field_validator("summary")
162
+ @classmethod
163
+ def _summary(cls, v: str) -> str:
164
+ if len(v) > 220:
165
+ raise ValueError("summary must be one sentence, at most 220 characters")
166
+ return v
167
+
168
+
169
+ REQUIRED_SECTIONS = {
170
+ "benchmark": ["What it measures", "How it is scored", "Dataset and licence", "Who publishes it", "Lineage",
171
+ "Saturation and contamination", "How to run it", "Reading the numbers"],
172
+ "family": ["What it measures", "How it is scored", "Dataset and licence", "Who publishes it", "Lineage",
173
+ "Saturation and contamination", "How to run it", "Reading the numbers"],
174
+ "subset": ["What it measures", "Reading the numbers"],
175
+ }