modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
api/classes.py
ADDED
|
@@ -0,0 +1,557 @@
|
|
|
1
|
+
"""What class of model a problem needs — the taxonomy (MODEL-100).
|
|
2
|
+
|
|
3
|
+
`ModelType` is a good publishing field and a bad selection axis. Its 35 values
|
|
4
|
+
sit on four axes at once: modality (`image-generation`), role (`reward-model`,
|
|
5
|
+
`router`), lineage (`distilled`, `merged`, `quantized-variant`) and domain
|
|
6
|
+
(`medical`, `legal`). A builder cannot choose on that — `quantized-variant` and
|
|
7
|
+
`llm-chat` are not alternatives, one is a provenance relation to the other.
|
|
8
|
+
|
|
9
|
+
So this module is a **view over `ModelType`, never a replacement for it**. The
|
|
10
|
+
enum stays exactly as published; MODEL-98 has just paid a major version for it.
|
|
11
|
+
Nothing here is written on a card, so nothing can drift per card — the same
|
|
12
|
+
call `schema/applicability.py` made one commit ago, for the same reason.
|
|
13
|
+
|
|
14
|
+
The axis is **what a model consumes, what it emits, and what decision it makes
|
|
15
|
+
on the caller's behalf**. `decides` does the real work, because it is the facet
|
|
16
|
+
that changes the caller's own code: prose a person reads, a permutation of
|
|
17
|
+
items you supplied, a label from a taxonomy the model defines, a value from a
|
|
18
|
+
set *you* defined, a preference between two outputs, or an action taken.
|
|
19
|
+
|
|
20
|
+
A class is **derived**, not hand-listed: `CLASS_BY_PAIR[(emits, decides)]`.
|
|
21
|
+
That pair is injective, checked at import and again by a test, so the mapping
|
|
22
|
+
is a function a sceptic can evaluate rather than a matter of taste.
|
|
23
|
+
|
|
24
|
+
Two deliberate properties:
|
|
25
|
+
|
|
26
|
+
* **Keyed on the enum's string values**, never on `ModelType` itself. The
|
|
27
|
+
module therefore imports nothing the Cloudflare Worker's Python bundle does
|
|
28
|
+
not already carry — `api/ranking/engine.py` is in it and `schema/` is not —
|
|
29
|
+
so it can be vendored unchanged the day `POST /v1/class-fit` is built
|
|
30
|
+
(`api/worker/vendor.py`). `tests/test_class_fit.py` holds the key set equal
|
|
31
|
+
to `{t.value for t in ModelType}`, so a new enum member is a red test on the
|
|
32
|
+
commit that adds it — which is where the argument about its class belongs.
|
|
33
|
+
* **`api/ranking/engine.py` never imports this module.** The dependency runs
|
|
34
|
+
one way only, and a test enforces it. Cost-to-correct (MODEL-99) is fitness
|
|
35
|
+
evidence for a *task*; `rank_score` is a within-class quality composite. If
|
|
36
|
+
the scorer could see this package, a measurement of one task could lift a
|
|
37
|
+
model in a ranking of another, invisibly.
|
|
38
|
+
|
|
39
|
+
The reasoning is `docs/design/class-selection.md`.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
from __future__ import annotations
|
|
43
|
+
|
|
44
|
+
from dataclasses import dataclass, field
|
|
45
|
+
from typing import Any
|
|
46
|
+
|
|
47
|
+
# `neutrality_commitment` is *called*, never transcribed, so the published
|
|
48
|
+
# terms keep their single source and `tests/test_legal.py` stays the one place
|
|
49
|
+
# prose and JSON are held together. `USE_CASE_PROFILES` is read to derive which
|
|
50
|
+
# ranking profiles cover a class, rather than hand-listing them here. This is
|
|
51
|
+
# the only import from the repository, and it points at a module the Worker
|
|
52
|
+
# bundle already carries.
|
|
53
|
+
from api.ranking.engine import USE_CASE_PROFILES, neutrality_commitment
|
|
54
|
+
|
|
55
|
+
# ── the vocabularies, published ─────────────────────────────────────────────
|
|
56
|
+
|
|
57
|
+
#: Input kinds a task can have. Not modalities: `structured_state` is the
|
|
58
|
+
#: typed state a decision model is handed, and `item_list` is the candidate
|
|
59
|
+
#: set a reranker orders.
|
|
60
|
+
CONSUMES: tuple[str, ...] = (
|
|
61
|
+
"text", "image", "audio", "video", "page_image", "structured_state",
|
|
62
|
+
"numeric_series", "model_output", "item_list", "observations",
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
#: What comes out. `open_text` is generated language a person or a parser
|
|
66
|
+
#: reads; `choice` is a value from a set the caller defined; `label` is a value
|
|
67
|
+
#: from a taxonomy the model defines. Those three are different products.
|
|
68
|
+
EMITS: tuple[str, ...] = (
|
|
69
|
+
"open_text", "media", "transcript", "vector", "ordering", "label",
|
|
70
|
+
"choice", "preference_score", "action", "series", "world_state",
|
|
71
|
+
"annotation",
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
#: The decision the model makes on the caller's behalf — the facet that decides
|
|
75
|
+
#: what the caller's code looks like afterwards.
|
|
76
|
+
DECIDES: tuple[str, ...] = (
|
|
77
|
+
"nothing", "what_order", "which_of_a_fixed_set",
|
|
78
|
+
"which_of_a_caller_defined_set", "how_good", "what_to_do_next",
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
#: Every refusal the rule can produce. Published, and `api/class_fit.py` reads
|
|
82
|
+
#: this tuple rather than repeating it.
|
|
83
|
+
REFUSAL_CODES: tuple[str, ...] = (
|
|
84
|
+
"empty_request", "no_term_matched", "unknown_facet_value",
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
#: Why a class is not a candidate. A closed set: an exclusion without a reason
|
|
88
|
+
#: is a guess wearing a verdict's clothes.
|
|
89
|
+
EXCLUDED_REASONS: tuple[str, ...] = (
|
|
90
|
+
"emits_wrong_kind", "consumes_unsupported", "decision_shape_mismatch",
|
|
91
|
+
# The task description used none of this class's terms. Only ever reached
|
|
92
|
+
# when the caller supplied no facets of their own: a term is a hint and may
|
|
93
|
+
# discover a class, but it may never remove one a constraint admits.
|
|
94
|
+
"not_described",
|
|
95
|
+
"not_a_class",
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
@dataclass(frozen=True)
|
|
100
|
+
class Adaptation:
|
|
101
|
+
"""One published way an input or an output can be made to fit.
|
|
102
|
+
|
|
103
|
+
Adaptations are the reason a text generator can answer a decision question
|
|
104
|
+
at all, and they are deliberately few: each one is a claim that a caller
|
|
105
|
+
can bridge two facet values in practice, and each one here is a claim
|
|
106
|
+
MODEL-99 actually made and measured.
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
from_: str
|
|
110
|
+
to: str
|
|
111
|
+
how: str
|
|
112
|
+
|
|
113
|
+
def to_json(self) -> dict[str, str]:
|
|
114
|
+
return {"from": self.from_, "to": self.to, "how": self.how}
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
#: Inputs that can be made to fit. Exactly one today.
|
|
118
|
+
CONSUMES_ADAPTATIONS: tuple[Adaptation, ...] = (
|
|
119
|
+
Adaptation(
|
|
120
|
+
"structured_state", "text",
|
|
121
|
+
"serialised as JSON — MODEL-99 handed the LLM arms the decision "
|
|
122
|
+
"model's own state and questions verbatim",
|
|
123
|
+
),
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
#: Outputs that can be made to fit. Exactly one today, and it carries its own
|
|
127
|
+
#: caveat: MODEL-99 scored a truncated or unparseable reply as a *wrong*
|
|
128
|
+
#: answer, not a missing one, because dropping them would flatter the arm that
|
|
129
|
+
#: produces them.
|
|
130
|
+
EMITS_ADAPTATIONS: tuple[Adaptation, ...] = (
|
|
131
|
+
Adaptation(
|
|
132
|
+
"open_text", "choice",
|
|
133
|
+
"parsed out of generated text; a malformed, truncated or refused "
|
|
134
|
+
"answer is a wrong answer, not a missing one",
|
|
135
|
+
),
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
@dataclass(frozen=True)
|
|
140
|
+
class ModelClass:
|
|
141
|
+
"""One class: its facets, the `ModelType` values that derive to it, and why."""
|
|
142
|
+
|
|
143
|
+
id: str
|
|
144
|
+
consumes: tuple[str, ...]
|
|
145
|
+
emits: str
|
|
146
|
+
decides: str
|
|
147
|
+
#: Whether this class has an "I cannot tell" output state. The precondition
|
|
148
|
+
#: for a cascade (see `COMPOSITION_RULE`), not a quality judgement.
|
|
149
|
+
abstains: bool
|
|
150
|
+
#: Why this `(emits, decides)` pair is its own class, in the words a
|
|
151
|
+
#: reviewer needs. Not published per class to sell it; published so it can
|
|
152
|
+
#: be argued with.
|
|
153
|
+
because: str
|
|
154
|
+
#: Words a task description might use. The weakest part of this design, and
|
|
155
|
+
#: the part published loudest for exactly that reason.
|
|
156
|
+
terms: tuple[str, ...]
|
|
157
|
+
#: The `ModelType` values that derive to this class.
|
|
158
|
+
model_types: tuple[str, ...]
|
|
159
|
+
#: A domain narrows the pool *inside* a class; it never names one.
|
|
160
|
+
domains: dict[str, str] = field(default_factory=dict)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
CLASSES: tuple[ModelClass, ...] = (
|
|
164
|
+
ModelClass(
|
|
165
|
+
id="actor",
|
|
166
|
+
consumes=("text", "image", "observations"),
|
|
167
|
+
emits="action", decides="what_to_do_next", abstains=False,
|
|
168
|
+
because=(
|
|
169
|
+
"it acts. Everything else hands the caller material or a verdict "
|
|
170
|
+
"and stops; this one changes the world, so the caller's code is an "
|
|
171
|
+
"executor rather than a reader"
|
|
172
|
+
),
|
|
173
|
+
terms=("agent", "agentic", "act", "action", "tool", "tools", "browse",
|
|
174
|
+
"automate", "robot", "robotics", "control", "execute"),
|
|
175
|
+
model_types=("agent-model", "robotics"),
|
|
176
|
+
),
|
|
177
|
+
ModelClass(
|
|
178
|
+
id="analyser",
|
|
179
|
+
consumes=("text", "image", "video"),
|
|
180
|
+
emits="annotation", decides="which_of_a_fixed_set", abstains=False,
|
|
181
|
+
because=(
|
|
182
|
+
"it labels the *parts* of its input — tokens, spans, regions, "
|
|
183
|
+
"depth — rather than the whole of it, so the caller gets a "
|
|
184
|
+
"structure over the input, not a verdict about it"
|
|
185
|
+
),
|
|
186
|
+
terms=("detect", "detection", "segment", "segmentation", "bounding box",
|
|
187
|
+
"depth", "named entity", "token classification", "annotate",
|
|
188
|
+
"keypoint", "fill mask"),
|
|
189
|
+
model_types=("vision-encoder", "text-encoder"),
|
|
190
|
+
),
|
|
191
|
+
ModelClass(
|
|
192
|
+
id="decider",
|
|
193
|
+
consumes=("text", "structured_state"),
|
|
194
|
+
emits="choice", decides="which_of_a_caller_defined_set", abstains=True,
|
|
195
|
+
because=(
|
|
196
|
+
"the answer set is defined by the caller at request time. That is "
|
|
197
|
+
"the sentence `schema/enums.py` already writes for `decision-model`"
|
|
198
|
+
" — a safety classifier has a fixed harm taxonomy, these labels do "
|
|
199
|
+
"not — promoted from a docstring to the axis"
|
|
200
|
+
),
|
|
201
|
+
terms=("decide", "decision", "classify", "classification", "choose",
|
|
202
|
+
"route", "routing", "triage", "judge", "judgement", "judgment",
|
|
203
|
+
"verdict", "determine", "attribute", "attribution", "calibrated",
|
|
204
|
+
"probability", "abstain"),
|
|
205
|
+
model_types=("decision-model", "router"),
|
|
206
|
+
),
|
|
207
|
+
ModelClass(
|
|
208
|
+
id="forecaster",
|
|
209
|
+
consumes=("numeric_series",),
|
|
210
|
+
emits="series", decides="nothing", abstains=False,
|
|
211
|
+
because="no tokens go in and none come out; the unit is a number over time",
|
|
212
|
+
terms=("forecast", "forecasting", "time series", "timeseries",
|
|
213
|
+
"seasonality", "demand", "horizon"),
|
|
214
|
+
model_types=("time-series",),
|
|
215
|
+
),
|
|
216
|
+
ModelClass(
|
|
217
|
+
id="labeller",
|
|
218
|
+
consumes=("text", "image"),
|
|
219
|
+
emits="label", decides="which_of_a_fixed_set", abstains=True,
|
|
220
|
+
because=(
|
|
221
|
+
"the taxonomy belongs to the model, not the caller. You get the "
|
|
222
|
+
"categories it was trained on and no others, which is a different "
|
|
223
|
+
"product from one that answers your questions"
|
|
224
|
+
),
|
|
225
|
+
terms=("moderate", "moderation", "safety", "toxicity", "toxic",
|
|
226
|
+
"harmful", "abuse", "spam", "guardrail", "classify",
|
|
227
|
+
"classification", "flag"),
|
|
228
|
+
model_types=("safety-classifier",),
|
|
229
|
+
),
|
|
230
|
+
ModelClass(
|
|
231
|
+
id="media-generator",
|
|
232
|
+
consumes=("text", "image", "audio"),
|
|
233
|
+
emits="media", decides="nothing", abstains=False,
|
|
234
|
+
because="the artefact is pixels or samples, so nothing downstream parses it",
|
|
235
|
+
terms=("image", "picture", "illustration", "render", "photo", "video",
|
|
236
|
+
"speech", "voice", "music", "synthesise", "synthesize", "inpaint"),
|
|
237
|
+
model_types=("image-generation", "image-editing", "video-generation",
|
|
238
|
+
"audio-tts", "audio-music"),
|
|
239
|
+
),
|
|
240
|
+
ModelClass(
|
|
241
|
+
id="orderer",
|
|
242
|
+
consumes=("text", "item_list"),
|
|
243
|
+
emits="ordering", decides="what_order", abstains=False,
|
|
244
|
+
because=(
|
|
245
|
+
"it is handed the candidates and returns a permutation of them. It "
|
|
246
|
+
"invents nothing and chooses nothing; it only sorts"
|
|
247
|
+
),
|
|
248
|
+
terms=("rerank", "reranking", "reorder", "relevance", "shortlist",
|
|
249
|
+
"ordering", "sort"),
|
|
250
|
+
model_types=("reranker",),
|
|
251
|
+
),
|
|
252
|
+
ModelClass(
|
|
253
|
+
id="scorer",
|
|
254
|
+
consumes=("text", "model_output"),
|
|
255
|
+
emits="preference_score", decides="how_good", abstains=False,
|
|
256
|
+
because=(
|
|
257
|
+
"its subject is another model's output, for training or selection. "
|
|
258
|
+
"It answers 'how good is this answer', never 'what is the answer'"
|
|
259
|
+
),
|
|
260
|
+
terms=("reward", "preference", "rlhf", "best of n", "rate the output",
|
|
261
|
+
"score the output"),
|
|
262
|
+
model_types=("reward-model",),
|
|
263
|
+
),
|
|
264
|
+
ModelClass(
|
|
265
|
+
id="simulator",
|
|
266
|
+
consumes=("observations",),
|
|
267
|
+
emits="world_state", decides="nothing", abstains=False,
|
|
268
|
+
because="it predicts what happens next in an environment, not what to say about it",
|
|
269
|
+
terms=("simulate", "simulation", "world model", "dynamics", "rollout",
|
|
270
|
+
"environment"),
|
|
271
|
+
model_types=("world-model",),
|
|
272
|
+
),
|
|
273
|
+
ModelClass(
|
|
274
|
+
id="text-generator",
|
|
275
|
+
consumes=("text", "image", "audio"),
|
|
276
|
+
emits="open_text", decides="nothing", abstains=False,
|
|
277
|
+
because=(
|
|
278
|
+
"it produces open-ended language and decides nothing: the caller "
|
|
279
|
+
"reads it, or parses it and decides. This is the class every other "
|
|
280
|
+
"catalogue already ranks, and the one MODEL-100 exists to stop "
|
|
281
|
+
"assuming"
|
|
282
|
+
),
|
|
283
|
+
terms=("write", "writing", "summarise", "summarize", "summary", "draft",
|
|
284
|
+
"explain", "chat", "conversation", "prose", "essay", "reply",
|
|
285
|
+
"translate", "rewrite", "compose", "reason", "code", "coding"),
|
|
286
|
+
model_types=("llm-chat", "llm-reasoning", "llm-code", "llm-base", "vlm",
|
|
287
|
+
"medical", "legal", "financial", "audio-realtime"),
|
|
288
|
+
domains={"medical": "medical", "legal": "legal", "financial": "financial"},
|
|
289
|
+
),
|
|
290
|
+
ModelClass(
|
|
291
|
+
id="transcriber",
|
|
292
|
+
consumes=("audio", "page_image"),
|
|
293
|
+
emits="transcript", decides="nothing", abstains=False,
|
|
294
|
+
because=(
|
|
295
|
+
"the output is a faithful rendering of the input in another form. "
|
|
296
|
+
"A generator may add or reorganise; this one may not, and that is "
|
|
297
|
+
"the whole difference to the caller"
|
|
298
|
+
),
|
|
299
|
+
terms=("transcribe", "transcript", "transcription", "ocr", "dictation",
|
|
300
|
+
"subtitle", "caption", "speech to text"),
|
|
301
|
+
model_types=("audio-asr", "document-ocr"),
|
|
302
|
+
),
|
|
303
|
+
ModelClass(
|
|
304
|
+
id="vectoriser",
|
|
305
|
+
consumes=("text", "image"),
|
|
306
|
+
emits="vector", decides="nothing", abstains=False,
|
|
307
|
+
because=(
|
|
308
|
+
"the output is a reusable representation the caller stores and "
|
|
309
|
+
"compares, not an answer about any one input"
|
|
310
|
+
),
|
|
311
|
+
terms=("embed", "embedding", "embeddings", "vector", "similarity",
|
|
312
|
+
"semantic search", "nearest neighbour", "nearest neighbor",
|
|
313
|
+
"retrieval", "index"),
|
|
314
|
+
model_types=("embedding-text", "embedding-multimodal", "embedding-code"),
|
|
315
|
+
),
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
#: `ModelType` values that name **no class**, and why.
|
|
320
|
+
#:
|
|
321
|
+
#: Lineage answers "where did these weights come from", not "what does this
|
|
322
|
+
#: do". A distilled model is a smaller instance of its base model's class, not
|
|
323
|
+
#: an alternative to it. `miscellaneous` disqualifies itself by its own
|
|
324
|
+
#: definition — "for models no specific type describes honestly".
|
|
325
|
+
NON_CLASSES: dict[str, tuple[str, ...]] = {
|
|
326
|
+
"derived": ("adapter", "quantized-variant", "distilled", "merged"),
|
|
327
|
+
"unclassified": ("miscellaneous",),
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
#: What a caller does instead. Named, because "we cannot place this" without a
|
|
331
|
+
#: next step is a dead end rather than a refusal.
|
|
332
|
+
NON_CLASS_RESOLUTION: dict[str, str] = {
|
|
333
|
+
"derived": (
|
|
334
|
+
"this value records a lineage relation, not a capability. Follow "
|
|
335
|
+
"`lineage.base_model` on the card and take that model's class."
|
|
336
|
+
),
|
|
337
|
+
"unclassified": (
|
|
338
|
+
"`miscellaneous` is the catalogue's own admission that no type "
|
|
339
|
+
"describes the model honestly. There is nothing to derive from it."
|
|
340
|
+
),
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def _build_class_by_pair() -> dict[tuple[str, str], str]:
|
|
345
|
+
"""`(emits, decides) -> class id`, and prove it is a function."""
|
|
346
|
+
table: dict[tuple[str, str], str] = {}
|
|
347
|
+
for model_class in CLASSES:
|
|
348
|
+
key = (model_class.emits, model_class.decides)
|
|
349
|
+
if key in table:
|
|
350
|
+
raise ValueError(
|
|
351
|
+
f"{model_class.id} and {table[key]} share the pair {key}: the "
|
|
352
|
+
"class derivation would stop being a function"
|
|
353
|
+
)
|
|
354
|
+
table[key] = model_class.id
|
|
355
|
+
return table
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
#: The derivation, printed. A class is looked up from its facets; it is never
|
|
359
|
+
#: assigned by hand.
|
|
360
|
+
CLASS_BY_PAIR: dict[tuple[str, str], str] = _build_class_by_pair()
|
|
361
|
+
|
|
362
|
+
CLASS_BY_ID: dict[str, ModelClass] = {c.id: c for c in CLASSES}
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def _build_placement() -> dict[str, str]:
|
|
366
|
+
placed: dict[str, str] = {}
|
|
367
|
+
for model_class in CLASSES:
|
|
368
|
+
for value in model_class.model_types:
|
|
369
|
+
placed[value] = model_class.id
|
|
370
|
+
for non_class, values in NON_CLASSES.items():
|
|
371
|
+
for value in values:
|
|
372
|
+
placed[value] = non_class
|
|
373
|
+
return placed
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
#: `model_type` value -> class id, or one of the `NON_CLASSES` keys.
|
|
377
|
+
MODEL_TYPE_PLACEMENT: dict[str, str] = _build_placement()
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def class_for_model_type(value: str | None) -> str | None:
|
|
381
|
+
"""The class id for a `model_type`, or None when it names no class."""
|
|
382
|
+
placed = MODEL_TYPE_PLACEMENT.get(value or "")
|
|
383
|
+
return placed if placed in CLASS_BY_ID else None
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def non_class_for_model_type(value: str | None) -> str | None:
|
|
387
|
+
"""`derived`, `unclassified`, or None when the value does name a class."""
|
|
388
|
+
placed = MODEL_TYPE_PLACEMENT.get(value or "")
|
|
389
|
+
return placed if placed in NON_CLASSES else None
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
def rank_profiles_for(class_id: str) -> list[str]:
|
|
393
|
+
"""Ranking profiles that cover a class, **derived** from `preferred_types`.
|
|
394
|
+
|
|
395
|
+
Not a hand list, so it cannot drift from the profiles. For `decider` it is
|
|
396
|
+
empty today, which is the true and useful statement that ranking stops at
|
|
397
|
+
that class boundary.
|
|
398
|
+
"""
|
|
399
|
+
model_class = CLASS_BY_ID.get(class_id)
|
|
400
|
+
if model_class is None:
|
|
401
|
+
return []
|
|
402
|
+
wanted = set(model_class.model_types)
|
|
403
|
+
return sorted(
|
|
404
|
+
key for key, profile in USE_CASE_PROFILES.items()
|
|
405
|
+
if wanted & set(profile.get("preferred_types") or ())
|
|
406
|
+
)
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
# ── the questions a `partial` answer hands back ─────────────────────────────
|
|
410
|
+
#
|
|
411
|
+
# Published, not generated. When two classes both survive, the facet they
|
|
412
|
+
# differ on is the thing the caller has to settle, and phrasing it is the
|
|
413
|
+
# difference between "I cannot tell you" and work somebody can do.
|
|
414
|
+
|
|
415
|
+
@dataclass(frozen=True)
|
|
416
|
+
class DistinguishingQuestion:
|
|
417
|
+
between: tuple[str, str]
|
|
418
|
+
ask: str
|
|
419
|
+
|
|
420
|
+
def to_json(self) -> dict[str, Any]:
|
|
421
|
+
return {"between": list(self.between), "ask": self.ask}
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
DISTINGUISHING_QUESTIONS: tuple[DistinguishingQuestion, ...] = (
|
|
425
|
+
DistinguishingQuestion(
|
|
426
|
+
("choice", "open_text"),
|
|
427
|
+
"Does your code branch on the answer, or does a person read it? A "
|
|
428
|
+
"value your code switches on wants a decider; prose a person reads "
|
|
429
|
+
"wants a text-generator. If you would parse the prose into a value, "
|
|
430
|
+
"you want a decider and are considering paying an LLM to be one.",
|
|
431
|
+
),
|
|
432
|
+
DistinguishingQuestion(
|
|
433
|
+
("choice", "label"),
|
|
434
|
+
"Is the set of answers fixed by the model, or defined by you per "
|
|
435
|
+
"request? A fixed taxonomy — harm categories, sentiment — is a "
|
|
436
|
+
"labeller. Your own options, changing per call, need a decider.",
|
|
437
|
+
),
|
|
438
|
+
DistinguishingQuestion(
|
|
439
|
+
("ordering", "vector"),
|
|
440
|
+
"Do you need a reusable representation you store and compare later, "
|
|
441
|
+
"or one ordering of items you already hold? Storing it is a "
|
|
442
|
+
"vectoriser; ordering it now is an orderer. Retrieval usually wants "
|
|
443
|
+
"both, in that order.",
|
|
444
|
+
),
|
|
445
|
+
DistinguishingQuestion(
|
|
446
|
+
("action", "open_text"),
|
|
447
|
+
"Does the model hand your code text to act on, or take the action "
|
|
448
|
+
"itself? If your code owns the loop you want a text-generator and a "
|
|
449
|
+
"harness, not an actor.",
|
|
450
|
+
),
|
|
451
|
+
DistinguishingQuestion(
|
|
452
|
+
("open_text", "transcript"),
|
|
453
|
+
"Must the output be a faithful rendering of the input, or may it add, "
|
|
454
|
+
"omit and reorganise? A transcriber may not; a generator may, and "
|
|
455
|
+
"will.",
|
|
456
|
+
),
|
|
457
|
+
DistinguishingQuestion(
|
|
458
|
+
("annotation", "label"),
|
|
459
|
+
"Do you need one answer about the whole input, or a structure over "
|
|
460
|
+
"its parts? Whole-input is a labeller; spans, boxes and masks are an "
|
|
461
|
+
"analyser.",
|
|
462
|
+
),
|
|
463
|
+
)
|
|
464
|
+
|
|
465
|
+
QUESTION_BY_PAIR: dict[frozenset[str], DistinguishingQuestion] = {
|
|
466
|
+
frozenset(q.between): q for q in DISTINGUISHING_QUESTIONS
|
|
467
|
+
}
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
#: When two candidate classes do not compete but chain. Published as the rule,
|
|
471
|
+
#: because "both, in this order" is a legitimate answer and has to be
|
|
472
|
+
#: distinguishable from an ordering, which it is not: neither class is placed
|
|
473
|
+
#: above the other.
|
|
474
|
+
COMPOSITION_RULE = (
|
|
475
|
+
"When a candidate class has an 'I cannot tell' output state (`abstains`) "
|
|
476
|
+
"and another candidate decides the same shape from the same inputs, the "
|
|
477
|
+
"pair is named as a sequence — the abstaining class first, the other on "
|
|
478
|
+
"its abstentions. This is not an ordering: neither class is better, and "
|
|
479
|
+
"no number is attached to either."
|
|
480
|
+
)
|
|
481
|
+
|
|
482
|
+
#: The one measured case, cited as a document. Its figures are deliberately
|
|
483
|
+
#: not served: a number printed beside a class name is a score whatever the
|
|
484
|
+
#: key is called, and that page's own "what this does not show" forbids
|
|
485
|
+
#: generalising one task to another.
|
|
486
|
+
COMPOSITION_EVIDENCE = "one measured task"
|
|
487
|
+
COMPOSITION_SEE = (
|
|
488
|
+
"https://github.com/turbobeest/modelspec/blob/main/"
|
|
489
|
+
"docs/research/cost-to-correct-attribution.md"
|
|
490
|
+
)
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
#: What the term matcher cannot do. In the published rule, not only in the
|
|
494
|
+
#: design document, so a caller reads it beside the answer it shaped.
|
|
495
|
+
CANNOT: tuple[str, ...] = (
|
|
496
|
+
"paraphrase: a task described in other words than the published terms is refused",
|
|
497
|
+
"negation: 'it must not write anything' matches the writing terms",
|
|
498
|
+
"any language other than English",
|
|
499
|
+
"constraints of volume, latency or budget, which are not in the words",
|
|
500
|
+
"choosing a model — that is ranking, and it runs within one class",
|
|
501
|
+
)
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
def class_fit_policy() -> dict[str, Any]:
|
|
505
|
+
"""The decision rule, as data a sceptic can read and re-run.
|
|
506
|
+
|
|
507
|
+
`ranking_policy()` publishes the floors and `neutrality_commitment()`
|
|
508
|
+
publishes the honest-broker promise, both keyless and both beside the
|
|
509
|
+
answer they shaped. This ships the same way, and carries the commitment by
|
|
510
|
+
calling it rather than copying its strings.
|
|
511
|
+
"""
|
|
512
|
+
return {
|
|
513
|
+
"version": "class-fit-v1",
|
|
514
|
+
"basis": "model_type",
|
|
515
|
+
"axis": ["consumes", "emits", "decides"],
|
|
516
|
+
"derivation": "class = CLASS_BY_PAIR[(emits, decides)]",
|
|
517
|
+
"orders_classes": False,
|
|
518
|
+
"cross_class_scores": "never",
|
|
519
|
+
"task_matching": (
|
|
520
|
+
"exact token match over the published terms; no model call, so the "
|
|
521
|
+
"same request always returns the same answer"
|
|
522
|
+
),
|
|
523
|
+
"request_text": (
|
|
524
|
+
"matched and discarded. Only the matched terms are echoed; the "
|
|
525
|
+
"text is never stored, logged or forwarded"
|
|
526
|
+
),
|
|
527
|
+
"cannot": list(CANNOT),
|
|
528
|
+
"refuses_when": [
|
|
529
|
+
{"code": "empty_request",
|
|
530
|
+
"meaning": "no task description and no facet was given"},
|
|
531
|
+
{"code": "no_term_matched",
|
|
532
|
+
"meaning": "a task description was given and no published term occurs in it"},
|
|
533
|
+
{"code": "unknown_facet_value",
|
|
534
|
+
"meaning": "a facet value outside the published vocabulary"},
|
|
535
|
+
],
|
|
536
|
+
"fit_statuses": ["resolved", "partial", "unavailable", "refused"],
|
|
537
|
+
"excluded_reasons": list(EXCLUDED_REASONS),
|
|
538
|
+
"adaptations": {
|
|
539
|
+
"consumes": [a.to_json() for a in CONSUMES_ADAPTATIONS],
|
|
540
|
+
"emits": [a.to_json() for a in EMITS_ADAPTATIONS],
|
|
541
|
+
},
|
|
542
|
+
"strict_facet": (
|
|
543
|
+
"decides — it admits no adaptation, because adapting it means the "
|
|
544
|
+
"caller writes the decision logic themselves, which is a different "
|
|
545
|
+
"architecture rather than a different model"
|
|
546
|
+
),
|
|
547
|
+
"composition_rule": COMPOSITION_RULE,
|
|
548
|
+
"non_classes": {key: NON_CLASS_RESOLUTION[key] for key in NON_CLASSES},
|
|
549
|
+
# Additive under the contract's own rule, and called rather than
|
|
550
|
+
# transcribed so the published terms keep one source.
|
|
551
|
+
"neutrality": neutrality_commitment(),
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def vocabulary() -> dict[str, list[str]]:
|
|
556
|
+
"""The three facet vocabularies, published with every answer."""
|
|
557
|
+
return {"consumes": list(CONSUMES), "emits": list(EMITS), "decides": list(DECIDES)}
|
api/ranking/__init__.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""ModelSpec Ranking Engine — 4-stage pipeline for model downselection.
|
|
2
|
+
|
|
3
|
+
Stages:
|
|
4
|
+
1. Filter — eliminate models that don't meet hard constraints
|
|
5
|
+
2. Score — compute weighted composite scores per use-case profile
|
|
6
|
+
3. Rank — sort by score, break ties
|
|
7
|
+
4. Explain — generate human-readable reasons
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from .engine import RankingEngine, USE_CASE_PROFILES
|
|
11
|
+
|
|
12
|
+
__all__ = ["RankingEngine", "USE_CASE_PROFILES"]
|