modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
api/classes.py ADDED
@@ -0,0 +1,557 @@
1
+ """What class of model a problem needs — the taxonomy (MODEL-100).
2
+
3
+ `ModelType` is a good publishing field and a bad selection axis. Its 35 values
4
+ sit on four axes at once: modality (`image-generation`), role (`reward-model`,
5
+ `router`), lineage (`distilled`, `merged`, `quantized-variant`) and domain
6
+ (`medical`, `legal`). A builder cannot choose on that — `quantized-variant` and
7
+ `llm-chat` are not alternatives, one is a provenance relation to the other.
8
+
9
+ So this module is a **view over `ModelType`, never a replacement for it**. The
10
+ enum stays exactly as published; MODEL-98 has just paid a major version for it.
11
+ Nothing here is written on a card, so nothing can drift per card — the same
12
+ call `schema/applicability.py` made one commit ago, for the same reason.
13
+
14
+ The axis is **what a model consumes, what it emits, and what decision it makes
15
+ on the caller's behalf**. `decides` does the real work, because it is the facet
16
+ that changes the caller's own code: prose a person reads, a permutation of
17
+ items you supplied, a label from a taxonomy the model defines, a value from a
18
+ set *you* defined, a preference between two outputs, or an action taken.
19
+
20
+ A class is **derived**, not hand-listed: `CLASS_BY_PAIR[(emits, decides)]`.
21
+ That pair is injective, checked at import and again by a test, so the mapping
22
+ is a function a sceptic can evaluate rather than a matter of taste.
23
+
24
+ Two deliberate properties:
25
+
26
+ * **Keyed on the enum's string values**, never on `ModelType` itself. The
27
+ module therefore imports nothing the Cloudflare Worker's Python bundle does
28
+ not already carry — `api/ranking/engine.py` is in it and `schema/` is not —
29
+ so it can be vendored unchanged the day `POST /v1/class-fit` is built
30
+ (`api/worker/vendor.py`). `tests/test_class_fit.py` holds the key set equal
31
+ to `{t.value for t in ModelType}`, so a new enum member is a red test on the
32
+ commit that adds it — which is where the argument about its class belongs.
33
+ * **`api/ranking/engine.py` never imports this module.** The dependency runs
34
+ one way only, and a test enforces it. Cost-to-correct (MODEL-99) is fitness
35
+ evidence for a *task*; `rank_score` is a within-class quality composite. If
36
+ the scorer could see this package, a measurement of one task could lift a
37
+ model in a ranking of another, invisibly.
38
+
39
+ The reasoning is `docs/design/class-selection.md`.
40
+ """
41
+
42
+ from __future__ import annotations
43
+
44
+ from dataclasses import dataclass, field
45
+ from typing import Any
46
+
47
+ # `neutrality_commitment` is *called*, never transcribed, so the published
48
+ # terms keep their single source and `tests/test_legal.py` stays the one place
49
+ # prose and JSON are held together. `USE_CASE_PROFILES` is read to derive which
50
+ # ranking profiles cover a class, rather than hand-listing them here. This is
51
+ # the only import from the repository, and it points at a module the Worker
52
+ # bundle already carries.
53
+ from api.ranking.engine import USE_CASE_PROFILES, neutrality_commitment
54
+
55
+ # ── the vocabularies, published ─────────────────────────────────────────────
56
+
57
+ #: Input kinds a task can have. Not modalities: `structured_state` is the
58
+ #: typed state a decision model is handed, and `item_list` is the candidate
59
+ #: set a reranker orders.
60
+ CONSUMES: tuple[str, ...] = (
61
+ "text", "image", "audio", "video", "page_image", "structured_state",
62
+ "numeric_series", "model_output", "item_list", "observations",
63
+ )
64
+
65
+ #: What comes out. `open_text` is generated language a person or a parser
66
+ #: reads; `choice` is a value from a set the caller defined; `label` is a value
67
+ #: from a taxonomy the model defines. Those three are different products.
68
+ EMITS: tuple[str, ...] = (
69
+ "open_text", "media", "transcript", "vector", "ordering", "label",
70
+ "choice", "preference_score", "action", "series", "world_state",
71
+ "annotation",
72
+ )
73
+
74
+ #: The decision the model makes on the caller's behalf — the facet that decides
75
+ #: what the caller's code looks like afterwards.
76
+ DECIDES: tuple[str, ...] = (
77
+ "nothing", "what_order", "which_of_a_fixed_set",
78
+ "which_of_a_caller_defined_set", "how_good", "what_to_do_next",
79
+ )
80
+
81
+ #: Every refusal the rule can produce. Published, and `api/class_fit.py` reads
82
+ #: this tuple rather than repeating it.
83
+ REFUSAL_CODES: tuple[str, ...] = (
84
+ "empty_request", "no_term_matched", "unknown_facet_value",
85
+ )
86
+
87
+ #: Why a class is not a candidate. A closed set: an exclusion without a reason
88
+ #: is a guess wearing a verdict's clothes.
89
+ EXCLUDED_REASONS: tuple[str, ...] = (
90
+ "emits_wrong_kind", "consumes_unsupported", "decision_shape_mismatch",
91
+ # The task description used none of this class's terms. Only ever reached
92
+ # when the caller supplied no facets of their own: a term is a hint and may
93
+ # discover a class, but it may never remove one a constraint admits.
94
+ "not_described",
95
+ "not_a_class",
96
+ )
97
+
98
+
99
+ @dataclass(frozen=True)
100
+ class Adaptation:
101
+ """One published way an input or an output can be made to fit.
102
+
103
+ Adaptations are the reason a text generator can answer a decision question
104
+ at all, and they are deliberately few: each one is a claim that a caller
105
+ can bridge two facet values in practice, and each one here is a claim
106
+ MODEL-99 actually made and measured.
107
+ """
108
+
109
+ from_: str
110
+ to: str
111
+ how: str
112
+
113
+ def to_json(self) -> dict[str, str]:
114
+ return {"from": self.from_, "to": self.to, "how": self.how}
115
+
116
+
117
+ #: Inputs that can be made to fit. Exactly one today.
118
+ CONSUMES_ADAPTATIONS: tuple[Adaptation, ...] = (
119
+ Adaptation(
120
+ "structured_state", "text",
121
+ "serialised as JSON — MODEL-99 handed the LLM arms the decision "
122
+ "model's own state and questions verbatim",
123
+ ),
124
+ )
125
+
126
+ #: Outputs that can be made to fit. Exactly one today, and it carries its own
127
+ #: caveat: MODEL-99 scored a truncated or unparseable reply as a *wrong*
128
+ #: answer, not a missing one, because dropping them would flatter the arm that
129
+ #: produces them.
130
+ EMITS_ADAPTATIONS: tuple[Adaptation, ...] = (
131
+ Adaptation(
132
+ "open_text", "choice",
133
+ "parsed out of generated text; a malformed, truncated or refused "
134
+ "answer is a wrong answer, not a missing one",
135
+ ),
136
+ )
137
+
138
+
139
+ @dataclass(frozen=True)
140
+ class ModelClass:
141
+ """One class: its facets, the `ModelType` values that derive to it, and why."""
142
+
143
+ id: str
144
+ consumes: tuple[str, ...]
145
+ emits: str
146
+ decides: str
147
+ #: Whether this class has an "I cannot tell" output state. The precondition
148
+ #: for a cascade (see `COMPOSITION_RULE`), not a quality judgement.
149
+ abstains: bool
150
+ #: Why this `(emits, decides)` pair is its own class, in the words a
151
+ #: reviewer needs. Not published per class to sell it; published so it can
152
+ #: be argued with.
153
+ because: str
154
+ #: Words a task description might use. The weakest part of this design, and
155
+ #: the part published loudest for exactly that reason.
156
+ terms: tuple[str, ...]
157
+ #: The `ModelType` values that derive to this class.
158
+ model_types: tuple[str, ...]
159
+ #: A domain narrows the pool *inside* a class; it never names one.
160
+ domains: dict[str, str] = field(default_factory=dict)
161
+
162
+
163
+ CLASSES: tuple[ModelClass, ...] = (
164
+ ModelClass(
165
+ id="actor",
166
+ consumes=("text", "image", "observations"),
167
+ emits="action", decides="what_to_do_next", abstains=False,
168
+ because=(
169
+ "it acts. Everything else hands the caller material or a verdict "
170
+ "and stops; this one changes the world, so the caller's code is an "
171
+ "executor rather than a reader"
172
+ ),
173
+ terms=("agent", "agentic", "act", "action", "tool", "tools", "browse",
174
+ "automate", "robot", "robotics", "control", "execute"),
175
+ model_types=("agent-model", "robotics"),
176
+ ),
177
+ ModelClass(
178
+ id="analyser",
179
+ consumes=("text", "image", "video"),
180
+ emits="annotation", decides="which_of_a_fixed_set", abstains=False,
181
+ because=(
182
+ "it labels the *parts* of its input — tokens, spans, regions, "
183
+ "depth — rather than the whole of it, so the caller gets a "
184
+ "structure over the input, not a verdict about it"
185
+ ),
186
+ terms=("detect", "detection", "segment", "segmentation", "bounding box",
187
+ "depth", "named entity", "token classification", "annotate",
188
+ "keypoint", "fill mask"),
189
+ model_types=("vision-encoder", "text-encoder"),
190
+ ),
191
+ ModelClass(
192
+ id="decider",
193
+ consumes=("text", "structured_state"),
194
+ emits="choice", decides="which_of_a_caller_defined_set", abstains=True,
195
+ because=(
196
+ "the answer set is defined by the caller at request time. That is "
197
+ "the sentence `schema/enums.py` already writes for `decision-model`"
198
+ " — a safety classifier has a fixed harm taxonomy, these labels do "
199
+ "not — promoted from a docstring to the axis"
200
+ ),
201
+ terms=("decide", "decision", "classify", "classification", "choose",
202
+ "route", "routing", "triage", "judge", "judgement", "judgment",
203
+ "verdict", "determine", "attribute", "attribution", "calibrated",
204
+ "probability", "abstain"),
205
+ model_types=("decision-model", "router"),
206
+ ),
207
+ ModelClass(
208
+ id="forecaster",
209
+ consumes=("numeric_series",),
210
+ emits="series", decides="nothing", abstains=False,
211
+ because="no tokens go in and none come out; the unit is a number over time",
212
+ terms=("forecast", "forecasting", "time series", "timeseries",
213
+ "seasonality", "demand", "horizon"),
214
+ model_types=("time-series",),
215
+ ),
216
+ ModelClass(
217
+ id="labeller",
218
+ consumes=("text", "image"),
219
+ emits="label", decides="which_of_a_fixed_set", abstains=True,
220
+ because=(
221
+ "the taxonomy belongs to the model, not the caller. You get the "
222
+ "categories it was trained on and no others, which is a different "
223
+ "product from one that answers your questions"
224
+ ),
225
+ terms=("moderate", "moderation", "safety", "toxicity", "toxic",
226
+ "harmful", "abuse", "spam", "guardrail", "classify",
227
+ "classification", "flag"),
228
+ model_types=("safety-classifier",),
229
+ ),
230
+ ModelClass(
231
+ id="media-generator",
232
+ consumes=("text", "image", "audio"),
233
+ emits="media", decides="nothing", abstains=False,
234
+ because="the artefact is pixels or samples, so nothing downstream parses it",
235
+ terms=("image", "picture", "illustration", "render", "photo", "video",
236
+ "speech", "voice", "music", "synthesise", "synthesize", "inpaint"),
237
+ model_types=("image-generation", "image-editing", "video-generation",
238
+ "audio-tts", "audio-music"),
239
+ ),
240
+ ModelClass(
241
+ id="orderer",
242
+ consumes=("text", "item_list"),
243
+ emits="ordering", decides="what_order", abstains=False,
244
+ because=(
245
+ "it is handed the candidates and returns a permutation of them. It "
246
+ "invents nothing and chooses nothing; it only sorts"
247
+ ),
248
+ terms=("rerank", "reranking", "reorder", "relevance", "shortlist",
249
+ "ordering", "sort"),
250
+ model_types=("reranker",),
251
+ ),
252
+ ModelClass(
253
+ id="scorer",
254
+ consumes=("text", "model_output"),
255
+ emits="preference_score", decides="how_good", abstains=False,
256
+ because=(
257
+ "its subject is another model's output, for training or selection. "
258
+ "It answers 'how good is this answer', never 'what is the answer'"
259
+ ),
260
+ terms=("reward", "preference", "rlhf", "best of n", "rate the output",
261
+ "score the output"),
262
+ model_types=("reward-model",),
263
+ ),
264
+ ModelClass(
265
+ id="simulator",
266
+ consumes=("observations",),
267
+ emits="world_state", decides="nothing", abstains=False,
268
+ because="it predicts what happens next in an environment, not what to say about it",
269
+ terms=("simulate", "simulation", "world model", "dynamics", "rollout",
270
+ "environment"),
271
+ model_types=("world-model",),
272
+ ),
273
+ ModelClass(
274
+ id="text-generator",
275
+ consumes=("text", "image", "audio"),
276
+ emits="open_text", decides="nothing", abstains=False,
277
+ because=(
278
+ "it produces open-ended language and decides nothing: the caller "
279
+ "reads it, or parses it and decides. This is the class every other "
280
+ "catalogue already ranks, and the one MODEL-100 exists to stop "
281
+ "assuming"
282
+ ),
283
+ terms=("write", "writing", "summarise", "summarize", "summary", "draft",
284
+ "explain", "chat", "conversation", "prose", "essay", "reply",
285
+ "translate", "rewrite", "compose", "reason", "code", "coding"),
286
+ model_types=("llm-chat", "llm-reasoning", "llm-code", "llm-base", "vlm",
287
+ "medical", "legal", "financial", "audio-realtime"),
288
+ domains={"medical": "medical", "legal": "legal", "financial": "financial"},
289
+ ),
290
+ ModelClass(
291
+ id="transcriber",
292
+ consumes=("audio", "page_image"),
293
+ emits="transcript", decides="nothing", abstains=False,
294
+ because=(
295
+ "the output is a faithful rendering of the input in another form. "
296
+ "A generator may add or reorganise; this one may not, and that is "
297
+ "the whole difference to the caller"
298
+ ),
299
+ terms=("transcribe", "transcript", "transcription", "ocr", "dictation",
300
+ "subtitle", "caption", "speech to text"),
301
+ model_types=("audio-asr", "document-ocr"),
302
+ ),
303
+ ModelClass(
304
+ id="vectoriser",
305
+ consumes=("text", "image"),
306
+ emits="vector", decides="nothing", abstains=False,
307
+ because=(
308
+ "the output is a reusable representation the caller stores and "
309
+ "compares, not an answer about any one input"
310
+ ),
311
+ terms=("embed", "embedding", "embeddings", "vector", "similarity",
312
+ "semantic search", "nearest neighbour", "nearest neighbor",
313
+ "retrieval", "index"),
314
+ model_types=("embedding-text", "embedding-multimodal", "embedding-code"),
315
+ ),
316
+ )
317
+
318
+
319
+ #: `ModelType` values that name **no class**, and why.
320
+ #:
321
+ #: Lineage answers "where did these weights come from", not "what does this
322
+ #: do". A distilled model is a smaller instance of its base model's class, not
323
+ #: an alternative to it. `miscellaneous` disqualifies itself by its own
324
+ #: definition — "for models no specific type describes honestly".
325
+ NON_CLASSES: dict[str, tuple[str, ...]] = {
326
+ "derived": ("adapter", "quantized-variant", "distilled", "merged"),
327
+ "unclassified": ("miscellaneous",),
328
+ }
329
+
330
+ #: What a caller does instead. Named, because "we cannot place this" without a
331
+ #: next step is a dead end rather than a refusal.
332
+ NON_CLASS_RESOLUTION: dict[str, str] = {
333
+ "derived": (
334
+ "this value records a lineage relation, not a capability. Follow "
335
+ "`lineage.base_model` on the card and take that model's class."
336
+ ),
337
+ "unclassified": (
338
+ "`miscellaneous` is the catalogue's own admission that no type "
339
+ "describes the model honestly. There is nothing to derive from it."
340
+ ),
341
+ }
342
+
343
+
344
+ def _build_class_by_pair() -> dict[tuple[str, str], str]:
345
+ """`(emits, decides) -> class id`, and prove it is a function."""
346
+ table: dict[tuple[str, str], str] = {}
347
+ for model_class in CLASSES:
348
+ key = (model_class.emits, model_class.decides)
349
+ if key in table:
350
+ raise ValueError(
351
+ f"{model_class.id} and {table[key]} share the pair {key}: the "
352
+ "class derivation would stop being a function"
353
+ )
354
+ table[key] = model_class.id
355
+ return table
356
+
357
+
358
+ #: The derivation, printed. A class is looked up from its facets; it is never
359
+ #: assigned by hand.
360
+ CLASS_BY_PAIR: dict[tuple[str, str], str] = _build_class_by_pair()
361
+
362
+ CLASS_BY_ID: dict[str, ModelClass] = {c.id: c for c in CLASSES}
363
+
364
+
365
+ def _build_placement() -> dict[str, str]:
366
+ placed: dict[str, str] = {}
367
+ for model_class in CLASSES:
368
+ for value in model_class.model_types:
369
+ placed[value] = model_class.id
370
+ for non_class, values in NON_CLASSES.items():
371
+ for value in values:
372
+ placed[value] = non_class
373
+ return placed
374
+
375
+
376
+ #: `model_type` value -> class id, or one of the `NON_CLASSES` keys.
377
+ MODEL_TYPE_PLACEMENT: dict[str, str] = _build_placement()
378
+
379
+
380
+ def class_for_model_type(value: str | None) -> str | None:
381
+ """The class id for a `model_type`, or None when it names no class."""
382
+ placed = MODEL_TYPE_PLACEMENT.get(value or "")
383
+ return placed if placed in CLASS_BY_ID else None
384
+
385
+
386
+ def non_class_for_model_type(value: str | None) -> str | None:
387
+ """`derived`, `unclassified`, or None when the value does name a class."""
388
+ placed = MODEL_TYPE_PLACEMENT.get(value or "")
389
+ return placed if placed in NON_CLASSES else None
390
+
391
+
392
+ def rank_profiles_for(class_id: str) -> list[str]:
393
+ """Ranking profiles that cover a class, **derived** from `preferred_types`.
394
+
395
+ Not a hand list, so it cannot drift from the profiles. For `decider` it is
396
+ empty today, which is the true and useful statement that ranking stops at
397
+ that class boundary.
398
+ """
399
+ model_class = CLASS_BY_ID.get(class_id)
400
+ if model_class is None:
401
+ return []
402
+ wanted = set(model_class.model_types)
403
+ return sorted(
404
+ key for key, profile in USE_CASE_PROFILES.items()
405
+ if wanted & set(profile.get("preferred_types") or ())
406
+ )
407
+
408
+
409
+ # ── the questions a `partial` answer hands back ─────────────────────────────
410
+ #
411
+ # Published, not generated. When two classes both survive, the facet they
412
+ # differ on is the thing the caller has to settle, and phrasing it is the
413
+ # difference between "I cannot tell you" and work somebody can do.
414
+
415
+ @dataclass(frozen=True)
416
+ class DistinguishingQuestion:
417
+ between: tuple[str, str]
418
+ ask: str
419
+
420
+ def to_json(self) -> dict[str, Any]:
421
+ return {"between": list(self.between), "ask": self.ask}
422
+
423
+
424
+ DISTINGUISHING_QUESTIONS: tuple[DistinguishingQuestion, ...] = (
425
+ DistinguishingQuestion(
426
+ ("choice", "open_text"),
427
+ "Does your code branch on the answer, or does a person read it? A "
428
+ "value your code switches on wants a decider; prose a person reads "
429
+ "wants a text-generator. If you would parse the prose into a value, "
430
+ "you want a decider and are considering paying an LLM to be one.",
431
+ ),
432
+ DistinguishingQuestion(
433
+ ("choice", "label"),
434
+ "Is the set of answers fixed by the model, or defined by you per "
435
+ "request? A fixed taxonomy — harm categories, sentiment — is a "
436
+ "labeller. Your own options, changing per call, need a decider.",
437
+ ),
438
+ DistinguishingQuestion(
439
+ ("ordering", "vector"),
440
+ "Do you need a reusable representation you store and compare later, "
441
+ "or one ordering of items you already hold? Storing it is a "
442
+ "vectoriser; ordering it now is an orderer. Retrieval usually wants "
443
+ "both, in that order.",
444
+ ),
445
+ DistinguishingQuestion(
446
+ ("action", "open_text"),
447
+ "Does the model hand your code text to act on, or take the action "
448
+ "itself? If your code owns the loop you want a text-generator and a "
449
+ "harness, not an actor.",
450
+ ),
451
+ DistinguishingQuestion(
452
+ ("open_text", "transcript"),
453
+ "Must the output be a faithful rendering of the input, or may it add, "
454
+ "omit and reorganise? A transcriber may not; a generator may, and "
455
+ "will.",
456
+ ),
457
+ DistinguishingQuestion(
458
+ ("annotation", "label"),
459
+ "Do you need one answer about the whole input, or a structure over "
460
+ "its parts? Whole-input is a labeller; spans, boxes and masks are an "
461
+ "analyser.",
462
+ ),
463
+ )
464
+
465
+ QUESTION_BY_PAIR: dict[frozenset[str], DistinguishingQuestion] = {
466
+ frozenset(q.between): q for q in DISTINGUISHING_QUESTIONS
467
+ }
468
+
469
+
470
+ #: When two candidate classes do not compete but chain. Published as the rule,
471
+ #: because "both, in this order" is a legitimate answer and has to be
472
+ #: distinguishable from an ordering, which it is not: neither class is placed
473
+ #: above the other.
474
+ COMPOSITION_RULE = (
475
+ "When a candidate class has an 'I cannot tell' output state (`abstains`) "
476
+ "and another candidate decides the same shape from the same inputs, the "
477
+ "pair is named as a sequence — the abstaining class first, the other on "
478
+ "its abstentions. This is not an ordering: neither class is better, and "
479
+ "no number is attached to either."
480
+ )
481
+
482
+ #: The one measured case, cited as a document. Its figures are deliberately
483
+ #: not served: a number printed beside a class name is a score whatever the
484
+ #: key is called, and that page's own "what this does not show" forbids
485
+ #: generalising one task to another.
486
+ COMPOSITION_EVIDENCE = "one measured task"
487
+ COMPOSITION_SEE = (
488
+ "https://github.com/turbobeest/modelspec/blob/main/"
489
+ "docs/research/cost-to-correct-attribution.md"
490
+ )
491
+
492
+
493
+ #: What the term matcher cannot do. In the published rule, not only in the
494
+ #: design document, so a caller reads it beside the answer it shaped.
495
+ CANNOT: tuple[str, ...] = (
496
+ "paraphrase: a task described in other words than the published terms is refused",
497
+ "negation: 'it must not write anything' matches the writing terms",
498
+ "any language other than English",
499
+ "constraints of volume, latency or budget, which are not in the words",
500
+ "choosing a model — that is ranking, and it runs within one class",
501
+ )
502
+
503
+
504
+ def class_fit_policy() -> dict[str, Any]:
505
+ """The decision rule, as data a sceptic can read and re-run.
506
+
507
+ `ranking_policy()` publishes the floors and `neutrality_commitment()`
508
+ publishes the honest-broker promise, both keyless and both beside the
509
+ answer they shaped. This ships the same way, and carries the commitment by
510
+ calling it rather than copying its strings.
511
+ """
512
+ return {
513
+ "version": "class-fit-v1",
514
+ "basis": "model_type",
515
+ "axis": ["consumes", "emits", "decides"],
516
+ "derivation": "class = CLASS_BY_PAIR[(emits, decides)]",
517
+ "orders_classes": False,
518
+ "cross_class_scores": "never",
519
+ "task_matching": (
520
+ "exact token match over the published terms; no model call, so the "
521
+ "same request always returns the same answer"
522
+ ),
523
+ "request_text": (
524
+ "matched and discarded. Only the matched terms are echoed; the "
525
+ "text is never stored, logged or forwarded"
526
+ ),
527
+ "cannot": list(CANNOT),
528
+ "refuses_when": [
529
+ {"code": "empty_request",
530
+ "meaning": "no task description and no facet was given"},
531
+ {"code": "no_term_matched",
532
+ "meaning": "a task description was given and no published term occurs in it"},
533
+ {"code": "unknown_facet_value",
534
+ "meaning": "a facet value outside the published vocabulary"},
535
+ ],
536
+ "fit_statuses": ["resolved", "partial", "unavailable", "refused"],
537
+ "excluded_reasons": list(EXCLUDED_REASONS),
538
+ "adaptations": {
539
+ "consumes": [a.to_json() for a in CONSUMES_ADAPTATIONS],
540
+ "emits": [a.to_json() for a in EMITS_ADAPTATIONS],
541
+ },
542
+ "strict_facet": (
543
+ "decides — it admits no adaptation, because adapting it means the "
544
+ "caller writes the decision logic themselves, which is a different "
545
+ "architecture rather than a different model"
546
+ ),
547
+ "composition_rule": COMPOSITION_RULE,
548
+ "non_classes": {key: NON_CLASS_RESOLUTION[key] for key in NON_CLASSES},
549
+ # Additive under the contract's own rule, and called rather than
550
+ # transcribed so the published terms keep one source.
551
+ "neutrality": neutrality_commitment(),
552
+ }
553
+
554
+
555
+ def vocabulary() -> dict[str, list[str]]:
556
+ """The three facet vocabularies, published with every answer."""
557
+ return {"consumes": list(CONSUMES), "emits": list(EMITS), "decides": list(DECIDES)}
@@ -0,0 +1,12 @@
1
+ """ModelSpec Ranking Engine — 4-stage pipeline for model downselection.
2
+
3
+ Stages:
4
+ 1. Filter — eliminate models that don't meet hard constraints
5
+ 2. Score — compute weighted composite scores per use-case profile
6
+ 3. Rank — sort by score, break ties
7
+ 4. Explain — generate human-readable reasons
8
+ """
9
+
10
+ from .engine import RankingEngine, USE_CASE_PROFILES
11
+
12
+ __all__ = ["RankingEngine", "USE_CASE_PROFILES"]