modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
api/__init__.py
ADDED
|
File without changes
|
api/class_fit.py
ADDED
|
@@ -0,0 +1,334 @@
|
|
|
1
|
+
"""Which class of model a task needs — the rule (MODEL-100).
|
|
2
|
+
|
|
3
|
+
`api/classes.py` is the taxonomy; this is the decision made over it. A pure
|
|
4
|
+
function: a request and whatever catalogue evidence the caller has in, an
|
|
5
|
+
answer out. It performs no I/O, calls no model, and never raises on caller
|
|
6
|
+
input — a refusal is an answer here, not an exception.
|
|
7
|
+
|
|
8
|
+
**The refusal was designed first**, because the honest inventory demands it.
|
|
9
|
+
ModelSpec holds, per class, the facets and how many cards claim it; per card,
|
|
10
|
+
benchmarks that are *within-class by construction*; and across classes, **one
|
|
11
|
+
measurement of one task** (MODEL-99), whose own page says nothing about it
|
|
12
|
+
generalises. So the rule that can be applied honestly is a **filter over
|
|
13
|
+
facets, not a comparison**. `partial` — two or more classes survive and we
|
|
14
|
+
will not order them — is therefore the expected answer, not a degraded one, in
|
|
15
|
+
exactly the way the ranker's `unranked` is not "scored low".
|
|
16
|
+
|
|
17
|
+
Two asymmetries worth reading before the code:
|
|
18
|
+
|
|
19
|
+
* **A facet is a constraint; a term is a hint.** Facets the caller supplied are
|
|
20
|
+
assertions about their own problem and they bind. Terms are our guess about
|
|
21
|
+
their words, so a term match may *discover* a class but may never remove one
|
|
22
|
+
the caller's own constraints admit. A guess never overrides an assertion.
|
|
23
|
+
* **`decides` is strict.** `emits` and `consumes` admit published adaptations —
|
|
24
|
+
a text generator can be parsed into a choice, a typed state can be serialised
|
|
25
|
+
into text, and MODEL-99 did both. `decides` admits none, because adapting it
|
|
26
|
+
means the caller writes the decision logic themselves, which is a different
|
|
27
|
+
architecture rather than a different model.
|
|
28
|
+
|
|
29
|
+
No number here orders one class above another, and MODEL-99's figures are not
|
|
30
|
+
served on any surface. Cost-to-correct is fitness evidence for a *task*;
|
|
31
|
+
`rank_score` is a within-class quality composite. `tests/test_class_fit.py`
|
|
32
|
+
holds the boundary in four places, including the import direction.
|
|
33
|
+
|
|
34
|
+
The reasoning is `docs/design/class-selection.md`.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import re
|
|
40
|
+
from dataclasses import dataclass, field
|
|
41
|
+
from typing import Any, Mapping, Sequence
|
|
42
|
+
|
|
43
|
+
from api import classes as cls
|
|
44
|
+
|
|
45
|
+
#: Single source; re-exported so a caller reads one vocabulary, not two.
|
|
46
|
+
REFUSAL_CODES = cls.REFUSAL_CODES
|
|
47
|
+
EXCLUDED_REASONS = cls.EXCLUDED_REASONS
|
|
48
|
+
|
|
49
|
+
#: At most this many example ids per class, sorted. A longer list, or an
|
|
50
|
+
#: unsorted one, would start to read as a recommendation.
|
|
51
|
+
MAX_EXAMPLES = 3
|
|
52
|
+
|
|
53
|
+
_WORD = re.compile(r"[^a-z0-9]+")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass(frozen=True)
|
|
57
|
+
class CatalogueEvidence:
|
|
58
|
+
"""What the caller knows about the catalogue, supplied rather than fetched.
|
|
59
|
+
|
|
60
|
+
Absence and zero are different answers — MODEL-97's lesson, reapplied. A
|
|
61
|
+
class missing from `card_counts` reports `unknown`; a class present with 0
|
|
62
|
+
reports `empty`, which is a real and useful thing to be told.
|
|
63
|
+
"""
|
|
64
|
+
|
|
65
|
+
#: class id -> number of cards
|
|
66
|
+
card_counts: Mapping[str, int] = field(default_factory=dict)
|
|
67
|
+
#: class id -> model ids; capped and sorted before publication
|
|
68
|
+
examples: Mapping[str, Sequence[str]] = field(default_factory=dict)
|
|
69
|
+
#: class id -> ranking profile keys. Omitted falls back to the derivation
|
|
70
|
+
#: in `api.classes.rank_profiles_for`.
|
|
71
|
+
rank_profiles: Mapping[str, Sequence[str]] = field(default_factory=dict)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _tokens(text: str) -> tuple[set[str], str]:
|
|
75
|
+
"""Lowercase word tokens, and the normalised string multi-word terms match."""
|
|
76
|
+
normalised = _WORD.sub(" ", text.lower()).strip()
|
|
77
|
+
return set(normalised.split()), f" {normalised} "
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _matched_terms(model_class: cls.ModelClass, words: set[str], blob: str) -> list[str]:
|
|
81
|
+
hits = []
|
|
82
|
+
for term in model_class.terms:
|
|
83
|
+
if " " in term:
|
|
84
|
+
if f" {term} " in blob:
|
|
85
|
+
hits.append(term)
|
|
86
|
+
elif term in words:
|
|
87
|
+
hits.append(term)
|
|
88
|
+
return sorted(hits)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _emits_check(model_class: cls.ModelClass, wanted: str) -> tuple[bool, dict | None]:
|
|
92
|
+
if model_class.emits == wanted:
|
|
93
|
+
return True, None
|
|
94
|
+
for adaptation in cls.EMITS_ADAPTATIONS:
|
|
95
|
+
if adaptation.from_ == model_class.emits and adaptation.to == wanted:
|
|
96
|
+
return True, adaptation.to_json()
|
|
97
|
+
return False, None
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _consumes_check(model_class: cls.ModelClass,
|
|
101
|
+
wanted: Sequence[str]) -> tuple[bool, list[dict]]:
|
|
102
|
+
adapted: list[dict] = []
|
|
103
|
+
for kind in wanted:
|
|
104
|
+
if kind in model_class.consumes:
|
|
105
|
+
continue
|
|
106
|
+
bridge = next(
|
|
107
|
+
(a for a in cls.CONSUMES_ADAPTATIONS
|
|
108
|
+
if a.from_ == kind and a.to in model_class.consumes),
|
|
109
|
+
None,
|
|
110
|
+
)
|
|
111
|
+
if bridge is None:
|
|
112
|
+
return False, adapted
|
|
113
|
+
adapted.append(bridge.to_json())
|
|
114
|
+
return True, adapted
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _catalogue(class_id: str, evidence: CatalogueEvidence | None) -> dict[str, Any]:
|
|
118
|
+
if evidence is None or class_id not in evidence.card_counts:
|
|
119
|
+
return {"evidence_state": "unknown"}
|
|
120
|
+
count = int(evidence.card_counts[class_id])
|
|
121
|
+
examples = sorted(evidence.examples.get(class_id) or ())[:MAX_EXAMPLES]
|
|
122
|
+
return {
|
|
123
|
+
"card_count": count,
|
|
124
|
+
"example_model_ids": examples,
|
|
125
|
+
"evidence_state": "populated" if count > 0 else "empty",
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _next_step(catalogue: dict[str, Any], profiles: Sequence[str]) -> str:
|
|
130
|
+
state = catalogue["evidence_state"]
|
|
131
|
+
if state == "unknown":
|
|
132
|
+
return ("Catalogue evidence was not supplied with this request. Fetch "
|
|
133
|
+
"/api/rank/class-fit.json for the per-class counts.")
|
|
134
|
+
if state == "empty":
|
|
135
|
+
return ("ModelSpec catalogues no model of this class today, so it "
|
|
136
|
+
"cannot name one for you — look outside the catalogue.")
|
|
137
|
+
if profiles:
|
|
138
|
+
return (f"Rank within this class: `modelspec offline rank {profiles[0]}`"
|
|
139
|
+
f" ({len(profiles)} profile(s) cover it).")
|
|
140
|
+
return ("No ranking profile covers this class, so ranking stops at this "
|
|
141
|
+
"class boundary. The catalogue can name the cards, not order them.")
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _answer(fit_status: str, **rest: Any) -> dict[str, Any]:
|
|
145
|
+
answer: dict[str, Any] = {
|
|
146
|
+
"fit_status": fit_status,
|
|
147
|
+
"policy": cls.class_fit_policy(),
|
|
148
|
+
"vocabulary": cls.vocabulary(),
|
|
149
|
+
"matched_terms": [],
|
|
150
|
+
"candidates": [],
|
|
151
|
+
"excluded": [],
|
|
152
|
+
"distinguishing_questions": [],
|
|
153
|
+
"composition": [],
|
|
154
|
+
}
|
|
155
|
+
answer.update(rest)
|
|
156
|
+
return answer
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _refused(code: str, message: str, **detail: Any) -> dict[str, Any]:
|
|
160
|
+
refusal = {"code": code, "message": message}
|
|
161
|
+
refusal.update(detail)
|
|
162
|
+
return _answer("refused", refusal=refusal)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def class_fit(*, task: str | None = None,
|
|
166
|
+
emits: str | None = None,
|
|
167
|
+
consumes: Sequence[str] | None = None,
|
|
168
|
+
decides: str | None = None,
|
|
169
|
+
evidence: CatalogueEvidence | None = None) -> dict[str, Any]:
|
|
170
|
+
"""Candidate classes for a task, or a refusal that says what would work.
|
|
171
|
+
|
|
172
|
+
`emits` / `consumes` / `decides` are the primary input: a caller who knows
|
|
173
|
+
what their system needs gets an exact, deterministic answer with no text
|
|
174
|
+
matching at all. `task` is a *term source only* — it is tokenised, matched
|
|
175
|
+
against the published terms, and discarded. Only the matched terms are
|
|
176
|
+
echoed; the text itself is never stored, logged or forwarded, which is what
|
|
177
|
+
keeps `neutrality_commitment()`'s `stores_customer_prompts: false` an
|
|
178
|
+
architectural fact rather than a promise.
|
|
179
|
+
"""
|
|
180
|
+
for label, value in (("emits", emits), ("decides", decides)):
|
|
181
|
+
vocab = cls.EMITS if label == "emits" else cls.DECIDES
|
|
182
|
+
if value is not None and value not in vocab:
|
|
183
|
+
return _refused(
|
|
184
|
+
"unknown_facet_value",
|
|
185
|
+
f"{label} must be one of the published vocabulary",
|
|
186
|
+
field=label, value=value, vocabulary=cls.vocabulary(),
|
|
187
|
+
)
|
|
188
|
+
for kind in consumes or ():
|
|
189
|
+
if kind not in cls.CONSUMES:
|
|
190
|
+
return _refused(
|
|
191
|
+
"unknown_facet_value",
|
|
192
|
+
"consumes must be drawn from the published vocabulary",
|
|
193
|
+
field="consumes", value=kind, vocabulary=cls.vocabulary(),
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
has_facets = bool(emits or decides or consumes)
|
|
197
|
+
if not has_facets and not (task or "").strip():
|
|
198
|
+
return _refused(
|
|
199
|
+
"empty_request",
|
|
200
|
+
"give a task description, or one of emits / consumes / decides",
|
|
201
|
+
vocabulary=cls.vocabulary(),
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
words, blob = _tokens(task) if task else (set(), " ")
|
|
205
|
+
terms_by_class = {
|
|
206
|
+
model_class.id: _matched_terms(model_class, words, blob)
|
|
207
|
+
for model_class in cls.CLASSES
|
|
208
|
+
}
|
|
209
|
+
all_matched = sorted({t for hits in terms_by_class.values() for t in hits})
|
|
210
|
+
|
|
211
|
+
if task and not has_facets and not all_matched:
|
|
212
|
+
return _refused(
|
|
213
|
+
"no_term_matched",
|
|
214
|
+
"no published term occurs in that description. The matcher does "
|
|
215
|
+
"not paraphrase; pick a class below, or send emits / consumes / "
|
|
216
|
+
"decides instead.",
|
|
217
|
+
classes=[
|
|
218
|
+
{"id": c.id, "emits": c.emits, "decides": c.decides,
|
|
219
|
+
"terms": list(c.terms)}
|
|
220
|
+
for c in cls.CLASSES
|
|
221
|
+
],
|
|
222
|
+
vocabulary=cls.vocabulary(),
|
|
223
|
+
)
|
|
224
|
+
|
|
225
|
+
candidates: list[dict[str, Any]] = []
|
|
226
|
+
excluded: list[dict[str, str]] = []
|
|
227
|
+
|
|
228
|
+
for model_class in cls.CLASSES:
|
|
229
|
+
emits_adapted: dict | None = None
|
|
230
|
+
consumes_adapted: list[dict] = []
|
|
231
|
+
|
|
232
|
+
if emits is not None:
|
|
233
|
+
ok, emits_adapted = _emits_check(model_class, emits)
|
|
234
|
+
if not ok:
|
|
235
|
+
excluded.append({"class": model_class.id,
|
|
236
|
+
"excluded_reason": "emits_wrong_kind"})
|
|
237
|
+
continue
|
|
238
|
+
if decides is not None and model_class.decides != decides:
|
|
239
|
+
excluded.append({"class": model_class.id,
|
|
240
|
+
"excluded_reason": "decision_shape_mismatch"})
|
|
241
|
+
continue
|
|
242
|
+
if consumes:
|
|
243
|
+
ok, consumes_adapted = _consumes_check(model_class, consumes)
|
|
244
|
+
if not ok:
|
|
245
|
+
excluded.append({"class": model_class.id,
|
|
246
|
+
"excluded_reason": "consumes_unsupported"})
|
|
247
|
+
continue
|
|
248
|
+
|
|
249
|
+
# A term is a hint, never a veto: it may only discover a class when the
|
|
250
|
+
# caller supplied no constraints of their own.
|
|
251
|
+
if not has_facets and not terms_by_class[model_class.id]:
|
|
252
|
+
excluded.append({"class": model_class.id,
|
|
253
|
+
"excluded_reason": "not_described"})
|
|
254
|
+
continue
|
|
255
|
+
|
|
256
|
+
catalogue = _catalogue(model_class.id, evidence)
|
|
257
|
+
profiles = list(
|
|
258
|
+
(evidence.rank_profiles.get(model_class.id) if evidence else None)
|
|
259
|
+
or cls.rank_profiles_for(model_class.id)
|
|
260
|
+
)
|
|
261
|
+
candidates.append({
|
|
262
|
+
"class": model_class.id,
|
|
263
|
+
"consumes": list(model_class.consumes),
|
|
264
|
+
"emits": model_class.emits,
|
|
265
|
+
"decides": model_class.decides,
|
|
266
|
+
"abstains": model_class.abstains,
|
|
267
|
+
"because": model_class.because,
|
|
268
|
+
"model_types": list(model_class.model_types),
|
|
269
|
+
"matched_terms": terms_by_class[model_class.id],
|
|
270
|
+
"emits_adapted": emits_adapted,
|
|
271
|
+
"consumes_adapted": consumes_adapted,
|
|
272
|
+
"rank_profiles": profiles,
|
|
273
|
+
"catalogue": catalogue,
|
|
274
|
+
"next": _next_step(catalogue, profiles),
|
|
275
|
+
})
|
|
276
|
+
|
|
277
|
+
# `derived` and `unclassified` are never candidates, and they say why and
|
|
278
|
+
# where to look instead rather than being silently absent.
|
|
279
|
+
for non_class in cls.NON_CLASSES:
|
|
280
|
+
excluded.append({"class": non_class, "excluded_reason": "not_a_class",
|
|
281
|
+
"resolution": cls.NON_CLASS_RESOLUTION[non_class]})
|
|
282
|
+
|
|
283
|
+
candidates.sort(key=lambda row: row["class"])
|
|
284
|
+
excluded.sort(key=lambda row: row["class"])
|
|
285
|
+
|
|
286
|
+
if not candidates:
|
|
287
|
+
return _answer("unavailable", matched_terms=all_matched, excluded=excluded)
|
|
288
|
+
|
|
289
|
+
fit_status = "resolved" if len(candidates) == 1 else "partial"
|
|
290
|
+
return _answer(
|
|
291
|
+
fit_status,
|
|
292
|
+
matched_terms=all_matched,
|
|
293
|
+
candidates=candidates,
|
|
294
|
+
excluded=excluded,
|
|
295
|
+
distinguishing_questions=_questions(candidates),
|
|
296
|
+
composition=_compositions(candidates),
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _questions(candidates: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
301
|
+
"""The facet the survivors differ on, phrased so the caller can settle it."""
|
|
302
|
+
kinds = sorted({row["emits"] for row in candidates})
|
|
303
|
+
found: dict[frozenset[str], cls.DistinguishingQuestion] = {}
|
|
304
|
+
for index, left in enumerate(kinds):
|
|
305
|
+
for right in kinds[index + 1:]:
|
|
306
|
+
question = cls.QUESTION_BY_PAIR.get(frozenset({left, right}))
|
|
307
|
+
if question is not None:
|
|
308
|
+
found[frozenset({left, right})] = question
|
|
309
|
+
return [q.to_json() for q in sorted(found.values(), key=lambda q: q.between)]
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _compositions(candidates: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
313
|
+
"""Pairs that chain rather than compete. Not an ordering — see the rule."""
|
|
314
|
+
chains: list[dict[str, Any]] = []
|
|
315
|
+
for first in candidates:
|
|
316
|
+
if not first["abstains"]:
|
|
317
|
+
continue
|
|
318
|
+
for second in candidates:
|
|
319
|
+
if second["class"] == first["class"]:
|
|
320
|
+
continue
|
|
321
|
+
bridged = second["emits"] == first["emits"] or any(
|
|
322
|
+
a.from_ == second["emits"] and a.to == first["emits"]
|
|
323
|
+
for a in cls.EMITS_ADAPTATIONS
|
|
324
|
+
)
|
|
325
|
+
if not bridged:
|
|
326
|
+
continue
|
|
327
|
+
chains.append({
|
|
328
|
+
"sequence": [first["class"], second["class"]],
|
|
329
|
+
"rule": cls.COMPOSITION_RULE,
|
|
330
|
+
"evidence": cls.COMPOSITION_EVIDENCE,
|
|
331
|
+
"see": cls.COMPOSITION_SEE,
|
|
332
|
+
})
|
|
333
|
+
chains.sort(key=lambda chain: chain["sequence"])
|
|
334
|
+
return chains
|