modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
cli/modelspec/offline.py
ADDED
|
@@ -0,0 +1,623 @@
|
|
|
1
|
+
"""Commands that answer from the local snapshot, with no database and no network.
|
|
2
|
+
|
|
3
|
+
This is the interface dpf's ticket author calls, so it is a contract rather than
|
|
4
|
+
a convenience:
|
|
5
|
+
|
|
6
|
+
* `--json` emits a versioned envelope. `schema_version` changes only when the
|
|
7
|
+
shape changes incompatibly.
|
|
8
|
+
* Every answer carries the freshness of the data behind it. A caller can record
|
|
9
|
+
which snapshot informed a decision, which is the honest-broker requirement.
|
|
10
|
+
* Exit codes distinguish the cases a caller must tell apart:
|
|
11
|
+
|
|
12
|
+
0 an answer
|
|
13
|
+
1 a usage or runtime error
|
|
14
|
+
2 no result matched the constraints, which is a real answer
|
|
15
|
+
3 no snapshot; run `modelspec snapshot fetch`
|
|
16
|
+
4 the snapshot is stale and --require-fresh was given
|
|
17
|
+
5 a keyed origin refused the credential (MODEL-71)
|
|
18
|
+
6 a keyed origin rate-limited the credential (MODEL-71)
|
|
19
|
+
|
|
20
|
+
The pre-existing graph commands exit 0 when FalkorDB is unreachable, so a
|
|
21
|
+
script cannot tell an answer from a failure. Nothing here does that.
|
|
22
|
+
|
|
23
|
+
5 and 6 belong to `snapshot fetch` against an origin that keys its export.
|
|
24
|
+
No unkeyed origin produces them, so no existing call can start seeing one:
|
|
25
|
+
the free path against `https://modelspec.dev` still exits only 0 or 1.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import json
|
|
31
|
+
import sys
|
|
32
|
+
from typing import Any, NoReturn
|
|
33
|
+
|
|
34
|
+
import click
|
|
35
|
+
import typer
|
|
36
|
+
from typer.core import TyperCommand, TyperGroup
|
|
37
|
+
|
|
38
|
+
from . import snapshot as snap
|
|
39
|
+
|
|
40
|
+
#: Bumped only for an incompatible change to the JSON envelope.
|
|
41
|
+
SCHEMA_VERSION = "1.0"
|
|
42
|
+
|
|
43
|
+
EXIT_OK = 0
|
|
44
|
+
EXIT_ERROR = 1
|
|
45
|
+
EXIT_NO_MATCH = 2
|
|
46
|
+
EXIT_NO_SNAPSHOT = 3
|
|
47
|
+
EXIT_STALE = 4
|
|
48
|
+
#: Only reachable from `snapshot fetch` against a keyed origin. A rejected key
|
|
49
|
+
#: and a spent quota are different problems — one is fixed by a new key, the
|
|
50
|
+
#: other by waiting — so a script must be able to tell them apart without
|
|
51
|
+
#: reading prose off stderr.
|
|
52
|
+
EXIT_KEY_REFUSED = 5
|
|
53
|
+
EXIT_RATE_LIMITED = 6
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class ContractCommand(TyperCommand):
|
|
57
|
+
"""Make Click's syntax/option failures use the CLI's documented code 1."""
|
|
58
|
+
|
|
59
|
+
def parse_args(self, ctx: click.Context, args: list[str]) -> list[str]:
|
|
60
|
+
try:
|
|
61
|
+
return super().parse_args(ctx, args)
|
|
62
|
+
except Exception as exc: # noqa: BLE001 - Typer wraps Click usage errors
|
|
63
|
+
if getattr(exc, "exit_code", None) == 2:
|
|
64
|
+
exc.exit_code = EXIT_ERROR
|
|
65
|
+
raise
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class ContractGroup(TyperGroup):
|
|
69
|
+
"""Apply the same usage-error exit code while resolving subcommands."""
|
|
70
|
+
|
|
71
|
+
def parse_args(self, ctx: click.Context, args: list[str]) -> list[str]:
|
|
72
|
+
try:
|
|
73
|
+
return super().parse_args(ctx, args)
|
|
74
|
+
except Exception as exc: # noqa: BLE001 - Typer wraps Click usage errors
|
|
75
|
+
if getattr(exc, "exit_code", None) == 2:
|
|
76
|
+
exc.exit_code = EXIT_ERROR
|
|
77
|
+
raise
|
|
78
|
+
|
|
79
|
+
def resolve_command(self, ctx: click.Context, args: list[str]) -> Any:
|
|
80
|
+
try:
|
|
81
|
+
return super().resolve_command(ctx, args)
|
|
82
|
+
except Exception as exc: # noqa: BLE001 - Typer wraps Click usage errors
|
|
83
|
+
if getattr(exc, "exit_code", None) == 2:
|
|
84
|
+
exc.exit_code = EXIT_ERROR
|
|
85
|
+
raise
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
app = typer.Typer(
|
|
89
|
+
cls=ContractGroup,
|
|
90
|
+
help="Answer from the local snapshot. No database, no network, no credential.",
|
|
91
|
+
)
|
|
92
|
+
snapshot_app = typer.Typer(cls=ContractGroup, help="Manage the local snapshot.")
|
|
93
|
+
app.add_typer(snapshot_app, name="snapshot")
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _emit(payload: dict[str, Any], as_json: bool) -> None:
|
|
97
|
+
if as_json:
|
|
98
|
+
typer.echo(json.dumps(payload, indent=2, default=str))
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _envelope(command: str, snapshot: snap.Snapshot, result: Any,
|
|
102
|
+
**metadata: Any) -> dict[str, Any]:
|
|
103
|
+
return {
|
|
104
|
+
"schema_version": SCHEMA_VERSION,
|
|
105
|
+
"command": command,
|
|
106
|
+
"freshness": snapshot.freshness(),
|
|
107
|
+
"result": result,
|
|
108
|
+
**metadata,
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _emit_error(command: str, message: str, as_json: bool) -> None:
|
|
113
|
+
if as_json:
|
|
114
|
+
typer.echo(json.dumps({
|
|
115
|
+
"schema_version": SCHEMA_VERSION,
|
|
116
|
+
"command": command,
|
|
117
|
+
"error": {"message": message},
|
|
118
|
+
}, indent=2), err=True)
|
|
119
|
+
else:
|
|
120
|
+
typer.echo(f"error: {message}", err=True)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _load_or_exit(require_fresh: bool, *, command: str, as_json: bool = False) -> snap.Snapshot:
|
|
124
|
+
try:
|
|
125
|
+
snapshot = snap.load()
|
|
126
|
+
except snap.SnapshotMissing as exc:
|
|
127
|
+
_emit_error(command, str(exc), as_json)
|
|
128
|
+
raise typer.Exit(EXIT_NO_SNAPSHOT) from exc
|
|
129
|
+
except snap.SnapshotInvalid as exc:
|
|
130
|
+
_emit_error(command, str(exc), as_json)
|
|
131
|
+
raise typer.Exit(EXIT_ERROR) from exc
|
|
132
|
+
if require_fresh and snapshot.is_stale:
|
|
133
|
+
_emit_error(
|
|
134
|
+
command,
|
|
135
|
+
f"snapshot is {snapshot.age_days:.0f} days old and --require-fresh was given. "
|
|
136
|
+
"Run `modelspec snapshot fetch`.",
|
|
137
|
+
as_json,
|
|
138
|
+
)
|
|
139
|
+
raise typer.Exit(EXIT_STALE)
|
|
140
|
+
if snapshot.is_stale:
|
|
141
|
+
typer.echo(
|
|
142
|
+
f"warning: snapshot is {snapshot.age_days:.0f} days old "
|
|
143
|
+
f"(stale after {snap.STALE_AFTER_DAYS}). Answers may be out of date.", err=True)
|
|
144
|
+
return snapshot
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _candidates(snapshot: snap.Snapshot) -> list[Any]:
|
|
148
|
+
"""Rebuild ranking records from the snapshot, reusing the tested scorer."""
|
|
149
|
+
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parents[2]))
|
|
150
|
+
from pipeline.ranking import Candidate
|
|
151
|
+
|
|
152
|
+
return [
|
|
153
|
+
Candidate(
|
|
154
|
+
model_id=c["model_id"], display_name=c["display_name"], provider=c["provider"],
|
|
155
|
+
model_type=c.get("model_type"), model_subtypes=c.get("model_subtypes") or [],
|
|
156
|
+
benchmark_scores=c.get("benchmark_scores") or {},
|
|
157
|
+
capability_tiers=c.get("capability_tiers") or {},
|
|
158
|
+
cost_input=c.get("cost_input"), context_window=c.get("context_window"),
|
|
159
|
+
open_weights=bool(c.get("open_weights")), scores_as_of=c.get("scores_as_of"),
|
|
160
|
+
fits=c.get("fits") or {},
|
|
161
|
+
verified_benchmarks=set(c.get("verified_benchmarks") or []),
|
|
162
|
+
rehost_of=c.get("rehost_of"), release_date=c.get("release_date"),
|
|
163
|
+
)
|
|
164
|
+
for c in snapshot.data["candidates"]["candidates"]
|
|
165
|
+
]
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _not_ranked_yet(block: dict[str, Any]) -> str | None:
|
|
169
|
+
"""One line under the rank table naming what the ranking could not order (MODEL-110)."""
|
|
170
|
+
count = block["count"]
|
|
171
|
+
if not count:
|
|
172
|
+
return None
|
|
173
|
+
named = [m["model_id"] for m in block["models"]]
|
|
174
|
+
rest = count - len(named)
|
|
175
|
+
head = (f"{count} model is not ranked yet" if count == 1
|
|
176
|
+
else f"{count} models are not ranked yet")
|
|
177
|
+
tail = f", and {rest} more." if rest else "."
|
|
178
|
+
return (f"{head} (not enough benchmark evidence), newest first: "
|
|
179
|
+
f"{', '.join(named)}{tail}")
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
@snapshot_app.command("fetch", cls=ContractCommand)
|
|
183
|
+
def snapshot_fetch(
|
|
184
|
+
origin: str = typer.Option(snap.DEFAULT_ORIGIN, help="Where to fetch from."),
|
|
185
|
+
api_key: str = typer.Option(
|
|
186
|
+
None, "--api-key",
|
|
187
|
+
help=f"Credential for an origin that keys its export. Prefer the "
|
|
188
|
+
f"{snap.API_KEY_ENV} environment variable: a key in argv is visible "
|
|
189
|
+
f"in shell history and in `ps`."),
|
|
190
|
+
as_json: bool = typer.Option(False, "--json", help="Machine-readable output."),
|
|
191
|
+
) -> None:
|
|
192
|
+
"""Download the published export. The only command that needs the network.
|
|
193
|
+
|
|
194
|
+
Unkeyed by default. With a credential — `MODELSPEC_API_KEY`, or `--api-key`
|
|
195
|
+
— the same command fetches from an origin that keys its export, and the
|
|
196
|
+
snapshot it writes is current rather than 90 days delayed. Nothing about
|
|
197
|
+
the snapshot, the envelope or any other command changes; only `fetched_at`.
|
|
198
|
+
"""
|
|
199
|
+
credential = snap.resolve_credential(api_key)
|
|
200
|
+
|
|
201
|
+
def fail(message: str, code: int) -> NoReturn:
|
|
202
|
+
# The backstop, not the plan: no message built above interpolates a key.
|
|
203
|
+
typer.echo(f"error: {credential.redact(message) if credential else message}",
|
|
204
|
+
err=True)
|
|
205
|
+
raise typer.Exit(code)
|
|
206
|
+
|
|
207
|
+
if credential is not None and credential.source == "flag":
|
|
208
|
+
typer.echo(
|
|
209
|
+
f"warning: --api-key is visible in shell history and in process "
|
|
210
|
+
f"listings. Prefer {snap.API_KEY_ENV} in the environment.", err=True)
|
|
211
|
+
try:
|
|
212
|
+
result = snap.fetch(origin, credential=credential)
|
|
213
|
+
except snap.KeyRequiredError as exc:
|
|
214
|
+
fail(str(exc), EXIT_KEY_REFUSED)
|
|
215
|
+
except snap.KeyRejectedError as exc:
|
|
216
|
+
fail(str(exc), EXIT_KEY_REFUSED)
|
|
217
|
+
except snap.RateLimitedError as exc:
|
|
218
|
+
fail(str(exc), EXIT_RATE_LIMITED)
|
|
219
|
+
except Exception as exc: # noqa: BLE001 - surfaced to the user, not swallowed
|
|
220
|
+
# Unchanged for every failure that is not a credential one, including an
|
|
221
|
+
# unreachable origin: same sentence, same exit code as before MODEL-71.
|
|
222
|
+
fail(f"could not fetch the snapshot: {exc}", EXIT_ERROR)
|
|
223
|
+
decision = result.decision_fetch or {"available": False, "error": "not attempted"}
|
|
224
|
+
if as_json:
|
|
225
|
+
typer.echo(json.dumps({
|
|
226
|
+
"schema_version": SCHEMA_VERSION,
|
|
227
|
+
"command": "fetch",
|
|
228
|
+
"freshness": result.freshness(),
|
|
229
|
+
"result": {
|
|
230
|
+
"model_count": len(result.data["candidates"]["candidates"]),
|
|
231
|
+
"path": str(result.path),
|
|
232
|
+
"decision_snapshot": decision,
|
|
233
|
+
},
|
|
234
|
+
}, indent=2))
|
|
235
|
+
return
|
|
236
|
+
typer.echo(f"fetched {len(result.data['candidates']['candidates'])} models "
|
|
237
|
+
f"from {origin} (build {result.build_commit[:12]})")
|
|
238
|
+
typer.echo(f"cached at {result.path}")
|
|
239
|
+
if decision["available"]:
|
|
240
|
+
typer.echo(f"decision cached from {decision['origin']}")
|
|
241
|
+
else:
|
|
242
|
+
typer.echo(f"decision unavailable: {decision['error']}")
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
@snapshot_app.command("status", cls=ContractCommand)
|
|
246
|
+
def snapshot_status(as_json: bool = typer.Option(False, "--json")) -> None:
|
|
247
|
+
"""Show what snapshot is cached and how old it is."""
|
|
248
|
+
try:
|
|
249
|
+
info = snap.status()
|
|
250
|
+
except snap.SnapshotInvalid as exc:
|
|
251
|
+
_emit_error("status", str(exc), as_json)
|
|
252
|
+
raise typer.Exit(EXIT_ERROR) from exc
|
|
253
|
+
if as_json:
|
|
254
|
+
freshness_keys = ("fetched_at", "age_days", "stale", "stale_after_days",
|
|
255
|
+
"origin", "build_commit", "built_at")
|
|
256
|
+
freshness = {key: info[key] for key in freshness_keys if key in info} or None
|
|
257
|
+
result = {key: value for key, value in info.items() if key not in freshness_keys}
|
|
258
|
+
typer.echo(json.dumps({
|
|
259
|
+
"schema_version": SCHEMA_VERSION,
|
|
260
|
+
"command": "status",
|
|
261
|
+
"freshness": freshness,
|
|
262
|
+
"result": result,
|
|
263
|
+
}, indent=2))
|
|
264
|
+
elif not info["present"]:
|
|
265
|
+
typer.echo(info["message"])
|
|
266
|
+
else:
|
|
267
|
+
typer.echo(f"snapshot {info['path']}")
|
|
268
|
+
typer.echo(f"fetched {info['fetched_at']} ({info['age_days']:.1f} days ago)")
|
|
269
|
+
typer.echo(f"build {info['build_commit'][:12]} from {info['origin']}")
|
|
270
|
+
typer.echo(f"stale {info['stale']}")
|
|
271
|
+
decision = info["decision_snapshot"]
|
|
272
|
+
if decision["present"] and not decision.get("valid", True):
|
|
273
|
+
typer.echo(f"decision invalid: {decision['error']}")
|
|
274
|
+
elif decision["present"]:
|
|
275
|
+
typer.echo(f"decision {decision['snapshot_id']} as of {decision['as_of']} "
|
|
276
|
+
f"({decision['age_days']:.1f} days cached; signature unverified)")
|
|
277
|
+
else:
|
|
278
|
+
typer.echo("decision not cached")
|
|
279
|
+
if not info["present"]:
|
|
280
|
+
raise typer.Exit(EXIT_NO_SNAPSHOT)
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
@app.command("rank", cls=ContractCommand)
|
|
284
|
+
def rank_offline(
|
|
285
|
+
use_case: str = typer.Argument(..., help="Use-case profile, e.g. coding."),
|
|
286
|
+
limit: int = typer.Option(10, "--limit", "-n"),
|
|
287
|
+
open_weights: bool = typer.Option(
|
|
288
|
+
False, "--open-weights", help="Only models you can download."
|
|
289
|
+
),
|
|
290
|
+
fits: str = typer.Option(None, "--fits", help="Hardware id the model must fit on."),
|
|
291
|
+
max_cost: float = typer.Option(None, "--max-cost", help="Maximum $ per million input tokens."),
|
|
292
|
+
price_sensitivity: float = typer.Option(
|
|
293
|
+
0.0, "--price-sensitivity",
|
|
294
|
+
help="0 ignores price; 0.25 weighs it heavily. Profiles ignore price by default."),
|
|
295
|
+
as_json: bool = typer.Option(False, "--json", help="Machine-readable output."),
|
|
296
|
+
require_fresh: bool = typer.Option(False, "--require-fresh", help="Fail on a stale snapshot."),
|
|
297
|
+
include_rehosts: bool = typer.Option(
|
|
298
|
+
False, "--include-rehosts", help="Keep repackaged copies of another model's weights."),
|
|
299
|
+
) -> None:
|
|
300
|
+
"""Rank models for a use case, entirely offline."""
|
|
301
|
+
if limit < 0:
|
|
302
|
+
_emit_error("rank", "--limit must be nonnegative", as_json)
|
|
303
|
+
raise typer.Exit(EXIT_ERROR)
|
|
304
|
+
if not 0.0 <= price_sensitivity <= 1.0:
|
|
305
|
+
_emit_error("rank", "--price-sensitivity must be between 0 and 1", as_json)
|
|
306
|
+
raise typer.Exit(EXIT_ERROR)
|
|
307
|
+
snapshot = _load_or_exit(require_fresh, command="rank", as_json=as_json)
|
|
308
|
+
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parents[2]))
|
|
309
|
+
from pipeline.ranking import rank_report
|
|
310
|
+
|
|
311
|
+
profiles = snapshot.data["profiles"]["profiles"]
|
|
312
|
+
if use_case not in profiles:
|
|
313
|
+
featured = ", ".join(snapshot.data["profiles"].get("featured") or [])
|
|
314
|
+
_emit_error("rank", f"unknown use case {use_case!r}. Try one of: {featured}", as_json)
|
|
315
|
+
raise typer.Exit(EXIT_ERROR)
|
|
316
|
+
|
|
317
|
+
if fits:
|
|
318
|
+
# A typo'd device id must not read as "nothing fits your GPU". That is a
|
|
319
|
+
# different answer, and a caller would act on it.
|
|
320
|
+
known = {n["id"] for n in snapshot.data["hardware"].get("nodes", [])
|
|
321
|
+
if n.get("label") == "Hardware"}
|
|
322
|
+
if fits not in known:
|
|
323
|
+
_emit_error("rank", f"unknown device {fits!r}. Run `modelspec offline fit` "
|
|
324
|
+
"with no argument to list them.", as_json)
|
|
325
|
+
raise typer.Exit(EXIT_ERROR)
|
|
326
|
+
|
|
327
|
+
try:
|
|
328
|
+
pool = _candidates(snapshot)
|
|
329
|
+
if max_cost is not None:
|
|
330
|
+
pool = [c for c in pool if c.cost_input is not None and c.cost_input <= max_cost]
|
|
331
|
+
|
|
332
|
+
report = rank_report(pool, use_case, limit=limit, open_weights_only=open_weights,
|
|
333
|
+
hardware_id=fits, cost_weight=price_sensitivity or None,
|
|
334
|
+
include_rehosts=include_rehosts)
|
|
335
|
+
except Exception as exc: # noqa: BLE001 - CLI must not leak a traceback to callers
|
|
336
|
+
_emit_error("rank", f"could not rank the snapshot: {exc}", as_json)
|
|
337
|
+
raise typer.Exit(EXIT_ERROR) from exc
|
|
338
|
+
results = report["ranked"]
|
|
339
|
+
ranking_status = report["ranking_status"]
|
|
340
|
+
ranked_count = report["ranked_count"]
|
|
341
|
+
unranked_count = report["unranked_count"]
|
|
342
|
+
|
|
343
|
+
if as_json:
|
|
344
|
+
typer.echo(json.dumps(_envelope(
|
|
345
|
+
"rank", snapshot, results,
|
|
346
|
+
ranking_status=ranking_status,
|
|
347
|
+
ranked_count=ranked_count,
|
|
348
|
+
unranked_count=unranked_count,
|
|
349
|
+
# MODEL-110. A new envelope field, always present; additive under 1.0.
|
|
350
|
+
unranked_candidates=report["unranked_candidates"],
|
|
351
|
+
), indent=2, default=str))
|
|
352
|
+
else:
|
|
353
|
+
typer.echo(f"{use_case}: {ranking_status} ordering; "
|
|
354
|
+
f"{ranked_count} ranked, {unranked_count} unranked.")
|
|
355
|
+
for i, r in enumerate(results, 1):
|
|
356
|
+
typer.echo(f"{i:>3}. {r['score']:>6.2f} {r['display_name']} ({r['provider']})")
|
|
357
|
+
if ranking_status == "empty":
|
|
358
|
+
typer.echo("No model matches those constraints.")
|
|
359
|
+
elif ranking_status == "unavailable":
|
|
360
|
+
typer.echo("No model has enough evidence to be ranked.")
|
|
361
|
+
withheld = _not_ranked_yet(report["unranked_candidates"])
|
|
362
|
+
if withheld:
|
|
363
|
+
typer.echo(f"\n{withheld}")
|
|
364
|
+
typer.echo(f"\nfrom a snapshot {snapshot.age_days:.0f} days old, "
|
|
365
|
+
f"build {snapshot.build_commit[:12]}")
|
|
366
|
+
|
|
367
|
+
if ranking_status in {"empty", "unavailable"}:
|
|
368
|
+
raise typer.Exit(EXIT_NO_MATCH)
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
@app.command("class-fit", cls=ContractCommand)
|
|
372
|
+
def class_fit_offline(
|
|
373
|
+
task: str = typer.Argument(
|
|
374
|
+
None, help="What the system has to do, in prose. Matched against the "
|
|
375
|
+
"published terms and discarded; never sent anywhere."),
|
|
376
|
+
emits: str = typer.Option(
|
|
377
|
+
None, "--emits", help="What the model must produce, e.g. choice, open_text."),
|
|
378
|
+
consumes: str = typer.Option(
|
|
379
|
+
None, "--consumes", help="Comma-separated input kinds, e.g. structured_state,text."),
|
|
380
|
+
decides: str = typer.Option(
|
|
381
|
+
None, "--decides", help="The decision it must make. Strict: no adaptation."),
|
|
382
|
+
as_json: bool = typer.Option(False, "--json", help="Machine-readable output."),
|
|
383
|
+
require_fresh: bool = typer.Option(False, "--require-fresh"),
|
|
384
|
+
) -> None:
|
|
385
|
+
"""Which *class* of model a task needs, before ranking within one.
|
|
386
|
+
|
|
387
|
+
Ranking answers "which model?" once you have decided you want an LLM. This
|
|
388
|
+
answers the question before that one, and it refuses rather than guessing:
|
|
389
|
+
when two classes both survive, it says so and hands back the question you
|
|
390
|
+
have to settle, because ModelSpec holds no measurement that orders one
|
|
391
|
+
class against another.
|
|
392
|
+
"""
|
|
393
|
+
snapshot = _load_or_exit(require_fresh, command="class-fit", as_json=as_json)
|
|
394
|
+
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parents[2]))
|
|
395
|
+
from api.class_fit import CatalogueEvidence, class_fit
|
|
396
|
+
from pipeline.class_export import counts_from_snapshot_candidates
|
|
397
|
+
|
|
398
|
+
counts, examples = counts_from_snapshot_candidates(
|
|
399
|
+
snapshot.data["candidates"]["candidates"])
|
|
400
|
+
answer = class_fit(
|
|
401
|
+
task=task,
|
|
402
|
+
emits=emits,
|
|
403
|
+
consumes=[c.strip() for c in consumes.split(",") if c.strip()] if consumes else None,
|
|
404
|
+
decides=decides,
|
|
405
|
+
evidence=CatalogueEvidence(card_counts=counts, examples=examples),
|
|
406
|
+
)
|
|
407
|
+
|
|
408
|
+
if as_json:
|
|
409
|
+
typer.echo(json.dumps(_envelope("class-fit", snapshot, answer,
|
|
410
|
+
fit_status=answer["fit_status"]),
|
|
411
|
+
indent=2, default=str))
|
|
412
|
+
else:
|
|
413
|
+
status = answer["fit_status"]
|
|
414
|
+
if status == "refused":
|
|
415
|
+
typer.echo(f"refused: {answer['refusal']['code']} — "
|
|
416
|
+
f"{answer['refusal']['message']}")
|
|
417
|
+
else:
|
|
418
|
+
typer.echo(f"{status}: {len(answer['candidates'])} candidate class(es). "
|
|
419
|
+
"No score orders them.")
|
|
420
|
+
for row in answer["candidates"]:
|
|
421
|
+
catalogue = row["catalogue"]
|
|
422
|
+
count = catalogue.get("card_count")
|
|
423
|
+
held = "evidence not supplied" if count is None else f"{count} card(s)"
|
|
424
|
+
typer.echo(f"\n {row['class']} — consumes {', '.join(row['consumes'])}; "
|
|
425
|
+
f"emits {row['emits']}; decides {row['decides']} [{held}]")
|
|
426
|
+
typer.echo(f" {row['because']}")
|
|
427
|
+
if row["emits_adapted"]:
|
|
428
|
+
typer.echo(f" only via an adaptation: {row['emits_adapted']['how']}")
|
|
429
|
+
typer.echo(f" next: {row['next']}")
|
|
430
|
+
for question in answer["distinguishing_questions"]:
|
|
431
|
+
typer.echo(f"\n you must settle: {question['ask']}")
|
|
432
|
+
for chain in answer["composition"]:
|
|
433
|
+
typer.echo(f"\n they compose: {' then '.join(chain['sequence'])} "
|
|
434
|
+
f"(evidence: {chain['evidence']} — {chain['see']})")
|
|
435
|
+
if status == "unavailable":
|
|
436
|
+
typer.echo("\nNo class in the published taxonomy produces that. "
|
|
437
|
+
"See the `emits` vocabulary in /api/rank/class-fit.json.")
|
|
438
|
+
typer.echo(f"\nfrom a snapshot {snapshot.age_days:.0f} days old, "
|
|
439
|
+
f"build {snapshot.build_commit[:12]}")
|
|
440
|
+
|
|
441
|
+
if answer["fit_status"] == "refused":
|
|
442
|
+
raise typer.Exit(EXIT_ERROR)
|
|
443
|
+
if answer["fit_status"] == "unavailable":
|
|
444
|
+
raise typer.Exit(EXIT_NO_MATCH)
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
@app.command("fit", cls=ContractCommand)
|
|
448
|
+
def fit_offline(
|
|
449
|
+
hardware: str = typer.Argument(None, help="Hardware id. Omit to list the devices."),
|
|
450
|
+
limit: int = typer.Option(20, "--limit", "-n"),
|
|
451
|
+
as_json: bool = typer.Option(False, "--json"),
|
|
452
|
+
require_fresh: bool = typer.Option(False, "--require-fresh"),
|
|
453
|
+
include_rehosts: bool = typer.Option(
|
|
454
|
+
False, "--include-rehosts", help="Keep repackaged copies of another model's weights."),
|
|
455
|
+
host: str = typer.Option(
|
|
456
|
+
None, "--host", help="Host profile id (hosts.json). Adds fit_state to every row."),
|
|
457
|
+
host_ram: float = typer.Option(
|
|
458
|
+
None, "--host-ram",
|
|
459
|
+
help="GB of RAM on this machine, before the fixed 8 GB OS reserve. Needs --host."),
|
|
460
|
+
include_offload: bool = typer.Option(
|
|
461
|
+
False, "--include-offload",
|
|
462
|
+
help="Append a separate offload tier: models that fit only by spilling to host RAM. "
|
|
463
|
+
"Needs --host."),
|
|
464
|
+
) -> None:
|
|
465
|
+
"""What can this machine actually run?"""
|
|
466
|
+
if limit < 0:
|
|
467
|
+
_emit_error("fit", "--limit must be nonnegative", as_json)
|
|
468
|
+
raise typer.Exit(EXIT_ERROR)
|
|
469
|
+
if include_offload and not host:
|
|
470
|
+
_emit_error("fit", "--include-offload requires --host", as_json)
|
|
471
|
+
raise typer.Exit(EXIT_ERROR)
|
|
472
|
+
if host_ram is not None and not host:
|
|
473
|
+
_emit_error("fit", "--host-ram requires --host", as_json)
|
|
474
|
+
raise typer.Exit(EXIT_ERROR)
|
|
475
|
+
if host_ram is not None and host_ram <= 0:
|
|
476
|
+
_emit_error("fit", "--host-ram must be positive", as_json)
|
|
477
|
+
raise typer.Exit(EXIT_ERROR)
|
|
478
|
+
snapshot = _load_or_exit(require_fresh, command="fit", as_json=as_json)
|
|
479
|
+
devices = [
|
|
480
|
+
n for n in snapshot.data["hardware"].get("nodes", []) if n.get("label") == "Hardware"
|
|
481
|
+
]
|
|
482
|
+
|
|
483
|
+
if not hardware:
|
|
484
|
+
rows = sorted(devices, key=lambda d: -(d.get("memory_bandwidth_gb_s") or 0))
|
|
485
|
+
if as_json:
|
|
486
|
+
typer.echo(json.dumps(_envelope("fit", snapshot, rows), indent=2, default=str))
|
|
487
|
+
else:
|
|
488
|
+
for d in rows:
|
|
489
|
+
typer.echo(
|
|
490
|
+
f" {d['id']:<28} {d.get('memory_gb', '?'):>6} GB "
|
|
491
|
+
f"{d.get('memory_bandwidth_gb_s', '?'):>6} GB/s {d.get('display_name', '')}"
|
|
492
|
+
)
|
|
493
|
+
return
|
|
494
|
+
|
|
495
|
+
known = {d["id"] for d in devices}
|
|
496
|
+
if hardware not in known:
|
|
497
|
+
_emit_error("fit", f"unknown device {hardware!r}. Run `modelspec offline fit` "
|
|
498
|
+
"with no argument to list them.", as_json)
|
|
499
|
+
raise typer.Exit(EXIT_ERROR)
|
|
500
|
+
|
|
501
|
+
host_profile = None
|
|
502
|
+
if host:
|
|
503
|
+
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parents[2]))
|
|
504
|
+
from pipeline import hosts as host_layer
|
|
505
|
+
|
|
506
|
+
profiles = {h.get("id"): h for h in (snapshot.data.get("hosts") or {}).get("hosts") or []}
|
|
507
|
+
if not profiles:
|
|
508
|
+
_emit_error("fit", "this snapshot has no host profiles (an export older than "
|
|
509
|
+
"MODEL-26 phase B). Run `modelspec snapshot fetch`.", as_json)
|
|
510
|
+
raise typer.Exit(EXIT_ERROR)
|
|
511
|
+
if host not in profiles:
|
|
512
|
+
_emit_error("fit", f"unknown host {host!r}. Known hosts: "
|
|
513
|
+
f"{', '.join(sorted(profiles))}", as_json)
|
|
514
|
+
raise typer.Exit(EXIT_ERROR)
|
|
515
|
+
try:
|
|
516
|
+
host_profile = host_layer.host_from_raw(profiles[host])
|
|
517
|
+
except Exception as exc: # noqa: BLE001 - CLI must not leak a traceback to callers
|
|
518
|
+
_emit_error("fit", f"host profile {host!r} is unreadable: {exc}", as_json)
|
|
519
|
+
raise typer.Exit(EXIT_ERROR) from exc
|
|
520
|
+
|
|
521
|
+
try:
|
|
522
|
+
pool = [c for c in _candidates(snapshot) if hardware in c.fits
|
|
523
|
+
and (include_rehosts or not c.rehost_of)]
|
|
524
|
+
except Exception as exc: # noqa: BLE001 - CLI must not leak a traceback to callers
|
|
525
|
+
_emit_error("fit", f"could not read the snapshot: {exc}", as_json)
|
|
526
|
+
raise typer.Exit(EXIT_ERROR) from exc
|
|
527
|
+
# A non-token model (or, rarely, a token model missing KV geometry) fits
|
|
528
|
+
# in memory but has no decode-speed prediction. It must not crash the
|
|
529
|
+
# sort (None has no ordering against a float) and must not lead the list
|
|
530
|
+
# by an absent rate: real rates first, fastest first; null-decode rows
|
|
531
|
+
# follow, alphabetically (MODEL-53).
|
|
532
|
+
pool.sort(key=lambda c: (
|
|
533
|
+
c.fits[hardware] is None,
|
|
534
|
+
-c.fits[hardware] if c.fits[hardware] is not None else 0.0,
|
|
535
|
+
c.display_name.lower(),
|
|
536
|
+
))
|
|
537
|
+
results = [{"model_id": c.model_id, "display_name": c.display_name,
|
|
538
|
+
"predicted_decode_tps": c.fits[hardware],
|
|
539
|
+
"prediction_basis": "computed, not measured"}
|
|
540
|
+
for c in pool[:limit]]
|
|
541
|
+
|
|
542
|
+
# Without --host nothing below runs, so the output is byte-identical to the
|
|
543
|
+
# pre-host CLI (MODEL-26 decision 1).
|
|
544
|
+
offload: list[dict[str, Any]] = []
|
|
545
|
+
if host_profile is not None:
|
|
546
|
+
for r in results:
|
|
547
|
+
r.update({
|
|
548
|
+
"fit_state": "accelerator", "offload_fraction": 0.0, "host_id": host,
|
|
549
|
+
"predicted_decode_tps_basis": (
|
|
550
|
+
"accelerator-roofline" if r["predicted_decode_tps"] is not None else None),
|
|
551
|
+
})
|
|
552
|
+
if include_offload and not host_profile.unified:
|
|
553
|
+
offload = _offload_tier(snapshot, hardware, host_profile, host_ram,
|
|
554
|
+
include_rehosts, limit)
|
|
555
|
+
|
|
556
|
+
if as_json:
|
|
557
|
+
typer.echo(json.dumps(_envelope("fit", snapshot, results + offload),
|
|
558
|
+
indent=2, default=str))
|
|
559
|
+
elif not results and not offload:
|
|
560
|
+
typer.echo(f"Nothing in the catalogue fits {hardware}.")
|
|
561
|
+
else:
|
|
562
|
+
for r in results:
|
|
563
|
+
tps = r["predicted_decode_tps"]
|
|
564
|
+
rate = f"~{tps:>7.1f}" if tps is not None else f"{'n/a':>8}"
|
|
565
|
+
typer.echo(f" {rate} tok/s {r['display_name']}")
|
|
566
|
+
if offload:
|
|
567
|
+
typer.echo(f"\noffload tier on {host} (spills to host RAM; "
|
|
568
|
+
"not ranked with the rows above):")
|
|
569
|
+
for r in offload:
|
|
570
|
+
tps = r["predicted_decode_tps"]
|
|
571
|
+
rate = f"~{tps:>7.1f}" if tps is not None else f"{'n/a':>8}"
|
|
572
|
+
typer.echo(f" {rate} tok/s {r['display_name']} "
|
|
573
|
+
f"({r['offload_fraction']:.0%} offloaded, {r['quantization']})")
|
|
574
|
+
typer.echo("\npredicted from memory bandwidth, not measured; "
|
|
575
|
+
"n/a means the weights fit but the model does not decode tokens")
|
|
576
|
+
|
|
577
|
+
if not results and not offload:
|
|
578
|
+
raise typer.Exit(EXIT_NO_MATCH)
|
|
579
|
+
|
|
580
|
+
|
|
581
|
+
def _offload_tier(snapshot: snap.Snapshot, hardware: str, host: Any,
|
|
582
|
+
host_ram: float | None, include_rehosts: bool,
|
|
583
|
+
limit: int) -> list[dict[str, Any]]:
|
|
584
|
+
"""Models that fit `hardware` only by spilling to `host` RAM, fastest first.
|
|
585
|
+
|
|
586
|
+
A separate tier, never interleaved with accelerator rows (decision 2).
|
|
587
|
+
"""
|
|
588
|
+
from pipeline import hosts as host_layer
|
|
589
|
+
|
|
590
|
+
device = next(n for n in snapshot.data["hardware"]["nodes"]
|
|
591
|
+
if n.get("label") == "Hardware" and n.get("id") == hardware)
|
|
592
|
+
capacity = device.get("memory_gb")
|
|
593
|
+
bandwidth = device.get("memory_bandwidth_gb_s")
|
|
594
|
+
raw = snapshot.data["candidates"]["candidates"]
|
|
595
|
+
# A device that answers no single-device fit at all (single_device_fit:
|
|
596
|
+
# false) must not grow an offload tier either.
|
|
597
|
+
if not capacity or not bandwidth or not any(hardware in (c.get("fits") or {}) for c in raw):
|
|
598
|
+
return []
|
|
599
|
+
rows = []
|
|
600
|
+
for c in raw:
|
|
601
|
+
if not c.get("open_weights") or hardware in (c.get("fits") or {}):
|
|
602
|
+
continue
|
|
603
|
+
if c.get("rehost_of") and not include_rehosts:
|
|
604
|
+
continue
|
|
605
|
+
if not c.get("total_parameters"):
|
|
606
|
+
continue
|
|
607
|
+
a = host_layer.assess(
|
|
608
|
+
total_params=float(c["total_parameters"]),
|
|
609
|
+
active_params=float(c["active_parameters"]) if c.get("active_parameters") else None,
|
|
610
|
+
model_type=c.get("model_type"), accelerator_gb=float(capacity),
|
|
611
|
+
accelerator_bandwidth=float(bandwidth), host=host, host_ram_gb=host_ram)
|
|
612
|
+
if a["fit_state"] != host_layer.FIT_OFFLOAD:
|
|
613
|
+
continue
|
|
614
|
+
rows.append({"model_id": c["model_id"], "display_name": c["display_name"],
|
|
615
|
+
"predicted_decode_tps": a["predicted_decode_tps"],
|
|
616
|
+
"prediction_basis": "computed, not measured",
|
|
617
|
+
"fit_state": a["fit_state"], "offload_fraction": a["offload_fraction"],
|
|
618
|
+
"host_id": host.id,
|
|
619
|
+
"predicted_decode_tps_basis": a["predicted_decode_tps_basis"],
|
|
620
|
+
"quantization": a["quantization"]})
|
|
621
|
+
rows.sort(key=lambda r: (r["predicted_decode_tps"] is None,
|
|
622
|
+
-(r["predicted_decode_tps"] or 0.0), r["display_name"].lower()))
|
|
623
|
+
return rows[:limit]
|