modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
pipeline/ranking.py ADDED
@@ -0,0 +1,551 @@
1
+ """Rank models against a use-case profile, without a database.
2
+
3
+ `api/ranking/engine.py` holds the profiles, the benchmark normalisation ranges
4
+ and the scoring constants. That module imports cleanly with no FalkorDB
5
+ dependency, so this reuses its tables and helpers directly rather than
6
+ transcribing them — a transcribed copy of 51 profiles and 170 benchmark ranges
7
+ would drift on the first edit.
8
+
9
+ What is reimplemented here is only the scoring itself, against a model built
10
+ from a card rather than from a Cypher row. The formulas follow
11
+ `RankingEngine._score` exactly: benchmarks to 40 points, capabilities to 20,
12
+ cost and context scaled by the profile's weights, a type-match bonus to 15.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import math
18
+ import re
19
+ from dataclasses import dataclass, field
20
+ from typing import TYPE_CHECKING, Any
21
+
22
+ from api.ranking.engine import (
23
+ BENCHMARK_RANGES,
24
+ USE_CASE_PROFILES,
25
+ RANKING_POLICY,
26
+ WIZARD_BENCHMARK_COVERAGE,
27
+ _benchmark_evidence,
28
+ _ranking_status,
29
+ _tier_points,
30
+ _tier_rank,
31
+ ranking_policy,
32
+ )
33
+
34
+ if TYPE_CHECKING: # pragma: no cover - annotations only
35
+ # Only `build_candidates` and `write_export` take a sink, and neither runs
36
+ # on a serving path. `from __future__ import annotations` already makes the
37
+ # annotation a string, so deferring the import costs nothing here and buys
38
+ # a great deal: this module, and therefore the whole scorer, imports with
39
+ # nothing but the standard library and `api.ranking.engine`. The Cloudflare
40
+ # rank Worker (MODEL-68) runs this exact file, and pydantic is not
41
+ # available to it. See docs/rank-api.md.
42
+ from schema.graph import CollectingSink
43
+
44
+ #: Profiles offered in the wizard. All 51 are exported for the API, but a
45
+ #: dropdown of 51 is a worse experience than a short list. speech_to_text
46
+ #: currently ranks nothing (MODEL-30); offered-and-empty is worse than hiding
47
+ #: it. image_generation stays out until clip_score's range is sourced.
48
+ FEATURED_PROFILES = (
49
+ "general", "coding", "reasoning", "chat", "agentic", "rag",
50
+ "vision", "multilingual", "math_competition", "writing_technical",
51
+ "summarization", "embedding", "text_to_speech",
52
+ )
53
+
54
+
55
+ @dataclass
56
+ class Candidate:
57
+ model_id: str
58
+ display_name: str
59
+ provider: str
60
+ model_type: str | None
61
+ model_subtypes: list[str] = field(default_factory=list)
62
+ benchmark_scores: dict[str, float] = field(default_factory=dict)
63
+ capability_tiers: dict[str, str] = field(default_factory=dict)
64
+ cost_input: float | None = None
65
+ context_window: int | None = None
66
+ open_weights: bool = False
67
+ scores_as_of: str | None = None
68
+ #: hardware id -> predicted decode tok/s, or None when the weights fit
69
+ #: but the model does not decode tokens (or, rarely, geometry is missing).
70
+ fits: dict[str, float | None] = field(default_factory=dict)
71
+ #: Benchmarks whose score came from a reviewed evidence record rather than
72
+ #: the card's undated flat block. A ranking is only as good as the weakest
73
+ #: evidence under it, so this is tracked per benchmark, not per model.
74
+ verified_benchmarks: set[str] = field(default_factory=set)
75
+ #: Canonical model id when this card re-hosts the same weights
76
+ #: (`lineage.base_model_relation: repackaged`). Default fit pools and the
77
+ #: featured rankings leave these out so one weight set has one id (MODEL-54).
78
+ rehost_of: str | None = None
79
+ #: Parameter counts, so a consumer can size a model that does not fit an
80
+ #: accelerator (offload fit, MODEL-26). Additive to candidates.json.
81
+ total_parameters: float | None = None
82
+ active_parameters: float | None = None
83
+ #: `identity.release_date` as the card has it, or None. Orders the
84
+ #: `unranked_candidates` disclosure (MODEL-110); never scored. Additive to
85
+ #: candidates.json: an export without it reads as None everywhere.
86
+ release_date: str | None = None
87
+
88
+ def to_json(self) -> dict[str, Any]:
89
+ return {
90
+ "model_id": self.model_id, "display_name": self.display_name,
91
+ "provider": self.provider, "model_type": self.model_type,
92
+ "model_subtypes": self.model_subtypes,
93
+ "benchmark_scores": self.benchmark_scores,
94
+ "capability_tiers": self.capability_tiers,
95
+ "cost_input": self.cost_input, "context_window": self.context_window,
96
+ "open_weights": self.open_weights, "scores_as_of": self.scores_as_of,
97
+ "fits": self.fits, "verified_benchmarks": sorted(self.verified_benchmarks),
98
+ "rehost_of": self.rehost_of,
99
+ "total_parameters": self.total_parameters,
100
+ "active_parameters": self.active_parameters,
101
+ "release_date": self.release_date,
102
+ }
103
+
104
+
105
+ def rehost_of(card: Any) -> str | None:
106
+ """The canonical id a repackaged card re-hosts, or None.
107
+
108
+ Only `repackaged` counts. A quantised or finetuned derivative has different
109
+ weights, fits differently, and stays in the pool.
110
+ """
111
+ lineage = card.lineage
112
+ relation = lineage.base_model_relation
113
+ value = getattr(relation, "value", relation)
114
+ if value == "repackaged" and lineage.base_model:
115
+ return lineage.base_model
116
+ return None
117
+
118
+
119
+ def build_candidates(cards: list[Any], sink: CollectingSink) -> list[Candidate]:
120
+ """One ranking record per card, with capability tiers and hardware fit attached."""
121
+ tiers: dict[str, dict[str, str]] = {}
122
+ fits: dict[str, dict[str, float]] = {}
123
+ for edge in sink.edges:
124
+ if edge["type"] == "HAS_CAPABILITY":
125
+ props = edge.get("props") or {}
126
+ if props.get("tier"):
127
+ tiers.setdefault(edge["from"], {})[edge["to"]] = str(props["tier"])
128
+ elif edge["type"] == "FITS_ON":
129
+ # The capacity fit is meaningful even when there is no decode
130
+ # speed (a non-token model, or a token model still missing
131
+ # geometry): record the edge either way, with tps as None rather
132
+ # than dropping the row. `--fits <device>` and the fit ranking
133
+ # both need to see "this model fits" separately from "at this
134
+ # speed" (MODEL-53).
135
+ props = edge.get("props") or {}
136
+ tps = props.get("fastest_predicted_decode_tps")
137
+ fits.setdefault(edge["from"], {})[edge["to"]] = float(tps) if tps else None
138
+
139
+ out = []
140
+ for card in cards:
141
+ ident = card.identity
142
+ # Reviewed evidence takes precedence over the flat block for the same
143
+ # benchmark: it is the same measurement, checked.
144
+ scores = {k: float(v) for k, v in card.benchmarks.scores.items()
145
+ if isinstance(v, (int, float))}
146
+ verified: set[str] = set()
147
+ for record in card.benchmarks.evidence:
148
+ scores[record.benchmark_id] = float(record.score)
149
+ verified.add(record.benchmark_id)
150
+ out.append(Candidate(
151
+ model_id=ident.model_id,
152
+ display_name=ident.display_name or ident.model_id,
153
+ provider=ident.provider_display or ident.provider or "",
154
+ model_type=ident.model_type.value if ident.model_type else None,
155
+ model_subtypes=[s.value for s in ident.model_subtypes] if ident.model_subtypes else [],
156
+ benchmark_scores=scores,
157
+ verified_benchmarks=verified,
158
+ capability_tiers=tiers.get(ident.model_id, {}),
159
+ cost_input=card.cost.input,
160
+ context_window=card.modalities.text.context_window,
161
+ open_weights=bool(card.licensing.open_weights),
162
+ scores_as_of=str(card.benchmarks.benchmark_as_of or "") or None,
163
+ fits=fits.get(ident.model_id, {}),
164
+ rehost_of=rehost_of(card),
165
+ total_parameters=card.architecture.total_parameters,
166
+ active_parameters=card.architecture.active_parameters,
167
+ release_date=str(ident.release_date or "").strip() or None,
168
+ ))
169
+ return out
170
+
171
+
172
+ def score(candidate: Candidate, profile: dict[str, Any],
173
+ cost_weight: float | None = None,
174
+ min_coverage: float | None = None) -> dict[str, Any]:
175
+ """Score one candidate. Mirrors RankingEngine._score.
176
+
177
+ `cost_weight` overrides the profile's own. Every shipped profile carries
178
+ 0.0, so price contributes nothing unless a caller asks for it — and how much
179
+ quality someone will trade for price is a property of the person, not of the
180
+ use case, so it belongs in the query rather than in the table. See MODEL-30.
181
+ """
182
+ evidence = _benchmark_evidence(candidate.benchmark_scores, profile,
183
+ min_coverage=min_coverage)
184
+ contributions = evidence["benchmark_contributions"]
185
+ contributing_verified = len(contributions.keys() & candidate.verified_benchmarks)
186
+ weights = profile.get("benchmark_weights", {})
187
+ total_weight = sum(weights.values())
188
+ verified_weight = sum(weights[b] for b in contributions if b in candidate.verified_benchmarks)
189
+ bench = evidence["benchmark_lower_bound"]
190
+
191
+ cap_weights = profile.get("capability_weights", {})
192
+ total_cap_weight = sum(cap_weights.values()) if cap_weights else 1.0
193
+ cap_raw = 0.0
194
+ for name, weight in cap_weights.items():
195
+ tier = candidate.capability_tiers.get(name)
196
+ if tier:
197
+ cap_raw += _tier_points(tier) * (weight / total_cap_weight)
198
+ else:
199
+ subs = [t for cid, t in candidate.capability_tiers.items()
200
+ if cid.startswith(f"{name}:")]
201
+ if subs:
202
+ best = min(subs, key=_tier_rank)
203
+ cap_raw += _tier_points(best) * 0.7 * (weight / total_cap_weight)
204
+ cap = cap_raw * 2.0
205
+
206
+ if cost_weight is None:
207
+ cost_weight = profile.get("cost_weight", 0.0)
208
+ # MODEL-48: None is unknown and must not take the free branch. A sourced
209
+ # 0.0 stays legal as "vendor published $0 / million tokens" and scores as
210
+ # free (cost_raw = 10.0). Cards must not store 0.0 for missing research.
211
+ cost_raw = 0.0
212
+ if candidate.cost_input is not None:
213
+ if candidate.cost_input == 0:
214
+ cost_raw = 10.0
215
+ else:
216
+ clamped = max(candidate.cost_input, 0.10)
217
+ cost_raw = max(0.0, min(9.5, 9.0 + 2.0 * math.log10(0.10 / clamped)))
218
+ cost = cost_raw * cost_weight * 10
219
+
220
+ ctx_weight = profile.get("context_weight", 0.10)
221
+ ctx_raw = 0.0
222
+ if candidate.context_window and candidate.context_window > 0:
223
+ ctx_raw = min(10.0, max(0.0, (math.log10(candidate.context_window) - 3.5) * 3.0))
224
+ ctx = ctx_raw * ctx_weight * 10
225
+
226
+ preferred = profile.get("preferred_types", [])
227
+ type_bonus = 0.0
228
+ if candidate.model_type:
229
+ best_idx = None
230
+ for t in [candidate.model_type] + candidate.model_subtypes:
231
+ if t in preferred:
232
+ idx = preferred.index(t)
233
+ best_idx = idx if best_idx is None else min(best_idx, idx)
234
+ if best_idx is not None:
235
+ type_bonus = max(5.0, 15.0 - best_idx * 3.0)
236
+
237
+ other = cap + cost + ctx + type_bonus
238
+ total = max(0.0, min(100.0, bench + other))
239
+ upper = max(0.0, min(100.0, evidence["benchmark_upper_bound"] + other))
240
+ return {
241
+ "model_id": candidate.model_id,
242
+ "display_name": candidate.display_name,
243
+ "provider": candidate.provider,
244
+ "score": round(total, 2) if evidence["rank_status"] == "ranked" else None,
245
+ "rank": None,
246
+ "score_lower_bound": round(total, 2),
247
+ "score_upper_bound": round(upper, 2),
248
+ "score_kind": RANKING_POLICY["ordering"],
249
+ "benchmark_score": round(bench, 2),
250
+ "capability_score": round(cap, 2),
251
+ "cost_score": round(cost, 2),
252
+ "context_score": round(ctx, 2),
253
+ "type_bonus": round(type_bonus, 2),
254
+ **evidence,
255
+ "context_window": candidate.context_window,
256
+ "cost_input": candidate.cost_input,
257
+ "open_weights": candidate.open_weights,
258
+ # A ranking is only as good as the evidence under it. Say which kind
259
+ # rather than averaging the two into a single reassuring label.
260
+ "evidence_basis": _basis(len(contributions), contributing_verified,
261
+ evidence["benchmark_coverage"]),
262
+ "verified_contributions": contributing_verified,
263
+ "verified_benchmark_coverage": verified_weight / total_weight if total_weight else 0.0,
264
+ "scores_as_of": candidate.scores_as_of,
265
+ }
266
+
267
+
268
+ def _basis(contributing: int, verified: int, coverage: float) -> str:
269
+ """Provenance of benchmark inputs, never a certification of the composite."""
270
+ if not contributing:
271
+ return "none"
272
+ if verified == contributing:
273
+ return "verified" if coverage >= 1.0 - 1e-12 else "partial-verified"
274
+ if verified:
275
+ return "mixed"
276
+ return "unverified-legacy"
277
+
278
+
279
+ def rank(candidates: list[Candidate], profile_key: str, limit: int = 25,
280
+ open_weights_only: bool = False, hardware_id: str | None = None,
281
+ cost_weight: float | None = None,
282
+ min_benchmark_coverage: float | None = None) -> list[dict[str, Any]]:
283
+ """Return the models that can honestly be ordered for this profile.
284
+
285
+ Unrankable models are the normal catalogue state, not an error. This
286
+ returns the ranked shortlist, which may be empty when nothing has enough
287
+ evidence. Use rank_report() to see withheld models and ranking_status.
288
+ Default coverage floor is the CLI floor (0.50).
289
+ """
290
+ return rank_report(candidates, profile_key, limit, open_weights_only,
291
+ hardware_id, cost_weight,
292
+ min_benchmark_coverage)["ranked"]
293
+
294
+
295
+ def rank_report(candidates: list[Candidate], profile_key: str, limit: int = 25,
296
+ open_weights_only: bool = False, hardware_id: str | None = None,
297
+ cost_weight: float | None = None,
298
+ min_benchmark_coverage: float | None = None,
299
+ include_rehosts: bool = False) -> dict[str, Any]:
300
+ """Rank sufficiently covered models and retain all others as unranked.
301
+
302
+ `limit` caps the ranked shortlist only. Unranked entries are alphabetical,
303
+ have null rank/score, and are never truncated or presented as ranked last.
304
+ `unranked_candidates` names, newest first and capped, the unranked models
305
+ whose type the profile prefers — the disclosure every rank surface carries
306
+ (MODEL-110); see `unranked_candidates()`. Default coverage floor is the CLI floor (0.50). Pass
307
+ `WIZARD_BENCHMARK_COVERAGE` for the browser surface.
308
+ """
309
+ if limit < 0:
310
+ raise ValueError("limit must be nonnegative")
311
+ profile = USE_CASE_PROFILES[profile_key]
312
+ pool = candidates
313
+ if not include_rehosts:
314
+ pool = [c for c in pool if not c.rehost_of]
315
+ if open_weights_only:
316
+ pool = [c for c in pool if c.open_weights]
317
+ if hardware_id:
318
+ pool = [c for c in pool if hardware_id in c.fits]
319
+ scored = [score(c, profile, cost_weight, min_benchmark_coverage) for c in pool]
320
+ disclosure = unranked_candidates(pool, scored, profile)
321
+ ranked = [r for r in scored if r["rank_status"] == "ranked"]
322
+ unranked = [r for r in scored if r["rank_status"] == "unranked"]
323
+ ranked.sort(key=lambda r: (-r["score"], r["display_name"].lower(), r["model_id"]))
324
+ unranked.sort(key=lambda r: (r["display_name"].lower(), r["model_id"]))
325
+ for position, result in enumerate(ranked, 1):
326
+ result["rank"] = position
327
+ return {
328
+ "ranking_status": _ranking_status(ranked, unranked),
329
+ "profile": profile_key,
330
+ "policy": ranking_policy(min_benchmark_coverage=min_benchmark_coverage),
331
+ "ranked_count": len(ranked), "unranked_count": len(unranked),
332
+ "ranked": ranked[:limit], "unranked": unranked,
333
+ "unranked_candidates": disclosure,
334
+ }
335
+
336
+
337
+ #: How many `unranked_candidates` are named. The count beside them is never capped.
338
+ UNRANKED_CANDIDATES_CAP = 10
339
+
340
+ #: Why a candidate could not be ordered, in the order they are checked: a model
341
+ #: that fails both floors is named for the count floor, the more basic one.
342
+ UNRANKED_REASONS = ("no_scores", "below_count_floor", "below_coverage_floor")
343
+
344
+ #: What the ordering accepts as a date: `YYYY`, `YYYY-MM` or `YYYY-MM-DD`. These
345
+ #: sort correctly as strings. Anything else is published verbatim but ordered
346
+ #: with the undated, so prose in a card cannot jump the queue.
347
+ _ORDERABLE_DATE = re.compile(r"\d{4}(-\d{2}(-\d{2})?)?")
348
+
349
+
350
+ def _unranked_reason(row: dict[str, Any]) -> str:
351
+ """Which floor withheld this row. Reads the evidence `score` already computed."""
352
+ if row["benchmark_count"] == 0:
353
+ return "no_scores"
354
+ if row["benchmark_count"] < row["required_benchmark_count"]:
355
+ return "below_count_floor"
356
+ return "below_coverage_floor"
357
+
358
+
359
+ def _matches_profile_type(candidate: Candidate, profile: dict[str, Any]) -> bool:
360
+ """The type test `score` uses for its type bonus, as a membership test."""
361
+ preferred = set(profile.get("preferred_types", []))
362
+ return bool(candidate.model_type) and bool(
363
+ preferred & {candidate.model_type, *candidate.model_subtypes})
364
+
365
+
366
+ def unranked_candidates(pool: list[Candidate], scored: list[dict[str, Any]],
367
+ profile: dict[str, Any]) -> dict[str, Any]:
368
+ """Name the models that would have been candidates but lack the evidence (MODEL-110).
369
+
370
+ `pool` is what survived the request's filters, `scored` its rows in the same
371
+ order. A model is named when it is unranked *and* its type is one the
372
+ profile prefers: an embedding model with no coding scores is not a coding
373
+ candidate waiting for evidence, and listing it would bury the ones that are.
374
+ Newest `release_date` first, undated last, then by id. Disclosure only:
375
+ nothing here reads or changes a ranked row.
376
+ """
377
+ rows = []
378
+ for candidate, row in zip(pool, scored):
379
+ if row["rank_status"] != "unranked" or not _matches_profile_type(candidate, profile):
380
+ continue
381
+ released = (candidate.release_date or "").strip() or None
382
+ rows.append({
383
+ "model_id": candidate.model_id,
384
+ "display_name": candidate.display_name,
385
+ "release_date": released,
386
+ "reason": _unranked_reason(row),
387
+ "missing_benchmarks": list(row["missing_benchmarks"]),
388
+ })
389
+ rows.sort(key=lambda r: r["model_id"])
390
+ # Stable, so equal dates keep id order; "" (undated or unparseable) sorts last.
391
+ rows.sort(key=lambda r: r["release_date"]
392
+ if r["release_date"] and _ORDERABLE_DATE.fullmatch(r["release_date"]) else "",
393
+ reverse=True)
394
+ return {"count": len(rows), "cap": UNRANKED_CANDIDATES_CAP,
395
+ "models": rows[:UNRANKED_CANDIDATES_CAP]}
396
+
397
+
398
+ #: Guide `status` values the export will carry. Anything else is dropped rather
399
+ #: than published as current: MODEL-65 marks version drift `stale`, and a
400
+ #: missing status is not a guide.
401
+ _EXPORTED_GUIDE_STATUSES = frozenset({"current", "stale"})
402
+
403
+
404
+ def authoring_guides_from_cards(cards: list[Any]) -> dict[str, dict[str, Any]]:
405
+ """Card authoring guides as JSON, keyed by `model_id`.
406
+
407
+ Cards with no guide are omitted — never an empty string, never invented
408
+ text. The dump is what the card already holds (MODEL-8 sources and dates);
409
+ this function does not generate claims. Build-time only: the Worker reads
410
+ the map from `candidates.json` and must not import pydantic to do it.
411
+ """
412
+ out: dict[str, dict[str, Any]] = {}
413
+ for card in cards:
414
+ ident = getattr(card, "identity", None)
415
+ model_id = getattr(ident, "model_id", None) if ident is not None else None
416
+ if not model_id:
417
+ continue
418
+ guide = getattr(card, "authoring_guide", None)
419
+ if guide is None:
420
+ continue
421
+ dump = getattr(guide, "model_dump", None)
422
+ payload = dump(mode="json") if callable(dump) else None
423
+ if isinstance(payload, dict) and payload.get("status") in _EXPORTED_GUIDE_STATUSES:
424
+ out[str(model_id)] = payload
425
+ return out
426
+
427
+
428
+ def write_export(out_dir: Any, cards: list[Any], sink: CollectingSink,
429
+ build_json: dict[str, Any]) -> dict[str, Any]:
430
+ """Emit the tables the browser needs, plus precomputed rankings."""
431
+ import json
432
+ from pathlib import Path
433
+
434
+ out = Path(out_dir)
435
+ out.mkdir(parents=True, exist_ok=True)
436
+
437
+ def dump(rel: str, payload: Any) -> None:
438
+ path = out / rel
439
+ path.parent.mkdir(parents=True, exist_ok=True)
440
+ path.write_text(json.dumps(payload, sort_keys=True, default=str), encoding="utf-8")
441
+
442
+ candidates = build_candidates(cards, sink)
443
+ dump("profiles.json", {
444
+ "build": build_json,
445
+ "featured": list(FEATURED_PROFILES),
446
+ "profiles": USE_CASE_PROFILES,
447
+ "benchmark_ranges": {k: list(v) for k, v in BENCHMARK_RANGES.items()},
448
+ "ranking_policy": RANKING_POLICY,
449
+ })
450
+ # `authoring_guides` is additive on this file (MODEL-81). Scoring rows stay
451
+ # the shape they were; a consumer that does not read the new key is unchanged.
452
+ # No `export_schema_version` bump: new optional map, not a widened field.
453
+ dump("candidates.json", {
454
+ "build": build_json,
455
+ "count": len(candidates),
456
+ "candidates": [c.to_json() for c in candidates],
457
+ "authoring_guides": authoring_guides_from_cards(cards),
458
+ })
459
+
460
+ # The device vocabulary, without the 7 MB of FITS_ON edges that the graph
461
+ # view carries. `offline rank --fits` refuses an unknown device id rather
462
+ # than answering "nothing fits your GPU", because those are different
463
+ # answers and a caller would act on them differently. The rank Worker
464
+ # (MODEL-68) has to make the same distinction and cannot afford the graph
465
+ # view per request, so the ids ship on their own. Additive: no consumer
466
+ # pinning `export_schema_version` mis-parses a new file.
467
+ devices = sorted(
468
+ (props for (label, _id), props in sink.nodes.items() if label == "Hardware"),
469
+ key=lambda d: str(d.get("id", "")),
470
+ )
471
+ dump("hardware.json", {
472
+ "build": build_json,
473
+ "count": len(devices),
474
+ "hardware": devices,
475
+ })
476
+
477
+ precomputed = {}
478
+ for key in FEATURED_PROFILES:
479
+ precomputed[key] = rank_report(
480
+ candidates, key, limit=25,
481
+ min_benchmark_coverage=WIZARD_BENCHMARK_COVERAGE,
482
+ )
483
+ dump("rankings.json", {"schema_version": "2.0", "build": build_json, "rankings": precomputed})
484
+
485
+ return {
486
+ "candidates": len(candidates),
487
+ "hardware": len(devices),
488
+ "profiles": len(USE_CASE_PROFILES),
489
+ "featured": len(FEATURED_PROFILES),
490
+ "precomputed": {k: len(v["ranked"]) for k, v in precomputed.items()},
491
+ }
492
+
493
+
494
+ def format_report(report: dict[str, Any]) -> str:
495
+ """Human-readable contract shared by the report CLI and its tests."""
496
+ policy = report['policy']
497
+ lines = [
498
+ f"{report['profile']}: {report['ranking_status']} ordering; "
499
+ f"{report['ranked_count']} ranked, {report['unranked_count']} unranked.",
500
+ "Ranked by conservative composite lower bound; observed benchmark averages "
501
+ "do not predict missing results.",
502
+ f"Eligibility: at least {policy['min_benchmark_coverage']:.0%} weighted benchmark coverage "
503
+ f"and {policy['min_benchmark_count']} benchmarks (or all for a smaller profile). "
504
+ "Bounds describe missing evidence, not statistical confidence.",
505
+ ]
506
+ for row in report["ranked"]:
507
+ lines.append(
508
+ f"{row['rank']:>3}. {row['score']:.2f} {row['display_name']} "
509
+ f"coverage {row['benchmark_coverage']:.0%}, "
510
+ f"bounds [{row['score_lower_bound']:.2f}, {row['score_upper_bound']:.2f}], "
511
+ f"benchmark basis: {row['evidence_basis']}"
512
+ )
513
+ if report["unranked"]:
514
+ lines.append("Unranked for insufficient benchmark evidence (alphabetical; not ranked low):")
515
+ for row in report["unranked"]:
516
+ estimate = row["benchmark_estimate"]
517
+ observed = "none" if estimate is None else f"{estimate:.2f}/100"
518
+ lines.append(
519
+ f" UNRANKED {row['display_name']} coverage {row['benchmark_coverage']:.0%}, "
520
+ f"observed benchmark average {observed}, benchmark basis: {row['evidence_basis']}"
521
+ )
522
+ return "\n".join(lines)
523
+
524
+
525
+ def main() -> None:
526
+ """Report CLI, usable without changes to the legacy list-only CLI."""
527
+ import argparse
528
+ import json
529
+ from pathlib import Path
530
+ from schema.card import ModelCard
531
+ from schema.graph import derive_graph
532
+
533
+ parser = argparse.ArgumentParser(description="Rank models with explicit incomplete evidence.")
534
+ parser.add_argument("profile", choices=sorted(USE_CASE_PROFILES))
535
+ parser.add_argument("--limit", type=int, default=10, help="Maximum ranked rows; all unranked models remain visible.")
536
+ parser.add_argument("--json", action="store_true", help="Emit the version 2 ranking report.")
537
+ args = parser.parse_args()
538
+ if args.limit < 0:
539
+ parser.error("--limit must be nonnegative")
540
+ root = Path(__file__).resolve().parents[1]
541
+ cards = [ModelCard.from_yaml_file(p) for p in sorted((root / "models").rglob("*.md"))
542
+ if p.name != "LICENSE.md"]
543
+ report = rank_report(build_candidates(cards, derive_graph(cards)), args.profile, args.limit)
544
+ if args.json:
545
+ print(json.dumps({"schema_version": "2.0", **report}, indent=2))
546
+ else:
547
+ print(format_report(report))
548
+
549
+
550
+ if __name__ == "__main__":
551
+ main()
registry/domains.yaml ADDED
@@ -0,0 +1,130 @@
1
+ # The domain registry (MODEL-133; design §4.1 and §8).
2
+ #
3
+ # A domain is an area of capability that benchmarks can measure. It is not a
4
+ # use case: a use case is what a user wants, and the engine decomposes it into
5
+ # domains at query time. The set is open; adding a domain is an entry here.
6
+ # Loader: decision/registry.py.
7
+ #
8
+ # Entry fields
9
+ # id snake_case, unique
10
+ # name short label
11
+ # definition exact: what a benchmark must measure to count as direct
12
+ # evidence for this domain
13
+ # proxy_only true when no benchmark measures the domain directly, so every
14
+ # tag on it must be `proxy` and every answer says so
15
+ #
16
+ # How a benchmark page declares its domains (`schema/benchmark.py`,
17
+ # `BenchmarkCard.domains`, optional and additive):
18
+ #
19
+ # domains:
20
+ # - {id: software_engineering, directness: direct}
21
+ # - {id: agentic_tool_use, directness: proxy}
22
+ #
23
+ # id a domain id from this file
24
+ # directness direct: the benchmark's tasks are instances of the domain as
25
+ # defined here, so its score is evidence of that capability.
26
+ # proxy: the benchmark measures something correlated with the
27
+ # domain, or covers only a slice of it, or measures preference
28
+ # rather than correctness, so its score is weaker evidence and
29
+ # is always labelled as proxy.
30
+ # (`none` is the absence of a tag, never written.)
31
+ #
32
+ # A page may carry several domains; a domain may appear once per page. A page
33
+ # with no `domains` key is untagged, not "no domain".
34
+
35
+ schema_version: 1
36
+
37
+ domains:
38
+ - id: software_engineering
39
+ name: Software engineering
40
+ # Preselected only when a user explicitly opens the "Measured by" benchmark
41
+ # drill-down. Domain capability is the ranking default (MODEL-167).
42
+ default_benchmark: swe_bench_pro
43
+ definition: >-
44
+ Producing or changing working software against an objective check:
45
+ resolving issues in real repositories, writing code that passes held-out
46
+ tests, or completing programming tasks whose result is executed and
47
+ graded.
48
+ - id: engineering_stem
49
+ name: Engineering and STEM
50
+ definition: >-
51
+ Answering or solving graduate- and professional-level problems in the
52
+ natural sciences and engineering (physics, chemistry, biology, earth
53
+ sciences, engineering disciplines) whose answers are objectively graded.
54
+ - id: maths
55
+ name: Maths
56
+ definition: >-
57
+ Solving mathematical problems whose final answer or proof is objectively
58
+ checked, from school word problems to competition and research-level
59
+ mathematics.
60
+ - id: reasoning
61
+ name: General reasoning
62
+ definition: >-
63
+ Solving novel problems by multi-step inference that does not depend on
64
+ specialist knowledge: abstraction, logical deduction, commonsense
65
+ inference and puzzle solving with objectively graded answers.
66
+ - id: legal
67
+ name: Legal
68
+ definition: >-
69
+ Legal tasks graded against a legal professional's answer or rubric:
70
+ reading statutes, contracts and cases, issue spotting, and legal
71
+ reasoning.
72
+ - id: medical
73
+ name: Medical
74
+ definition: >-
75
+ Clinical and biomedical tasks graded against clinicians' answers or
76
+ rubrics: diagnosis, treatment questions, medical knowledge and
77
+ patient-facing health conversations.
78
+ - id: finance
79
+ name: Finance
80
+ definition: >-
81
+ Financial analysis tasks graded against expert answers: reading filings
82
+ and statements, financial calculations, and domain questions in
83
+ accounting, markets and banking.
84
+ - id: writing
85
+ name: Writing
86
+ definition: >-
87
+ Producing prose judged for quality against a rubric or by human raters:
88
+ creative writing, editing, summarisation and long-form composition.
89
+ - id: marketing_seo
90
+ name: Marketing and SEO
91
+ definition: >-
92
+ Marketing copy, positioning and search-engine optimisation. No benchmark
93
+ measures commercial effect, so all evidence here is proxy and every
94
+ answer says so.
95
+ proxy_only: true
96
+ - id: agentic_tool_use
97
+ name: Agentic and tool use
98
+ definition: >-
99
+ Completing multi-step tasks by calling tools or acting in an environment
100
+ (shells, browsers, operating systems, APIs, simulated businesses), graded
101
+ on the end state rather than on the text of the answer.
102
+ - id: vision_documents
103
+ name: Vision and documents
104
+ definition: >-
105
+ Understanding images and documents given as input: charts, diagrams,
106
+ scanned pages, screenshots, photographs and text in images, graded
107
+ against reference answers.
108
+ - id: multilingual
109
+ name: Multilingual
110
+ definition: >-
111
+ Performing tasks in languages other than English, or across languages,
112
+ graded per language against reference answers or native raters.
113
+ - id: chat_preference
114
+ name: Chat and preference
115
+ definition: >-
116
+ Which of two models' responses to the same open-ended prompt people
117
+ prefer, aggregated into a rating by pairwise comparison. Measures
118
+ preference, not correctness.
119
+ # The overall text board is the preselected explicit drill-down, not the
120
+ # domain's default ranking basis.
121
+ default_benchmark: arena_elo_style_control
122
+ - id: retrieval
123
+ name: Retrieval
124
+ definition: >-
125
+ Ranking documents or passages by relevance to a query, as an embedding or
126
+ reranking model does, graded against relevance judgements with ranking
127
+ metrics such as nDCG or recall at k.
128
+ # Preselect the retrieval task type, not the reranking task, when the user
129
+ # explicitly drills down to one benchmark.
130
+ default_benchmark: mteb_v2_retrieval