modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
pipeline/ranking.py
ADDED
|
@@ -0,0 +1,551 @@
|
|
|
1
|
+
"""Rank models against a use-case profile, without a database.
|
|
2
|
+
|
|
3
|
+
`api/ranking/engine.py` holds the profiles, the benchmark normalisation ranges
|
|
4
|
+
and the scoring constants. That module imports cleanly with no FalkorDB
|
|
5
|
+
dependency, so this reuses its tables and helpers directly rather than
|
|
6
|
+
transcribing them — a transcribed copy of 51 profiles and 170 benchmark ranges
|
|
7
|
+
would drift on the first edit.
|
|
8
|
+
|
|
9
|
+
What is reimplemented here is only the scoring itself, against a model built
|
|
10
|
+
from a card rather than from a Cypher row. The formulas follow
|
|
11
|
+
`RankingEngine._score` exactly: benchmarks to 40 points, capabilities to 20,
|
|
12
|
+
cost and context scaled by the profile's weights, a type-match bonus to 15.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import math
|
|
18
|
+
import re
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
from typing import TYPE_CHECKING, Any
|
|
21
|
+
|
|
22
|
+
from api.ranking.engine import (
|
|
23
|
+
BENCHMARK_RANGES,
|
|
24
|
+
USE_CASE_PROFILES,
|
|
25
|
+
RANKING_POLICY,
|
|
26
|
+
WIZARD_BENCHMARK_COVERAGE,
|
|
27
|
+
_benchmark_evidence,
|
|
28
|
+
_ranking_status,
|
|
29
|
+
_tier_points,
|
|
30
|
+
_tier_rank,
|
|
31
|
+
ranking_policy,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
if TYPE_CHECKING: # pragma: no cover - annotations only
|
|
35
|
+
# Only `build_candidates` and `write_export` take a sink, and neither runs
|
|
36
|
+
# on a serving path. `from __future__ import annotations` already makes the
|
|
37
|
+
# annotation a string, so deferring the import costs nothing here and buys
|
|
38
|
+
# a great deal: this module, and therefore the whole scorer, imports with
|
|
39
|
+
# nothing but the standard library and `api.ranking.engine`. The Cloudflare
|
|
40
|
+
# rank Worker (MODEL-68) runs this exact file, and pydantic is not
|
|
41
|
+
# available to it. See docs/rank-api.md.
|
|
42
|
+
from schema.graph import CollectingSink
|
|
43
|
+
|
|
44
|
+
#: Profiles offered in the wizard. All 51 are exported for the API, but a
|
|
45
|
+
#: dropdown of 51 is a worse experience than a short list. speech_to_text
|
|
46
|
+
#: currently ranks nothing (MODEL-30); offered-and-empty is worse than hiding
|
|
47
|
+
#: it. image_generation stays out until clip_score's range is sourced.
|
|
48
|
+
FEATURED_PROFILES = (
|
|
49
|
+
"general", "coding", "reasoning", "chat", "agentic", "rag",
|
|
50
|
+
"vision", "multilingual", "math_competition", "writing_technical",
|
|
51
|
+
"summarization", "embedding", "text_to_speech",
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass
|
|
56
|
+
class Candidate:
|
|
57
|
+
model_id: str
|
|
58
|
+
display_name: str
|
|
59
|
+
provider: str
|
|
60
|
+
model_type: str | None
|
|
61
|
+
model_subtypes: list[str] = field(default_factory=list)
|
|
62
|
+
benchmark_scores: dict[str, float] = field(default_factory=dict)
|
|
63
|
+
capability_tiers: dict[str, str] = field(default_factory=dict)
|
|
64
|
+
cost_input: float | None = None
|
|
65
|
+
context_window: int | None = None
|
|
66
|
+
open_weights: bool = False
|
|
67
|
+
scores_as_of: str | None = None
|
|
68
|
+
#: hardware id -> predicted decode tok/s, or None when the weights fit
|
|
69
|
+
#: but the model does not decode tokens (or, rarely, geometry is missing).
|
|
70
|
+
fits: dict[str, float | None] = field(default_factory=dict)
|
|
71
|
+
#: Benchmarks whose score came from a reviewed evidence record rather than
|
|
72
|
+
#: the card's undated flat block. A ranking is only as good as the weakest
|
|
73
|
+
#: evidence under it, so this is tracked per benchmark, not per model.
|
|
74
|
+
verified_benchmarks: set[str] = field(default_factory=set)
|
|
75
|
+
#: Canonical model id when this card re-hosts the same weights
|
|
76
|
+
#: (`lineage.base_model_relation: repackaged`). Default fit pools and the
|
|
77
|
+
#: featured rankings leave these out so one weight set has one id (MODEL-54).
|
|
78
|
+
rehost_of: str | None = None
|
|
79
|
+
#: Parameter counts, so a consumer can size a model that does not fit an
|
|
80
|
+
#: accelerator (offload fit, MODEL-26). Additive to candidates.json.
|
|
81
|
+
total_parameters: float | None = None
|
|
82
|
+
active_parameters: float | None = None
|
|
83
|
+
#: `identity.release_date` as the card has it, or None. Orders the
|
|
84
|
+
#: `unranked_candidates` disclosure (MODEL-110); never scored. Additive to
|
|
85
|
+
#: candidates.json: an export without it reads as None everywhere.
|
|
86
|
+
release_date: str | None = None
|
|
87
|
+
|
|
88
|
+
def to_json(self) -> dict[str, Any]:
|
|
89
|
+
return {
|
|
90
|
+
"model_id": self.model_id, "display_name": self.display_name,
|
|
91
|
+
"provider": self.provider, "model_type": self.model_type,
|
|
92
|
+
"model_subtypes": self.model_subtypes,
|
|
93
|
+
"benchmark_scores": self.benchmark_scores,
|
|
94
|
+
"capability_tiers": self.capability_tiers,
|
|
95
|
+
"cost_input": self.cost_input, "context_window": self.context_window,
|
|
96
|
+
"open_weights": self.open_weights, "scores_as_of": self.scores_as_of,
|
|
97
|
+
"fits": self.fits, "verified_benchmarks": sorted(self.verified_benchmarks),
|
|
98
|
+
"rehost_of": self.rehost_of,
|
|
99
|
+
"total_parameters": self.total_parameters,
|
|
100
|
+
"active_parameters": self.active_parameters,
|
|
101
|
+
"release_date": self.release_date,
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def rehost_of(card: Any) -> str | None:
|
|
106
|
+
"""The canonical id a repackaged card re-hosts, or None.
|
|
107
|
+
|
|
108
|
+
Only `repackaged` counts. A quantised or finetuned derivative has different
|
|
109
|
+
weights, fits differently, and stays in the pool.
|
|
110
|
+
"""
|
|
111
|
+
lineage = card.lineage
|
|
112
|
+
relation = lineage.base_model_relation
|
|
113
|
+
value = getattr(relation, "value", relation)
|
|
114
|
+
if value == "repackaged" and lineage.base_model:
|
|
115
|
+
return lineage.base_model
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def build_candidates(cards: list[Any], sink: CollectingSink) -> list[Candidate]:
|
|
120
|
+
"""One ranking record per card, with capability tiers and hardware fit attached."""
|
|
121
|
+
tiers: dict[str, dict[str, str]] = {}
|
|
122
|
+
fits: dict[str, dict[str, float]] = {}
|
|
123
|
+
for edge in sink.edges:
|
|
124
|
+
if edge["type"] == "HAS_CAPABILITY":
|
|
125
|
+
props = edge.get("props") or {}
|
|
126
|
+
if props.get("tier"):
|
|
127
|
+
tiers.setdefault(edge["from"], {})[edge["to"]] = str(props["tier"])
|
|
128
|
+
elif edge["type"] == "FITS_ON":
|
|
129
|
+
# The capacity fit is meaningful even when there is no decode
|
|
130
|
+
# speed (a non-token model, or a token model still missing
|
|
131
|
+
# geometry): record the edge either way, with tps as None rather
|
|
132
|
+
# than dropping the row. `--fits <device>` and the fit ranking
|
|
133
|
+
# both need to see "this model fits" separately from "at this
|
|
134
|
+
# speed" (MODEL-53).
|
|
135
|
+
props = edge.get("props") or {}
|
|
136
|
+
tps = props.get("fastest_predicted_decode_tps")
|
|
137
|
+
fits.setdefault(edge["from"], {})[edge["to"]] = float(tps) if tps else None
|
|
138
|
+
|
|
139
|
+
out = []
|
|
140
|
+
for card in cards:
|
|
141
|
+
ident = card.identity
|
|
142
|
+
# Reviewed evidence takes precedence over the flat block for the same
|
|
143
|
+
# benchmark: it is the same measurement, checked.
|
|
144
|
+
scores = {k: float(v) for k, v in card.benchmarks.scores.items()
|
|
145
|
+
if isinstance(v, (int, float))}
|
|
146
|
+
verified: set[str] = set()
|
|
147
|
+
for record in card.benchmarks.evidence:
|
|
148
|
+
scores[record.benchmark_id] = float(record.score)
|
|
149
|
+
verified.add(record.benchmark_id)
|
|
150
|
+
out.append(Candidate(
|
|
151
|
+
model_id=ident.model_id,
|
|
152
|
+
display_name=ident.display_name or ident.model_id,
|
|
153
|
+
provider=ident.provider_display or ident.provider or "",
|
|
154
|
+
model_type=ident.model_type.value if ident.model_type else None,
|
|
155
|
+
model_subtypes=[s.value for s in ident.model_subtypes] if ident.model_subtypes else [],
|
|
156
|
+
benchmark_scores=scores,
|
|
157
|
+
verified_benchmarks=verified,
|
|
158
|
+
capability_tiers=tiers.get(ident.model_id, {}),
|
|
159
|
+
cost_input=card.cost.input,
|
|
160
|
+
context_window=card.modalities.text.context_window,
|
|
161
|
+
open_weights=bool(card.licensing.open_weights),
|
|
162
|
+
scores_as_of=str(card.benchmarks.benchmark_as_of or "") or None,
|
|
163
|
+
fits=fits.get(ident.model_id, {}),
|
|
164
|
+
rehost_of=rehost_of(card),
|
|
165
|
+
total_parameters=card.architecture.total_parameters,
|
|
166
|
+
active_parameters=card.architecture.active_parameters,
|
|
167
|
+
release_date=str(ident.release_date or "").strip() or None,
|
|
168
|
+
))
|
|
169
|
+
return out
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def score(candidate: Candidate, profile: dict[str, Any],
|
|
173
|
+
cost_weight: float | None = None,
|
|
174
|
+
min_coverage: float | None = None) -> dict[str, Any]:
|
|
175
|
+
"""Score one candidate. Mirrors RankingEngine._score.
|
|
176
|
+
|
|
177
|
+
`cost_weight` overrides the profile's own. Every shipped profile carries
|
|
178
|
+
0.0, so price contributes nothing unless a caller asks for it — and how much
|
|
179
|
+
quality someone will trade for price is a property of the person, not of the
|
|
180
|
+
use case, so it belongs in the query rather than in the table. See MODEL-30.
|
|
181
|
+
"""
|
|
182
|
+
evidence = _benchmark_evidence(candidate.benchmark_scores, profile,
|
|
183
|
+
min_coverage=min_coverage)
|
|
184
|
+
contributions = evidence["benchmark_contributions"]
|
|
185
|
+
contributing_verified = len(contributions.keys() & candidate.verified_benchmarks)
|
|
186
|
+
weights = profile.get("benchmark_weights", {})
|
|
187
|
+
total_weight = sum(weights.values())
|
|
188
|
+
verified_weight = sum(weights[b] for b in contributions if b in candidate.verified_benchmarks)
|
|
189
|
+
bench = evidence["benchmark_lower_bound"]
|
|
190
|
+
|
|
191
|
+
cap_weights = profile.get("capability_weights", {})
|
|
192
|
+
total_cap_weight = sum(cap_weights.values()) if cap_weights else 1.0
|
|
193
|
+
cap_raw = 0.0
|
|
194
|
+
for name, weight in cap_weights.items():
|
|
195
|
+
tier = candidate.capability_tiers.get(name)
|
|
196
|
+
if tier:
|
|
197
|
+
cap_raw += _tier_points(tier) * (weight / total_cap_weight)
|
|
198
|
+
else:
|
|
199
|
+
subs = [t for cid, t in candidate.capability_tiers.items()
|
|
200
|
+
if cid.startswith(f"{name}:")]
|
|
201
|
+
if subs:
|
|
202
|
+
best = min(subs, key=_tier_rank)
|
|
203
|
+
cap_raw += _tier_points(best) * 0.7 * (weight / total_cap_weight)
|
|
204
|
+
cap = cap_raw * 2.0
|
|
205
|
+
|
|
206
|
+
if cost_weight is None:
|
|
207
|
+
cost_weight = profile.get("cost_weight", 0.0)
|
|
208
|
+
# MODEL-48: None is unknown and must not take the free branch. A sourced
|
|
209
|
+
# 0.0 stays legal as "vendor published $0 / million tokens" and scores as
|
|
210
|
+
# free (cost_raw = 10.0). Cards must not store 0.0 for missing research.
|
|
211
|
+
cost_raw = 0.0
|
|
212
|
+
if candidate.cost_input is not None:
|
|
213
|
+
if candidate.cost_input == 0:
|
|
214
|
+
cost_raw = 10.0
|
|
215
|
+
else:
|
|
216
|
+
clamped = max(candidate.cost_input, 0.10)
|
|
217
|
+
cost_raw = max(0.0, min(9.5, 9.0 + 2.0 * math.log10(0.10 / clamped)))
|
|
218
|
+
cost = cost_raw * cost_weight * 10
|
|
219
|
+
|
|
220
|
+
ctx_weight = profile.get("context_weight", 0.10)
|
|
221
|
+
ctx_raw = 0.0
|
|
222
|
+
if candidate.context_window and candidate.context_window > 0:
|
|
223
|
+
ctx_raw = min(10.0, max(0.0, (math.log10(candidate.context_window) - 3.5) * 3.0))
|
|
224
|
+
ctx = ctx_raw * ctx_weight * 10
|
|
225
|
+
|
|
226
|
+
preferred = profile.get("preferred_types", [])
|
|
227
|
+
type_bonus = 0.0
|
|
228
|
+
if candidate.model_type:
|
|
229
|
+
best_idx = None
|
|
230
|
+
for t in [candidate.model_type] + candidate.model_subtypes:
|
|
231
|
+
if t in preferred:
|
|
232
|
+
idx = preferred.index(t)
|
|
233
|
+
best_idx = idx if best_idx is None else min(best_idx, idx)
|
|
234
|
+
if best_idx is not None:
|
|
235
|
+
type_bonus = max(5.0, 15.0 - best_idx * 3.0)
|
|
236
|
+
|
|
237
|
+
other = cap + cost + ctx + type_bonus
|
|
238
|
+
total = max(0.0, min(100.0, bench + other))
|
|
239
|
+
upper = max(0.0, min(100.0, evidence["benchmark_upper_bound"] + other))
|
|
240
|
+
return {
|
|
241
|
+
"model_id": candidate.model_id,
|
|
242
|
+
"display_name": candidate.display_name,
|
|
243
|
+
"provider": candidate.provider,
|
|
244
|
+
"score": round(total, 2) if evidence["rank_status"] == "ranked" else None,
|
|
245
|
+
"rank": None,
|
|
246
|
+
"score_lower_bound": round(total, 2),
|
|
247
|
+
"score_upper_bound": round(upper, 2),
|
|
248
|
+
"score_kind": RANKING_POLICY["ordering"],
|
|
249
|
+
"benchmark_score": round(bench, 2),
|
|
250
|
+
"capability_score": round(cap, 2),
|
|
251
|
+
"cost_score": round(cost, 2),
|
|
252
|
+
"context_score": round(ctx, 2),
|
|
253
|
+
"type_bonus": round(type_bonus, 2),
|
|
254
|
+
**evidence,
|
|
255
|
+
"context_window": candidate.context_window,
|
|
256
|
+
"cost_input": candidate.cost_input,
|
|
257
|
+
"open_weights": candidate.open_weights,
|
|
258
|
+
# A ranking is only as good as the evidence under it. Say which kind
|
|
259
|
+
# rather than averaging the two into a single reassuring label.
|
|
260
|
+
"evidence_basis": _basis(len(contributions), contributing_verified,
|
|
261
|
+
evidence["benchmark_coverage"]),
|
|
262
|
+
"verified_contributions": contributing_verified,
|
|
263
|
+
"verified_benchmark_coverage": verified_weight / total_weight if total_weight else 0.0,
|
|
264
|
+
"scores_as_of": candidate.scores_as_of,
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _basis(contributing: int, verified: int, coverage: float) -> str:
|
|
269
|
+
"""Provenance of benchmark inputs, never a certification of the composite."""
|
|
270
|
+
if not contributing:
|
|
271
|
+
return "none"
|
|
272
|
+
if verified == contributing:
|
|
273
|
+
return "verified" if coverage >= 1.0 - 1e-12 else "partial-verified"
|
|
274
|
+
if verified:
|
|
275
|
+
return "mixed"
|
|
276
|
+
return "unverified-legacy"
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def rank(candidates: list[Candidate], profile_key: str, limit: int = 25,
|
|
280
|
+
open_weights_only: bool = False, hardware_id: str | None = None,
|
|
281
|
+
cost_weight: float | None = None,
|
|
282
|
+
min_benchmark_coverage: float | None = None) -> list[dict[str, Any]]:
|
|
283
|
+
"""Return the models that can honestly be ordered for this profile.
|
|
284
|
+
|
|
285
|
+
Unrankable models are the normal catalogue state, not an error. This
|
|
286
|
+
returns the ranked shortlist, which may be empty when nothing has enough
|
|
287
|
+
evidence. Use rank_report() to see withheld models and ranking_status.
|
|
288
|
+
Default coverage floor is the CLI floor (0.50).
|
|
289
|
+
"""
|
|
290
|
+
return rank_report(candidates, profile_key, limit, open_weights_only,
|
|
291
|
+
hardware_id, cost_weight,
|
|
292
|
+
min_benchmark_coverage)["ranked"]
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def rank_report(candidates: list[Candidate], profile_key: str, limit: int = 25,
|
|
296
|
+
open_weights_only: bool = False, hardware_id: str | None = None,
|
|
297
|
+
cost_weight: float | None = None,
|
|
298
|
+
min_benchmark_coverage: float | None = None,
|
|
299
|
+
include_rehosts: bool = False) -> dict[str, Any]:
|
|
300
|
+
"""Rank sufficiently covered models and retain all others as unranked.
|
|
301
|
+
|
|
302
|
+
`limit` caps the ranked shortlist only. Unranked entries are alphabetical,
|
|
303
|
+
have null rank/score, and are never truncated or presented as ranked last.
|
|
304
|
+
`unranked_candidates` names, newest first and capped, the unranked models
|
|
305
|
+
whose type the profile prefers — the disclosure every rank surface carries
|
|
306
|
+
(MODEL-110); see `unranked_candidates()`. Default coverage floor is the CLI floor (0.50). Pass
|
|
307
|
+
`WIZARD_BENCHMARK_COVERAGE` for the browser surface.
|
|
308
|
+
"""
|
|
309
|
+
if limit < 0:
|
|
310
|
+
raise ValueError("limit must be nonnegative")
|
|
311
|
+
profile = USE_CASE_PROFILES[profile_key]
|
|
312
|
+
pool = candidates
|
|
313
|
+
if not include_rehosts:
|
|
314
|
+
pool = [c for c in pool if not c.rehost_of]
|
|
315
|
+
if open_weights_only:
|
|
316
|
+
pool = [c for c in pool if c.open_weights]
|
|
317
|
+
if hardware_id:
|
|
318
|
+
pool = [c for c in pool if hardware_id in c.fits]
|
|
319
|
+
scored = [score(c, profile, cost_weight, min_benchmark_coverage) for c in pool]
|
|
320
|
+
disclosure = unranked_candidates(pool, scored, profile)
|
|
321
|
+
ranked = [r for r in scored if r["rank_status"] == "ranked"]
|
|
322
|
+
unranked = [r for r in scored if r["rank_status"] == "unranked"]
|
|
323
|
+
ranked.sort(key=lambda r: (-r["score"], r["display_name"].lower(), r["model_id"]))
|
|
324
|
+
unranked.sort(key=lambda r: (r["display_name"].lower(), r["model_id"]))
|
|
325
|
+
for position, result in enumerate(ranked, 1):
|
|
326
|
+
result["rank"] = position
|
|
327
|
+
return {
|
|
328
|
+
"ranking_status": _ranking_status(ranked, unranked),
|
|
329
|
+
"profile": profile_key,
|
|
330
|
+
"policy": ranking_policy(min_benchmark_coverage=min_benchmark_coverage),
|
|
331
|
+
"ranked_count": len(ranked), "unranked_count": len(unranked),
|
|
332
|
+
"ranked": ranked[:limit], "unranked": unranked,
|
|
333
|
+
"unranked_candidates": disclosure,
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
#: How many `unranked_candidates` are named. The count beside them is never capped.
|
|
338
|
+
UNRANKED_CANDIDATES_CAP = 10
|
|
339
|
+
|
|
340
|
+
#: Why a candidate could not be ordered, in the order they are checked: a model
|
|
341
|
+
#: that fails both floors is named for the count floor, the more basic one.
|
|
342
|
+
UNRANKED_REASONS = ("no_scores", "below_count_floor", "below_coverage_floor")
|
|
343
|
+
|
|
344
|
+
#: What the ordering accepts as a date: `YYYY`, `YYYY-MM` or `YYYY-MM-DD`. These
|
|
345
|
+
#: sort correctly as strings. Anything else is published verbatim but ordered
|
|
346
|
+
#: with the undated, so prose in a card cannot jump the queue.
|
|
347
|
+
_ORDERABLE_DATE = re.compile(r"\d{4}(-\d{2}(-\d{2})?)?")
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _unranked_reason(row: dict[str, Any]) -> str:
|
|
351
|
+
"""Which floor withheld this row. Reads the evidence `score` already computed."""
|
|
352
|
+
if row["benchmark_count"] == 0:
|
|
353
|
+
return "no_scores"
|
|
354
|
+
if row["benchmark_count"] < row["required_benchmark_count"]:
|
|
355
|
+
return "below_count_floor"
|
|
356
|
+
return "below_coverage_floor"
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def _matches_profile_type(candidate: Candidate, profile: dict[str, Any]) -> bool:
|
|
360
|
+
"""The type test `score` uses for its type bonus, as a membership test."""
|
|
361
|
+
preferred = set(profile.get("preferred_types", []))
|
|
362
|
+
return bool(candidate.model_type) and bool(
|
|
363
|
+
preferred & {candidate.model_type, *candidate.model_subtypes})
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def unranked_candidates(pool: list[Candidate], scored: list[dict[str, Any]],
|
|
367
|
+
profile: dict[str, Any]) -> dict[str, Any]:
|
|
368
|
+
"""Name the models that would have been candidates but lack the evidence (MODEL-110).
|
|
369
|
+
|
|
370
|
+
`pool` is what survived the request's filters, `scored` its rows in the same
|
|
371
|
+
order. A model is named when it is unranked *and* its type is one the
|
|
372
|
+
profile prefers: an embedding model with no coding scores is not a coding
|
|
373
|
+
candidate waiting for evidence, and listing it would bury the ones that are.
|
|
374
|
+
Newest `release_date` first, undated last, then by id. Disclosure only:
|
|
375
|
+
nothing here reads or changes a ranked row.
|
|
376
|
+
"""
|
|
377
|
+
rows = []
|
|
378
|
+
for candidate, row in zip(pool, scored):
|
|
379
|
+
if row["rank_status"] != "unranked" or not _matches_profile_type(candidate, profile):
|
|
380
|
+
continue
|
|
381
|
+
released = (candidate.release_date or "").strip() or None
|
|
382
|
+
rows.append({
|
|
383
|
+
"model_id": candidate.model_id,
|
|
384
|
+
"display_name": candidate.display_name,
|
|
385
|
+
"release_date": released,
|
|
386
|
+
"reason": _unranked_reason(row),
|
|
387
|
+
"missing_benchmarks": list(row["missing_benchmarks"]),
|
|
388
|
+
})
|
|
389
|
+
rows.sort(key=lambda r: r["model_id"])
|
|
390
|
+
# Stable, so equal dates keep id order; "" (undated or unparseable) sorts last.
|
|
391
|
+
rows.sort(key=lambda r: r["release_date"]
|
|
392
|
+
if r["release_date"] and _ORDERABLE_DATE.fullmatch(r["release_date"]) else "",
|
|
393
|
+
reverse=True)
|
|
394
|
+
return {"count": len(rows), "cap": UNRANKED_CANDIDATES_CAP,
|
|
395
|
+
"models": rows[:UNRANKED_CANDIDATES_CAP]}
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
#: Guide `status` values the export will carry. Anything else is dropped rather
|
|
399
|
+
#: than published as current: MODEL-65 marks version drift `stale`, and a
|
|
400
|
+
#: missing status is not a guide.
|
|
401
|
+
_EXPORTED_GUIDE_STATUSES = frozenset({"current", "stale"})
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
def authoring_guides_from_cards(cards: list[Any]) -> dict[str, dict[str, Any]]:
|
|
405
|
+
"""Card authoring guides as JSON, keyed by `model_id`.
|
|
406
|
+
|
|
407
|
+
Cards with no guide are omitted — never an empty string, never invented
|
|
408
|
+
text. The dump is what the card already holds (MODEL-8 sources and dates);
|
|
409
|
+
this function does not generate claims. Build-time only: the Worker reads
|
|
410
|
+
the map from `candidates.json` and must not import pydantic to do it.
|
|
411
|
+
"""
|
|
412
|
+
out: dict[str, dict[str, Any]] = {}
|
|
413
|
+
for card in cards:
|
|
414
|
+
ident = getattr(card, "identity", None)
|
|
415
|
+
model_id = getattr(ident, "model_id", None) if ident is not None else None
|
|
416
|
+
if not model_id:
|
|
417
|
+
continue
|
|
418
|
+
guide = getattr(card, "authoring_guide", None)
|
|
419
|
+
if guide is None:
|
|
420
|
+
continue
|
|
421
|
+
dump = getattr(guide, "model_dump", None)
|
|
422
|
+
payload = dump(mode="json") if callable(dump) else None
|
|
423
|
+
if isinstance(payload, dict) and payload.get("status") in _EXPORTED_GUIDE_STATUSES:
|
|
424
|
+
out[str(model_id)] = payload
|
|
425
|
+
return out
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def write_export(out_dir: Any, cards: list[Any], sink: CollectingSink,
|
|
429
|
+
build_json: dict[str, Any]) -> dict[str, Any]:
|
|
430
|
+
"""Emit the tables the browser needs, plus precomputed rankings."""
|
|
431
|
+
import json
|
|
432
|
+
from pathlib import Path
|
|
433
|
+
|
|
434
|
+
out = Path(out_dir)
|
|
435
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
436
|
+
|
|
437
|
+
def dump(rel: str, payload: Any) -> None:
|
|
438
|
+
path = out / rel
|
|
439
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
440
|
+
path.write_text(json.dumps(payload, sort_keys=True, default=str), encoding="utf-8")
|
|
441
|
+
|
|
442
|
+
candidates = build_candidates(cards, sink)
|
|
443
|
+
dump("profiles.json", {
|
|
444
|
+
"build": build_json,
|
|
445
|
+
"featured": list(FEATURED_PROFILES),
|
|
446
|
+
"profiles": USE_CASE_PROFILES,
|
|
447
|
+
"benchmark_ranges": {k: list(v) for k, v in BENCHMARK_RANGES.items()},
|
|
448
|
+
"ranking_policy": RANKING_POLICY,
|
|
449
|
+
})
|
|
450
|
+
# `authoring_guides` is additive on this file (MODEL-81). Scoring rows stay
|
|
451
|
+
# the shape they were; a consumer that does not read the new key is unchanged.
|
|
452
|
+
# No `export_schema_version` bump: new optional map, not a widened field.
|
|
453
|
+
dump("candidates.json", {
|
|
454
|
+
"build": build_json,
|
|
455
|
+
"count": len(candidates),
|
|
456
|
+
"candidates": [c.to_json() for c in candidates],
|
|
457
|
+
"authoring_guides": authoring_guides_from_cards(cards),
|
|
458
|
+
})
|
|
459
|
+
|
|
460
|
+
# The device vocabulary, without the 7 MB of FITS_ON edges that the graph
|
|
461
|
+
# view carries. `offline rank --fits` refuses an unknown device id rather
|
|
462
|
+
# than answering "nothing fits your GPU", because those are different
|
|
463
|
+
# answers and a caller would act on them differently. The rank Worker
|
|
464
|
+
# (MODEL-68) has to make the same distinction and cannot afford the graph
|
|
465
|
+
# view per request, so the ids ship on their own. Additive: no consumer
|
|
466
|
+
# pinning `export_schema_version` mis-parses a new file.
|
|
467
|
+
devices = sorted(
|
|
468
|
+
(props for (label, _id), props in sink.nodes.items() if label == "Hardware"),
|
|
469
|
+
key=lambda d: str(d.get("id", "")),
|
|
470
|
+
)
|
|
471
|
+
dump("hardware.json", {
|
|
472
|
+
"build": build_json,
|
|
473
|
+
"count": len(devices),
|
|
474
|
+
"hardware": devices,
|
|
475
|
+
})
|
|
476
|
+
|
|
477
|
+
precomputed = {}
|
|
478
|
+
for key in FEATURED_PROFILES:
|
|
479
|
+
precomputed[key] = rank_report(
|
|
480
|
+
candidates, key, limit=25,
|
|
481
|
+
min_benchmark_coverage=WIZARD_BENCHMARK_COVERAGE,
|
|
482
|
+
)
|
|
483
|
+
dump("rankings.json", {"schema_version": "2.0", "build": build_json, "rankings": precomputed})
|
|
484
|
+
|
|
485
|
+
return {
|
|
486
|
+
"candidates": len(candidates),
|
|
487
|
+
"hardware": len(devices),
|
|
488
|
+
"profiles": len(USE_CASE_PROFILES),
|
|
489
|
+
"featured": len(FEATURED_PROFILES),
|
|
490
|
+
"precomputed": {k: len(v["ranked"]) for k, v in precomputed.items()},
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def format_report(report: dict[str, Any]) -> str:
|
|
495
|
+
"""Human-readable contract shared by the report CLI and its tests."""
|
|
496
|
+
policy = report['policy']
|
|
497
|
+
lines = [
|
|
498
|
+
f"{report['profile']}: {report['ranking_status']} ordering; "
|
|
499
|
+
f"{report['ranked_count']} ranked, {report['unranked_count']} unranked.",
|
|
500
|
+
"Ranked by conservative composite lower bound; observed benchmark averages "
|
|
501
|
+
"do not predict missing results.",
|
|
502
|
+
f"Eligibility: at least {policy['min_benchmark_coverage']:.0%} weighted benchmark coverage "
|
|
503
|
+
f"and {policy['min_benchmark_count']} benchmarks (or all for a smaller profile). "
|
|
504
|
+
"Bounds describe missing evidence, not statistical confidence.",
|
|
505
|
+
]
|
|
506
|
+
for row in report["ranked"]:
|
|
507
|
+
lines.append(
|
|
508
|
+
f"{row['rank']:>3}. {row['score']:.2f} {row['display_name']} "
|
|
509
|
+
f"coverage {row['benchmark_coverage']:.0%}, "
|
|
510
|
+
f"bounds [{row['score_lower_bound']:.2f}, {row['score_upper_bound']:.2f}], "
|
|
511
|
+
f"benchmark basis: {row['evidence_basis']}"
|
|
512
|
+
)
|
|
513
|
+
if report["unranked"]:
|
|
514
|
+
lines.append("Unranked for insufficient benchmark evidence (alphabetical; not ranked low):")
|
|
515
|
+
for row in report["unranked"]:
|
|
516
|
+
estimate = row["benchmark_estimate"]
|
|
517
|
+
observed = "none" if estimate is None else f"{estimate:.2f}/100"
|
|
518
|
+
lines.append(
|
|
519
|
+
f" UNRANKED {row['display_name']} coverage {row['benchmark_coverage']:.0%}, "
|
|
520
|
+
f"observed benchmark average {observed}, benchmark basis: {row['evidence_basis']}"
|
|
521
|
+
)
|
|
522
|
+
return "\n".join(lines)
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def main() -> None:
|
|
526
|
+
"""Report CLI, usable without changes to the legacy list-only CLI."""
|
|
527
|
+
import argparse
|
|
528
|
+
import json
|
|
529
|
+
from pathlib import Path
|
|
530
|
+
from schema.card import ModelCard
|
|
531
|
+
from schema.graph import derive_graph
|
|
532
|
+
|
|
533
|
+
parser = argparse.ArgumentParser(description="Rank models with explicit incomplete evidence.")
|
|
534
|
+
parser.add_argument("profile", choices=sorted(USE_CASE_PROFILES))
|
|
535
|
+
parser.add_argument("--limit", type=int, default=10, help="Maximum ranked rows; all unranked models remain visible.")
|
|
536
|
+
parser.add_argument("--json", action="store_true", help="Emit the version 2 ranking report.")
|
|
537
|
+
args = parser.parse_args()
|
|
538
|
+
if args.limit < 0:
|
|
539
|
+
parser.error("--limit must be nonnegative")
|
|
540
|
+
root = Path(__file__).resolve().parents[1]
|
|
541
|
+
cards = [ModelCard.from_yaml_file(p) for p in sorted((root / "models").rglob("*.md"))
|
|
542
|
+
if p.name != "LICENSE.md"]
|
|
543
|
+
report = rank_report(build_candidates(cards, derive_graph(cards)), args.profile, args.limit)
|
|
544
|
+
if args.json:
|
|
545
|
+
print(json.dumps({"schema_version": "2.0", **report}, indent=2))
|
|
546
|
+
else:
|
|
547
|
+
print(format_report(report))
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
if __name__ == "__main__":
|
|
551
|
+
main()
|
registry/domains.yaml
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# The domain registry (MODEL-133; design §4.1 and §8).
|
|
2
|
+
#
|
|
3
|
+
# A domain is an area of capability that benchmarks can measure. It is not a
|
|
4
|
+
# use case: a use case is what a user wants, and the engine decomposes it into
|
|
5
|
+
# domains at query time. The set is open; adding a domain is an entry here.
|
|
6
|
+
# Loader: decision/registry.py.
|
|
7
|
+
#
|
|
8
|
+
# Entry fields
|
|
9
|
+
# id snake_case, unique
|
|
10
|
+
# name short label
|
|
11
|
+
# definition exact: what a benchmark must measure to count as direct
|
|
12
|
+
# evidence for this domain
|
|
13
|
+
# proxy_only true when no benchmark measures the domain directly, so every
|
|
14
|
+
# tag on it must be `proxy` and every answer says so
|
|
15
|
+
#
|
|
16
|
+
# How a benchmark page declares its domains (`schema/benchmark.py`,
|
|
17
|
+
# `BenchmarkCard.domains`, optional and additive):
|
|
18
|
+
#
|
|
19
|
+
# domains:
|
|
20
|
+
# - {id: software_engineering, directness: direct}
|
|
21
|
+
# - {id: agentic_tool_use, directness: proxy}
|
|
22
|
+
#
|
|
23
|
+
# id a domain id from this file
|
|
24
|
+
# directness direct: the benchmark's tasks are instances of the domain as
|
|
25
|
+
# defined here, so its score is evidence of that capability.
|
|
26
|
+
# proxy: the benchmark measures something correlated with the
|
|
27
|
+
# domain, or covers only a slice of it, or measures preference
|
|
28
|
+
# rather than correctness, so its score is weaker evidence and
|
|
29
|
+
# is always labelled as proxy.
|
|
30
|
+
# (`none` is the absence of a tag, never written.)
|
|
31
|
+
#
|
|
32
|
+
# A page may carry several domains; a domain may appear once per page. A page
|
|
33
|
+
# with no `domains` key is untagged, not "no domain".
|
|
34
|
+
|
|
35
|
+
schema_version: 1
|
|
36
|
+
|
|
37
|
+
domains:
|
|
38
|
+
- id: software_engineering
|
|
39
|
+
name: Software engineering
|
|
40
|
+
# Preselected only when a user explicitly opens the "Measured by" benchmark
|
|
41
|
+
# drill-down. Domain capability is the ranking default (MODEL-167).
|
|
42
|
+
default_benchmark: swe_bench_pro
|
|
43
|
+
definition: >-
|
|
44
|
+
Producing or changing working software against an objective check:
|
|
45
|
+
resolving issues in real repositories, writing code that passes held-out
|
|
46
|
+
tests, or completing programming tasks whose result is executed and
|
|
47
|
+
graded.
|
|
48
|
+
- id: engineering_stem
|
|
49
|
+
name: Engineering and STEM
|
|
50
|
+
definition: >-
|
|
51
|
+
Answering or solving graduate- and professional-level problems in the
|
|
52
|
+
natural sciences and engineering (physics, chemistry, biology, earth
|
|
53
|
+
sciences, engineering disciplines) whose answers are objectively graded.
|
|
54
|
+
- id: maths
|
|
55
|
+
name: Maths
|
|
56
|
+
definition: >-
|
|
57
|
+
Solving mathematical problems whose final answer or proof is objectively
|
|
58
|
+
checked, from school word problems to competition and research-level
|
|
59
|
+
mathematics.
|
|
60
|
+
- id: reasoning
|
|
61
|
+
name: General reasoning
|
|
62
|
+
definition: >-
|
|
63
|
+
Solving novel problems by multi-step inference that does not depend on
|
|
64
|
+
specialist knowledge: abstraction, logical deduction, commonsense
|
|
65
|
+
inference and puzzle solving with objectively graded answers.
|
|
66
|
+
- id: legal
|
|
67
|
+
name: Legal
|
|
68
|
+
definition: >-
|
|
69
|
+
Legal tasks graded against a legal professional's answer or rubric:
|
|
70
|
+
reading statutes, contracts and cases, issue spotting, and legal
|
|
71
|
+
reasoning.
|
|
72
|
+
- id: medical
|
|
73
|
+
name: Medical
|
|
74
|
+
definition: >-
|
|
75
|
+
Clinical and biomedical tasks graded against clinicians' answers or
|
|
76
|
+
rubrics: diagnosis, treatment questions, medical knowledge and
|
|
77
|
+
patient-facing health conversations.
|
|
78
|
+
- id: finance
|
|
79
|
+
name: Finance
|
|
80
|
+
definition: >-
|
|
81
|
+
Financial analysis tasks graded against expert answers: reading filings
|
|
82
|
+
and statements, financial calculations, and domain questions in
|
|
83
|
+
accounting, markets and banking.
|
|
84
|
+
- id: writing
|
|
85
|
+
name: Writing
|
|
86
|
+
definition: >-
|
|
87
|
+
Producing prose judged for quality against a rubric or by human raters:
|
|
88
|
+
creative writing, editing, summarisation and long-form composition.
|
|
89
|
+
- id: marketing_seo
|
|
90
|
+
name: Marketing and SEO
|
|
91
|
+
definition: >-
|
|
92
|
+
Marketing copy, positioning and search-engine optimisation. No benchmark
|
|
93
|
+
measures commercial effect, so all evidence here is proxy and every
|
|
94
|
+
answer says so.
|
|
95
|
+
proxy_only: true
|
|
96
|
+
- id: agentic_tool_use
|
|
97
|
+
name: Agentic and tool use
|
|
98
|
+
definition: >-
|
|
99
|
+
Completing multi-step tasks by calling tools or acting in an environment
|
|
100
|
+
(shells, browsers, operating systems, APIs, simulated businesses), graded
|
|
101
|
+
on the end state rather than on the text of the answer.
|
|
102
|
+
- id: vision_documents
|
|
103
|
+
name: Vision and documents
|
|
104
|
+
definition: >-
|
|
105
|
+
Understanding images and documents given as input: charts, diagrams,
|
|
106
|
+
scanned pages, screenshots, photographs and text in images, graded
|
|
107
|
+
against reference answers.
|
|
108
|
+
- id: multilingual
|
|
109
|
+
name: Multilingual
|
|
110
|
+
definition: >-
|
|
111
|
+
Performing tasks in languages other than English, or across languages,
|
|
112
|
+
graded per language against reference answers or native raters.
|
|
113
|
+
- id: chat_preference
|
|
114
|
+
name: Chat and preference
|
|
115
|
+
definition: >-
|
|
116
|
+
Which of two models' responses to the same open-ended prompt people
|
|
117
|
+
prefer, aggregated into a rating by pairwise comparison. Measures
|
|
118
|
+
preference, not correctness.
|
|
119
|
+
# The overall text board is the preselected explicit drill-down, not the
|
|
120
|
+
# domain's default ranking basis.
|
|
121
|
+
default_benchmark: arena_elo_style_control
|
|
122
|
+
- id: retrieval
|
|
123
|
+
name: Retrieval
|
|
124
|
+
definition: >-
|
|
125
|
+
Ranking documents or passages by relevance to a query, as an embedding or
|
|
126
|
+
reranking model does, graded against relevance judgements with ranking
|
|
127
|
+
metrics such as nDCG or recall at k.
|
|
128
|
+
# Preselect the retrieval task type, not the reranking task, when the user
|
|
129
|
+
# explicitly drills down to one benchmark.
|
|
130
|
+
default_benchmark: mteb_v2_retrieval
|