modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
api/ranking/engine.py
ADDED
|
@@ -0,0 +1,1943 @@
|
|
|
1
|
+
"""ModelSpec Ranking Engine — 4-stage pipeline.
|
|
2
|
+
|
|
3
|
+
Queries FalkorDB directly, scores models against use-case profiles,
|
|
4
|
+
and returns ranked results with human-readable explanations.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import logging
|
|
10
|
+
import math
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger("modelspec.ranking")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
# ═══════════════════════════════════════════════════════════════
|
|
18
|
+
# Benchmark normalization ranges
|
|
19
|
+
# ═══════════════════════════════════════════════════════════════
|
|
20
|
+
# Maps benchmark_id -> (min_plausible, max_plausible) for 0-100 normalization.
|
|
21
|
+
# Scores below min map to 0, above max map to 100.
|
|
22
|
+
# For ELO-based scores, the range is wider; for percentage-based, it's 0-100.
|
|
23
|
+
|
|
24
|
+
BENCHMARK_RANGES: dict[str, tuple[float, float]] = {
|
|
25
|
+
# Knowledge & Reasoning (percentage-based, 0-100)
|
|
26
|
+
"mmlu_pro": (20.0, 90.0),
|
|
27
|
+
"gpqa_diamond": (20.0, 80.0),
|
|
28
|
+
"hle": (0.0, 50.0),
|
|
29
|
+
"arc_challenge": (40.0, 100.0),
|
|
30
|
+
"hellaswag": (40.0, 100.0),
|
|
31
|
+
"truthfulqa": (20.0, 90.0),
|
|
32
|
+
"bbh": (20.0, 95.0),
|
|
33
|
+
"ifeval": (20.0, 95.0),
|
|
34
|
+
"musr": (10.0, 80.0),
|
|
35
|
+
"winogrande": (50.0, 100.0),
|
|
36
|
+
# Math
|
|
37
|
+
"math_500": (10.0, 100.0),
|
|
38
|
+
"aime_2025": (0.0, 80.0),
|
|
39
|
+
"aime_2026": (0.0, 80.0),
|
|
40
|
+
"gsm8k": (20.0, 100.0),
|
|
41
|
+
"mgsm": (10.0, 100.0),
|
|
42
|
+
# Coding
|
|
43
|
+
"humaneval": (10.0, 100.0),
|
|
44
|
+
"humaneval_plus": (10.0, 100.0),
|
|
45
|
+
"swe_bench_verified": (0.0, 70.0),
|
|
46
|
+
"live_code_bench": (0.0, 60.0),
|
|
47
|
+
"aider_polyglot": (0.0, 90.0),
|
|
48
|
+
"terminal_bench": (0.0, 80.0),
|
|
49
|
+
"mbpp": (20.0, 100.0),
|
|
50
|
+
"multipl_e": (10.0, 100.0),
|
|
51
|
+
# Multimodal
|
|
52
|
+
"mmmu": (20.0, 80.0),
|
|
53
|
+
"mathvista": (20.0, 80.0),
|
|
54
|
+
"docvqa": (40.0, 100.0),
|
|
55
|
+
"chartqa": (40.0, 100.0),
|
|
56
|
+
# Safety
|
|
57
|
+
"helm_safety": (30.0, 100.0),
|
|
58
|
+
"bbq": (30.0, 100.0),
|
|
59
|
+
"toxigen": (30.0, 100.0),
|
|
60
|
+
# Human preference (ELO-based: typical range 900-1400)
|
|
61
|
+
"arena_elo_overall": (1000.0, 1400.0),
|
|
62
|
+
"arena_elo_coding": (1000.0, 1400.0),
|
|
63
|
+
"arena_elo_math": (1000.0, 1400.0),
|
|
64
|
+
"arena_elo_vision": (1000.0, 1400.0),
|
|
65
|
+
"arena_elo_hard_prompts": (1000.0, 1400.0),
|
|
66
|
+
"arena_elo_style_control": (1000.0, 1400.0),
|
|
67
|
+
"mt_bench": (5.0, 10.0),
|
|
68
|
+
"alpaca_eval": (0.0, 60.0),
|
|
69
|
+
"wildbench": (-100.0, 100.0),
|
|
70
|
+
# Embedding (percentage or ratio-based)
|
|
71
|
+
"mteb_overall": (30.0, 80.0),
|
|
72
|
+
"mteb_retrieval": (20.0, 70.0),
|
|
73
|
+
"mteb_classification": (40.0, 90.0),
|
|
74
|
+
"mteb_clustering": (20.0, 60.0),
|
|
75
|
+
"mteb_reranking": (20.0, 70.0),
|
|
76
|
+
"mteb_sts": (40.0, 90.0),
|
|
77
|
+
"mteb_pair_classification": (50.0, 95.0),
|
|
78
|
+
"mteb_summarization": (20.0, 50.0),
|
|
79
|
+
"beir": (20.0, 70.0),
|
|
80
|
+
"miracl": (10.0, 70.0),
|
|
81
|
+
# Agentic
|
|
82
|
+
"swe_bench_agent": (0.0, 60.0),
|
|
83
|
+
"swe_bench_pro": (0.0, 80.0),
|
|
84
|
+
"swe_bench_multilingual": (0.0, 90.0),
|
|
85
|
+
"swe_bench_multimodal": (0.0, 60.0),
|
|
86
|
+
"tau_bench": (0.0, 80.0),
|
|
87
|
+
"web_arena": (0.0, 50.0),
|
|
88
|
+
"osworld": (0.0, 80.0),
|
|
89
|
+
# Agentic search
|
|
90
|
+
"browsecomp": (0.0, 90.0),
|
|
91
|
+
"hle": (0.0, 65.0),
|
|
92
|
+
"hle_tools": (0.0, 70.0),
|
|
93
|
+
# Multimodal (advanced)
|
|
94
|
+
"charxiv_reasoning": (20.0, 95.0),
|
|
95
|
+
"charxiv_reasoning_tools": (20.0, 95.0),
|
|
96
|
+
"lab_bench_figqa": (20.0, 90.0),
|
|
97
|
+
"lab_bench_figqa_tools": (20.0, 90.0),
|
|
98
|
+
"screenspot_pro": (10.0, 95.0),
|
|
99
|
+
"screenspot_pro_tools": (10.0, 95.0),
|
|
100
|
+
# Long context
|
|
101
|
+
"graphwalks_bfs_256k_1m": (0.0, 85.0),
|
|
102
|
+
"graphwalks_parents_256k_1m": (0.0, 100.0),
|
|
103
|
+
# Math (competition)
|
|
104
|
+
"usamo_2026": (0.0, 100.0),
|
|
105
|
+
"ipho_2025_theory": (0.0, 100.0),
|
|
106
|
+
# Agentic research
|
|
107
|
+
"deepsearchqa": (0.0, 80.0),
|
|
108
|
+
"frontierscience_research": (0.0, 50.0),
|
|
109
|
+
# Health / Medical (advanced)
|
|
110
|
+
"healthbench_hard": (0.0, 50.0),
|
|
111
|
+
"medxpertqa_multimodal": (20.0, 85.0),
|
|
112
|
+
# Visual reasoning
|
|
113
|
+
"zerobench": (0.0, 50.0),
|
|
114
|
+
"ai2d": (40.0, 100.0),
|
|
115
|
+
"ocrbench": (0.0, 100.0),
|
|
116
|
+
"realworldqa": (30.0, 90.0),
|
|
117
|
+
# AGI benchmarks
|
|
118
|
+
"arc_agi_2": (0.0, 80.0),
|
|
119
|
+
# Multilingual knowledge
|
|
120
|
+
"mmmlu": (40.0, 95.0),
|
|
121
|
+
# Per-language MultiPL-E (percentage-based, 0-100)
|
|
122
|
+
"multipl_e_python": (10.0, 100.0),
|
|
123
|
+
"multipl_e_rust": (10.0, 100.0),
|
|
124
|
+
"multipl_e_cpp": (10.0, 100.0),
|
|
125
|
+
"multipl_e_java": (10.0, 100.0),
|
|
126
|
+
"multipl_e_typescript": (10.0, 100.0),
|
|
127
|
+
"multipl_e_go": (10.0, 100.0),
|
|
128
|
+
"multipl_e_javascript": (10.0, 100.0),
|
|
129
|
+
"multipl_e_csharp": (10.0, 100.0),
|
|
130
|
+
"multipl_e_php": (10.0, 100.0),
|
|
131
|
+
"multipl_e_ruby": (10.0, 100.0),
|
|
132
|
+
"multipl_e_swift": (10.0, 100.0),
|
|
133
|
+
"multipl_e_kotlin": (10.0, 100.0),
|
|
134
|
+
"multipl_e_scala": (10.0, 100.0),
|
|
135
|
+
"multipl_e_r": (10.0, 100.0),
|
|
136
|
+
"multipl_e_julia": (10.0, 100.0),
|
|
137
|
+
"multipl_e_perl": (10.0, 100.0),
|
|
138
|
+
"multipl_e_lua": (10.0, 100.0),
|
|
139
|
+
# Terminal-Bench 2.0 (percentage-based, 0-100)
|
|
140
|
+
"terminal_bench_2": (0.0, 80.0),
|
|
141
|
+
# MMLU subject scores (percentage-based, 0-100)
|
|
142
|
+
# All 57 MMLU (hendrycksTest) subjects from the evaluation harness
|
|
143
|
+
"mmlu_abstract_algebra": (20.0, 75.0),
|
|
144
|
+
"mmlu_anatomy": (20.0, 90.0),
|
|
145
|
+
"mmlu_astronomy": (20.0, 95.0),
|
|
146
|
+
"mmlu_business_ethics": (20.0, 90.0),
|
|
147
|
+
"mmlu_clinical_knowledge": (20.0, 95.0),
|
|
148
|
+
"mmlu_college_biology": (20.0, 95.0),
|
|
149
|
+
"mmlu_college_chemistry": (20.0, 80.0),
|
|
150
|
+
"mmlu_college_computer_science": (20.0, 95.0),
|
|
151
|
+
"mmlu_college_mathematics": (20.0, 80.0),
|
|
152
|
+
"mmlu_college_medicine": (20.0, 90.0),
|
|
153
|
+
"mmlu_college_physics": (20.0, 80.0),
|
|
154
|
+
"mmlu_computer_security": (20.0, 95.0),
|
|
155
|
+
"mmlu_conceptual_physics": (20.0, 95.0),
|
|
156
|
+
"mmlu_econometrics": (20.0, 85.0),
|
|
157
|
+
"mmlu_electrical_engineering": (20.0, 90.0),
|
|
158
|
+
"mmlu_elementary_mathematics": (20.0, 90.0),
|
|
159
|
+
"mmlu_formal_logic": (20.0, 80.0),
|
|
160
|
+
"mmlu_global_facts": (20.0, 75.0),
|
|
161
|
+
"mmlu_high_school_biology": (20.0, 95.0),
|
|
162
|
+
"mmlu_high_school_chemistry": (20.0, 90.0),
|
|
163
|
+
"mmlu_high_school_computer_science": (20.0, 95.0),
|
|
164
|
+
"mmlu_high_school_european_history": (20.0, 95.0),
|
|
165
|
+
"mmlu_high_school_geography": (20.0, 95.0),
|
|
166
|
+
"mmlu_high_school_government_and_politics": (20.0, 100.0),
|
|
167
|
+
"mmlu_high_school_macroeconomics": (20.0, 95.0),
|
|
168
|
+
"mmlu_high_school_mathematics": (20.0, 80.0),
|
|
169
|
+
"mmlu_high_school_microeconomics": (20.0, 95.0),
|
|
170
|
+
"mmlu_high_school_physics": (20.0, 85.0),
|
|
171
|
+
"mmlu_high_school_psychology": (20.0, 98.0),
|
|
172
|
+
"mmlu_high_school_statistics": (20.0, 90.0),
|
|
173
|
+
"mmlu_high_school_us_history": (20.0, 95.0),
|
|
174
|
+
"mmlu_high_school_world_history": (20.0, 95.0),
|
|
175
|
+
"mmlu_human_aging": (20.0, 90.0),
|
|
176
|
+
"mmlu_human_sexuality": (20.0, 95.0),
|
|
177
|
+
"mmlu_international_law": (20.0, 95.0),
|
|
178
|
+
"mmlu_jurisprudence": (20.0, 90.0),
|
|
179
|
+
"mmlu_logical_fallacies": (20.0, 95.0),
|
|
180
|
+
"mmlu_machine_learning": (20.0, 85.0),
|
|
181
|
+
"mmlu_management": (20.0, 95.0),
|
|
182
|
+
"mmlu_marketing": (20.0, 95.0),
|
|
183
|
+
"mmlu_medical_genetics": (20.0, 95.0),
|
|
184
|
+
"mmlu_miscellaneous": (20.0, 95.0),
|
|
185
|
+
"mmlu_moral_disputes": (20.0, 90.0),
|
|
186
|
+
"mmlu_moral_scenarios": (20.0, 85.0),
|
|
187
|
+
"mmlu_nutrition": (20.0, 95.0),
|
|
188
|
+
"mmlu_philosophy": (20.0, 90.0),
|
|
189
|
+
"mmlu_prehistory": (20.0, 95.0),
|
|
190
|
+
"mmlu_professional_accounting": (20.0, 85.0),
|
|
191
|
+
"mmlu_professional_law": (20.0, 90.0),
|
|
192
|
+
"mmlu_professional_medicine": (20.0, 95.0),
|
|
193
|
+
"mmlu_professional_psychology": (20.0, 95.0),
|
|
194
|
+
"mmlu_public_relations": (20.0, 90.0),
|
|
195
|
+
"mmlu_security_studies": (20.0, 90.0),
|
|
196
|
+
"mmlu_sociology": (20.0, 95.0),
|
|
197
|
+
"mmlu_us_foreign_policy": (20.0, 95.0),
|
|
198
|
+
"mmlu_virology": (20.0, 80.0),
|
|
199
|
+
"mmlu_world_religions": (20.0, 95.0),
|
|
200
|
+
# Legacy aliases (mapped to new names for backward compatibility)
|
|
201
|
+
"mmlu_chemistry": (20.0, 95.0),
|
|
202
|
+
"mmlu_physics": (20.0, 95.0),
|
|
203
|
+
"mmlu_biology": (20.0, 95.0),
|
|
204
|
+
"mmlu_computer_science": (20.0, 95.0),
|
|
205
|
+
# Domain-specific extras
|
|
206
|
+
"pubmedqa": (30.0, 90.0),
|
|
207
|
+
"medmcqa": (20.0, 80.0),
|
|
208
|
+
"bioasq": (20.0, 80.0),
|
|
209
|
+
"finqa": (20.0, 90.0),
|
|
210
|
+
"convfinqa": (20.0, 80.0),
|
|
211
|
+
"fpb": (20.0, 90.0),
|
|
212
|
+
# Translation
|
|
213
|
+
"flores": (10.0, 70.0),
|
|
214
|
+
"flores_en_zh": (10.0, 50.0),
|
|
215
|
+
"flores_en_de": (10.0, 50.0),
|
|
216
|
+
"flores_en_fr": (10.0, 50.0),
|
|
217
|
+
"flores_en_es": (10.0, 50.0),
|
|
218
|
+
"flores_en_ja": (10.0, 50.0),
|
|
219
|
+
"flores_en_ko": (10.0, 50.0),
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
# ═══════════════════════════════════════════════════════════════
|
|
224
|
+
# Use case profiles
|
|
225
|
+
# ═══════════════════════════════════════════════════════════════
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
#: Benchmarks whose score is better when lower. `_normalize_benchmark` assumed
|
|
229
|
+
#: higher-is-better for everything, which silently inverted these: a worse
|
|
230
|
+
#: transcriber outranked a better one. See MODEL-30.
|
|
231
|
+
BENCHMARK_DIRECTIONS: dict[str, str] = {
|
|
232
|
+
"wer_librispeech": "lower_is_better",
|
|
233
|
+
"fid": "lower_is_better",
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
#: Ranges added in MODEL-30 for benchmarks that profiles weight but that had no
|
|
237
|
+
#: entry above, so normalisation fell back to "assume 0-100, higher is better".
|
|
238
|
+
#:
|
|
239
|
+
#: medqa, finbench and legalbench are confirmed from their benchgraph pages —
|
|
240
|
+
#: all three declare higher_is_better with unit % and max_score 100, so the old
|
|
241
|
+
#: fallback happened to be right and this simply makes it explicit.
|
|
242
|
+
#:
|
|
243
|
+
#: wer_librispeech, fid and mos_tts have no benchgraph page. Their *direction*
|
|
244
|
+
#: and scale are not in doubt, and fixing those removes the inversion. The exact
|
|
245
|
+
#: bounds are judgement and are marked for confirmation alongside MODEL-30.
|
|
246
|
+
BENCHMARK_RANGES.update({
|
|
247
|
+
"medqa": (0.0, 100.0), # confirmed from page
|
|
248
|
+
"finbench": (0.0, 100.0), # confirmed from page
|
|
249
|
+
"legalbench": (0.0, 100.0), # confirmed from page
|
|
250
|
+
"wer_librispeech": (1.5, 25.0), # word error rate %, bounds unconfirmed
|
|
251
|
+
"fid": (1.0, 100.0), # Frechet distance, bounds unconfirmed
|
|
252
|
+
"mos_tts": (1.0, 5.0), # mean opinion score, 1-5 by definition
|
|
253
|
+
})
|
|
254
|
+
|
|
255
|
+
USE_CASE_PROFILES: dict[str, dict[str, Any]] = {
|
|
256
|
+
"coding": {
|
|
257
|
+
"preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
|
|
258
|
+
"benchmark_weights": {
|
|
259
|
+
"humaneval": 0.20, "swe_bench_verified": 0.20, "live_code_bench": 0.15,
|
|
260
|
+
"aider_polyglot": 0.15, "arena_elo_coding": 0.15, "arena_elo_overall": 0.10,
|
|
261
|
+
"terminal_bench": 0.05,
|
|
262
|
+
},
|
|
263
|
+
"capability_weights": {
|
|
264
|
+
"coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
|
|
265
|
+
},
|
|
266
|
+
"cost_weight": 0.0,
|
|
267
|
+
"context_weight": 0.10,
|
|
268
|
+
},
|
|
269
|
+
"reasoning": {
|
|
270
|
+
"preferred_types": ["llm-reasoning", "llm-chat"],
|
|
271
|
+
"benchmark_weights": {
|
|
272
|
+
"gpqa_diamond": 0.20, "math_500": 0.20, "aime_2025": 0.15,
|
|
273
|
+
"mmlu_pro": 0.15, "arena_elo_overall": 0.15, "bbh": 0.10,
|
|
274
|
+
"ifeval": 0.05,
|
|
275
|
+
},
|
|
276
|
+
"capability_weights": {
|
|
277
|
+
"reasoning": 0.35, "coding": 0.15, "tool_use": 0.10,
|
|
278
|
+
},
|
|
279
|
+
"cost_weight": 0.0,
|
|
280
|
+
"context_weight": 0.10,
|
|
281
|
+
},
|
|
282
|
+
"chat": {
|
|
283
|
+
"preferred_types": ["llm-chat", "vlm", "llm-reasoning"],
|
|
284
|
+
"benchmark_weights": {
|
|
285
|
+
"arena_elo_overall": 0.30, "mt_bench": 0.15, "alpaca_eval": 0.15,
|
|
286
|
+
"ifeval": 0.15, "mmlu_pro": 0.10, "arena_elo_style_control": 0.10,
|
|
287
|
+
"wildbench": 0.05,
|
|
288
|
+
},
|
|
289
|
+
"capability_weights": {
|
|
290
|
+
"creative": 0.20, "language": 0.20, "reasoning": 0.15, "tool_use": 0.10,
|
|
291
|
+
},
|
|
292
|
+
"cost_weight": 0.0,
|
|
293
|
+
"context_weight": 0.10,
|
|
294
|
+
},
|
|
295
|
+
"embedding": {
|
|
296
|
+
"preferred_types": ["embedding-text", "embedding-multimodal"],
|
|
297
|
+
"benchmark_weights": {
|
|
298
|
+
"mteb_overall": 0.30, "mteb_retrieval": 0.25, "mteb_classification": 0.15,
|
|
299
|
+
"beir": 0.15, "miracl": 0.10, "mteb_clustering": 0.05,
|
|
300
|
+
},
|
|
301
|
+
"capability_weights": {},
|
|
302
|
+
"cost_weight": 0.0,
|
|
303
|
+
"context_weight": 0.05,
|
|
304
|
+
},
|
|
305
|
+
"vision": {
|
|
306
|
+
"preferred_types": ["vlm", "llm-chat"],
|
|
307
|
+
"benchmark_weights": {
|
|
308
|
+
"mmmu": 0.25, "mathvista": 0.20, "docvqa": 0.15, "chartqa": 0.15,
|
|
309
|
+
"arena_elo_vision": 0.15, "arena_elo_overall": 0.10,
|
|
310
|
+
},
|
|
311
|
+
"capability_weights": {
|
|
312
|
+
"reasoning": 0.15, "creative": 0.10,
|
|
313
|
+
},
|
|
314
|
+
"cost_weight": 0.0,
|
|
315
|
+
"context_weight": 0.10,
|
|
316
|
+
},
|
|
317
|
+
"agentic": {
|
|
318
|
+
"preferred_types": ["llm-reasoning", "llm-code", "llm-chat"],
|
|
319
|
+
"benchmark_weights": {
|
|
320
|
+
"swe_bench_agent": 0.20, "tau_bench": 0.15, "web_arena": 0.15,
|
|
321
|
+
"swe_bench_verified": 0.15, "arena_elo_overall": 0.15,
|
|
322
|
+
"terminal_bench": 0.10, "ifeval": 0.10,
|
|
323
|
+
},
|
|
324
|
+
"capability_weights": {
|
|
325
|
+
"tool_use": 0.25, "coding": 0.20, "reasoning": 0.20,
|
|
326
|
+
},
|
|
327
|
+
"cost_weight": 0.0,
|
|
328
|
+
"context_weight": 0.15,
|
|
329
|
+
},
|
|
330
|
+
"rag": {
|
|
331
|
+
"preferred_types": ["embedding-text", "reranker", "llm-chat"],
|
|
332
|
+
"benchmark_weights": {
|
|
333
|
+
"mteb_retrieval": 0.25, "beir": 0.20, "mteb_overall": 0.15,
|
|
334
|
+
"arena_elo_overall": 0.15, "ifeval": 0.10, "mmlu_pro": 0.10,
|
|
335
|
+
"miracl": 0.05,
|
|
336
|
+
},
|
|
337
|
+
"capability_weights": {
|
|
338
|
+
"language": 0.15, "reasoning": 0.10,
|
|
339
|
+
},
|
|
340
|
+
"cost_weight": 0.0,
|
|
341
|
+
"context_weight": 0.15,
|
|
342
|
+
},
|
|
343
|
+
"safety": {
|
|
344
|
+
"preferred_types": ["safety-classifier", "reward-model"],
|
|
345
|
+
"benchmark_weights": {
|
|
346
|
+
"helm_safety": 0.30, "bbq": 0.25, "toxigen": 0.25,
|
|
347
|
+
"arena_elo_overall": 0.20,
|
|
348
|
+
},
|
|
349
|
+
"capability_weights": {},
|
|
350
|
+
"cost_weight": 0.0,
|
|
351
|
+
"context_weight": 0.05,
|
|
352
|
+
},
|
|
353
|
+
"general": {
|
|
354
|
+
"preferred_types": ["llm-chat", "llm-reasoning", "vlm"],
|
|
355
|
+
"benchmark_weights": {
|
|
356
|
+
"arena_elo_overall": 0.25, "mmlu_pro": 0.15, "gpqa_diamond": 0.10,
|
|
357
|
+
"humaneval": 0.10, "math_500": 0.10, "ifeval": 0.10,
|
|
358
|
+
"mt_bench": 0.10, "swe_bench_verified": 0.10,
|
|
359
|
+
},
|
|
360
|
+
"capability_weights": {
|
|
361
|
+
"reasoning": 0.15, "coding": 0.15, "tool_use": 0.10, "creative": 0.10,
|
|
362
|
+
},
|
|
363
|
+
"cost_weight": 0.0,
|
|
364
|
+
"context_weight": 0.10,
|
|
365
|
+
},
|
|
366
|
+
|
|
367
|
+
# ─── Sub-domain: Coding by language ───────────────────────
|
|
368
|
+
"coding_python": {
|
|
369
|
+
"preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
|
|
370
|
+
"benchmark_weights": {
|
|
371
|
+
"multipl_e_python": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
|
|
372
|
+
"aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
|
|
373
|
+
"arena_elo_overall": 0.05,
|
|
374
|
+
},
|
|
375
|
+
"capability_weights": {
|
|
376
|
+
"coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
|
|
377
|
+
},
|
|
378
|
+
"cost_weight": 0.0,
|
|
379
|
+
"context_weight": 0.10,
|
|
380
|
+
},
|
|
381
|
+
"coding_rust": {
|
|
382
|
+
"preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
|
|
383
|
+
"benchmark_weights": {
|
|
384
|
+
"multipl_e_rust": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
|
|
385
|
+
"aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
|
|
386
|
+
"arena_elo_overall": 0.05,
|
|
387
|
+
},
|
|
388
|
+
"capability_weights": {
|
|
389
|
+
"coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
|
|
390
|
+
},
|
|
391
|
+
"cost_weight": 0.0,
|
|
392
|
+
"context_weight": 0.10,
|
|
393
|
+
},
|
|
394
|
+
"coding_go": {
|
|
395
|
+
"preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
|
|
396
|
+
"benchmark_weights": {
|
|
397
|
+
"multipl_e_go": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
|
|
398
|
+
"aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
|
|
399
|
+
"arena_elo_overall": 0.05,
|
|
400
|
+
},
|
|
401
|
+
"capability_weights": {
|
|
402
|
+
"coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
|
|
403
|
+
},
|
|
404
|
+
"cost_weight": 0.0,
|
|
405
|
+
"context_weight": 0.10,
|
|
406
|
+
},
|
|
407
|
+
"coding_typescript": {
|
|
408
|
+
"preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
|
|
409
|
+
"benchmark_weights": {
|
|
410
|
+
"multipl_e_typescript": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
|
|
411
|
+
"aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
|
|
412
|
+
"arena_elo_overall": 0.05,
|
|
413
|
+
},
|
|
414
|
+
"capability_weights": {
|
|
415
|
+
"coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
|
|
416
|
+
},
|
|
417
|
+
"cost_weight": 0.0,
|
|
418
|
+
"context_weight": 0.10,
|
|
419
|
+
},
|
|
420
|
+
"coding_cpp": {
|
|
421
|
+
"preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
|
|
422
|
+
"benchmark_weights": {
|
|
423
|
+
"multipl_e_cpp": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
|
|
424
|
+
"aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
|
|
425
|
+
"arena_elo_overall": 0.05,
|
|
426
|
+
},
|
|
427
|
+
"capability_weights": {
|
|
428
|
+
"coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
|
|
429
|
+
},
|
|
430
|
+
"cost_weight": 0.0,
|
|
431
|
+
"context_weight": 0.10,
|
|
432
|
+
},
|
|
433
|
+
"coding_java": {
|
|
434
|
+
"preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
|
|
435
|
+
"benchmark_weights": {
|
|
436
|
+
"multipl_e_java": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
|
|
437
|
+
"aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
|
|
438
|
+
"arena_elo_overall": 0.05,
|
|
439
|
+
},
|
|
440
|
+
"capability_weights": {
|
|
441
|
+
"coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
|
|
442
|
+
},
|
|
443
|
+
"cost_weight": 0.0,
|
|
444
|
+
"context_weight": 0.10,
|
|
445
|
+
},
|
|
446
|
+
"coding_javascript": {
|
|
447
|
+
"preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
|
|
448
|
+
"benchmark_weights": {
|
|
449
|
+
"multipl_e_javascript": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
|
|
450
|
+
"aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
|
|
451
|
+
"arena_elo_overall": 0.05,
|
|
452
|
+
},
|
|
453
|
+
"capability_weights": {
|
|
454
|
+
"coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
|
|
455
|
+
},
|
|
456
|
+
"cost_weight": 0.0,
|
|
457
|
+
"context_weight": 0.10,
|
|
458
|
+
},
|
|
459
|
+
|
|
460
|
+
# ─── Sub-domain: Medical ──────────────────────────────────
|
|
461
|
+
"medical": {
|
|
462
|
+
"preferred_types": ["llm-chat", "llm-reasoning"],
|
|
463
|
+
"benchmark_weights": {
|
|
464
|
+
"medqa": 0.30, "pubmedqa": 0.20, "medmcqa": 0.15,
|
|
465
|
+
"mmlu_clinical_knowledge": 0.15, "mmlu_pro": 0.10,
|
|
466
|
+
"arena_elo_overall": 0.10,
|
|
467
|
+
},
|
|
468
|
+
"capability_weights": {
|
|
469
|
+
"reasoning": 0.25, "domain": 0.20, "language": 0.10,
|
|
470
|
+
},
|
|
471
|
+
"cost_weight": 0.0,
|
|
472
|
+
"context_weight": 0.15,
|
|
473
|
+
},
|
|
474
|
+
"medical_clinical": {
|
|
475
|
+
"preferred_types": ["llm-chat", "llm-reasoning"],
|
|
476
|
+
"benchmark_weights": {
|
|
477
|
+
"medqa": 0.25, "mmlu_clinical_knowledge": 0.25, "pubmedqa": 0.15,
|
|
478
|
+
"medmcqa": 0.15, "mmlu_pro": 0.10, "arena_elo_overall": 0.10,
|
|
479
|
+
},
|
|
480
|
+
"capability_weights": {
|
|
481
|
+
"reasoning": 0.25, "domain": 0.20, "language": 0.10,
|
|
482
|
+
},
|
|
483
|
+
"cost_weight": 0.0,
|
|
484
|
+
"context_weight": 0.15,
|
|
485
|
+
},
|
|
486
|
+
"medical_radiology": {
|
|
487
|
+
"preferred_types": ["vlm", "llm-reasoning", "llm-chat"],
|
|
488
|
+
"benchmark_weights": {
|
|
489
|
+
"medqa": 0.20, "mmmu": 0.20, "mmlu_clinical_knowledge": 0.15,
|
|
490
|
+
"pubmedqa": 0.15, "arena_elo_vision": 0.15, "arena_elo_overall": 0.15,
|
|
491
|
+
},
|
|
492
|
+
"capability_weights": {
|
|
493
|
+
"reasoning": 0.25, "domain": 0.15, "creative": 0.10,
|
|
494
|
+
},
|
|
495
|
+
"cost_weight": 0.0,
|
|
496
|
+
"context_weight": 0.10,
|
|
497
|
+
},
|
|
498
|
+
|
|
499
|
+
# ─── Sub-domain: Legal ────────────────────────────────────
|
|
500
|
+
"legal": {
|
|
501
|
+
"preferred_types": ["llm-chat", "llm-reasoning"],
|
|
502
|
+
"benchmark_weights": {
|
|
503
|
+
"legalbench": 0.35, "mmlu_professional_law": 0.20,
|
|
504
|
+
"mmlu_jurisprudence": 0.15, "ifeval": 0.10,
|
|
505
|
+
"mmlu_pro": 0.10, "arena_elo_overall": 0.10,
|
|
506
|
+
},
|
|
507
|
+
"capability_weights": {
|
|
508
|
+
"reasoning": 0.25, "domain": 0.15, "language": 0.15,
|
|
509
|
+
},
|
|
510
|
+
"cost_weight": 0.0,
|
|
511
|
+
"context_weight": 0.15,
|
|
512
|
+
},
|
|
513
|
+
|
|
514
|
+
# ─── Sub-domain: Financial ────────────────────────────────
|
|
515
|
+
"financial": {
|
|
516
|
+
"preferred_types": ["llm-chat", "llm-reasoning"],
|
|
517
|
+
"benchmark_weights": {
|
|
518
|
+
"finbench": 0.30, "finqa": 0.20, "mmlu_pro": 0.15,
|
|
519
|
+
"ifeval": 0.10, "math_500": 0.10, "arena_elo_overall": 0.15,
|
|
520
|
+
},
|
|
521
|
+
"capability_weights": {
|
|
522
|
+
"reasoning": 0.25, "domain": 0.15, "language": 0.10,
|
|
523
|
+
},
|
|
524
|
+
"cost_weight": 0.0,
|
|
525
|
+
"context_weight": 0.15,
|
|
526
|
+
},
|
|
527
|
+
|
|
528
|
+
# ─── Sub-domain: Science ──────────────────────────────────
|
|
529
|
+
"science": {
|
|
530
|
+
"preferred_types": ["llm-reasoning", "llm-chat"],
|
|
531
|
+
"benchmark_weights": {
|
|
532
|
+
"gpqa_diamond": 0.25, "mmlu_pro": 0.20, "math_500": 0.15,
|
|
533
|
+
"mmlu_chemistry": 0.10, "mmlu_physics": 0.10, "mmlu_biology": 0.10,
|
|
534
|
+
"arena_elo_overall": 0.10,
|
|
535
|
+
},
|
|
536
|
+
"capability_weights": {
|
|
537
|
+
"reasoning": 0.30, "domain": 0.15, "language": 0.10,
|
|
538
|
+
},
|
|
539
|
+
"cost_weight": 0.0,
|
|
540
|
+
"context_weight": 0.10,
|
|
541
|
+
},
|
|
542
|
+
"science_chemistry": {
|
|
543
|
+
"preferred_types": ["llm-reasoning", "llm-chat"],
|
|
544
|
+
"benchmark_weights": {
|
|
545
|
+
"mmlu_chemistry": 0.30, "gpqa_diamond": 0.20, "mmlu_pro": 0.15,
|
|
546
|
+
"math_500": 0.15, "arena_elo_overall": 0.10, "ifeval": 0.10,
|
|
547
|
+
},
|
|
548
|
+
"capability_weights": {
|
|
549
|
+
"reasoning": 0.30, "domain": 0.15, "language": 0.10,
|
|
550
|
+
},
|
|
551
|
+
"cost_weight": 0.0,
|
|
552
|
+
"context_weight": 0.10,
|
|
553
|
+
},
|
|
554
|
+
"science_physics": {
|
|
555
|
+
"preferred_types": ["llm-reasoning", "llm-chat"],
|
|
556
|
+
"benchmark_weights": {
|
|
557
|
+
"mmlu_physics": 0.30, "gpqa_diamond": 0.20, "math_500": 0.15,
|
|
558
|
+
"mmlu_pro": 0.15, "arena_elo_overall": 0.10, "ifeval": 0.10,
|
|
559
|
+
},
|
|
560
|
+
"capability_weights": {
|
|
561
|
+
"reasoning": 0.30, "domain": 0.15, "language": 0.10,
|
|
562
|
+
},
|
|
563
|
+
"cost_weight": 0.0,
|
|
564
|
+
"context_weight": 0.10,
|
|
565
|
+
},
|
|
566
|
+
"science_biology": {
|
|
567
|
+
"preferred_types": ["llm-reasoning", "llm-chat"],
|
|
568
|
+
"benchmark_weights": {
|
|
569
|
+
"mmlu_biology": 0.30, "gpqa_diamond": 0.20, "mmlu_pro": 0.15,
|
|
570
|
+
"mmlu_clinical_knowledge": 0.10, "arena_elo_overall": 0.15,
|
|
571
|
+
"ifeval": 0.10,
|
|
572
|
+
},
|
|
573
|
+
"capability_weights": {
|
|
574
|
+
"reasoning": 0.30, "domain": 0.15, "language": 0.10,
|
|
575
|
+
},
|
|
576
|
+
"cost_weight": 0.0,
|
|
577
|
+
"context_weight": 0.10,
|
|
578
|
+
},
|
|
579
|
+
|
|
580
|
+
# ─── Sub-domain: Translation ──────────────────────────────
|
|
581
|
+
"translation": {
|
|
582
|
+
"preferred_types": ["llm-chat", "vlm"],
|
|
583
|
+
"benchmark_weights": {
|
|
584
|
+
"flores": 0.30, "mgsm": 0.20, "miracl": 0.20,
|
|
585
|
+
"mmlu_pro": 0.10, "arena_elo_overall": 0.10, "ifeval": 0.10,
|
|
586
|
+
},
|
|
587
|
+
"capability_weights": {
|
|
588
|
+
"language": 0.30, "creative": 0.15, "reasoning": 0.10,
|
|
589
|
+
},
|
|
590
|
+
"cost_weight": 0.0,
|
|
591
|
+
"context_weight": 0.15,
|
|
592
|
+
},
|
|
593
|
+
|
|
594
|
+
# ─── Sub-domain: Creative Writing ─────────────────────────
|
|
595
|
+
"writing_creative": {
|
|
596
|
+
"preferred_types": ["llm-chat", "llm-reasoning"],
|
|
597
|
+
"benchmark_weights": {
|
|
598
|
+
"alpaca_eval": 0.25, "wildbench": 0.20, "mt_bench": 0.20,
|
|
599
|
+
"arena_elo_style_control": 0.15, "arena_elo_overall": 0.10,
|
|
600
|
+
"ifeval": 0.10,
|
|
601
|
+
},
|
|
602
|
+
"capability_weights": {
|
|
603
|
+
"creative": 0.30, "language": 0.20, "reasoning": 0.10,
|
|
604
|
+
},
|
|
605
|
+
"cost_weight": 0.0,
|
|
606
|
+
"context_weight": 0.10,
|
|
607
|
+
},
|
|
608
|
+
"writing_technical": {
|
|
609
|
+
"preferred_types": ["llm-chat", "llm-reasoning", "llm-code"],
|
|
610
|
+
"benchmark_weights": {
|
|
611
|
+
"mt_bench": 0.20, "ifeval": 0.20, "alpaca_eval": 0.15,
|
|
612
|
+
"mmlu_pro": 0.15, "arena_elo_overall": 0.15, "wildbench": 0.15,
|
|
613
|
+
},
|
|
614
|
+
"capability_weights": {
|
|
615
|
+
"creative": 0.25, "reasoning": 0.20, "language": 0.15,
|
|
616
|
+
},
|
|
617
|
+
"cost_weight": 0.0,
|
|
618
|
+
"context_weight": 0.15,
|
|
619
|
+
},
|
|
620
|
+
"summarization": {
|
|
621
|
+
"preferred_types": ["llm-chat", "llm-reasoning"],
|
|
622
|
+
"benchmark_weights": {
|
|
623
|
+
"mt_bench": 0.20, "alpaca_eval": 0.20, "arena_elo_overall": 0.15,
|
|
624
|
+
"ifeval": 0.15, "wildbench": 0.15, "arena_elo_style_control": 0.15,
|
|
625
|
+
},
|
|
626
|
+
"capability_weights": {
|
|
627
|
+
"creative": 0.25, "language": 0.20, "reasoning": 0.15,
|
|
628
|
+
},
|
|
629
|
+
"cost_weight": 0.0,
|
|
630
|
+
"context_weight": 0.20,
|
|
631
|
+
},
|
|
632
|
+
|
|
633
|
+
# ─── Sub-domain: Math (competitive / advanced) ────────
|
|
634
|
+
"math_competition": {
|
|
635
|
+
"preferred_types": ["llm-reasoning", "llm-chat"],
|
|
636
|
+
"benchmark_weights": {
|
|
637
|
+
"aime_2025": 0.30, "math_500": 0.25, "aime_2026": 0.15,
|
|
638
|
+
"gpqa_diamond": 0.10, "arena_elo_math": 0.10, "gsm8k": 0.10,
|
|
639
|
+
},
|
|
640
|
+
"capability_weights": {
|
|
641
|
+
"reasoning": 0.35, "coding": 0.10,
|
|
642
|
+
},
|
|
643
|
+
"cost_weight": 0.0,
|
|
644
|
+
"context_weight": 0.05,
|
|
645
|
+
},
|
|
646
|
+
|
|
647
|
+
# ─── Sub-domain: Education ────────────────────────────
|
|
648
|
+
"education": {
|
|
649
|
+
"preferred_types": ["llm-chat", "llm-reasoning", "vlm"],
|
|
650
|
+
"benchmark_weights": {
|
|
651
|
+
"mmlu_pro": 0.25, "arc_challenge": 0.15, "hellaswag": 0.10,
|
|
652
|
+
"mt_bench": 0.15, "ifeval": 0.10, "arena_elo_overall": 0.15,
|
|
653
|
+
"truthfulqa": 0.10,
|
|
654
|
+
},
|
|
655
|
+
"capability_weights": {
|
|
656
|
+
"reasoning": 0.20, "language": 0.20, "creative": 0.15,
|
|
657
|
+
},
|
|
658
|
+
"cost_weight": 0.0,
|
|
659
|
+
"context_weight": 0.10,
|
|
660
|
+
},
|
|
661
|
+
"education_stem": {
|
|
662
|
+
"preferred_types": ["llm-reasoning", "llm-chat"],
|
|
663
|
+
"benchmark_weights": {
|
|
664
|
+
"mmlu_pro": 0.15, "mmlu_physics": 0.15, "mmlu_chemistry": 0.15,
|
|
665
|
+
"mmlu_biology": 0.10, "math_500": 0.15, "gpqa_diamond": 0.15,
|
|
666
|
+
"arena_elo_overall": 0.15,
|
|
667
|
+
},
|
|
668
|
+
"capability_weights": {
|
|
669
|
+
"reasoning": 0.30, "domain": 0.15, "language": 0.10,
|
|
670
|
+
},
|
|
671
|
+
"cost_weight": 0.0,
|
|
672
|
+
"context_weight": 0.10,
|
|
673
|
+
},
|
|
674
|
+
"education_humanities": {
|
|
675
|
+
"preferred_types": ["llm-chat", "llm-reasoning"],
|
|
676
|
+
"benchmark_weights": {
|
|
677
|
+
"mmlu_pro": 0.15, "mmlu_professional_law": 0.10,
|
|
678
|
+
"mmlu_business_ethics": 0.10, "truthfulqa": 0.15,
|
|
679
|
+
"alpaca_eval": 0.15, "mt_bench": 0.15, "arena_elo_overall": 0.20,
|
|
680
|
+
},
|
|
681
|
+
"capability_weights": {
|
|
682
|
+
"language": 0.25, "creative": 0.20, "reasoning": 0.15,
|
|
683
|
+
},
|
|
684
|
+
"cost_weight": 0.0,
|
|
685
|
+
"context_weight": 0.10,
|
|
686
|
+
},
|
|
687
|
+
|
|
688
|
+
# ─── Sub-domain: Data Science / Analytics ─────────────
|
|
689
|
+
"data_science": {
|
|
690
|
+
"preferred_types": ["llm-code", "llm-reasoning", "llm-chat"],
|
|
691
|
+
"benchmark_weights": {
|
|
692
|
+
"multipl_e_python": 0.20, "humaneval": 0.15, "math_500": 0.15,
|
|
693
|
+
"mmlu_pro": 0.10, "aider_polyglot": 0.10, "arena_elo_coding": 0.15,
|
|
694
|
+
"gpqa_diamond": 0.10, "arena_elo_overall": 0.05,
|
|
695
|
+
},
|
|
696
|
+
"capability_weights": {
|
|
697
|
+
"coding": 0.25, "reasoning": 0.25, "tool_use": 0.15,
|
|
698
|
+
},
|
|
699
|
+
"cost_weight": 0.0,
|
|
700
|
+
"context_weight": 0.15,
|
|
701
|
+
},
|
|
702
|
+
|
|
703
|
+
# ─── Sub-domain: Customer Support / Chatbot ───────────
|
|
704
|
+
"customer_support": {
|
|
705
|
+
"preferred_types": ["llm-chat", "vlm"],
|
|
706
|
+
"benchmark_weights": {
|
|
707
|
+
"arena_elo_overall": 0.20, "mt_bench": 0.20, "ifeval": 0.20,
|
|
708
|
+
"alpaca_eval": 0.15, "arena_elo_style_control": 0.15,
|
|
709
|
+
"truthfulqa": 0.10,
|
|
710
|
+
},
|
|
711
|
+
"capability_weights": {
|
|
712
|
+
"language": 0.25, "tool_use": 0.20, "creative": 0.10,
|
|
713
|
+
},
|
|
714
|
+
"cost_weight": 0.0,
|
|
715
|
+
"context_weight": 0.10,
|
|
716
|
+
},
|
|
717
|
+
|
|
718
|
+
# ─── Sub-domain: Content Moderation ───────────────────
|
|
719
|
+
"content_moderation": {
|
|
720
|
+
"preferred_types": ["safety-classifier", "llm-chat", "reward-model"],
|
|
721
|
+
"benchmark_weights": {
|
|
722
|
+
"helm_safety": 0.30, "toxigen": 0.25, "bbq": 0.20,
|
|
723
|
+
"ifeval": 0.10, "arena_elo_overall": 0.15,
|
|
724
|
+
},
|
|
725
|
+
"capability_weights": {
|
|
726
|
+
"safety": 0.30, "language": 0.15,
|
|
727
|
+
},
|
|
728
|
+
"cost_weight": 0.0,
|
|
729
|
+
"context_weight": 0.05,
|
|
730
|
+
},
|
|
731
|
+
|
|
732
|
+
# ─── Sub-domain: Research Assistant ───────────────────
|
|
733
|
+
"research_assistant": {
|
|
734
|
+
"preferred_types": ["llm-reasoning", "llm-chat", "vlm"],
|
|
735
|
+
"benchmark_weights": {
|
|
736
|
+
"gpqa_diamond": 0.20, "mmlu_pro": 0.15, "math_500": 0.10,
|
|
737
|
+
"arena_elo_overall": 0.15, "mt_bench": 0.10, "ifeval": 0.10,
|
|
738
|
+
"truthfulqa": 0.10, "arena_elo_hard_prompts": 0.10,
|
|
739
|
+
},
|
|
740
|
+
"capability_weights": {
|
|
741
|
+
"reasoning": 0.25, "language": 0.15, "tool_use": 0.15,
|
|
742
|
+
},
|
|
743
|
+
"cost_weight": 0.0,
|
|
744
|
+
"context_weight": 0.20,
|
|
745
|
+
},
|
|
746
|
+
|
|
747
|
+
# ─── Sub-domain: Roleplay / Character ─────────────────
|
|
748
|
+
"roleplay": {
|
|
749
|
+
"preferred_types": ["llm-chat"],
|
|
750
|
+
"benchmark_weights": {
|
|
751
|
+
"arena_elo_style_control": 0.25, "alpaca_eval": 0.20,
|
|
752
|
+
"wildbench": 0.15, "mt_bench": 0.15, "arena_elo_overall": 0.15,
|
|
753
|
+
"ifeval": 0.10,
|
|
754
|
+
},
|
|
755
|
+
"capability_weights": {
|
|
756
|
+
"creative": 0.35, "language": 0.25,
|
|
757
|
+
},
|
|
758
|
+
"cost_weight": 0.0,
|
|
759
|
+
"context_weight": 0.15,
|
|
760
|
+
},
|
|
761
|
+
|
|
762
|
+
# ─── Sub-domain: Audio ────────────────────────────────
|
|
763
|
+
"speech_to_text": {
|
|
764
|
+
"preferred_types": ["audio-stt", "audio-multimodal"],
|
|
765
|
+
"benchmark_weights": {
|
|
766
|
+
"wer_librispeech": 0.50, "arena_elo_overall": 0.20,
|
|
767
|
+
"miracl": 0.15, "mgsm": 0.15,
|
|
768
|
+
},
|
|
769
|
+
"capability_weights": {
|
|
770
|
+
"language": 0.20,
|
|
771
|
+
},
|
|
772
|
+
"cost_weight": 0.0,
|
|
773
|
+
"context_weight": 0.05,
|
|
774
|
+
},
|
|
775
|
+
"text_to_speech": {
|
|
776
|
+
"preferred_types": ["audio-tts", "audio-multimodal"],
|
|
777
|
+
"benchmark_weights": {
|
|
778
|
+
"mos_tts": 0.50, "arena_elo_overall": 0.20,
|
|
779
|
+
"arena_elo_style_control": 0.15, "mt_bench": 0.15,
|
|
780
|
+
},
|
|
781
|
+
"capability_weights": {
|
|
782
|
+
"creative": 0.15, "language": 0.15,
|
|
783
|
+
},
|
|
784
|
+
"cost_weight": 0.0,
|
|
785
|
+
"context_weight": 0.05,
|
|
786
|
+
},
|
|
787
|
+
|
|
788
|
+
# ─── Sub-domain: Image Generation ─────────────────────
|
|
789
|
+
"image_generation": {
|
|
790
|
+
"preferred_types": ["image-gen", "vlm"],
|
|
791
|
+
"benchmark_weights": {
|
|
792
|
+
"fid": 0.30, "clip_score": 0.30,
|
|
793
|
+
"arena_elo_overall": 0.20, "arena_elo_vision": 0.20,
|
|
794
|
+
},
|
|
795
|
+
"capability_weights": {
|
|
796
|
+
"creative": 0.30,
|
|
797
|
+
},
|
|
798
|
+
"cost_weight": 0.0,
|
|
799
|
+
"context_weight": 0.05,
|
|
800
|
+
},
|
|
801
|
+
|
|
802
|
+
# ─── Sub-domain: Cybersecurity ────────────────────────
|
|
803
|
+
"cybersecurity": {
|
|
804
|
+
"preferred_types": ["llm-code", "llm-reasoning", "llm-chat"],
|
|
805
|
+
"benchmark_weights": {
|
|
806
|
+
"humaneval": 0.15, "swe_bench_verified": 0.15,
|
|
807
|
+
"mmlu_computer_science": 0.15, "terminal_bench": 0.15,
|
|
808
|
+
"arena_elo_coding": 0.10, "gpqa_diamond": 0.10,
|
|
809
|
+
"arena_elo_overall": 0.10, "ifeval": 0.10,
|
|
810
|
+
},
|
|
811
|
+
"capability_weights": {
|
|
812
|
+
"coding": 0.25, "reasoning": 0.25, "tool_use": 0.20,
|
|
813
|
+
},
|
|
814
|
+
"cost_weight": 0.0,
|
|
815
|
+
"context_weight": 0.15,
|
|
816
|
+
},
|
|
817
|
+
|
|
818
|
+
# ─── Sub-domain: DevOps / Infrastructure ──────────────
|
|
819
|
+
"devops": {
|
|
820
|
+
"preferred_types": ["llm-code", "llm-chat", "llm-reasoning"],
|
|
821
|
+
"benchmark_weights": {
|
|
822
|
+
"terminal_bench": 0.25, "humaneval": 0.15, "swe_bench_verified": 0.15,
|
|
823
|
+
"aider_polyglot": 0.10, "arena_elo_coding": 0.15,
|
|
824
|
+
"ifeval": 0.10, "arena_elo_overall": 0.10,
|
|
825
|
+
},
|
|
826
|
+
"capability_weights": {
|
|
827
|
+
"coding": 0.25, "tool_use": 0.25, "reasoning": 0.15,
|
|
828
|
+
},
|
|
829
|
+
"cost_weight": 0.0,
|
|
830
|
+
"context_weight": 0.10,
|
|
831
|
+
},
|
|
832
|
+
|
|
833
|
+
# ─── Sub-domain: Multilingual ─────────────────────────
|
|
834
|
+
"multilingual": {
|
|
835
|
+
"preferred_types": ["llm-chat", "vlm"],
|
|
836
|
+
"benchmark_weights": {
|
|
837
|
+
"mgsm": 0.25, "miracl": 0.20, "flores": 0.20,
|
|
838
|
+
"mmlu_pro": 0.10, "arena_elo_overall": 0.15, "ifeval": 0.10,
|
|
839
|
+
},
|
|
840
|
+
"capability_weights": {
|
|
841
|
+
"language": 0.35, "creative": 0.10, "reasoning": 0.10,
|
|
842
|
+
},
|
|
843
|
+
"cost_weight": 0.0,
|
|
844
|
+
"context_weight": 0.15,
|
|
845
|
+
},
|
|
846
|
+
|
|
847
|
+
# ─── Sub-domain: Financial (specialties) ──────────────
|
|
848
|
+
"financial_analysis": {
|
|
849
|
+
"preferred_types": ["llm-reasoning", "llm-chat"],
|
|
850
|
+
"benchmark_weights": {
|
|
851
|
+
"finbench": 0.25, "finqa": 0.25, "math_500": 0.15,
|
|
852
|
+
"mmlu_professional_accounting": 0.10, "mmlu_pro": 0.10,
|
|
853
|
+
"arena_elo_overall": 0.15,
|
|
854
|
+
},
|
|
855
|
+
"capability_weights": {
|
|
856
|
+
"reasoning": 0.30, "domain": 0.15, "language": 0.10,
|
|
857
|
+
},
|
|
858
|
+
"cost_weight": 0.0,
|
|
859
|
+
"context_weight": 0.15,
|
|
860
|
+
},
|
|
861
|
+
"financial_compliance": {
|
|
862
|
+
"preferred_types": ["llm-chat", "llm-reasoning"],
|
|
863
|
+
"benchmark_weights": {
|
|
864
|
+
"finbench": 0.20, "legalbench": 0.20, "mmlu_professional_law": 0.15,
|
|
865
|
+
"mmlu_professional_accounting": 0.15, "ifeval": 0.15,
|
|
866
|
+
"arena_elo_overall": 0.15,
|
|
867
|
+
},
|
|
868
|
+
"capability_weights": {
|
|
869
|
+
"reasoning": 0.25, "domain": 0.15, "language": 0.15,
|
|
870
|
+
},
|
|
871
|
+
"cost_weight": 0.0,
|
|
872
|
+
"context_weight": 0.15,
|
|
873
|
+
},
|
|
874
|
+
|
|
875
|
+
# ─── Sub-domain: Science (more specialties) ───────────
|
|
876
|
+
"science_astronomy": {
|
|
877
|
+
"preferred_types": ["llm-reasoning", "llm-chat"],
|
|
878
|
+
"benchmark_weights": {
|
|
879
|
+
"mmlu_astronomy": 0.30, "gpqa_diamond": 0.20, "mmlu_physics": 0.15,
|
|
880
|
+
"math_500": 0.15, "mmlu_pro": 0.10, "arena_elo_overall": 0.10,
|
|
881
|
+
},
|
|
882
|
+
"capability_weights": {
|
|
883
|
+
"reasoning": 0.30, "domain": 0.15, "language": 0.10,
|
|
884
|
+
},
|
|
885
|
+
"cost_weight": 0.0,
|
|
886
|
+
"context_weight": 0.10,
|
|
887
|
+
},
|
|
888
|
+
|
|
889
|
+
# ─── Sub-domain: Legal (specialties) ──────────────────
|
|
890
|
+
"legal_contract_review": {
|
|
891
|
+
"preferred_types": ["llm-chat", "llm-reasoning"],
|
|
892
|
+
"benchmark_weights": {
|
|
893
|
+
"legalbench": 0.30, "mmlu_professional_law": 0.20,
|
|
894
|
+
"ifeval": 0.15, "arena_elo_overall": 0.10,
|
|
895
|
+
"mt_bench": 0.10, "mmlu_pro": 0.15,
|
|
896
|
+
},
|
|
897
|
+
"capability_weights": {
|
|
898
|
+
"reasoning": 0.25, "language": 0.20, "domain": 0.15,
|
|
899
|
+
},
|
|
900
|
+
"cost_weight": 0.0,
|
|
901
|
+
"context_weight": 0.20,
|
|
902
|
+
},
|
|
903
|
+
|
|
904
|
+
# ─── Sub-domain: Biotech / Life Sciences ──────────────
|
|
905
|
+
"biotech": {
|
|
906
|
+
"preferred_types": ["llm-reasoning", "llm-chat"],
|
|
907
|
+
"benchmark_weights": {
|
|
908
|
+
"mmlu_biology": 0.20, "mmlu_chemistry": 0.15, "medqa": 0.15,
|
|
909
|
+
"pubmedqa": 0.15, "gpqa_diamond": 0.15, "mmlu_pro": 0.10,
|
|
910
|
+
"arena_elo_overall": 0.10,
|
|
911
|
+
},
|
|
912
|
+
"capability_weights": {
|
|
913
|
+
"reasoning": 0.30, "domain": 0.20, "language": 0.10,
|
|
914
|
+
},
|
|
915
|
+
"cost_weight": 0.0,
|
|
916
|
+
"context_weight": 0.15,
|
|
917
|
+
},
|
|
918
|
+
|
|
919
|
+
# ─── Sub-domain: Accounting ───────────────────────────
|
|
920
|
+
"accounting": {
|
|
921
|
+
"preferred_types": ["llm-chat", "llm-reasoning"],
|
|
922
|
+
"benchmark_weights": {
|
|
923
|
+
"mmlu_professional_accounting": 0.30, "finbench": 0.20,
|
|
924
|
+
"finqa": 0.15, "math_500": 0.10, "ifeval": 0.10,
|
|
925
|
+
"arena_elo_overall": 0.15,
|
|
926
|
+
},
|
|
927
|
+
"capability_weights": {
|
|
928
|
+
"reasoning": 0.25, "domain": 0.15, "language": 0.10,
|
|
929
|
+
},
|
|
930
|
+
"cost_weight": 0.0,
|
|
931
|
+
"context_weight": 0.10,
|
|
932
|
+
},
|
|
933
|
+
|
|
934
|
+
# ─── Sub-domain: Code Review ──────────────────────────
|
|
935
|
+
"code_review": {
|
|
936
|
+
"preferred_types": ["llm-code", "llm-reasoning", "llm-chat"],
|
|
937
|
+
"benchmark_weights": {
|
|
938
|
+
"swe_bench_verified": 0.25, "humaneval": 0.15, "aider_polyglot": 0.15,
|
|
939
|
+
"arena_elo_coding": 0.15, "terminal_bench": 0.10,
|
|
940
|
+
"ifeval": 0.10, "arena_elo_overall": 0.10,
|
|
941
|
+
},
|
|
942
|
+
"capability_weights": {
|
|
943
|
+
"coding": 0.30, "reasoning": 0.25, "language": 0.10,
|
|
944
|
+
},
|
|
945
|
+
"cost_weight": 0.0,
|
|
946
|
+
"context_weight": 0.15,
|
|
947
|
+
},
|
|
948
|
+
}
|
|
949
|
+
|
|
950
|
+
|
|
951
|
+
# ═══════════════════════════════════════════════════════════════
|
|
952
|
+
# Platform classifications for hosting filter
|
|
953
|
+
# ═══════════════════════════════════════════════════════════════
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
# ═══════════════════════════════════════════════════════════════
|
|
957
|
+
# Verified evidence in the profiles (MODEL-32, option B)
|
|
958
|
+
# ═══════════════════════════════════════════════════════════════
|
|
959
|
+
#
|
|
960
|
+
# The census verifies benchmarks that publish current, dated results for
|
|
961
|
+
# current models, and few do. The profiles below weight the classic
|
|
962
|
+
# benchmarks a reader expects — HumanEval, SWE-bench Verified, GPQA Diamond —
|
|
963
|
+
# and the two sets did not overlap at all, so no reviewed evidence could move
|
|
964
|
+
# any ranking.
|
|
965
|
+
#
|
|
966
|
+
# Operator decision 2026-09-09: take the verified benchmarks into the profiles
|
|
967
|
+
# now, and keep verifying the classic ones as the standing goal (MODEL-33).
|
|
968
|
+
#
|
|
969
|
+
# The cap is the point. One evaluator's index carries at most this share of any
|
|
970
|
+
# profile, so reviewed evidence is load-bearing without a single source
|
|
971
|
+
# deciding what "best" means. Existing weights are scaled down proportionally,
|
|
972
|
+
# so a profile still sums to what it did before.
|
|
973
|
+
|
|
974
|
+
#: Maximum share of any profile that one evaluator's index may carry.
|
|
975
|
+
VERIFIED_INDEX_WEIGHT = 0.20
|
|
976
|
+
|
|
977
|
+
#: Verified benchmarks, mapped onto profiles by the category their benchgraph
|
|
978
|
+
#: page declares. Both declare higher_is_better, unit %, max_score 100,
|
|
979
|
+
#: which is where their ranges below come from — read, not assumed.
|
|
980
|
+
VERIFIED_ADDITIONS: dict[str, list[str]] = {
|
|
981
|
+
"coding": ["scicode"],
|
|
982
|
+
"reasoning": ["critpt"],
|
|
983
|
+
"science": ["scicode", "critpt"],
|
|
984
|
+
}
|
|
985
|
+
|
|
986
|
+
BENCHMARK_RANGES.update({
|
|
987
|
+
# Confirmed from each benchmark's page: higher_is_better, %, max 100.
|
|
988
|
+
"critpt": (0.0, 100.0),
|
|
989
|
+
"scicode": (0.0, 100.0),
|
|
990
|
+
})
|
|
991
|
+
|
|
992
|
+
|
|
993
|
+
def _apply_verified_additions() -> None:
|
|
994
|
+
"""Fold the verified benchmarks into the profiles, under the cap."""
|
|
995
|
+
for profile_key, benchmarks in VERIFIED_ADDITIONS.items():
|
|
996
|
+
profile = USE_CASE_PROFILES.get(profile_key)
|
|
997
|
+
if not profile:
|
|
998
|
+
continue
|
|
999
|
+
weights = profile.get("benchmark_weights") or {}
|
|
1000
|
+
# Anything already weighted is left alone; only genuinely new entries
|
|
1001
|
+
# take from the existing budget.
|
|
1002
|
+
new = [b for b in benchmarks if b not in weights]
|
|
1003
|
+
if not new:
|
|
1004
|
+
continue
|
|
1005
|
+
scale = 1.0 - VERIFIED_INDEX_WEIGHT
|
|
1006
|
+
rescaled = {k: round(v * scale, 4) for k, v in weights.items()}
|
|
1007
|
+
share = round(VERIFIED_INDEX_WEIGHT / len(new), 4)
|
|
1008
|
+
for benchmark in new:
|
|
1009
|
+
rescaled[benchmark] = share
|
|
1010
|
+
profile["benchmark_weights"] = rescaled
|
|
1011
|
+
|
|
1012
|
+
|
|
1013
|
+
_apply_verified_additions()
|
|
1014
|
+
|
|
1015
|
+
# Product defaults, not statistical confidence thresholds. The benchmark set
|
|
1016
|
+
# stays fixed, including when a candidate or an evaluator has sparse coverage.
|
|
1017
|
+
# CLI / API stay conservative. The wizard is a different surface: a browser
|
|
1018
|
+
# wants breadth, dpf wants a shortlist it can defend. MODEL-34, 2026-09-11.
|
|
1019
|
+
MIN_BENCHMARK_COVERAGE = 0.50
|
|
1020
|
+
WIZARD_BENCHMARK_COVERAGE = 0.25
|
|
1021
|
+
MIN_BENCHMARK_COUNT = 2
|
|
1022
|
+
|
|
1023
|
+
|
|
1024
|
+
# ── the neutrality commitment (MODEL-70) ─────────────────────────────────────
|
|
1025
|
+
#
|
|
1026
|
+
# A floor is checkable because it is published as a number next to the answer it
|
|
1027
|
+
# shaped. The neutrality claim was only ever prose, which means a caller had to
|
|
1028
|
+
# trust it. It ships here, beside the floors, for the same reason the floors
|
|
1029
|
+
# ship: so an agent can read the commitment out of the same object that carries
|
|
1030
|
+
# the policy it acted on, rather than believe a page it never fetched.
|
|
1031
|
+
#
|
|
1032
|
+
# This constant is the single source. `ranking_policy()` carries it into
|
|
1033
|
+
# `/api/rank/profiles.json`, into every `rank_report()` and therefore into
|
|
1034
|
+
# `rankings.json`, and into the `policy` block of every `POST /v1/rank`
|
|
1035
|
+
# response. `pipeline/legal.py` renders the published pages from the same
|
|
1036
|
+
# values, so the prose and the JSON cannot drift apart.
|
|
1037
|
+
#
|
|
1038
|
+
# Source of the rule: `docs/agent-commerce-assessment.md` §3, which requires it
|
|
1039
|
+
# to be written into the terms before any money moves.
|
|
1040
|
+
|
|
1041
|
+
#: Verbatim. Quoted in `docs/legal/terms-of-service.md` and rendered on
|
|
1042
|
+
#: https://modelspec.dev/legal/terms/ — three copies of one string, checked by
|
|
1043
|
+
#: `tests/test_legal.py`. Editing it here is editing the published terms.
|
|
1044
|
+
HONEST_BROKER_RULE = (
|
|
1045
|
+
"Charging the consumer of a recommendation is compatible with being an "
|
|
1046
|
+
"honest broker. Charging the subjects of one is not."
|
|
1047
|
+
)
|
|
1048
|
+
|
|
1049
|
+
#: Verbatim, and the word "permanently" is load-bearing: it is the difference
|
|
1050
|
+
#: between a current price list and a commitment.
|
|
1051
|
+
NEUTRALITY_PLEDGE = (
|
|
1052
|
+
"No referral fees, no paid placement, no provider-paid visibility, "
|
|
1053
|
+
"permanently."
|
|
1054
|
+
)
|
|
1055
|
+
|
|
1056
|
+
#: Verbatim, neutrality 1.1 (2026-09-23). The case the pledge does not reach:
|
|
1057
|
+
#: money flowing *to* a vendor we catalogue rather than from one. Quoted in
|
|
1058
|
+
#: `docs/legal/neutrality.md`; `tests/test_legal.py` holds the two equal. The
|
|
1059
|
+
#: mechanisms are `schema/suppliers.py` (the vendors we pay), the disclosure the
|
|
1060
|
+
#: card page prints from it (`pipeline/render.py`), and the refusals in
|
|
1061
|
+
#: `scripts/attribution.py` (`supplier_conflict`, `apply_policy`). MODEL-101.
|
|
1062
|
+
VENDOR_PURCHASE_RULE = (
|
|
1063
|
+
"We may be a paying customer of a vendor whose models we catalogue. When we "
|
|
1064
|
+
"are, the card says so, and no field on that vendor's card is ever set by "
|
|
1065
|
+
"that vendor's own model."
|
|
1066
|
+
)
|
|
1067
|
+
|
|
1068
|
+
#: Where a machine reads the long forms. Static Pages, no key, no account.
|
|
1069
|
+
LEGAL_BASE_URL = "https://modelspec.dev/legal"
|
|
1070
|
+
|
|
1071
|
+
|
|
1072
|
+
def neutrality_commitment() -> dict[str, Any]:
|
|
1073
|
+
"""What ModelSpec will not take money for, in a shape an agent can check.
|
|
1074
|
+
|
|
1075
|
+
Every `assertion` is negative on purpose. The positioning is a set of things
|
|
1076
|
+
the service is structurally unable to do, not a set of things it promises
|
|
1077
|
+
not to do, so each one is either contradicted by an observable fact or it
|
|
1078
|
+
holds. A new assertion is additive; flipping one of these `false` values to
|
|
1079
|
+
`true` is not a version bump, it is a different product.
|
|
1080
|
+
"""
|
|
1081
|
+
return {
|
|
1082
|
+
"version": "neutrality-v1",
|
|
1083
|
+
"operator": "Sparks and Sawdust LLC",
|
|
1084
|
+
"rule": HONEST_BROKER_RULE,
|
|
1085
|
+
"pledge": NEUTRALITY_PLEDGE,
|
|
1086
|
+
"permanent": True,
|
|
1087
|
+
"assertions": {
|
|
1088
|
+
# §4.5: never charge the subjects of a recommendation.
|
|
1089
|
+
"accepts_referral_fees": False,
|
|
1090
|
+
"accepts_paid_placement": False,
|
|
1091
|
+
"accepts_provider_paid_visibility": False,
|
|
1092
|
+
# §10.1: recommend and hand off. A router earns on token volume,
|
|
1093
|
+
# and margin that grows with volume is a steering incentive.
|
|
1094
|
+
"proxies_inference_tokens": False,
|
|
1095
|
+
# §10.2: a request carries a profile, not a prompt. There is
|
|
1096
|
+
# nothing to retain, which is why this is architecture and not a
|
|
1097
|
+
# retention promise.
|
|
1098
|
+
"stores_customer_prompts": False,
|
|
1099
|
+
# Neutrality 1.1: buying from a vendor we catalogue. The card
|
|
1100
|
+
# page discloses it (`pipeline/render.py`, from
|
|
1101
|
+
# `schema/suppliers.py`), and `scripts/attribution.py` refuses a
|
|
1102
|
+
# supplier's model any field on that supplier's card.
|
|
1103
|
+
"conceals_purchases_from_catalogued_vendors": False,
|
|
1104
|
+
"lets_supplier_models_write_supplier_cards": False,
|
|
1105
|
+
},
|
|
1106
|
+
#: §10.3: neutrality past money, into sourcing. Naming the stages is the
|
|
1107
|
+
#: point — "neutral ranking" would leave the tie-break and the hosting
|
|
1108
|
+
#: suggestion unclaimed, and those are where a lean is cheapest to hide.
|
|
1109
|
+
"source_neutral_at": [
|
|
1110
|
+
"ranking",
|
|
1111
|
+
"tie_breaks",
|
|
1112
|
+
"hosting_suggestions",
|
|
1113
|
+
"route_advice",
|
|
1114
|
+
],
|
|
1115
|
+
"charges": "the consumer of a recommendation, never its subjects",
|
|
1116
|
+
"vendor_purchases": VENDOR_PURCHASE_RULE,
|
|
1117
|
+
"method_source": (
|
|
1118
|
+
"https://github.com/turbobeest/modelspec/blob/main/api/ranking/engine.py"
|
|
1119
|
+
),
|
|
1120
|
+
"terms_url": f"{LEGAL_BASE_URL}/terms/",
|
|
1121
|
+
"neutrality_url": f"{LEGAL_BASE_URL}/neutrality/",
|
|
1122
|
+
"privacy_url": f"{LEGAL_BASE_URL}/privacy/",
|
|
1123
|
+
}
|
|
1124
|
+
|
|
1125
|
+
|
|
1126
|
+
def ranking_policy(*, min_benchmark_coverage: float | None = None) -> dict[str, Any]:
|
|
1127
|
+
"""Policy for one ranking surface. Default is the CLI floor."""
|
|
1128
|
+
coverage = MIN_BENCHMARK_COVERAGE if min_benchmark_coverage is None else min_benchmark_coverage
|
|
1129
|
+
return {
|
|
1130
|
+
"version": "incomplete-evidence-v1",
|
|
1131
|
+
"ordering": "conservative_lower_bound",
|
|
1132
|
+
"min_benchmark_coverage": coverage,
|
|
1133
|
+
"cli_min_benchmark_coverage": MIN_BENCHMARK_COVERAGE,
|
|
1134
|
+
"wizard_min_benchmark_coverage": WIZARD_BENCHMARK_COVERAGE,
|
|
1135
|
+
"min_benchmark_count": MIN_BENCHMARK_COUNT,
|
|
1136
|
+
"limit_applies_to": "ranked_only",
|
|
1137
|
+
"uncertainty": "missing-benchmark bounds, not statistical confidence intervals",
|
|
1138
|
+
# Additive under the contract's own rule (docs/cli-contract.md: "New
|
|
1139
|
+
# fields may be added to any object"). No existing field widens.
|
|
1140
|
+
"neutrality": neutrality_commitment(),
|
|
1141
|
+
}
|
|
1142
|
+
|
|
1143
|
+
|
|
1144
|
+
RANKING_POLICY = ranking_policy()
|
|
1145
|
+
|
|
1146
|
+
|
|
1147
|
+
class IncompleteEvidenceError(ValueError):
|
|
1148
|
+
"""Optional complete-ordering signal wrapping a rank_report().
|
|
1149
|
+
|
|
1150
|
+
`rank()` returns the ranked shortlist and does not raise when other models
|
|
1151
|
+
lack evidence. Callers that require every candidate to be ordered may raise
|
|
1152
|
+
this themselves after inspecting rank_report(). `report` preserves both the
|
|
1153
|
+
ranked shortlist and every unranked candidate.
|
|
1154
|
+
"""
|
|
1155
|
+
|
|
1156
|
+
def __init__(self, report: dict[str, Any]):
|
|
1157
|
+
self.report = report
|
|
1158
|
+
names = []
|
|
1159
|
+
for row in report["unranked"]:
|
|
1160
|
+
if isinstance(row, dict):
|
|
1161
|
+
names.append(f"{row['display_name']} ({row['benchmark_coverage']:.0%} coverage)")
|
|
1162
|
+
else:
|
|
1163
|
+
names.append(f"{row.display_name} ({row.benchmark_coverage:.0%} coverage)")
|
|
1164
|
+
super().__init__(
|
|
1165
|
+
"Cannot return a total ordering. Unranked for insufficient evidence: "
|
|
1166
|
+
+ "; ".join(names)
|
|
1167
|
+
+ ". These models are not ranked low. Use rank_report() or "
|
|
1168
|
+
"`.venv/bin/python -m pipeline.ranking PROFILE` to see ranked and unranked results."
|
|
1169
|
+
)
|
|
1170
|
+
|
|
1171
|
+
|
|
1172
|
+
def _benchmark_evidence(scores: dict[str, float], profile: dict[str, Any],
|
|
1173
|
+
min_coverage: float | None = None) -> dict[str, Any]:
|
|
1174
|
+
"""Bound the fixed profile without guessing unmeasured benchmark values.
|
|
1175
|
+
|
|
1176
|
+
All normalized benchmarks lie in [0, 100]. Missing weight therefore spans
|
|
1177
|
+
[0, weight * 100], rather than being a measurement of zero. Other composite
|
|
1178
|
+
components are held fixed; these are not bounds on real-world ability.
|
|
1179
|
+
"""
|
|
1180
|
+
weights = profile.get("benchmark_weights", {})
|
|
1181
|
+
if any(not math.isfinite(w) or w < 0 for w in weights.values()):
|
|
1182
|
+
raise ValueError("Benchmark weights must be finite and nonnegative")
|
|
1183
|
+
weights = {b: w for b, w in weights.items() if w > 0}
|
|
1184
|
+
present = {
|
|
1185
|
+
b: _normalize_benchmark(b, scores[b]) for b in weights
|
|
1186
|
+
if scores.get(b) is not None and math.isfinite(scores[b])
|
|
1187
|
+
}
|
|
1188
|
+
total_weight = sum(weights.values())
|
|
1189
|
+
present_weight = sum(weights[b] for b in present)
|
|
1190
|
+
missing = sorted(weights.keys() - present.keys())
|
|
1191
|
+
missing_weight = sum(weights[b] for b in missing)
|
|
1192
|
+
coverage = present_weight / total_weight if total_weight else 0.0
|
|
1193
|
+
required = min(MIN_BENCHMARK_COUNT, len(weights))
|
|
1194
|
+
floor = MIN_BENCHMARK_COVERAGE if min_coverage is None else min_coverage
|
|
1195
|
+
rankable = (total_weight > 0 and coverage + 1e-12 >= floor
|
|
1196
|
+
and len(present) >= required)
|
|
1197
|
+
lower = sum(present[b] * weights[b] for b in present) * 0.40
|
|
1198
|
+
return {
|
|
1199
|
+
"rank_status": "ranked" if rankable else "unranked",
|
|
1200
|
+
"unranked_reason": None if rankable else "insufficient_benchmark_evidence",
|
|
1201
|
+
"benchmark_coverage": coverage,
|
|
1202
|
+
"benchmark_count": len(present),
|
|
1203
|
+
"required_benchmark_count": required,
|
|
1204
|
+
"missing_benchmarks": missing,
|
|
1205
|
+
"benchmark_estimate": lower / (0.40 * present_weight) if present_weight else None,
|
|
1206
|
+
"benchmark_lower_bound": lower,
|
|
1207
|
+
"benchmark_upper_bound": lower + missing_weight * 40.0,
|
|
1208
|
+
"benchmark_contributions": {b: round(present[b] * weights[b], 2) for b in present},
|
|
1209
|
+
}
|
|
1210
|
+
|
|
1211
|
+
|
|
1212
|
+
def _ranking_status(ranked: list, unranked: list) -> str:
|
|
1213
|
+
if unranked:
|
|
1214
|
+
return "partial" if ranked else "unavailable"
|
|
1215
|
+
return "complete" if ranked else "empty"
|
|
1216
|
+
|
|
1217
|
+
|
|
1218
|
+
CLOUD_PLATFORMS = {
|
|
1219
|
+
"aws_bedrock", "azure_ai_foundry", "google_vertex_ai", "nvidia_nim",
|
|
1220
|
+
"ibm_watsonx", "snowflake_cortex", "groq", "together_ai", "fireworks_ai",
|
|
1221
|
+
"replicate", "deepinfra", "cerebras", "sambanova", "openrouter",
|
|
1222
|
+
}
|
|
1223
|
+
|
|
1224
|
+
PROVIDER_PLATFORMS = {
|
|
1225
|
+
"anthropic", "openai", "google", "mistral", "cohere", "xai", "deepseek",
|
|
1226
|
+
"claude_ai", "chatgpt", "gemini_app", "grok_xai", "meta_ai",
|
|
1227
|
+
"mistral_plateforme", "ai21_labs",
|
|
1228
|
+
}
|
|
1229
|
+
|
|
1230
|
+
LOCAL_PLATFORMS = {
|
|
1231
|
+
"ollama", "lm_studio", "gpt4all", "jan_ai", "mlx_community", "open_webui",
|
|
1232
|
+
}
|
|
1233
|
+
|
|
1234
|
+
# Platforms that map to runtime identifiers (used to derive runtimes from graph edges)
|
|
1235
|
+
RUNTIME_PLATFORMS = {
|
|
1236
|
+
"ollama", "lm_studio", "gpt4all",
|
|
1237
|
+
}
|
|
1238
|
+
|
|
1239
|
+
|
|
1240
|
+
# ═══════════════════════════════════════════════════════════════
|
|
1241
|
+
# Data classes
|
|
1242
|
+
# ═══════════════════════════════════════════════════════════════
|
|
1243
|
+
|
|
1244
|
+
@dataclass
|
|
1245
|
+
class ModelData:
|
|
1246
|
+
"""All data gathered for one model during ranking."""
|
|
1247
|
+
model_id: str
|
|
1248
|
+
display_name: str
|
|
1249
|
+
model_type: str | None = None
|
|
1250
|
+
model_subtypes: list[str] = field(default_factory=list)
|
|
1251
|
+
status: str | None = None
|
|
1252
|
+
open_weights: bool | None = None
|
|
1253
|
+
origin_country: str | None = None
|
|
1254
|
+
total_parameters: int | None = None
|
|
1255
|
+
active_parameters: int | None = None
|
|
1256
|
+
context_window: int | None = None
|
|
1257
|
+
cost_input: float | None = None
|
|
1258
|
+
cost_output: float | None = None
|
|
1259
|
+
arena_elo_overall: float | None = None
|
|
1260
|
+
release_date: str | None = None
|
|
1261
|
+
reasoning: bool = False
|
|
1262
|
+
tool_call: bool = False
|
|
1263
|
+
vision_input: bool = False
|
|
1264
|
+
# Populated from graph edges
|
|
1265
|
+
benchmark_scores: dict[str, float] = field(default_factory=dict)
|
|
1266
|
+
capability_tiers: dict[str, str] = field(default_factory=dict)
|
|
1267
|
+
available_platforms: set[str] = field(default_factory=set)
|
|
1268
|
+
estimated_tps: float | None = None # Estimated tok/s on target hardware
|
|
1269
|
+
concurrent_instances: int | None = None # How many instances fit on target hardware
|
|
1270
|
+
hardware_fits: dict[str, dict] = field(default_factory=dict)
|
|
1271
|
+
# Tags and runtimes from the graph
|
|
1272
|
+
tags: set[str] = field(default_factory=set)
|
|
1273
|
+
runtimes: set[str] = field(default_factory=set)
|
|
1274
|
+
# Extra node props for pass-through
|
|
1275
|
+
provider: str | None = None
|
|
1276
|
+
|
|
1277
|
+
|
|
1278
|
+
@dataclass
|
|
1279
|
+
class ScoredModel:
|
|
1280
|
+
"""Result of scoring a model."""
|
|
1281
|
+
model_id: str
|
|
1282
|
+
display_name: str
|
|
1283
|
+
model_type: str | None
|
|
1284
|
+
score: float | None
|
|
1285
|
+
benchmark_score: float
|
|
1286
|
+
capability_score: float
|
|
1287
|
+
cost_score: float
|
|
1288
|
+
context_score: float
|
|
1289
|
+
type_bonus: float
|
|
1290
|
+
speed_score: float = 0.0
|
|
1291
|
+
estimated_tps: float | None = None
|
|
1292
|
+
concurrent_instances: int | None = None
|
|
1293
|
+
reasons: list[str] = field(default_factory=list)
|
|
1294
|
+
benchmark_contributions: dict[str, float] = field(default_factory=dict)
|
|
1295
|
+
# Pass-through for the response
|
|
1296
|
+
arena_elo_overall: float | None = None
|
|
1297
|
+
total_parameters: int | None = None
|
|
1298
|
+
context_window: int | None = None
|
|
1299
|
+
cost_input: float | None = None
|
|
1300
|
+
cost_output: float | None = None
|
|
1301
|
+
open_weights: bool | None = None
|
|
1302
|
+
provider: str | None = None
|
|
1303
|
+
status: str | None = None
|
|
1304
|
+
rank: int | None = None
|
|
1305
|
+
rank_status: str = "unranked"
|
|
1306
|
+
unranked_reason: str | None = None
|
|
1307
|
+
benchmark_coverage: float = 0.0
|
|
1308
|
+
benchmark_count: int = 0
|
|
1309
|
+
required_benchmark_count: int = 0
|
|
1310
|
+
missing_benchmarks: list[str] = field(default_factory=list)
|
|
1311
|
+
benchmark_estimate: float | None = None
|
|
1312
|
+
benchmark_lower_bound: float = 0.0
|
|
1313
|
+
benchmark_upper_bound: float = 0.0
|
|
1314
|
+
score_lower_bound: float = 0.0
|
|
1315
|
+
score_upper_bound: float = 0.0
|
|
1316
|
+
|
|
1317
|
+
|
|
1318
|
+
# ═══════════════════════════════════════════════════════════════
|
|
1319
|
+
# Ranking Engine
|
|
1320
|
+
# ═══════════════════════════════════════════════════════════════
|
|
1321
|
+
|
|
1322
|
+
class RankingEngine:
|
|
1323
|
+
"""4-stage ranking pipeline: Filter -> Score -> Rank -> Explain."""
|
|
1324
|
+
|
|
1325
|
+
def __init__(self, graph):
|
|
1326
|
+
self.graph = graph
|
|
1327
|
+
|
|
1328
|
+
# ─── Main entry point ────────────────────────────────────
|
|
1329
|
+
|
|
1330
|
+
def rank(
|
|
1331
|
+
self,
|
|
1332
|
+
use_case: str | None = None,
|
|
1333
|
+
hardware: str | None = None,
|
|
1334
|
+
constraints: dict[str, Any] | None = None,
|
|
1335
|
+
limit: int = 10,
|
|
1336
|
+
) -> list[ScoredModel]:
|
|
1337
|
+
"""Return the models that can honestly be ordered for this profile.
|
|
1338
|
+
|
|
1339
|
+
Unrankable models are the normal catalogue state, not an error. This
|
|
1340
|
+
returns the ranked shortlist, which may be empty when nothing has
|
|
1341
|
+
enough evidence. Use rank_report() to see withheld models and
|
|
1342
|
+
ranking_status.
|
|
1343
|
+
"""
|
|
1344
|
+
return self.rank_report(use_case, hardware, constraints, limit)["ranked"]
|
|
1345
|
+
|
|
1346
|
+
def rank_report(
|
|
1347
|
+
self,
|
|
1348
|
+
use_case: str | None = None,
|
|
1349
|
+
hardware: str | None = None,
|
|
1350
|
+
constraints: dict[str, Any] | None = None,
|
|
1351
|
+
limit: int = 10,
|
|
1352
|
+
) -> dict[str, Any]:
|
|
1353
|
+
"""Return a bounded shortlist and all unranked candidates separately."""
|
|
1354
|
+
if limit < 0:
|
|
1355
|
+
raise ValueError("limit must be nonnegative")
|
|
1356
|
+
constraints = constraints or {}
|
|
1357
|
+
profile = USE_CASE_PROFILES.get(use_case or "general", USE_CASE_PROFILES["general"])
|
|
1358
|
+
|
|
1359
|
+
# Stage 0: Fetch all candidate models from the graph
|
|
1360
|
+
candidates = self._fetch_candidates(hardware, constraints)
|
|
1361
|
+
logger.info(f"Fetched {len(candidates)} candidates from graph")
|
|
1362
|
+
|
|
1363
|
+
# Stage 1: Filter
|
|
1364
|
+
filtered = self._filter(candidates, constraints, profile)
|
|
1365
|
+
logger.info(f"After filtering: {len(filtered)} models remain")
|
|
1366
|
+
|
|
1367
|
+
# Stage 2: Score
|
|
1368
|
+
scored = [self._score(m, profile) for m in filtered]
|
|
1369
|
+
|
|
1370
|
+
# Stage 3: Rank (sort by score, tie-break by ELO then params)
|
|
1371
|
+
ranked = [s for s in scored if s.rank_status == "ranked"]
|
|
1372
|
+
unranked = [s for s in scored if s.rank_status == "unranked"]
|
|
1373
|
+
ranked.sort(key=lambda s: (
|
|
1374
|
+
s.score,
|
|
1375
|
+
s.arena_elo_overall or 0,
|
|
1376
|
+
s.total_parameters or 0,
|
|
1377
|
+
), reverse=True)
|
|
1378
|
+
unranked.sort(key=lambda s: (s.display_name.lower(), s.model_id))
|
|
1379
|
+
for position, result in enumerate(ranked, 1):
|
|
1380
|
+
result.rank = position
|
|
1381
|
+
|
|
1382
|
+
# Stage 4: Explain (already built into scoring, but add rank-relative info)
|
|
1383
|
+
self._explain(scored, profile, use_case)
|
|
1384
|
+
|
|
1385
|
+
return {
|
|
1386
|
+
"ranking_status": _ranking_status(ranked, unranked),
|
|
1387
|
+
"policy": dict(RANKING_POLICY),
|
|
1388
|
+
"ranked_count": len(ranked), "unranked_count": len(unranked),
|
|
1389
|
+
"ranked": ranked[:limit], "unranked": unranked,
|
|
1390
|
+
}
|
|
1391
|
+
|
|
1392
|
+
# ─── Stage 0: Fetch candidates ───────────────────────────
|
|
1393
|
+
|
|
1394
|
+
def _fetch_candidates(
|
|
1395
|
+
self,
|
|
1396
|
+
hardware: str | None,
|
|
1397
|
+
constraints: dict[str, Any],
|
|
1398
|
+
) -> list[ModelData]:
|
|
1399
|
+
"""Query FalkorDB for all model nodes with their edges."""
|
|
1400
|
+
# Build the base query with optional hardware join
|
|
1401
|
+
if hardware:
|
|
1402
|
+
base_q = (
|
|
1403
|
+
"MATCH (m:Model)-[fit:FITS_ON]->(h:Hardware {id: $hw_id}) "
|
|
1404
|
+
"RETURN m"
|
|
1405
|
+
)
|
|
1406
|
+
params: dict[str, Any] = {"hw_id": hardware}
|
|
1407
|
+
else:
|
|
1408
|
+
base_q = "MATCH (m:Model) RETURN m"
|
|
1409
|
+
params = {}
|
|
1410
|
+
|
|
1411
|
+
result = self.graph.query(base_q, params)
|
|
1412
|
+
model_ids: list[str] = []
|
|
1413
|
+
models_by_id: dict[str, ModelData] = {}
|
|
1414
|
+
|
|
1415
|
+
for row in result.result_set:
|
|
1416
|
+
node = row[0]
|
|
1417
|
+
props = dict(node.properties) if node.properties else {}
|
|
1418
|
+
mid = props.get("id", "")
|
|
1419
|
+
if not mid:
|
|
1420
|
+
continue
|
|
1421
|
+
|
|
1422
|
+
md = ModelData(
|
|
1423
|
+
model_id=mid,
|
|
1424
|
+
display_name=props.get("display_name", mid),
|
|
1425
|
+
model_type=props.get("model_type"),
|
|
1426
|
+
status=props.get("status"),
|
|
1427
|
+
open_weights=props.get("open_weights"),
|
|
1428
|
+
origin_country=props.get("origin_country") or None,
|
|
1429
|
+
total_parameters=_safe_int(props.get("total_parameters")),
|
|
1430
|
+
active_parameters=_safe_int(props.get("active_parameters")),
|
|
1431
|
+
context_window=_safe_int(props.get("context_window")),
|
|
1432
|
+
cost_input=_safe_float(props.get("cost_input")),
|
|
1433
|
+
cost_output=_safe_float(props.get("cost_output")),
|
|
1434
|
+
arena_elo_overall=_safe_float(props.get("arena_elo_overall")),
|
|
1435
|
+
release_date=props.get("release_date"),
|
|
1436
|
+
reasoning=bool(props.get("reasoning")),
|
|
1437
|
+
tool_call=bool(props.get("tool_call")),
|
|
1438
|
+
vision_input=bool(props.get("vision_input")),
|
|
1439
|
+
provider=mid.split("/")[0] if "/" in mid else None,
|
|
1440
|
+
model_subtypes=props.get("model_subtypes", "").split(",") if props.get("model_subtypes") else [],
|
|
1441
|
+
)
|
|
1442
|
+
models_by_id[mid] = md
|
|
1443
|
+
model_ids.append(mid)
|
|
1444
|
+
|
|
1445
|
+
if not model_ids:
|
|
1446
|
+
return []
|
|
1447
|
+
|
|
1448
|
+
# Batch-fetch benchmark scores via SCORED_ON edges
|
|
1449
|
+
bench_q = (
|
|
1450
|
+
"MATCH (m:Model)-[e:SCORED_ON]->(b:Benchmark) "
|
|
1451
|
+
"RETURN m.id, b.id, e.value"
|
|
1452
|
+
)
|
|
1453
|
+
bench_result = self.graph.query(bench_q)
|
|
1454
|
+
for row in bench_result.result_set:
|
|
1455
|
+
mid, bench_id, value = row
|
|
1456
|
+
if mid in models_by_id and value is not None:
|
|
1457
|
+
models_by_id[mid].benchmark_scores[bench_id] = float(value)
|
|
1458
|
+
|
|
1459
|
+
# Batch-fetch capability tiers via HAS_CAPABILITY edges
|
|
1460
|
+
cap_q = (
|
|
1461
|
+
"MATCH (m:Model)-[e:HAS_CAPABILITY]->(c:Capability) "
|
|
1462
|
+
"RETURN m.id, c.id, e.tier"
|
|
1463
|
+
)
|
|
1464
|
+
cap_result = self.graph.query(cap_q)
|
|
1465
|
+
for row in cap_result.result_set:
|
|
1466
|
+
mid, cap_id, tier = row
|
|
1467
|
+
if mid in models_by_id and tier:
|
|
1468
|
+
# Keep the best tier if multiple edges exist
|
|
1469
|
+
existing = models_by_id[mid].capability_tiers.get(cap_id)
|
|
1470
|
+
if existing is None or _tier_rank(tier) < _tier_rank(existing):
|
|
1471
|
+
models_by_id[mid].capability_tiers[cap_id] = tier
|
|
1472
|
+
|
|
1473
|
+
# Batch-fetch platform availability via AVAILABLE_ON edges
|
|
1474
|
+
plat_q = (
|
|
1475
|
+
"MATCH (m:Model)-[:AVAILABLE_ON]->(p:Platform) "
|
|
1476
|
+
"RETURN m.id, p.id"
|
|
1477
|
+
)
|
|
1478
|
+
plat_result = self.graph.query(plat_q)
|
|
1479
|
+
for row in plat_result.result_set:
|
|
1480
|
+
mid, plat_id = row
|
|
1481
|
+
if mid in models_by_id and plat_id:
|
|
1482
|
+
models_by_id[mid].available_platforms.add(plat_id)
|
|
1483
|
+
# Derive runtime info from platform availability
|
|
1484
|
+
if plat_id in RUNTIME_PLATFORMS:
|
|
1485
|
+
models_by_id[mid].runtimes.add(plat_id)
|
|
1486
|
+
|
|
1487
|
+
# Batch-fetch tags via TAGGED_WITH edges
|
|
1488
|
+
tag_q = (
|
|
1489
|
+
"MATCH (m:Model)-[:TAGGED_WITH]->(t:Tag) "
|
|
1490
|
+
"RETURN m.id, t.id"
|
|
1491
|
+
)
|
|
1492
|
+
tag_result = self.graph.query(tag_q)
|
|
1493
|
+
for row in tag_result.result_set:
|
|
1494
|
+
mid, tag_id = row
|
|
1495
|
+
if mid in models_by_id and tag_id:
|
|
1496
|
+
models_by_id[mid].tags.add(tag_id)
|
|
1497
|
+
|
|
1498
|
+
return list(models_by_id.values())
|
|
1499
|
+
|
|
1500
|
+
# ─── Stage 1: Filter ─────────────────────────────────────
|
|
1501
|
+
|
|
1502
|
+
def _filter(
|
|
1503
|
+
self,
|
|
1504
|
+
candidates: list[ModelData],
|
|
1505
|
+
constraints: dict[str, Any],
|
|
1506
|
+
profile: dict[str, Any],
|
|
1507
|
+
) -> list[ModelData]:
|
|
1508
|
+
"""Eliminate models that fail hard constraints."""
|
|
1509
|
+
result = []
|
|
1510
|
+
|
|
1511
|
+
model_type_filter = constraints.get("model_type")
|
|
1512
|
+
open_weights_req = constraints.get("open_weights")
|
|
1513
|
+
max_cost = _safe_float(constraints.get("max_cost_input"))
|
|
1514
|
+
min_context = _safe_int(constraints.get("min_context"))
|
|
1515
|
+
min_params = _safe_int(constraints.get("min_params"))
|
|
1516
|
+
max_params = _safe_int(constraints.get("max_params"))
|
|
1517
|
+
origin_whitelist = constraints.get("origin_countries") # list of country codes
|
|
1518
|
+
origin_blacklist = constraints.get("origin_blacklist") # list of country codes
|
|
1519
|
+
require_reasoning = constraints.get("reasoning")
|
|
1520
|
+
require_tools = constraints.get("tool_use")
|
|
1521
|
+
require_vision = constraints.get("vision")
|
|
1522
|
+
provider_filter = constraints.get("provider")
|
|
1523
|
+
hw_memory_gb = _safe_float(constraints.get("hw_memory_gb"))
|
|
1524
|
+
hw_bandwidth_gbps = _safe_float(constraints.get("hw_bandwidth_gbps"))
|
|
1525
|
+
hw_quant = constraints.get("hw_quant", "")
|
|
1526
|
+
hw_tops = _safe_float(constraints.get("hw_tops"))
|
|
1527
|
+
|
|
1528
|
+
for m in candidates:
|
|
1529
|
+
# Skip deprecated/sunset unless explicitly requested
|
|
1530
|
+
if m.status in ("deprecated", "sunset"):
|
|
1531
|
+
continue
|
|
1532
|
+
|
|
1533
|
+
# Model type filter
|
|
1534
|
+
if model_type_filter:
|
|
1535
|
+
if m.model_type != model_type_filter:
|
|
1536
|
+
continue
|
|
1537
|
+
|
|
1538
|
+
# Open weights requirement
|
|
1539
|
+
if open_weights_req is True:
|
|
1540
|
+
if m.open_weights is not True:
|
|
1541
|
+
continue
|
|
1542
|
+
|
|
1543
|
+
# Max cost threshold
|
|
1544
|
+
if max_cost is not None and m.cost_input is not None:
|
|
1545
|
+
if m.cost_input > max_cost:
|
|
1546
|
+
continue
|
|
1547
|
+
|
|
1548
|
+
# Min context window
|
|
1549
|
+
if min_context is not None and m.context_window is not None:
|
|
1550
|
+
if m.context_window < min_context:
|
|
1551
|
+
continue
|
|
1552
|
+
|
|
1553
|
+
# Parameter bounds
|
|
1554
|
+
if min_params is not None and m.total_parameters is not None:
|
|
1555
|
+
if m.total_parameters < min_params:
|
|
1556
|
+
continue
|
|
1557
|
+
if max_params is not None and m.total_parameters is not None:
|
|
1558
|
+
if m.total_parameters > max_params:
|
|
1559
|
+
continue
|
|
1560
|
+
|
|
1561
|
+
# Origin country whitelist — exclude models with unknown origin too
|
|
1562
|
+
if origin_whitelist:
|
|
1563
|
+
if not m.origin_country or m.origin_country not in origin_whitelist:
|
|
1564
|
+
continue
|
|
1565
|
+
|
|
1566
|
+
# Origin country blacklist
|
|
1567
|
+
if origin_blacklist and m.origin_country:
|
|
1568
|
+
if m.origin_country in origin_blacklist:
|
|
1569
|
+
continue
|
|
1570
|
+
|
|
1571
|
+
# Capability requirements
|
|
1572
|
+
if require_reasoning and not m.reasoning:
|
|
1573
|
+
continue
|
|
1574
|
+
if require_tools and not m.tool_call:
|
|
1575
|
+
continue
|
|
1576
|
+
if require_vision and not m.vision_input:
|
|
1577
|
+
continue
|
|
1578
|
+
|
|
1579
|
+
# Provider filter
|
|
1580
|
+
if provider_filter and m.provider != provider_filter:
|
|
1581
|
+
continue
|
|
1582
|
+
|
|
1583
|
+
# Hosting mode filter
|
|
1584
|
+
hosting_modes = constraints.get("hosting") # list: ["local", "cloud", "provider"]
|
|
1585
|
+
if hosting_modes:
|
|
1586
|
+
passes_hosting = False
|
|
1587
|
+
for mode in hosting_modes:
|
|
1588
|
+
if mode == "local":
|
|
1589
|
+
# Must be open-weights to run locally
|
|
1590
|
+
if m.open_weights:
|
|
1591
|
+
passes_hosting = True
|
|
1592
|
+
elif mode == "cloud":
|
|
1593
|
+
# Check if available on any cloud/inference platform
|
|
1594
|
+
if m.available_platforms & CLOUD_PLATFORMS:
|
|
1595
|
+
passes_hosting = True
|
|
1596
|
+
elif m.open_weights:
|
|
1597
|
+
# Open-weight models can always be deployed to cloud
|
|
1598
|
+
passes_hosting = True
|
|
1599
|
+
elif mode == "provider":
|
|
1600
|
+
# Check if available via provider's own API
|
|
1601
|
+
if m.available_platforms & PROVIDER_PLATFORMS:
|
|
1602
|
+
passes_hosting = True
|
|
1603
|
+
elif not m.open_weights:
|
|
1604
|
+
# API-only models are always provider-hosted
|
|
1605
|
+
passes_hosting = True
|
|
1606
|
+
if not passes_hosting:
|
|
1607
|
+
continue
|
|
1608
|
+
|
|
1609
|
+
# Specific platform filter
|
|
1610
|
+
required_platforms = constraints.get("platforms") # list of platform IDs
|
|
1611
|
+
if required_platforms:
|
|
1612
|
+
if not m.available_platforms & set(required_platforms):
|
|
1613
|
+
# Also check by provider slug for provider platforms
|
|
1614
|
+
provider_match = m.provider in required_platforms
|
|
1615
|
+
if not provider_match:
|
|
1616
|
+
continue
|
|
1617
|
+
|
|
1618
|
+
# Runtime filter — require model to support specific runtimes
|
|
1619
|
+
required_runtimes = constraints.get("runtime") # list of runtime IDs
|
|
1620
|
+
if required_runtimes:
|
|
1621
|
+
# Check explicit runtime platforms
|
|
1622
|
+
model_runtimes = m.runtimes | (m.available_platforms & {
|
|
1623
|
+
"ollama", "lm_studio", "gpt4all", "vllm", "mlx",
|
|
1624
|
+
"llama_cpp", "transformers",
|
|
1625
|
+
})
|
|
1626
|
+
# Infer runtime compatibility for open-weights models:
|
|
1627
|
+
# If on HuggingFace or open-weights, they can run on vllm/transformers/llama_cpp
|
|
1628
|
+
if m.open_weights:
|
|
1629
|
+
model_runtimes |= {"vllm", "transformers", "llama_cpp"}
|
|
1630
|
+
if m.available_platforms & {"ollama", "lm_studio", "gpt4all"}:
|
|
1631
|
+
model_runtimes |= {"ollama", "lm_studio", "gpt4all"}
|
|
1632
|
+
if not model_runtimes & set(required_runtimes):
|
|
1633
|
+
continue
|
|
1634
|
+
|
|
1635
|
+
# OpenAI SDK compatibility filter — require "openai-compatible" tag
|
|
1636
|
+
if constraints.get("openai_compatible"):
|
|
1637
|
+
if "openai-compatible" not in m.tags:
|
|
1638
|
+
continue
|
|
1639
|
+
|
|
1640
|
+
# Hardware fit check — estimate if model fits and how fast it would run
|
|
1641
|
+
if hw_memory_gb and hw_memory_gb > 0:
|
|
1642
|
+
bytes_per_param = {"Q2": 0.25, "Q4": 0.5, "Q8": 1.0, "FP16": 2.0}.get(hw_quant, 0.5)
|
|
1643
|
+
usable_mem = hw_memory_gb * 0.85 # reserve 15% for OS/KV cache
|
|
1644
|
+
|
|
1645
|
+
if m.total_parameters:
|
|
1646
|
+
model_mem_gb = m.total_parameters * bytes_per_param / 1e9
|
|
1647
|
+
# Hard filter: won't fit in memory
|
|
1648
|
+
if model_mem_gb > usable_mem:
|
|
1649
|
+
continue
|
|
1650
|
+
# Concurrent instances that fit
|
|
1651
|
+
m.concurrent_instances = max(1, int(usable_mem / model_mem_gb))
|
|
1652
|
+
# Estimate tokens/sec from memory bandwidth
|
|
1653
|
+
# tok/s ≈ bandwidth / model_memory (memory-bandwidth-bound)
|
|
1654
|
+
if hw_bandwidth_gbps and hw_bandwidth_gbps > 0:
|
|
1655
|
+
est_tps = hw_bandwidth_gbps / model_mem_gb
|
|
1656
|
+
m.estimated_tps = round(est_tps, 1)
|
|
1657
|
+
# Filter out unusably slow models (< 1 tok/s)
|
|
1658
|
+
if est_tps < 1.0:
|
|
1659
|
+
continue
|
|
1660
|
+
else:
|
|
1661
|
+
# No parameter data — use heuristic: API-only models are fine,
|
|
1662
|
+
# but for local hosting, unknown-size models on small hardware are risky
|
|
1663
|
+
if hw_memory_gb <= 24 and m.open_weights:
|
|
1664
|
+
# Small device + open weights + unknown size = skip
|
|
1665
|
+
# (likely too big for an RPi or MacBook Air)
|
|
1666
|
+
continue
|
|
1667
|
+
|
|
1668
|
+
result.append(m)
|
|
1669
|
+
|
|
1670
|
+
return result
|
|
1671
|
+
|
|
1672
|
+
# ─── Stage 2: Score ──────────────────────────────────────
|
|
1673
|
+
|
|
1674
|
+
def _score(self, model: ModelData, profile: dict[str, Any]) -> ScoredModel:
|
|
1675
|
+
"""Compute weighted composite score for a model against a profile."""
|
|
1676
|
+
|
|
1677
|
+
# --- Benchmark scoring (up to 40 points) ---
|
|
1678
|
+
evidence = _benchmark_evidence(model.benchmark_scores, profile)
|
|
1679
|
+
bench_score_scaled = evidence["benchmark_lower_bound"]
|
|
1680
|
+
|
|
1681
|
+
# --- Capability scoring (up to 20 points) ---
|
|
1682
|
+
cap_weights = profile.get("capability_weights", {})
|
|
1683
|
+
cap_score = 0.0
|
|
1684
|
+
total_cap_weight = sum(cap_weights.values()) if cap_weights else 1.0
|
|
1685
|
+
|
|
1686
|
+
for cap_name, weight in cap_weights.items():
|
|
1687
|
+
# Look for the main capability tier (e.g., "coding", "reasoning")
|
|
1688
|
+
tier = model.capability_tiers.get(cap_name)
|
|
1689
|
+
if tier:
|
|
1690
|
+
tier_pts = _tier_points(tier)
|
|
1691
|
+
cap_score += tier_pts * (weight / total_cap_weight)
|
|
1692
|
+
else:
|
|
1693
|
+
# Check for sub-capabilities (e.g., "coding:debugging")
|
|
1694
|
+
sub_tiers = [
|
|
1695
|
+
t for cid, t in model.capability_tiers.items()
|
|
1696
|
+
if cid.startswith(f"{cap_name}:")
|
|
1697
|
+
]
|
|
1698
|
+
if sub_tiers:
|
|
1699
|
+
# Average the sub-capability tiers
|
|
1700
|
+
best_tier = min(sub_tiers, key=_tier_rank)
|
|
1701
|
+
tier_pts = _tier_points(best_tier) * 0.7 # Discount vs explicit overall
|
|
1702
|
+
cap_score += tier_pts * (weight / total_cap_weight)
|
|
1703
|
+
|
|
1704
|
+
cap_score_scaled = cap_score * 2.0 # Scale to max ~20 points
|
|
1705
|
+
|
|
1706
|
+
# --- Cost efficiency scoring (up to profile's cost_weight * 100 points) ---
|
|
1707
|
+
cost_weight = profile.get("cost_weight", 0.10)
|
|
1708
|
+
cost_score = 0.0
|
|
1709
|
+
if model.cost_input is not None:
|
|
1710
|
+
if model.cost_input == 0:
|
|
1711
|
+
cost_score = 10.0 # Free is the best
|
|
1712
|
+
else:
|
|
1713
|
+
# Log scale with floor at $0.10 to keep free always on top.
|
|
1714
|
+
# $0.10 -> ~9pts, $1 -> ~7pts, $5 -> ~5.6pts, $15 -> ~4.6pts
|
|
1715
|
+
clamped = max(model.cost_input, 0.10)
|
|
1716
|
+
cost_score = max(0.0, min(9.5, 9.0 + 2.0 * math.log10(0.10 / clamped)))
|
|
1717
|
+
cost_score_scaled = cost_score * cost_weight * 10 # Up to ~10 points
|
|
1718
|
+
|
|
1719
|
+
# --- Context window scoring (log-scaled, up to profile weight) ---
|
|
1720
|
+
ctx_weight = profile.get("context_weight", 0.10)
|
|
1721
|
+
ctx_score = 0.0
|
|
1722
|
+
if model.context_window and model.context_window > 0:
|
|
1723
|
+
# log10(4K)=3.6, log10(32K)=4.5, log10(128K)=5.1, log10(1M)=6.0, log10(10M)=7.0
|
|
1724
|
+
ctx_score = min(10.0, max(0, (math.log10(model.context_window) - 3.5) * 3.0))
|
|
1725
|
+
ctx_score_scaled = ctx_score * ctx_weight * 10 # Up to ~10 points
|
|
1726
|
+
|
|
1727
|
+
# --- Type match bonus (up to 15 points) ---
|
|
1728
|
+
# Check primary type AND subtypes against preferred types
|
|
1729
|
+
preferred = profile.get("preferred_types", [])
|
|
1730
|
+
type_bonus = 0.0
|
|
1731
|
+
if model.model_type:
|
|
1732
|
+
# Collect all types this model claims (primary + subtypes)
|
|
1733
|
+
all_types = [model.model_type] + list(model.model_subtypes)
|
|
1734
|
+
best_idx = None
|
|
1735
|
+
for t in all_types:
|
|
1736
|
+
if t in preferred:
|
|
1737
|
+
idx = preferred.index(t)
|
|
1738
|
+
if best_idx is None or idx < best_idx:
|
|
1739
|
+
best_idx = idx
|
|
1740
|
+
if best_idx is not None:
|
|
1741
|
+
# First preferred type gets full bonus, descending
|
|
1742
|
+
type_bonus = max(5.0, 15.0 - best_idx * 3.0)
|
|
1743
|
+
|
|
1744
|
+
# --- Speed bonus (up to 10 points, only when hardware specified) ---
|
|
1745
|
+
speed_score = 0.0
|
|
1746
|
+
if model.estimated_tps is not None:
|
|
1747
|
+
# Log-scale: 1 tok/s = 0, 10 tok/s = 5, 100 tok/s = 10
|
|
1748
|
+
speed_score = min(10.0, max(0.0, math.log10(max(1.0, model.estimated_tps)) * 5.0))
|
|
1749
|
+
|
|
1750
|
+
# --- Composite score (0-100 scale) ---
|
|
1751
|
+
raw_total = (
|
|
1752
|
+
bench_score_scaled
|
|
1753
|
+
+ cap_score_scaled
|
|
1754
|
+
+ cost_score_scaled
|
|
1755
|
+
+ ctx_score_scaled
|
|
1756
|
+
+ type_bonus
|
|
1757
|
+
+ speed_score
|
|
1758
|
+
)
|
|
1759
|
+
# Clamp to 0-100
|
|
1760
|
+
final_score = max(0.0, min(100.0, raw_total))
|
|
1761
|
+
upper_score = max(0.0, min(100.0, raw_total
|
|
1762
|
+
+ evidence["benchmark_upper_bound"] - bench_score_scaled))
|
|
1763
|
+
|
|
1764
|
+
return ScoredModel(
|
|
1765
|
+
model_id=model.model_id,
|
|
1766
|
+
display_name=model.display_name,
|
|
1767
|
+
model_type=model.model_type,
|
|
1768
|
+
score=round(final_score, 2) if evidence["rank_status"] == "ranked" else None,
|
|
1769
|
+
score_lower_bound=round(final_score, 2),
|
|
1770
|
+
score_upper_bound=round(upper_score, 2),
|
|
1771
|
+
benchmark_score=round(bench_score_scaled, 2),
|
|
1772
|
+
capability_score=round(cap_score_scaled, 2),
|
|
1773
|
+
cost_score=round(cost_score_scaled, 2),
|
|
1774
|
+
context_score=round(ctx_score_scaled, 2),
|
|
1775
|
+
type_bonus=round(type_bonus, 2),
|
|
1776
|
+
speed_score=round(speed_score, 2),
|
|
1777
|
+
estimated_tps=model.estimated_tps,
|
|
1778
|
+
concurrent_instances=model.concurrent_instances,
|
|
1779
|
+
**evidence,
|
|
1780
|
+
arena_elo_overall=model.arena_elo_overall,
|
|
1781
|
+
total_parameters=model.total_parameters,
|
|
1782
|
+
context_window=model.context_window,
|
|
1783
|
+
cost_input=model.cost_input,
|
|
1784
|
+
cost_output=model.cost_output,
|
|
1785
|
+
open_weights=model.open_weights,
|
|
1786
|
+
provider=model.provider,
|
|
1787
|
+
status=model.status,
|
|
1788
|
+
)
|
|
1789
|
+
|
|
1790
|
+
# ─── Stage 4: Explain ────────────────────────────────────
|
|
1791
|
+
|
|
1792
|
+
def _explain(
|
|
1793
|
+
self,
|
|
1794
|
+
scored: list[ScoredModel],
|
|
1795
|
+
profile: dict[str, Any],
|
|
1796
|
+
use_case: str | None,
|
|
1797
|
+
) -> None:
|
|
1798
|
+
"""Generate human-readable reasons for each model's ranking. Mutates in place."""
|
|
1799
|
+
if not scored:
|
|
1800
|
+
return
|
|
1801
|
+
|
|
1802
|
+
for sm in scored:
|
|
1803
|
+
reasons = [
|
|
1804
|
+
f"Benchmark coverage: {sm.benchmark_coverage:.0%}; "
|
|
1805
|
+
f"composite bounds {sm.score_lower_bound:.2f}–{sm.score_upper_bound:.2f} "
|
|
1806
|
+
"(missing benchmarks, not statistical confidence)."
|
|
1807
|
+
]
|
|
1808
|
+
if sm.rank_status == "unranked":
|
|
1809
|
+
sm.reasons = ["Unranked for insufficient benchmark evidence; not ranked low.", *reasons]
|
|
1810
|
+
continue
|
|
1811
|
+
reasons.append("Ordered by conservative lower bound, not an estimate of ability.")
|
|
1812
|
+
|
|
1813
|
+
# Benchmark highlights
|
|
1814
|
+
if sm.benchmark_contributions:
|
|
1815
|
+
# Find top contributing benchmarks
|
|
1816
|
+
sorted_benches = sorted(
|
|
1817
|
+
sm.benchmark_contributions.items(),
|
|
1818
|
+
key=lambda x: x[1],
|
|
1819
|
+
reverse=True,
|
|
1820
|
+
)
|
|
1821
|
+
for bench_id, contrib in sorted_benches[:3]:
|
|
1822
|
+
if contrib > 0.5:
|
|
1823
|
+
nice_name = bench_id.replace("_", " ").title()
|
|
1824
|
+
reasons.append(f"Strong {nice_name} performance")
|
|
1825
|
+
|
|
1826
|
+
# Type match
|
|
1827
|
+
if sm.type_bonus >= 12:
|
|
1828
|
+
reasons.append(f"Ideal model type for {use_case or 'general'}")
|
|
1829
|
+
elif sm.type_bonus >= 5:
|
|
1830
|
+
reasons.append(f"Good model type match")
|
|
1831
|
+
|
|
1832
|
+
# Capability highlights
|
|
1833
|
+
if sm.capability_score >= 15:
|
|
1834
|
+
reasons.append("Top-tier capabilities")
|
|
1835
|
+
elif sm.capability_score >= 8:
|
|
1836
|
+
reasons.append("Strong capabilities")
|
|
1837
|
+
|
|
1838
|
+
# Cost efficiency
|
|
1839
|
+
if sm.cost_score >= 8:
|
|
1840
|
+
if sm.cost_input == 0:
|
|
1841
|
+
reasons.append("Free tier available")
|
|
1842
|
+
else:
|
|
1843
|
+
reasons.append("Excellent cost efficiency")
|
|
1844
|
+
elif sm.cost_score >= 4:
|
|
1845
|
+
reasons.append("Good cost/quality ratio")
|
|
1846
|
+
|
|
1847
|
+
# Context window
|
|
1848
|
+
if sm.context_window:
|
|
1849
|
+
if sm.context_window >= 1_000_000:
|
|
1850
|
+
reasons.append(f"{sm.context_window // 1_000_000}M token context")
|
|
1851
|
+
elif sm.context_window >= 128_000:
|
|
1852
|
+
reasons.append(f"{sm.context_window // 1000}K token context")
|
|
1853
|
+
|
|
1854
|
+
# Arena ELO
|
|
1855
|
+
if sm.arena_elo_overall and sm.arena_elo_overall >= 1300:
|
|
1856
|
+
reasons.append(f"Arena ELO: {int(sm.arena_elo_overall)}")
|
|
1857
|
+
|
|
1858
|
+
# Speed estimate
|
|
1859
|
+
if sm.estimated_tps is not None:
|
|
1860
|
+
if sm.estimated_tps >= 50:
|
|
1861
|
+
reasons.append(f"~{sm.estimated_tps:.0f} tok/s (fast)")
|
|
1862
|
+
elif sm.estimated_tps >= 10:
|
|
1863
|
+
reasons.append(f"~{sm.estimated_tps:.0f} tok/s")
|
|
1864
|
+
elif sm.estimated_tps >= 1:
|
|
1865
|
+
reasons.append(f"~{sm.estimated_tps:.1f} tok/s (slow)")
|
|
1866
|
+
|
|
1867
|
+
# Open weights
|
|
1868
|
+
if sm.open_weights:
|
|
1869
|
+
reasons.append("Open weights")
|
|
1870
|
+
|
|
1871
|
+
# If no reasons generated, add a generic one
|
|
1872
|
+
if not reasons:
|
|
1873
|
+
if sm.score > 0:
|
|
1874
|
+
reasons.append("Matches basic criteria")
|
|
1875
|
+
else:
|
|
1876
|
+
reasons.append("Limited data available")
|
|
1877
|
+
|
|
1878
|
+
sm.reasons = reasons
|
|
1879
|
+
|
|
1880
|
+
|
|
1881
|
+
# ═══════════════════════════════════════════════════════════════
|
|
1882
|
+
# Helper functions
|
|
1883
|
+
# ═══════════════════════════════════════════════════════════════
|
|
1884
|
+
|
|
1885
|
+
def _normalize_benchmark(bench_id: str, raw_value: float) -> float:
|
|
1886
|
+
"""Normalize a benchmark score to 0-100, higher always meaning better.
|
|
1887
|
+
|
|
1888
|
+
Direction is explicit rather than assumed. A lower-is-better metric such as
|
|
1889
|
+
word error rate is inverted here, so that the rest of the pipeline can treat
|
|
1890
|
+
every normalised score the same way.
|
|
1891
|
+
"""
|
|
1892
|
+
range_info = BENCHMARK_RANGES.get(bench_id)
|
|
1893
|
+
if range_info is None:
|
|
1894
|
+
# Unknown benchmark: assume a 0-100 scale. This is a guess, and it is
|
|
1895
|
+
# wrong for any lower-is-better metric — add the benchmark to
|
|
1896
|
+
# BENCHMARK_RANGES and BENCHMARK_DIRECTIONS rather than relying on it.
|
|
1897
|
+
return max(0.0, min(100.0, raw_value))
|
|
1898
|
+
|
|
1899
|
+
low, high = range_info
|
|
1900
|
+
if high <= low:
|
|
1901
|
+
return 50.0 # Degenerate range
|
|
1902
|
+
normalized = ((raw_value - low) / (high - low)) * 100.0
|
|
1903
|
+
if BENCHMARK_DIRECTIONS.get(bench_id) == "lower_is_better":
|
|
1904
|
+
normalized = 100.0 - normalized
|
|
1905
|
+
return max(0.0, min(100.0, normalized))
|
|
1906
|
+
|
|
1907
|
+
|
|
1908
|
+
def _tier_points(tier: str) -> float:
|
|
1909
|
+
"""Convert a tier string to point value."""
|
|
1910
|
+
return {
|
|
1911
|
+
"tier-1": 10.0,
|
|
1912
|
+
"tier-2": 6.0,
|
|
1913
|
+
"tier-3": 3.0,
|
|
1914
|
+
"n/a": 0.0,
|
|
1915
|
+
}.get(tier, 0.0)
|
|
1916
|
+
|
|
1917
|
+
|
|
1918
|
+
def _tier_rank(tier: str) -> int:
|
|
1919
|
+
"""Numeric rank for tier comparison (lower = better)."""
|
|
1920
|
+
return {
|
|
1921
|
+
"tier-1": 1,
|
|
1922
|
+
"tier-2": 2,
|
|
1923
|
+
"tier-3": 3,
|
|
1924
|
+
"n/a": 99,
|
|
1925
|
+
}.get(tier, 50)
|
|
1926
|
+
|
|
1927
|
+
|
|
1928
|
+
def _safe_int(val) -> int | None:
|
|
1929
|
+
if val is None:
|
|
1930
|
+
return None
|
|
1931
|
+
try:
|
|
1932
|
+
return int(val)
|
|
1933
|
+
except (ValueError, TypeError):
|
|
1934
|
+
return None
|
|
1935
|
+
|
|
1936
|
+
|
|
1937
|
+
def _safe_float(val) -> float | None:
|
|
1938
|
+
if val is None:
|
|
1939
|
+
return None
|
|
1940
|
+
try:
|
|
1941
|
+
return float(val)
|
|
1942
|
+
except (ValueError, TypeError):
|
|
1943
|
+
return None
|