modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
api/ranking/engine.py ADDED
@@ -0,0 +1,1943 @@
1
+ """ModelSpec Ranking Engine — 4-stage pipeline.
2
+
3
+ Queries FalkorDB directly, scores models against use-case profiles,
4
+ and returns ranked results with human-readable explanations.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import logging
10
+ import math
11
+ from dataclasses import dataclass, field
12
+ from typing import Any
13
+
14
+ logger = logging.getLogger("modelspec.ranking")
15
+
16
+
17
+ # ═══════════════════════════════════════════════════════════════
18
+ # Benchmark normalization ranges
19
+ # ═══════════════════════════════════════════════════════════════
20
+ # Maps benchmark_id -> (min_plausible, max_plausible) for 0-100 normalization.
21
+ # Scores below min map to 0, above max map to 100.
22
+ # For ELO-based scores, the range is wider; for percentage-based, it's 0-100.
23
+
24
+ BENCHMARK_RANGES: dict[str, tuple[float, float]] = {
25
+ # Knowledge & Reasoning (percentage-based, 0-100)
26
+ "mmlu_pro": (20.0, 90.0),
27
+ "gpqa_diamond": (20.0, 80.0),
28
+ "hle": (0.0, 50.0),
29
+ "arc_challenge": (40.0, 100.0),
30
+ "hellaswag": (40.0, 100.0),
31
+ "truthfulqa": (20.0, 90.0),
32
+ "bbh": (20.0, 95.0),
33
+ "ifeval": (20.0, 95.0),
34
+ "musr": (10.0, 80.0),
35
+ "winogrande": (50.0, 100.0),
36
+ # Math
37
+ "math_500": (10.0, 100.0),
38
+ "aime_2025": (0.0, 80.0),
39
+ "aime_2026": (0.0, 80.0),
40
+ "gsm8k": (20.0, 100.0),
41
+ "mgsm": (10.0, 100.0),
42
+ # Coding
43
+ "humaneval": (10.0, 100.0),
44
+ "humaneval_plus": (10.0, 100.0),
45
+ "swe_bench_verified": (0.0, 70.0),
46
+ "live_code_bench": (0.0, 60.0),
47
+ "aider_polyglot": (0.0, 90.0),
48
+ "terminal_bench": (0.0, 80.0),
49
+ "mbpp": (20.0, 100.0),
50
+ "multipl_e": (10.0, 100.0),
51
+ # Multimodal
52
+ "mmmu": (20.0, 80.0),
53
+ "mathvista": (20.0, 80.0),
54
+ "docvqa": (40.0, 100.0),
55
+ "chartqa": (40.0, 100.0),
56
+ # Safety
57
+ "helm_safety": (30.0, 100.0),
58
+ "bbq": (30.0, 100.0),
59
+ "toxigen": (30.0, 100.0),
60
+ # Human preference (ELO-based: typical range 900-1400)
61
+ "arena_elo_overall": (1000.0, 1400.0),
62
+ "arena_elo_coding": (1000.0, 1400.0),
63
+ "arena_elo_math": (1000.0, 1400.0),
64
+ "arena_elo_vision": (1000.0, 1400.0),
65
+ "arena_elo_hard_prompts": (1000.0, 1400.0),
66
+ "arena_elo_style_control": (1000.0, 1400.0),
67
+ "mt_bench": (5.0, 10.0),
68
+ "alpaca_eval": (0.0, 60.0),
69
+ "wildbench": (-100.0, 100.0),
70
+ # Embedding (percentage or ratio-based)
71
+ "mteb_overall": (30.0, 80.0),
72
+ "mteb_retrieval": (20.0, 70.0),
73
+ "mteb_classification": (40.0, 90.0),
74
+ "mteb_clustering": (20.0, 60.0),
75
+ "mteb_reranking": (20.0, 70.0),
76
+ "mteb_sts": (40.0, 90.0),
77
+ "mteb_pair_classification": (50.0, 95.0),
78
+ "mteb_summarization": (20.0, 50.0),
79
+ "beir": (20.0, 70.0),
80
+ "miracl": (10.0, 70.0),
81
+ # Agentic
82
+ "swe_bench_agent": (0.0, 60.0),
83
+ "swe_bench_pro": (0.0, 80.0),
84
+ "swe_bench_multilingual": (0.0, 90.0),
85
+ "swe_bench_multimodal": (0.0, 60.0),
86
+ "tau_bench": (0.0, 80.0),
87
+ "web_arena": (0.0, 50.0),
88
+ "osworld": (0.0, 80.0),
89
+ # Agentic search
90
+ "browsecomp": (0.0, 90.0),
91
+ "hle": (0.0, 65.0),
92
+ "hle_tools": (0.0, 70.0),
93
+ # Multimodal (advanced)
94
+ "charxiv_reasoning": (20.0, 95.0),
95
+ "charxiv_reasoning_tools": (20.0, 95.0),
96
+ "lab_bench_figqa": (20.0, 90.0),
97
+ "lab_bench_figqa_tools": (20.0, 90.0),
98
+ "screenspot_pro": (10.0, 95.0),
99
+ "screenspot_pro_tools": (10.0, 95.0),
100
+ # Long context
101
+ "graphwalks_bfs_256k_1m": (0.0, 85.0),
102
+ "graphwalks_parents_256k_1m": (0.0, 100.0),
103
+ # Math (competition)
104
+ "usamo_2026": (0.0, 100.0),
105
+ "ipho_2025_theory": (0.0, 100.0),
106
+ # Agentic research
107
+ "deepsearchqa": (0.0, 80.0),
108
+ "frontierscience_research": (0.0, 50.0),
109
+ # Health / Medical (advanced)
110
+ "healthbench_hard": (0.0, 50.0),
111
+ "medxpertqa_multimodal": (20.0, 85.0),
112
+ # Visual reasoning
113
+ "zerobench": (0.0, 50.0),
114
+ "ai2d": (40.0, 100.0),
115
+ "ocrbench": (0.0, 100.0),
116
+ "realworldqa": (30.0, 90.0),
117
+ # AGI benchmarks
118
+ "arc_agi_2": (0.0, 80.0),
119
+ # Multilingual knowledge
120
+ "mmmlu": (40.0, 95.0),
121
+ # Per-language MultiPL-E (percentage-based, 0-100)
122
+ "multipl_e_python": (10.0, 100.0),
123
+ "multipl_e_rust": (10.0, 100.0),
124
+ "multipl_e_cpp": (10.0, 100.0),
125
+ "multipl_e_java": (10.0, 100.0),
126
+ "multipl_e_typescript": (10.0, 100.0),
127
+ "multipl_e_go": (10.0, 100.0),
128
+ "multipl_e_javascript": (10.0, 100.0),
129
+ "multipl_e_csharp": (10.0, 100.0),
130
+ "multipl_e_php": (10.0, 100.0),
131
+ "multipl_e_ruby": (10.0, 100.0),
132
+ "multipl_e_swift": (10.0, 100.0),
133
+ "multipl_e_kotlin": (10.0, 100.0),
134
+ "multipl_e_scala": (10.0, 100.0),
135
+ "multipl_e_r": (10.0, 100.0),
136
+ "multipl_e_julia": (10.0, 100.0),
137
+ "multipl_e_perl": (10.0, 100.0),
138
+ "multipl_e_lua": (10.0, 100.0),
139
+ # Terminal-Bench 2.0 (percentage-based, 0-100)
140
+ "terminal_bench_2": (0.0, 80.0),
141
+ # MMLU subject scores (percentage-based, 0-100)
142
+ # All 57 MMLU (hendrycksTest) subjects from the evaluation harness
143
+ "mmlu_abstract_algebra": (20.0, 75.0),
144
+ "mmlu_anatomy": (20.0, 90.0),
145
+ "mmlu_astronomy": (20.0, 95.0),
146
+ "mmlu_business_ethics": (20.0, 90.0),
147
+ "mmlu_clinical_knowledge": (20.0, 95.0),
148
+ "mmlu_college_biology": (20.0, 95.0),
149
+ "mmlu_college_chemistry": (20.0, 80.0),
150
+ "mmlu_college_computer_science": (20.0, 95.0),
151
+ "mmlu_college_mathematics": (20.0, 80.0),
152
+ "mmlu_college_medicine": (20.0, 90.0),
153
+ "mmlu_college_physics": (20.0, 80.0),
154
+ "mmlu_computer_security": (20.0, 95.0),
155
+ "mmlu_conceptual_physics": (20.0, 95.0),
156
+ "mmlu_econometrics": (20.0, 85.0),
157
+ "mmlu_electrical_engineering": (20.0, 90.0),
158
+ "mmlu_elementary_mathematics": (20.0, 90.0),
159
+ "mmlu_formal_logic": (20.0, 80.0),
160
+ "mmlu_global_facts": (20.0, 75.0),
161
+ "mmlu_high_school_biology": (20.0, 95.0),
162
+ "mmlu_high_school_chemistry": (20.0, 90.0),
163
+ "mmlu_high_school_computer_science": (20.0, 95.0),
164
+ "mmlu_high_school_european_history": (20.0, 95.0),
165
+ "mmlu_high_school_geography": (20.0, 95.0),
166
+ "mmlu_high_school_government_and_politics": (20.0, 100.0),
167
+ "mmlu_high_school_macroeconomics": (20.0, 95.0),
168
+ "mmlu_high_school_mathematics": (20.0, 80.0),
169
+ "mmlu_high_school_microeconomics": (20.0, 95.0),
170
+ "mmlu_high_school_physics": (20.0, 85.0),
171
+ "mmlu_high_school_psychology": (20.0, 98.0),
172
+ "mmlu_high_school_statistics": (20.0, 90.0),
173
+ "mmlu_high_school_us_history": (20.0, 95.0),
174
+ "mmlu_high_school_world_history": (20.0, 95.0),
175
+ "mmlu_human_aging": (20.0, 90.0),
176
+ "mmlu_human_sexuality": (20.0, 95.0),
177
+ "mmlu_international_law": (20.0, 95.0),
178
+ "mmlu_jurisprudence": (20.0, 90.0),
179
+ "mmlu_logical_fallacies": (20.0, 95.0),
180
+ "mmlu_machine_learning": (20.0, 85.0),
181
+ "mmlu_management": (20.0, 95.0),
182
+ "mmlu_marketing": (20.0, 95.0),
183
+ "mmlu_medical_genetics": (20.0, 95.0),
184
+ "mmlu_miscellaneous": (20.0, 95.0),
185
+ "mmlu_moral_disputes": (20.0, 90.0),
186
+ "mmlu_moral_scenarios": (20.0, 85.0),
187
+ "mmlu_nutrition": (20.0, 95.0),
188
+ "mmlu_philosophy": (20.0, 90.0),
189
+ "mmlu_prehistory": (20.0, 95.0),
190
+ "mmlu_professional_accounting": (20.0, 85.0),
191
+ "mmlu_professional_law": (20.0, 90.0),
192
+ "mmlu_professional_medicine": (20.0, 95.0),
193
+ "mmlu_professional_psychology": (20.0, 95.0),
194
+ "mmlu_public_relations": (20.0, 90.0),
195
+ "mmlu_security_studies": (20.0, 90.0),
196
+ "mmlu_sociology": (20.0, 95.0),
197
+ "mmlu_us_foreign_policy": (20.0, 95.0),
198
+ "mmlu_virology": (20.0, 80.0),
199
+ "mmlu_world_religions": (20.0, 95.0),
200
+ # Legacy aliases (mapped to new names for backward compatibility)
201
+ "mmlu_chemistry": (20.0, 95.0),
202
+ "mmlu_physics": (20.0, 95.0),
203
+ "mmlu_biology": (20.0, 95.0),
204
+ "mmlu_computer_science": (20.0, 95.0),
205
+ # Domain-specific extras
206
+ "pubmedqa": (30.0, 90.0),
207
+ "medmcqa": (20.0, 80.0),
208
+ "bioasq": (20.0, 80.0),
209
+ "finqa": (20.0, 90.0),
210
+ "convfinqa": (20.0, 80.0),
211
+ "fpb": (20.0, 90.0),
212
+ # Translation
213
+ "flores": (10.0, 70.0),
214
+ "flores_en_zh": (10.0, 50.0),
215
+ "flores_en_de": (10.0, 50.0),
216
+ "flores_en_fr": (10.0, 50.0),
217
+ "flores_en_es": (10.0, 50.0),
218
+ "flores_en_ja": (10.0, 50.0),
219
+ "flores_en_ko": (10.0, 50.0),
220
+ }
221
+
222
+
223
+ # ═══════════════════════════════════════════════════════════════
224
+ # Use case profiles
225
+ # ═══════════════════════════════════════════════════════════════
226
+
227
+
228
+ #: Benchmarks whose score is better when lower. `_normalize_benchmark` assumed
229
+ #: higher-is-better for everything, which silently inverted these: a worse
230
+ #: transcriber outranked a better one. See MODEL-30.
231
+ BENCHMARK_DIRECTIONS: dict[str, str] = {
232
+ "wer_librispeech": "lower_is_better",
233
+ "fid": "lower_is_better",
234
+ }
235
+
236
+ #: Ranges added in MODEL-30 for benchmarks that profiles weight but that had no
237
+ #: entry above, so normalisation fell back to "assume 0-100, higher is better".
238
+ #:
239
+ #: medqa, finbench and legalbench are confirmed from their benchgraph pages —
240
+ #: all three declare higher_is_better with unit % and max_score 100, so the old
241
+ #: fallback happened to be right and this simply makes it explicit.
242
+ #:
243
+ #: wer_librispeech, fid and mos_tts have no benchgraph page. Their *direction*
244
+ #: and scale are not in doubt, and fixing those removes the inversion. The exact
245
+ #: bounds are judgement and are marked for confirmation alongside MODEL-30.
246
+ BENCHMARK_RANGES.update({
247
+ "medqa": (0.0, 100.0), # confirmed from page
248
+ "finbench": (0.0, 100.0), # confirmed from page
249
+ "legalbench": (0.0, 100.0), # confirmed from page
250
+ "wer_librispeech": (1.5, 25.0), # word error rate %, bounds unconfirmed
251
+ "fid": (1.0, 100.0), # Frechet distance, bounds unconfirmed
252
+ "mos_tts": (1.0, 5.0), # mean opinion score, 1-5 by definition
253
+ })
254
+
255
+ USE_CASE_PROFILES: dict[str, dict[str, Any]] = {
256
+ "coding": {
257
+ "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
258
+ "benchmark_weights": {
259
+ "humaneval": 0.20, "swe_bench_verified": 0.20, "live_code_bench": 0.15,
260
+ "aider_polyglot": 0.15, "arena_elo_coding": 0.15, "arena_elo_overall": 0.10,
261
+ "terminal_bench": 0.05,
262
+ },
263
+ "capability_weights": {
264
+ "coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
265
+ },
266
+ "cost_weight": 0.0,
267
+ "context_weight": 0.10,
268
+ },
269
+ "reasoning": {
270
+ "preferred_types": ["llm-reasoning", "llm-chat"],
271
+ "benchmark_weights": {
272
+ "gpqa_diamond": 0.20, "math_500": 0.20, "aime_2025": 0.15,
273
+ "mmlu_pro": 0.15, "arena_elo_overall": 0.15, "bbh": 0.10,
274
+ "ifeval": 0.05,
275
+ },
276
+ "capability_weights": {
277
+ "reasoning": 0.35, "coding": 0.15, "tool_use": 0.10,
278
+ },
279
+ "cost_weight": 0.0,
280
+ "context_weight": 0.10,
281
+ },
282
+ "chat": {
283
+ "preferred_types": ["llm-chat", "vlm", "llm-reasoning"],
284
+ "benchmark_weights": {
285
+ "arena_elo_overall": 0.30, "mt_bench": 0.15, "alpaca_eval": 0.15,
286
+ "ifeval": 0.15, "mmlu_pro": 0.10, "arena_elo_style_control": 0.10,
287
+ "wildbench": 0.05,
288
+ },
289
+ "capability_weights": {
290
+ "creative": 0.20, "language": 0.20, "reasoning": 0.15, "tool_use": 0.10,
291
+ },
292
+ "cost_weight": 0.0,
293
+ "context_weight": 0.10,
294
+ },
295
+ "embedding": {
296
+ "preferred_types": ["embedding-text", "embedding-multimodal"],
297
+ "benchmark_weights": {
298
+ "mteb_overall": 0.30, "mteb_retrieval": 0.25, "mteb_classification": 0.15,
299
+ "beir": 0.15, "miracl": 0.10, "mteb_clustering": 0.05,
300
+ },
301
+ "capability_weights": {},
302
+ "cost_weight": 0.0,
303
+ "context_weight": 0.05,
304
+ },
305
+ "vision": {
306
+ "preferred_types": ["vlm", "llm-chat"],
307
+ "benchmark_weights": {
308
+ "mmmu": 0.25, "mathvista": 0.20, "docvqa": 0.15, "chartqa": 0.15,
309
+ "arena_elo_vision": 0.15, "arena_elo_overall": 0.10,
310
+ },
311
+ "capability_weights": {
312
+ "reasoning": 0.15, "creative": 0.10,
313
+ },
314
+ "cost_weight": 0.0,
315
+ "context_weight": 0.10,
316
+ },
317
+ "agentic": {
318
+ "preferred_types": ["llm-reasoning", "llm-code", "llm-chat"],
319
+ "benchmark_weights": {
320
+ "swe_bench_agent": 0.20, "tau_bench": 0.15, "web_arena": 0.15,
321
+ "swe_bench_verified": 0.15, "arena_elo_overall": 0.15,
322
+ "terminal_bench": 0.10, "ifeval": 0.10,
323
+ },
324
+ "capability_weights": {
325
+ "tool_use": 0.25, "coding": 0.20, "reasoning": 0.20,
326
+ },
327
+ "cost_weight": 0.0,
328
+ "context_weight": 0.15,
329
+ },
330
+ "rag": {
331
+ "preferred_types": ["embedding-text", "reranker", "llm-chat"],
332
+ "benchmark_weights": {
333
+ "mteb_retrieval": 0.25, "beir": 0.20, "mteb_overall": 0.15,
334
+ "arena_elo_overall": 0.15, "ifeval": 0.10, "mmlu_pro": 0.10,
335
+ "miracl": 0.05,
336
+ },
337
+ "capability_weights": {
338
+ "language": 0.15, "reasoning": 0.10,
339
+ },
340
+ "cost_weight": 0.0,
341
+ "context_weight": 0.15,
342
+ },
343
+ "safety": {
344
+ "preferred_types": ["safety-classifier", "reward-model"],
345
+ "benchmark_weights": {
346
+ "helm_safety": 0.30, "bbq": 0.25, "toxigen": 0.25,
347
+ "arena_elo_overall": 0.20,
348
+ },
349
+ "capability_weights": {},
350
+ "cost_weight": 0.0,
351
+ "context_weight": 0.05,
352
+ },
353
+ "general": {
354
+ "preferred_types": ["llm-chat", "llm-reasoning", "vlm"],
355
+ "benchmark_weights": {
356
+ "arena_elo_overall": 0.25, "mmlu_pro": 0.15, "gpqa_diamond": 0.10,
357
+ "humaneval": 0.10, "math_500": 0.10, "ifeval": 0.10,
358
+ "mt_bench": 0.10, "swe_bench_verified": 0.10,
359
+ },
360
+ "capability_weights": {
361
+ "reasoning": 0.15, "coding": 0.15, "tool_use": 0.10, "creative": 0.10,
362
+ },
363
+ "cost_weight": 0.0,
364
+ "context_weight": 0.10,
365
+ },
366
+
367
+ # ─── Sub-domain: Coding by language ───────────────────────
368
+ "coding_python": {
369
+ "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
370
+ "benchmark_weights": {
371
+ "multipl_e_python": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
372
+ "aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
373
+ "arena_elo_overall": 0.05,
374
+ },
375
+ "capability_weights": {
376
+ "coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
377
+ },
378
+ "cost_weight": 0.0,
379
+ "context_weight": 0.10,
380
+ },
381
+ "coding_rust": {
382
+ "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
383
+ "benchmark_weights": {
384
+ "multipl_e_rust": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
385
+ "aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
386
+ "arena_elo_overall": 0.05,
387
+ },
388
+ "capability_weights": {
389
+ "coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
390
+ },
391
+ "cost_weight": 0.0,
392
+ "context_weight": 0.10,
393
+ },
394
+ "coding_go": {
395
+ "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
396
+ "benchmark_weights": {
397
+ "multipl_e_go": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
398
+ "aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
399
+ "arena_elo_overall": 0.05,
400
+ },
401
+ "capability_weights": {
402
+ "coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
403
+ },
404
+ "cost_weight": 0.0,
405
+ "context_weight": 0.10,
406
+ },
407
+ "coding_typescript": {
408
+ "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
409
+ "benchmark_weights": {
410
+ "multipl_e_typescript": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
411
+ "aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
412
+ "arena_elo_overall": 0.05,
413
+ },
414
+ "capability_weights": {
415
+ "coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
416
+ },
417
+ "cost_weight": 0.0,
418
+ "context_weight": 0.10,
419
+ },
420
+ "coding_cpp": {
421
+ "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
422
+ "benchmark_weights": {
423
+ "multipl_e_cpp": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
424
+ "aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
425
+ "arena_elo_overall": 0.05,
426
+ },
427
+ "capability_weights": {
428
+ "coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
429
+ },
430
+ "cost_weight": 0.0,
431
+ "context_weight": 0.10,
432
+ },
433
+ "coding_java": {
434
+ "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
435
+ "benchmark_weights": {
436
+ "multipl_e_java": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
437
+ "aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
438
+ "arena_elo_overall": 0.05,
439
+ },
440
+ "capability_weights": {
441
+ "coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
442
+ },
443
+ "cost_weight": 0.0,
444
+ "context_weight": 0.10,
445
+ },
446
+ "coding_javascript": {
447
+ "preferred_types": ["llm-code", "llm-chat", "llm-reasoning", "vlm"],
448
+ "benchmark_weights": {
449
+ "multipl_e_javascript": 0.30, "humaneval": 0.15, "swe_bench_verified": 0.15,
450
+ "aider_polyglot": 0.15, "arena_elo_coding": 0.10, "multipl_e": 0.10,
451
+ "arena_elo_overall": 0.05,
452
+ },
453
+ "capability_weights": {
454
+ "coding": 0.30, "reasoning": 0.20, "tool_use": 0.15,
455
+ },
456
+ "cost_weight": 0.0,
457
+ "context_weight": 0.10,
458
+ },
459
+
460
+ # ─── Sub-domain: Medical ──────────────────────────────────
461
+ "medical": {
462
+ "preferred_types": ["llm-chat", "llm-reasoning"],
463
+ "benchmark_weights": {
464
+ "medqa": 0.30, "pubmedqa": 0.20, "medmcqa": 0.15,
465
+ "mmlu_clinical_knowledge": 0.15, "mmlu_pro": 0.10,
466
+ "arena_elo_overall": 0.10,
467
+ },
468
+ "capability_weights": {
469
+ "reasoning": 0.25, "domain": 0.20, "language": 0.10,
470
+ },
471
+ "cost_weight": 0.0,
472
+ "context_weight": 0.15,
473
+ },
474
+ "medical_clinical": {
475
+ "preferred_types": ["llm-chat", "llm-reasoning"],
476
+ "benchmark_weights": {
477
+ "medqa": 0.25, "mmlu_clinical_knowledge": 0.25, "pubmedqa": 0.15,
478
+ "medmcqa": 0.15, "mmlu_pro": 0.10, "arena_elo_overall": 0.10,
479
+ },
480
+ "capability_weights": {
481
+ "reasoning": 0.25, "domain": 0.20, "language": 0.10,
482
+ },
483
+ "cost_weight": 0.0,
484
+ "context_weight": 0.15,
485
+ },
486
+ "medical_radiology": {
487
+ "preferred_types": ["vlm", "llm-reasoning", "llm-chat"],
488
+ "benchmark_weights": {
489
+ "medqa": 0.20, "mmmu": 0.20, "mmlu_clinical_knowledge": 0.15,
490
+ "pubmedqa": 0.15, "arena_elo_vision": 0.15, "arena_elo_overall": 0.15,
491
+ },
492
+ "capability_weights": {
493
+ "reasoning": 0.25, "domain": 0.15, "creative": 0.10,
494
+ },
495
+ "cost_weight": 0.0,
496
+ "context_weight": 0.10,
497
+ },
498
+
499
+ # ─── Sub-domain: Legal ────────────────────────────────────
500
+ "legal": {
501
+ "preferred_types": ["llm-chat", "llm-reasoning"],
502
+ "benchmark_weights": {
503
+ "legalbench": 0.35, "mmlu_professional_law": 0.20,
504
+ "mmlu_jurisprudence": 0.15, "ifeval": 0.10,
505
+ "mmlu_pro": 0.10, "arena_elo_overall": 0.10,
506
+ },
507
+ "capability_weights": {
508
+ "reasoning": 0.25, "domain": 0.15, "language": 0.15,
509
+ },
510
+ "cost_weight": 0.0,
511
+ "context_weight": 0.15,
512
+ },
513
+
514
+ # ─── Sub-domain: Financial ────────────────────────────────
515
+ "financial": {
516
+ "preferred_types": ["llm-chat", "llm-reasoning"],
517
+ "benchmark_weights": {
518
+ "finbench": 0.30, "finqa": 0.20, "mmlu_pro": 0.15,
519
+ "ifeval": 0.10, "math_500": 0.10, "arena_elo_overall": 0.15,
520
+ },
521
+ "capability_weights": {
522
+ "reasoning": 0.25, "domain": 0.15, "language": 0.10,
523
+ },
524
+ "cost_weight": 0.0,
525
+ "context_weight": 0.15,
526
+ },
527
+
528
+ # ─── Sub-domain: Science ──────────────────────────────────
529
+ "science": {
530
+ "preferred_types": ["llm-reasoning", "llm-chat"],
531
+ "benchmark_weights": {
532
+ "gpqa_diamond": 0.25, "mmlu_pro": 0.20, "math_500": 0.15,
533
+ "mmlu_chemistry": 0.10, "mmlu_physics": 0.10, "mmlu_biology": 0.10,
534
+ "arena_elo_overall": 0.10,
535
+ },
536
+ "capability_weights": {
537
+ "reasoning": 0.30, "domain": 0.15, "language": 0.10,
538
+ },
539
+ "cost_weight": 0.0,
540
+ "context_weight": 0.10,
541
+ },
542
+ "science_chemistry": {
543
+ "preferred_types": ["llm-reasoning", "llm-chat"],
544
+ "benchmark_weights": {
545
+ "mmlu_chemistry": 0.30, "gpqa_diamond": 0.20, "mmlu_pro": 0.15,
546
+ "math_500": 0.15, "arena_elo_overall": 0.10, "ifeval": 0.10,
547
+ },
548
+ "capability_weights": {
549
+ "reasoning": 0.30, "domain": 0.15, "language": 0.10,
550
+ },
551
+ "cost_weight": 0.0,
552
+ "context_weight": 0.10,
553
+ },
554
+ "science_physics": {
555
+ "preferred_types": ["llm-reasoning", "llm-chat"],
556
+ "benchmark_weights": {
557
+ "mmlu_physics": 0.30, "gpqa_diamond": 0.20, "math_500": 0.15,
558
+ "mmlu_pro": 0.15, "arena_elo_overall": 0.10, "ifeval": 0.10,
559
+ },
560
+ "capability_weights": {
561
+ "reasoning": 0.30, "domain": 0.15, "language": 0.10,
562
+ },
563
+ "cost_weight": 0.0,
564
+ "context_weight": 0.10,
565
+ },
566
+ "science_biology": {
567
+ "preferred_types": ["llm-reasoning", "llm-chat"],
568
+ "benchmark_weights": {
569
+ "mmlu_biology": 0.30, "gpqa_diamond": 0.20, "mmlu_pro": 0.15,
570
+ "mmlu_clinical_knowledge": 0.10, "arena_elo_overall": 0.15,
571
+ "ifeval": 0.10,
572
+ },
573
+ "capability_weights": {
574
+ "reasoning": 0.30, "domain": 0.15, "language": 0.10,
575
+ },
576
+ "cost_weight": 0.0,
577
+ "context_weight": 0.10,
578
+ },
579
+
580
+ # ─── Sub-domain: Translation ──────────────────────────────
581
+ "translation": {
582
+ "preferred_types": ["llm-chat", "vlm"],
583
+ "benchmark_weights": {
584
+ "flores": 0.30, "mgsm": 0.20, "miracl": 0.20,
585
+ "mmlu_pro": 0.10, "arena_elo_overall": 0.10, "ifeval": 0.10,
586
+ },
587
+ "capability_weights": {
588
+ "language": 0.30, "creative": 0.15, "reasoning": 0.10,
589
+ },
590
+ "cost_weight": 0.0,
591
+ "context_weight": 0.15,
592
+ },
593
+
594
+ # ─── Sub-domain: Creative Writing ─────────────────────────
595
+ "writing_creative": {
596
+ "preferred_types": ["llm-chat", "llm-reasoning"],
597
+ "benchmark_weights": {
598
+ "alpaca_eval": 0.25, "wildbench": 0.20, "mt_bench": 0.20,
599
+ "arena_elo_style_control": 0.15, "arena_elo_overall": 0.10,
600
+ "ifeval": 0.10,
601
+ },
602
+ "capability_weights": {
603
+ "creative": 0.30, "language": 0.20, "reasoning": 0.10,
604
+ },
605
+ "cost_weight": 0.0,
606
+ "context_weight": 0.10,
607
+ },
608
+ "writing_technical": {
609
+ "preferred_types": ["llm-chat", "llm-reasoning", "llm-code"],
610
+ "benchmark_weights": {
611
+ "mt_bench": 0.20, "ifeval": 0.20, "alpaca_eval": 0.15,
612
+ "mmlu_pro": 0.15, "arena_elo_overall": 0.15, "wildbench": 0.15,
613
+ },
614
+ "capability_weights": {
615
+ "creative": 0.25, "reasoning": 0.20, "language": 0.15,
616
+ },
617
+ "cost_weight": 0.0,
618
+ "context_weight": 0.15,
619
+ },
620
+ "summarization": {
621
+ "preferred_types": ["llm-chat", "llm-reasoning"],
622
+ "benchmark_weights": {
623
+ "mt_bench": 0.20, "alpaca_eval": 0.20, "arena_elo_overall": 0.15,
624
+ "ifeval": 0.15, "wildbench": 0.15, "arena_elo_style_control": 0.15,
625
+ },
626
+ "capability_weights": {
627
+ "creative": 0.25, "language": 0.20, "reasoning": 0.15,
628
+ },
629
+ "cost_weight": 0.0,
630
+ "context_weight": 0.20,
631
+ },
632
+
633
+ # ─── Sub-domain: Math (competitive / advanced) ────────
634
+ "math_competition": {
635
+ "preferred_types": ["llm-reasoning", "llm-chat"],
636
+ "benchmark_weights": {
637
+ "aime_2025": 0.30, "math_500": 0.25, "aime_2026": 0.15,
638
+ "gpqa_diamond": 0.10, "arena_elo_math": 0.10, "gsm8k": 0.10,
639
+ },
640
+ "capability_weights": {
641
+ "reasoning": 0.35, "coding": 0.10,
642
+ },
643
+ "cost_weight": 0.0,
644
+ "context_weight": 0.05,
645
+ },
646
+
647
+ # ─── Sub-domain: Education ────────────────────────────
648
+ "education": {
649
+ "preferred_types": ["llm-chat", "llm-reasoning", "vlm"],
650
+ "benchmark_weights": {
651
+ "mmlu_pro": 0.25, "arc_challenge": 0.15, "hellaswag": 0.10,
652
+ "mt_bench": 0.15, "ifeval": 0.10, "arena_elo_overall": 0.15,
653
+ "truthfulqa": 0.10,
654
+ },
655
+ "capability_weights": {
656
+ "reasoning": 0.20, "language": 0.20, "creative": 0.15,
657
+ },
658
+ "cost_weight": 0.0,
659
+ "context_weight": 0.10,
660
+ },
661
+ "education_stem": {
662
+ "preferred_types": ["llm-reasoning", "llm-chat"],
663
+ "benchmark_weights": {
664
+ "mmlu_pro": 0.15, "mmlu_physics": 0.15, "mmlu_chemistry": 0.15,
665
+ "mmlu_biology": 0.10, "math_500": 0.15, "gpqa_diamond": 0.15,
666
+ "arena_elo_overall": 0.15,
667
+ },
668
+ "capability_weights": {
669
+ "reasoning": 0.30, "domain": 0.15, "language": 0.10,
670
+ },
671
+ "cost_weight": 0.0,
672
+ "context_weight": 0.10,
673
+ },
674
+ "education_humanities": {
675
+ "preferred_types": ["llm-chat", "llm-reasoning"],
676
+ "benchmark_weights": {
677
+ "mmlu_pro": 0.15, "mmlu_professional_law": 0.10,
678
+ "mmlu_business_ethics": 0.10, "truthfulqa": 0.15,
679
+ "alpaca_eval": 0.15, "mt_bench": 0.15, "arena_elo_overall": 0.20,
680
+ },
681
+ "capability_weights": {
682
+ "language": 0.25, "creative": 0.20, "reasoning": 0.15,
683
+ },
684
+ "cost_weight": 0.0,
685
+ "context_weight": 0.10,
686
+ },
687
+
688
+ # ─── Sub-domain: Data Science / Analytics ─────────────
689
+ "data_science": {
690
+ "preferred_types": ["llm-code", "llm-reasoning", "llm-chat"],
691
+ "benchmark_weights": {
692
+ "multipl_e_python": 0.20, "humaneval": 0.15, "math_500": 0.15,
693
+ "mmlu_pro": 0.10, "aider_polyglot": 0.10, "arena_elo_coding": 0.15,
694
+ "gpqa_diamond": 0.10, "arena_elo_overall": 0.05,
695
+ },
696
+ "capability_weights": {
697
+ "coding": 0.25, "reasoning": 0.25, "tool_use": 0.15,
698
+ },
699
+ "cost_weight": 0.0,
700
+ "context_weight": 0.15,
701
+ },
702
+
703
+ # ─── Sub-domain: Customer Support / Chatbot ───────────
704
+ "customer_support": {
705
+ "preferred_types": ["llm-chat", "vlm"],
706
+ "benchmark_weights": {
707
+ "arena_elo_overall": 0.20, "mt_bench": 0.20, "ifeval": 0.20,
708
+ "alpaca_eval": 0.15, "arena_elo_style_control": 0.15,
709
+ "truthfulqa": 0.10,
710
+ },
711
+ "capability_weights": {
712
+ "language": 0.25, "tool_use": 0.20, "creative": 0.10,
713
+ },
714
+ "cost_weight": 0.0,
715
+ "context_weight": 0.10,
716
+ },
717
+
718
+ # ─── Sub-domain: Content Moderation ───────────────────
719
+ "content_moderation": {
720
+ "preferred_types": ["safety-classifier", "llm-chat", "reward-model"],
721
+ "benchmark_weights": {
722
+ "helm_safety": 0.30, "toxigen": 0.25, "bbq": 0.20,
723
+ "ifeval": 0.10, "arena_elo_overall": 0.15,
724
+ },
725
+ "capability_weights": {
726
+ "safety": 0.30, "language": 0.15,
727
+ },
728
+ "cost_weight": 0.0,
729
+ "context_weight": 0.05,
730
+ },
731
+
732
+ # ─── Sub-domain: Research Assistant ───────────────────
733
+ "research_assistant": {
734
+ "preferred_types": ["llm-reasoning", "llm-chat", "vlm"],
735
+ "benchmark_weights": {
736
+ "gpqa_diamond": 0.20, "mmlu_pro": 0.15, "math_500": 0.10,
737
+ "arena_elo_overall": 0.15, "mt_bench": 0.10, "ifeval": 0.10,
738
+ "truthfulqa": 0.10, "arena_elo_hard_prompts": 0.10,
739
+ },
740
+ "capability_weights": {
741
+ "reasoning": 0.25, "language": 0.15, "tool_use": 0.15,
742
+ },
743
+ "cost_weight": 0.0,
744
+ "context_weight": 0.20,
745
+ },
746
+
747
+ # ─── Sub-domain: Roleplay / Character ─────────────────
748
+ "roleplay": {
749
+ "preferred_types": ["llm-chat"],
750
+ "benchmark_weights": {
751
+ "arena_elo_style_control": 0.25, "alpaca_eval": 0.20,
752
+ "wildbench": 0.15, "mt_bench": 0.15, "arena_elo_overall": 0.15,
753
+ "ifeval": 0.10,
754
+ },
755
+ "capability_weights": {
756
+ "creative": 0.35, "language": 0.25,
757
+ },
758
+ "cost_weight": 0.0,
759
+ "context_weight": 0.15,
760
+ },
761
+
762
+ # ─── Sub-domain: Audio ────────────────────────────────
763
+ "speech_to_text": {
764
+ "preferred_types": ["audio-stt", "audio-multimodal"],
765
+ "benchmark_weights": {
766
+ "wer_librispeech": 0.50, "arena_elo_overall": 0.20,
767
+ "miracl": 0.15, "mgsm": 0.15,
768
+ },
769
+ "capability_weights": {
770
+ "language": 0.20,
771
+ },
772
+ "cost_weight": 0.0,
773
+ "context_weight": 0.05,
774
+ },
775
+ "text_to_speech": {
776
+ "preferred_types": ["audio-tts", "audio-multimodal"],
777
+ "benchmark_weights": {
778
+ "mos_tts": 0.50, "arena_elo_overall": 0.20,
779
+ "arena_elo_style_control": 0.15, "mt_bench": 0.15,
780
+ },
781
+ "capability_weights": {
782
+ "creative": 0.15, "language": 0.15,
783
+ },
784
+ "cost_weight": 0.0,
785
+ "context_weight": 0.05,
786
+ },
787
+
788
+ # ─── Sub-domain: Image Generation ─────────────────────
789
+ "image_generation": {
790
+ "preferred_types": ["image-gen", "vlm"],
791
+ "benchmark_weights": {
792
+ "fid": 0.30, "clip_score": 0.30,
793
+ "arena_elo_overall": 0.20, "arena_elo_vision": 0.20,
794
+ },
795
+ "capability_weights": {
796
+ "creative": 0.30,
797
+ },
798
+ "cost_weight": 0.0,
799
+ "context_weight": 0.05,
800
+ },
801
+
802
+ # ─── Sub-domain: Cybersecurity ────────────────────────
803
+ "cybersecurity": {
804
+ "preferred_types": ["llm-code", "llm-reasoning", "llm-chat"],
805
+ "benchmark_weights": {
806
+ "humaneval": 0.15, "swe_bench_verified": 0.15,
807
+ "mmlu_computer_science": 0.15, "terminal_bench": 0.15,
808
+ "arena_elo_coding": 0.10, "gpqa_diamond": 0.10,
809
+ "arena_elo_overall": 0.10, "ifeval": 0.10,
810
+ },
811
+ "capability_weights": {
812
+ "coding": 0.25, "reasoning": 0.25, "tool_use": 0.20,
813
+ },
814
+ "cost_weight": 0.0,
815
+ "context_weight": 0.15,
816
+ },
817
+
818
+ # ─── Sub-domain: DevOps / Infrastructure ──────────────
819
+ "devops": {
820
+ "preferred_types": ["llm-code", "llm-chat", "llm-reasoning"],
821
+ "benchmark_weights": {
822
+ "terminal_bench": 0.25, "humaneval": 0.15, "swe_bench_verified": 0.15,
823
+ "aider_polyglot": 0.10, "arena_elo_coding": 0.15,
824
+ "ifeval": 0.10, "arena_elo_overall": 0.10,
825
+ },
826
+ "capability_weights": {
827
+ "coding": 0.25, "tool_use": 0.25, "reasoning": 0.15,
828
+ },
829
+ "cost_weight": 0.0,
830
+ "context_weight": 0.10,
831
+ },
832
+
833
+ # ─── Sub-domain: Multilingual ─────────────────────────
834
+ "multilingual": {
835
+ "preferred_types": ["llm-chat", "vlm"],
836
+ "benchmark_weights": {
837
+ "mgsm": 0.25, "miracl": 0.20, "flores": 0.20,
838
+ "mmlu_pro": 0.10, "arena_elo_overall": 0.15, "ifeval": 0.10,
839
+ },
840
+ "capability_weights": {
841
+ "language": 0.35, "creative": 0.10, "reasoning": 0.10,
842
+ },
843
+ "cost_weight": 0.0,
844
+ "context_weight": 0.15,
845
+ },
846
+
847
+ # ─── Sub-domain: Financial (specialties) ──────────────
848
+ "financial_analysis": {
849
+ "preferred_types": ["llm-reasoning", "llm-chat"],
850
+ "benchmark_weights": {
851
+ "finbench": 0.25, "finqa": 0.25, "math_500": 0.15,
852
+ "mmlu_professional_accounting": 0.10, "mmlu_pro": 0.10,
853
+ "arena_elo_overall": 0.15,
854
+ },
855
+ "capability_weights": {
856
+ "reasoning": 0.30, "domain": 0.15, "language": 0.10,
857
+ },
858
+ "cost_weight": 0.0,
859
+ "context_weight": 0.15,
860
+ },
861
+ "financial_compliance": {
862
+ "preferred_types": ["llm-chat", "llm-reasoning"],
863
+ "benchmark_weights": {
864
+ "finbench": 0.20, "legalbench": 0.20, "mmlu_professional_law": 0.15,
865
+ "mmlu_professional_accounting": 0.15, "ifeval": 0.15,
866
+ "arena_elo_overall": 0.15,
867
+ },
868
+ "capability_weights": {
869
+ "reasoning": 0.25, "domain": 0.15, "language": 0.15,
870
+ },
871
+ "cost_weight": 0.0,
872
+ "context_weight": 0.15,
873
+ },
874
+
875
+ # ─── Sub-domain: Science (more specialties) ───────────
876
+ "science_astronomy": {
877
+ "preferred_types": ["llm-reasoning", "llm-chat"],
878
+ "benchmark_weights": {
879
+ "mmlu_astronomy": 0.30, "gpqa_diamond": 0.20, "mmlu_physics": 0.15,
880
+ "math_500": 0.15, "mmlu_pro": 0.10, "arena_elo_overall": 0.10,
881
+ },
882
+ "capability_weights": {
883
+ "reasoning": 0.30, "domain": 0.15, "language": 0.10,
884
+ },
885
+ "cost_weight": 0.0,
886
+ "context_weight": 0.10,
887
+ },
888
+
889
+ # ─── Sub-domain: Legal (specialties) ──────────────────
890
+ "legal_contract_review": {
891
+ "preferred_types": ["llm-chat", "llm-reasoning"],
892
+ "benchmark_weights": {
893
+ "legalbench": 0.30, "mmlu_professional_law": 0.20,
894
+ "ifeval": 0.15, "arena_elo_overall": 0.10,
895
+ "mt_bench": 0.10, "mmlu_pro": 0.15,
896
+ },
897
+ "capability_weights": {
898
+ "reasoning": 0.25, "language": 0.20, "domain": 0.15,
899
+ },
900
+ "cost_weight": 0.0,
901
+ "context_weight": 0.20,
902
+ },
903
+
904
+ # ─── Sub-domain: Biotech / Life Sciences ──────────────
905
+ "biotech": {
906
+ "preferred_types": ["llm-reasoning", "llm-chat"],
907
+ "benchmark_weights": {
908
+ "mmlu_biology": 0.20, "mmlu_chemistry": 0.15, "medqa": 0.15,
909
+ "pubmedqa": 0.15, "gpqa_diamond": 0.15, "mmlu_pro": 0.10,
910
+ "arena_elo_overall": 0.10,
911
+ },
912
+ "capability_weights": {
913
+ "reasoning": 0.30, "domain": 0.20, "language": 0.10,
914
+ },
915
+ "cost_weight": 0.0,
916
+ "context_weight": 0.15,
917
+ },
918
+
919
+ # ─── Sub-domain: Accounting ───────────────────────────
920
+ "accounting": {
921
+ "preferred_types": ["llm-chat", "llm-reasoning"],
922
+ "benchmark_weights": {
923
+ "mmlu_professional_accounting": 0.30, "finbench": 0.20,
924
+ "finqa": 0.15, "math_500": 0.10, "ifeval": 0.10,
925
+ "arena_elo_overall": 0.15,
926
+ },
927
+ "capability_weights": {
928
+ "reasoning": 0.25, "domain": 0.15, "language": 0.10,
929
+ },
930
+ "cost_weight": 0.0,
931
+ "context_weight": 0.10,
932
+ },
933
+
934
+ # ─── Sub-domain: Code Review ──────────────────────────
935
+ "code_review": {
936
+ "preferred_types": ["llm-code", "llm-reasoning", "llm-chat"],
937
+ "benchmark_weights": {
938
+ "swe_bench_verified": 0.25, "humaneval": 0.15, "aider_polyglot": 0.15,
939
+ "arena_elo_coding": 0.15, "terminal_bench": 0.10,
940
+ "ifeval": 0.10, "arena_elo_overall": 0.10,
941
+ },
942
+ "capability_weights": {
943
+ "coding": 0.30, "reasoning": 0.25, "language": 0.10,
944
+ },
945
+ "cost_weight": 0.0,
946
+ "context_weight": 0.15,
947
+ },
948
+ }
949
+
950
+
951
+ # ═══════════════════════════════════════════════════════════════
952
+ # Platform classifications for hosting filter
953
+ # ═══════════════════════════════════════════════════════════════
954
+
955
+
956
+ # ═══════════════════════════════════════════════════════════════
957
+ # Verified evidence in the profiles (MODEL-32, option B)
958
+ # ═══════════════════════════════════════════════════════════════
959
+ #
960
+ # The census verifies benchmarks that publish current, dated results for
961
+ # current models, and few do. The profiles below weight the classic
962
+ # benchmarks a reader expects — HumanEval, SWE-bench Verified, GPQA Diamond —
963
+ # and the two sets did not overlap at all, so no reviewed evidence could move
964
+ # any ranking.
965
+ #
966
+ # Operator decision 2026-09-09: take the verified benchmarks into the profiles
967
+ # now, and keep verifying the classic ones as the standing goal (MODEL-33).
968
+ #
969
+ # The cap is the point. One evaluator's index carries at most this share of any
970
+ # profile, so reviewed evidence is load-bearing without a single source
971
+ # deciding what "best" means. Existing weights are scaled down proportionally,
972
+ # so a profile still sums to what it did before.
973
+
974
+ #: Maximum share of any profile that one evaluator's index may carry.
975
+ VERIFIED_INDEX_WEIGHT = 0.20
976
+
977
+ #: Verified benchmarks, mapped onto profiles by the category their benchgraph
978
+ #: page declares. Both declare higher_is_better, unit %, max_score 100,
979
+ #: which is where their ranges below come from — read, not assumed.
980
+ VERIFIED_ADDITIONS: dict[str, list[str]] = {
981
+ "coding": ["scicode"],
982
+ "reasoning": ["critpt"],
983
+ "science": ["scicode", "critpt"],
984
+ }
985
+
986
+ BENCHMARK_RANGES.update({
987
+ # Confirmed from each benchmark's page: higher_is_better, %, max 100.
988
+ "critpt": (0.0, 100.0),
989
+ "scicode": (0.0, 100.0),
990
+ })
991
+
992
+
993
+ def _apply_verified_additions() -> None:
994
+ """Fold the verified benchmarks into the profiles, under the cap."""
995
+ for profile_key, benchmarks in VERIFIED_ADDITIONS.items():
996
+ profile = USE_CASE_PROFILES.get(profile_key)
997
+ if not profile:
998
+ continue
999
+ weights = profile.get("benchmark_weights") or {}
1000
+ # Anything already weighted is left alone; only genuinely new entries
1001
+ # take from the existing budget.
1002
+ new = [b for b in benchmarks if b not in weights]
1003
+ if not new:
1004
+ continue
1005
+ scale = 1.0 - VERIFIED_INDEX_WEIGHT
1006
+ rescaled = {k: round(v * scale, 4) for k, v in weights.items()}
1007
+ share = round(VERIFIED_INDEX_WEIGHT / len(new), 4)
1008
+ for benchmark in new:
1009
+ rescaled[benchmark] = share
1010
+ profile["benchmark_weights"] = rescaled
1011
+
1012
+
1013
+ _apply_verified_additions()
1014
+
1015
+ # Product defaults, not statistical confidence thresholds. The benchmark set
1016
+ # stays fixed, including when a candidate or an evaluator has sparse coverage.
1017
+ # CLI / API stay conservative. The wizard is a different surface: a browser
1018
+ # wants breadth, dpf wants a shortlist it can defend. MODEL-34, 2026-09-11.
1019
+ MIN_BENCHMARK_COVERAGE = 0.50
1020
+ WIZARD_BENCHMARK_COVERAGE = 0.25
1021
+ MIN_BENCHMARK_COUNT = 2
1022
+
1023
+
1024
+ # ── the neutrality commitment (MODEL-70) ─────────────────────────────────────
1025
+ #
1026
+ # A floor is checkable because it is published as a number next to the answer it
1027
+ # shaped. The neutrality claim was only ever prose, which means a caller had to
1028
+ # trust it. It ships here, beside the floors, for the same reason the floors
1029
+ # ship: so an agent can read the commitment out of the same object that carries
1030
+ # the policy it acted on, rather than believe a page it never fetched.
1031
+ #
1032
+ # This constant is the single source. `ranking_policy()` carries it into
1033
+ # `/api/rank/profiles.json`, into every `rank_report()` and therefore into
1034
+ # `rankings.json`, and into the `policy` block of every `POST /v1/rank`
1035
+ # response. `pipeline/legal.py` renders the published pages from the same
1036
+ # values, so the prose and the JSON cannot drift apart.
1037
+ #
1038
+ # Source of the rule: `docs/agent-commerce-assessment.md` §3, which requires it
1039
+ # to be written into the terms before any money moves.
1040
+
1041
+ #: Verbatim. Quoted in `docs/legal/terms-of-service.md` and rendered on
1042
+ #: https://modelspec.dev/legal/terms/ — three copies of one string, checked by
1043
+ #: `tests/test_legal.py`. Editing it here is editing the published terms.
1044
+ HONEST_BROKER_RULE = (
1045
+ "Charging the consumer of a recommendation is compatible with being an "
1046
+ "honest broker. Charging the subjects of one is not."
1047
+ )
1048
+
1049
+ #: Verbatim, and the word "permanently" is load-bearing: it is the difference
1050
+ #: between a current price list and a commitment.
1051
+ NEUTRALITY_PLEDGE = (
1052
+ "No referral fees, no paid placement, no provider-paid visibility, "
1053
+ "permanently."
1054
+ )
1055
+
1056
+ #: Verbatim, neutrality 1.1 (2026-09-23). The case the pledge does not reach:
1057
+ #: money flowing *to* a vendor we catalogue rather than from one. Quoted in
1058
+ #: `docs/legal/neutrality.md`; `tests/test_legal.py` holds the two equal. The
1059
+ #: mechanisms are `schema/suppliers.py` (the vendors we pay), the disclosure the
1060
+ #: card page prints from it (`pipeline/render.py`), and the refusals in
1061
+ #: `scripts/attribution.py` (`supplier_conflict`, `apply_policy`). MODEL-101.
1062
+ VENDOR_PURCHASE_RULE = (
1063
+ "We may be a paying customer of a vendor whose models we catalogue. When we "
1064
+ "are, the card says so, and no field on that vendor's card is ever set by "
1065
+ "that vendor's own model."
1066
+ )
1067
+
1068
+ #: Where a machine reads the long forms. Static Pages, no key, no account.
1069
+ LEGAL_BASE_URL = "https://modelspec.dev/legal"
1070
+
1071
+
1072
+ def neutrality_commitment() -> dict[str, Any]:
1073
+ """What ModelSpec will not take money for, in a shape an agent can check.
1074
+
1075
+ Every `assertion` is negative on purpose. The positioning is a set of things
1076
+ the service is structurally unable to do, not a set of things it promises
1077
+ not to do, so each one is either contradicted by an observable fact or it
1078
+ holds. A new assertion is additive; flipping one of these `false` values to
1079
+ `true` is not a version bump, it is a different product.
1080
+ """
1081
+ return {
1082
+ "version": "neutrality-v1",
1083
+ "operator": "Sparks and Sawdust LLC",
1084
+ "rule": HONEST_BROKER_RULE,
1085
+ "pledge": NEUTRALITY_PLEDGE,
1086
+ "permanent": True,
1087
+ "assertions": {
1088
+ # §4.5: never charge the subjects of a recommendation.
1089
+ "accepts_referral_fees": False,
1090
+ "accepts_paid_placement": False,
1091
+ "accepts_provider_paid_visibility": False,
1092
+ # §10.1: recommend and hand off. A router earns on token volume,
1093
+ # and margin that grows with volume is a steering incentive.
1094
+ "proxies_inference_tokens": False,
1095
+ # §10.2: a request carries a profile, not a prompt. There is
1096
+ # nothing to retain, which is why this is architecture and not a
1097
+ # retention promise.
1098
+ "stores_customer_prompts": False,
1099
+ # Neutrality 1.1: buying from a vendor we catalogue. The card
1100
+ # page discloses it (`pipeline/render.py`, from
1101
+ # `schema/suppliers.py`), and `scripts/attribution.py` refuses a
1102
+ # supplier's model any field on that supplier's card.
1103
+ "conceals_purchases_from_catalogued_vendors": False,
1104
+ "lets_supplier_models_write_supplier_cards": False,
1105
+ },
1106
+ #: §10.3: neutrality past money, into sourcing. Naming the stages is the
1107
+ #: point — "neutral ranking" would leave the tie-break and the hosting
1108
+ #: suggestion unclaimed, and those are where a lean is cheapest to hide.
1109
+ "source_neutral_at": [
1110
+ "ranking",
1111
+ "tie_breaks",
1112
+ "hosting_suggestions",
1113
+ "route_advice",
1114
+ ],
1115
+ "charges": "the consumer of a recommendation, never its subjects",
1116
+ "vendor_purchases": VENDOR_PURCHASE_RULE,
1117
+ "method_source": (
1118
+ "https://github.com/turbobeest/modelspec/blob/main/api/ranking/engine.py"
1119
+ ),
1120
+ "terms_url": f"{LEGAL_BASE_URL}/terms/",
1121
+ "neutrality_url": f"{LEGAL_BASE_URL}/neutrality/",
1122
+ "privacy_url": f"{LEGAL_BASE_URL}/privacy/",
1123
+ }
1124
+
1125
+
1126
+ def ranking_policy(*, min_benchmark_coverage: float | None = None) -> dict[str, Any]:
1127
+ """Policy for one ranking surface. Default is the CLI floor."""
1128
+ coverage = MIN_BENCHMARK_COVERAGE if min_benchmark_coverage is None else min_benchmark_coverage
1129
+ return {
1130
+ "version": "incomplete-evidence-v1",
1131
+ "ordering": "conservative_lower_bound",
1132
+ "min_benchmark_coverage": coverage,
1133
+ "cli_min_benchmark_coverage": MIN_BENCHMARK_COVERAGE,
1134
+ "wizard_min_benchmark_coverage": WIZARD_BENCHMARK_COVERAGE,
1135
+ "min_benchmark_count": MIN_BENCHMARK_COUNT,
1136
+ "limit_applies_to": "ranked_only",
1137
+ "uncertainty": "missing-benchmark bounds, not statistical confidence intervals",
1138
+ # Additive under the contract's own rule (docs/cli-contract.md: "New
1139
+ # fields may be added to any object"). No existing field widens.
1140
+ "neutrality": neutrality_commitment(),
1141
+ }
1142
+
1143
+
1144
+ RANKING_POLICY = ranking_policy()
1145
+
1146
+
1147
+ class IncompleteEvidenceError(ValueError):
1148
+ """Optional complete-ordering signal wrapping a rank_report().
1149
+
1150
+ `rank()` returns the ranked shortlist and does not raise when other models
1151
+ lack evidence. Callers that require every candidate to be ordered may raise
1152
+ this themselves after inspecting rank_report(). `report` preserves both the
1153
+ ranked shortlist and every unranked candidate.
1154
+ """
1155
+
1156
+ def __init__(self, report: dict[str, Any]):
1157
+ self.report = report
1158
+ names = []
1159
+ for row in report["unranked"]:
1160
+ if isinstance(row, dict):
1161
+ names.append(f"{row['display_name']} ({row['benchmark_coverage']:.0%} coverage)")
1162
+ else:
1163
+ names.append(f"{row.display_name} ({row.benchmark_coverage:.0%} coverage)")
1164
+ super().__init__(
1165
+ "Cannot return a total ordering. Unranked for insufficient evidence: "
1166
+ + "; ".join(names)
1167
+ + ". These models are not ranked low. Use rank_report() or "
1168
+ "`.venv/bin/python -m pipeline.ranking PROFILE` to see ranked and unranked results."
1169
+ )
1170
+
1171
+
1172
+ def _benchmark_evidence(scores: dict[str, float], profile: dict[str, Any],
1173
+ min_coverage: float | None = None) -> dict[str, Any]:
1174
+ """Bound the fixed profile without guessing unmeasured benchmark values.
1175
+
1176
+ All normalized benchmarks lie in [0, 100]. Missing weight therefore spans
1177
+ [0, weight * 100], rather than being a measurement of zero. Other composite
1178
+ components are held fixed; these are not bounds on real-world ability.
1179
+ """
1180
+ weights = profile.get("benchmark_weights", {})
1181
+ if any(not math.isfinite(w) or w < 0 for w in weights.values()):
1182
+ raise ValueError("Benchmark weights must be finite and nonnegative")
1183
+ weights = {b: w for b, w in weights.items() if w > 0}
1184
+ present = {
1185
+ b: _normalize_benchmark(b, scores[b]) for b in weights
1186
+ if scores.get(b) is not None and math.isfinite(scores[b])
1187
+ }
1188
+ total_weight = sum(weights.values())
1189
+ present_weight = sum(weights[b] for b in present)
1190
+ missing = sorted(weights.keys() - present.keys())
1191
+ missing_weight = sum(weights[b] for b in missing)
1192
+ coverage = present_weight / total_weight if total_weight else 0.0
1193
+ required = min(MIN_BENCHMARK_COUNT, len(weights))
1194
+ floor = MIN_BENCHMARK_COVERAGE if min_coverage is None else min_coverage
1195
+ rankable = (total_weight > 0 and coverage + 1e-12 >= floor
1196
+ and len(present) >= required)
1197
+ lower = sum(present[b] * weights[b] for b in present) * 0.40
1198
+ return {
1199
+ "rank_status": "ranked" if rankable else "unranked",
1200
+ "unranked_reason": None if rankable else "insufficient_benchmark_evidence",
1201
+ "benchmark_coverage": coverage,
1202
+ "benchmark_count": len(present),
1203
+ "required_benchmark_count": required,
1204
+ "missing_benchmarks": missing,
1205
+ "benchmark_estimate": lower / (0.40 * present_weight) if present_weight else None,
1206
+ "benchmark_lower_bound": lower,
1207
+ "benchmark_upper_bound": lower + missing_weight * 40.0,
1208
+ "benchmark_contributions": {b: round(present[b] * weights[b], 2) for b in present},
1209
+ }
1210
+
1211
+
1212
+ def _ranking_status(ranked: list, unranked: list) -> str:
1213
+ if unranked:
1214
+ return "partial" if ranked else "unavailable"
1215
+ return "complete" if ranked else "empty"
1216
+
1217
+
1218
+ CLOUD_PLATFORMS = {
1219
+ "aws_bedrock", "azure_ai_foundry", "google_vertex_ai", "nvidia_nim",
1220
+ "ibm_watsonx", "snowflake_cortex", "groq", "together_ai", "fireworks_ai",
1221
+ "replicate", "deepinfra", "cerebras", "sambanova", "openrouter",
1222
+ }
1223
+
1224
+ PROVIDER_PLATFORMS = {
1225
+ "anthropic", "openai", "google", "mistral", "cohere", "xai", "deepseek",
1226
+ "claude_ai", "chatgpt", "gemini_app", "grok_xai", "meta_ai",
1227
+ "mistral_plateforme", "ai21_labs",
1228
+ }
1229
+
1230
+ LOCAL_PLATFORMS = {
1231
+ "ollama", "lm_studio", "gpt4all", "jan_ai", "mlx_community", "open_webui",
1232
+ }
1233
+
1234
+ # Platforms that map to runtime identifiers (used to derive runtimes from graph edges)
1235
+ RUNTIME_PLATFORMS = {
1236
+ "ollama", "lm_studio", "gpt4all",
1237
+ }
1238
+
1239
+
1240
+ # ═══════════════════════════════════════════════════════════════
1241
+ # Data classes
1242
+ # ═══════════════════════════════════════════════════════════════
1243
+
1244
+ @dataclass
1245
+ class ModelData:
1246
+ """All data gathered for one model during ranking."""
1247
+ model_id: str
1248
+ display_name: str
1249
+ model_type: str | None = None
1250
+ model_subtypes: list[str] = field(default_factory=list)
1251
+ status: str | None = None
1252
+ open_weights: bool | None = None
1253
+ origin_country: str | None = None
1254
+ total_parameters: int | None = None
1255
+ active_parameters: int | None = None
1256
+ context_window: int | None = None
1257
+ cost_input: float | None = None
1258
+ cost_output: float | None = None
1259
+ arena_elo_overall: float | None = None
1260
+ release_date: str | None = None
1261
+ reasoning: bool = False
1262
+ tool_call: bool = False
1263
+ vision_input: bool = False
1264
+ # Populated from graph edges
1265
+ benchmark_scores: dict[str, float] = field(default_factory=dict)
1266
+ capability_tiers: dict[str, str] = field(default_factory=dict)
1267
+ available_platforms: set[str] = field(default_factory=set)
1268
+ estimated_tps: float | None = None # Estimated tok/s on target hardware
1269
+ concurrent_instances: int | None = None # How many instances fit on target hardware
1270
+ hardware_fits: dict[str, dict] = field(default_factory=dict)
1271
+ # Tags and runtimes from the graph
1272
+ tags: set[str] = field(default_factory=set)
1273
+ runtimes: set[str] = field(default_factory=set)
1274
+ # Extra node props for pass-through
1275
+ provider: str | None = None
1276
+
1277
+
1278
+ @dataclass
1279
+ class ScoredModel:
1280
+ """Result of scoring a model."""
1281
+ model_id: str
1282
+ display_name: str
1283
+ model_type: str | None
1284
+ score: float | None
1285
+ benchmark_score: float
1286
+ capability_score: float
1287
+ cost_score: float
1288
+ context_score: float
1289
+ type_bonus: float
1290
+ speed_score: float = 0.0
1291
+ estimated_tps: float | None = None
1292
+ concurrent_instances: int | None = None
1293
+ reasons: list[str] = field(default_factory=list)
1294
+ benchmark_contributions: dict[str, float] = field(default_factory=dict)
1295
+ # Pass-through for the response
1296
+ arena_elo_overall: float | None = None
1297
+ total_parameters: int | None = None
1298
+ context_window: int | None = None
1299
+ cost_input: float | None = None
1300
+ cost_output: float | None = None
1301
+ open_weights: bool | None = None
1302
+ provider: str | None = None
1303
+ status: str | None = None
1304
+ rank: int | None = None
1305
+ rank_status: str = "unranked"
1306
+ unranked_reason: str | None = None
1307
+ benchmark_coverage: float = 0.0
1308
+ benchmark_count: int = 0
1309
+ required_benchmark_count: int = 0
1310
+ missing_benchmarks: list[str] = field(default_factory=list)
1311
+ benchmark_estimate: float | None = None
1312
+ benchmark_lower_bound: float = 0.0
1313
+ benchmark_upper_bound: float = 0.0
1314
+ score_lower_bound: float = 0.0
1315
+ score_upper_bound: float = 0.0
1316
+
1317
+
1318
+ # ═══════════════════════════════════════════════════════════════
1319
+ # Ranking Engine
1320
+ # ═══════════════════════════════════════════════════════════════
1321
+
1322
+ class RankingEngine:
1323
+ """4-stage ranking pipeline: Filter -> Score -> Rank -> Explain."""
1324
+
1325
+ def __init__(self, graph):
1326
+ self.graph = graph
1327
+
1328
+ # ─── Main entry point ────────────────────────────────────
1329
+
1330
+ def rank(
1331
+ self,
1332
+ use_case: str | None = None,
1333
+ hardware: str | None = None,
1334
+ constraints: dict[str, Any] | None = None,
1335
+ limit: int = 10,
1336
+ ) -> list[ScoredModel]:
1337
+ """Return the models that can honestly be ordered for this profile.
1338
+
1339
+ Unrankable models are the normal catalogue state, not an error. This
1340
+ returns the ranked shortlist, which may be empty when nothing has
1341
+ enough evidence. Use rank_report() to see withheld models and
1342
+ ranking_status.
1343
+ """
1344
+ return self.rank_report(use_case, hardware, constraints, limit)["ranked"]
1345
+
1346
+ def rank_report(
1347
+ self,
1348
+ use_case: str | None = None,
1349
+ hardware: str | None = None,
1350
+ constraints: dict[str, Any] | None = None,
1351
+ limit: int = 10,
1352
+ ) -> dict[str, Any]:
1353
+ """Return a bounded shortlist and all unranked candidates separately."""
1354
+ if limit < 0:
1355
+ raise ValueError("limit must be nonnegative")
1356
+ constraints = constraints or {}
1357
+ profile = USE_CASE_PROFILES.get(use_case or "general", USE_CASE_PROFILES["general"])
1358
+
1359
+ # Stage 0: Fetch all candidate models from the graph
1360
+ candidates = self._fetch_candidates(hardware, constraints)
1361
+ logger.info(f"Fetched {len(candidates)} candidates from graph")
1362
+
1363
+ # Stage 1: Filter
1364
+ filtered = self._filter(candidates, constraints, profile)
1365
+ logger.info(f"After filtering: {len(filtered)} models remain")
1366
+
1367
+ # Stage 2: Score
1368
+ scored = [self._score(m, profile) for m in filtered]
1369
+
1370
+ # Stage 3: Rank (sort by score, tie-break by ELO then params)
1371
+ ranked = [s for s in scored if s.rank_status == "ranked"]
1372
+ unranked = [s for s in scored if s.rank_status == "unranked"]
1373
+ ranked.sort(key=lambda s: (
1374
+ s.score,
1375
+ s.arena_elo_overall or 0,
1376
+ s.total_parameters or 0,
1377
+ ), reverse=True)
1378
+ unranked.sort(key=lambda s: (s.display_name.lower(), s.model_id))
1379
+ for position, result in enumerate(ranked, 1):
1380
+ result.rank = position
1381
+
1382
+ # Stage 4: Explain (already built into scoring, but add rank-relative info)
1383
+ self._explain(scored, profile, use_case)
1384
+
1385
+ return {
1386
+ "ranking_status": _ranking_status(ranked, unranked),
1387
+ "policy": dict(RANKING_POLICY),
1388
+ "ranked_count": len(ranked), "unranked_count": len(unranked),
1389
+ "ranked": ranked[:limit], "unranked": unranked,
1390
+ }
1391
+
1392
+ # ─── Stage 0: Fetch candidates ───────────────────────────
1393
+
1394
+ def _fetch_candidates(
1395
+ self,
1396
+ hardware: str | None,
1397
+ constraints: dict[str, Any],
1398
+ ) -> list[ModelData]:
1399
+ """Query FalkorDB for all model nodes with their edges."""
1400
+ # Build the base query with optional hardware join
1401
+ if hardware:
1402
+ base_q = (
1403
+ "MATCH (m:Model)-[fit:FITS_ON]->(h:Hardware {id: $hw_id}) "
1404
+ "RETURN m"
1405
+ )
1406
+ params: dict[str, Any] = {"hw_id": hardware}
1407
+ else:
1408
+ base_q = "MATCH (m:Model) RETURN m"
1409
+ params = {}
1410
+
1411
+ result = self.graph.query(base_q, params)
1412
+ model_ids: list[str] = []
1413
+ models_by_id: dict[str, ModelData] = {}
1414
+
1415
+ for row in result.result_set:
1416
+ node = row[0]
1417
+ props = dict(node.properties) if node.properties else {}
1418
+ mid = props.get("id", "")
1419
+ if not mid:
1420
+ continue
1421
+
1422
+ md = ModelData(
1423
+ model_id=mid,
1424
+ display_name=props.get("display_name", mid),
1425
+ model_type=props.get("model_type"),
1426
+ status=props.get("status"),
1427
+ open_weights=props.get("open_weights"),
1428
+ origin_country=props.get("origin_country") or None,
1429
+ total_parameters=_safe_int(props.get("total_parameters")),
1430
+ active_parameters=_safe_int(props.get("active_parameters")),
1431
+ context_window=_safe_int(props.get("context_window")),
1432
+ cost_input=_safe_float(props.get("cost_input")),
1433
+ cost_output=_safe_float(props.get("cost_output")),
1434
+ arena_elo_overall=_safe_float(props.get("arena_elo_overall")),
1435
+ release_date=props.get("release_date"),
1436
+ reasoning=bool(props.get("reasoning")),
1437
+ tool_call=bool(props.get("tool_call")),
1438
+ vision_input=bool(props.get("vision_input")),
1439
+ provider=mid.split("/")[0] if "/" in mid else None,
1440
+ model_subtypes=props.get("model_subtypes", "").split(",") if props.get("model_subtypes") else [],
1441
+ )
1442
+ models_by_id[mid] = md
1443
+ model_ids.append(mid)
1444
+
1445
+ if not model_ids:
1446
+ return []
1447
+
1448
+ # Batch-fetch benchmark scores via SCORED_ON edges
1449
+ bench_q = (
1450
+ "MATCH (m:Model)-[e:SCORED_ON]->(b:Benchmark) "
1451
+ "RETURN m.id, b.id, e.value"
1452
+ )
1453
+ bench_result = self.graph.query(bench_q)
1454
+ for row in bench_result.result_set:
1455
+ mid, bench_id, value = row
1456
+ if mid in models_by_id and value is not None:
1457
+ models_by_id[mid].benchmark_scores[bench_id] = float(value)
1458
+
1459
+ # Batch-fetch capability tiers via HAS_CAPABILITY edges
1460
+ cap_q = (
1461
+ "MATCH (m:Model)-[e:HAS_CAPABILITY]->(c:Capability) "
1462
+ "RETURN m.id, c.id, e.tier"
1463
+ )
1464
+ cap_result = self.graph.query(cap_q)
1465
+ for row in cap_result.result_set:
1466
+ mid, cap_id, tier = row
1467
+ if mid in models_by_id and tier:
1468
+ # Keep the best tier if multiple edges exist
1469
+ existing = models_by_id[mid].capability_tiers.get(cap_id)
1470
+ if existing is None or _tier_rank(tier) < _tier_rank(existing):
1471
+ models_by_id[mid].capability_tiers[cap_id] = tier
1472
+
1473
+ # Batch-fetch platform availability via AVAILABLE_ON edges
1474
+ plat_q = (
1475
+ "MATCH (m:Model)-[:AVAILABLE_ON]->(p:Platform) "
1476
+ "RETURN m.id, p.id"
1477
+ )
1478
+ plat_result = self.graph.query(plat_q)
1479
+ for row in plat_result.result_set:
1480
+ mid, plat_id = row
1481
+ if mid in models_by_id and plat_id:
1482
+ models_by_id[mid].available_platforms.add(plat_id)
1483
+ # Derive runtime info from platform availability
1484
+ if plat_id in RUNTIME_PLATFORMS:
1485
+ models_by_id[mid].runtimes.add(plat_id)
1486
+
1487
+ # Batch-fetch tags via TAGGED_WITH edges
1488
+ tag_q = (
1489
+ "MATCH (m:Model)-[:TAGGED_WITH]->(t:Tag) "
1490
+ "RETURN m.id, t.id"
1491
+ )
1492
+ tag_result = self.graph.query(tag_q)
1493
+ for row in tag_result.result_set:
1494
+ mid, tag_id = row
1495
+ if mid in models_by_id and tag_id:
1496
+ models_by_id[mid].tags.add(tag_id)
1497
+
1498
+ return list(models_by_id.values())
1499
+
1500
+ # ─── Stage 1: Filter ─────────────────────────────────────
1501
+
1502
+ def _filter(
1503
+ self,
1504
+ candidates: list[ModelData],
1505
+ constraints: dict[str, Any],
1506
+ profile: dict[str, Any],
1507
+ ) -> list[ModelData]:
1508
+ """Eliminate models that fail hard constraints."""
1509
+ result = []
1510
+
1511
+ model_type_filter = constraints.get("model_type")
1512
+ open_weights_req = constraints.get("open_weights")
1513
+ max_cost = _safe_float(constraints.get("max_cost_input"))
1514
+ min_context = _safe_int(constraints.get("min_context"))
1515
+ min_params = _safe_int(constraints.get("min_params"))
1516
+ max_params = _safe_int(constraints.get("max_params"))
1517
+ origin_whitelist = constraints.get("origin_countries") # list of country codes
1518
+ origin_blacklist = constraints.get("origin_blacklist") # list of country codes
1519
+ require_reasoning = constraints.get("reasoning")
1520
+ require_tools = constraints.get("tool_use")
1521
+ require_vision = constraints.get("vision")
1522
+ provider_filter = constraints.get("provider")
1523
+ hw_memory_gb = _safe_float(constraints.get("hw_memory_gb"))
1524
+ hw_bandwidth_gbps = _safe_float(constraints.get("hw_bandwidth_gbps"))
1525
+ hw_quant = constraints.get("hw_quant", "")
1526
+ hw_tops = _safe_float(constraints.get("hw_tops"))
1527
+
1528
+ for m in candidates:
1529
+ # Skip deprecated/sunset unless explicitly requested
1530
+ if m.status in ("deprecated", "sunset"):
1531
+ continue
1532
+
1533
+ # Model type filter
1534
+ if model_type_filter:
1535
+ if m.model_type != model_type_filter:
1536
+ continue
1537
+
1538
+ # Open weights requirement
1539
+ if open_weights_req is True:
1540
+ if m.open_weights is not True:
1541
+ continue
1542
+
1543
+ # Max cost threshold
1544
+ if max_cost is not None and m.cost_input is not None:
1545
+ if m.cost_input > max_cost:
1546
+ continue
1547
+
1548
+ # Min context window
1549
+ if min_context is not None and m.context_window is not None:
1550
+ if m.context_window < min_context:
1551
+ continue
1552
+
1553
+ # Parameter bounds
1554
+ if min_params is not None and m.total_parameters is not None:
1555
+ if m.total_parameters < min_params:
1556
+ continue
1557
+ if max_params is not None and m.total_parameters is not None:
1558
+ if m.total_parameters > max_params:
1559
+ continue
1560
+
1561
+ # Origin country whitelist — exclude models with unknown origin too
1562
+ if origin_whitelist:
1563
+ if not m.origin_country or m.origin_country not in origin_whitelist:
1564
+ continue
1565
+
1566
+ # Origin country blacklist
1567
+ if origin_blacklist and m.origin_country:
1568
+ if m.origin_country in origin_blacklist:
1569
+ continue
1570
+
1571
+ # Capability requirements
1572
+ if require_reasoning and not m.reasoning:
1573
+ continue
1574
+ if require_tools and not m.tool_call:
1575
+ continue
1576
+ if require_vision and not m.vision_input:
1577
+ continue
1578
+
1579
+ # Provider filter
1580
+ if provider_filter and m.provider != provider_filter:
1581
+ continue
1582
+
1583
+ # Hosting mode filter
1584
+ hosting_modes = constraints.get("hosting") # list: ["local", "cloud", "provider"]
1585
+ if hosting_modes:
1586
+ passes_hosting = False
1587
+ for mode in hosting_modes:
1588
+ if mode == "local":
1589
+ # Must be open-weights to run locally
1590
+ if m.open_weights:
1591
+ passes_hosting = True
1592
+ elif mode == "cloud":
1593
+ # Check if available on any cloud/inference platform
1594
+ if m.available_platforms & CLOUD_PLATFORMS:
1595
+ passes_hosting = True
1596
+ elif m.open_weights:
1597
+ # Open-weight models can always be deployed to cloud
1598
+ passes_hosting = True
1599
+ elif mode == "provider":
1600
+ # Check if available via provider's own API
1601
+ if m.available_platforms & PROVIDER_PLATFORMS:
1602
+ passes_hosting = True
1603
+ elif not m.open_weights:
1604
+ # API-only models are always provider-hosted
1605
+ passes_hosting = True
1606
+ if not passes_hosting:
1607
+ continue
1608
+
1609
+ # Specific platform filter
1610
+ required_platforms = constraints.get("platforms") # list of platform IDs
1611
+ if required_platforms:
1612
+ if not m.available_platforms & set(required_platforms):
1613
+ # Also check by provider slug for provider platforms
1614
+ provider_match = m.provider in required_platforms
1615
+ if not provider_match:
1616
+ continue
1617
+
1618
+ # Runtime filter — require model to support specific runtimes
1619
+ required_runtimes = constraints.get("runtime") # list of runtime IDs
1620
+ if required_runtimes:
1621
+ # Check explicit runtime platforms
1622
+ model_runtimes = m.runtimes | (m.available_platforms & {
1623
+ "ollama", "lm_studio", "gpt4all", "vllm", "mlx",
1624
+ "llama_cpp", "transformers",
1625
+ })
1626
+ # Infer runtime compatibility for open-weights models:
1627
+ # If on HuggingFace or open-weights, they can run on vllm/transformers/llama_cpp
1628
+ if m.open_weights:
1629
+ model_runtimes |= {"vllm", "transformers", "llama_cpp"}
1630
+ if m.available_platforms & {"ollama", "lm_studio", "gpt4all"}:
1631
+ model_runtimes |= {"ollama", "lm_studio", "gpt4all"}
1632
+ if not model_runtimes & set(required_runtimes):
1633
+ continue
1634
+
1635
+ # OpenAI SDK compatibility filter — require "openai-compatible" tag
1636
+ if constraints.get("openai_compatible"):
1637
+ if "openai-compatible" not in m.tags:
1638
+ continue
1639
+
1640
+ # Hardware fit check — estimate if model fits and how fast it would run
1641
+ if hw_memory_gb and hw_memory_gb > 0:
1642
+ bytes_per_param = {"Q2": 0.25, "Q4": 0.5, "Q8": 1.0, "FP16": 2.0}.get(hw_quant, 0.5)
1643
+ usable_mem = hw_memory_gb * 0.85 # reserve 15% for OS/KV cache
1644
+
1645
+ if m.total_parameters:
1646
+ model_mem_gb = m.total_parameters * bytes_per_param / 1e9
1647
+ # Hard filter: won't fit in memory
1648
+ if model_mem_gb > usable_mem:
1649
+ continue
1650
+ # Concurrent instances that fit
1651
+ m.concurrent_instances = max(1, int(usable_mem / model_mem_gb))
1652
+ # Estimate tokens/sec from memory bandwidth
1653
+ # tok/s ≈ bandwidth / model_memory (memory-bandwidth-bound)
1654
+ if hw_bandwidth_gbps and hw_bandwidth_gbps > 0:
1655
+ est_tps = hw_bandwidth_gbps / model_mem_gb
1656
+ m.estimated_tps = round(est_tps, 1)
1657
+ # Filter out unusably slow models (< 1 tok/s)
1658
+ if est_tps < 1.0:
1659
+ continue
1660
+ else:
1661
+ # No parameter data — use heuristic: API-only models are fine,
1662
+ # but for local hosting, unknown-size models on small hardware are risky
1663
+ if hw_memory_gb <= 24 and m.open_weights:
1664
+ # Small device + open weights + unknown size = skip
1665
+ # (likely too big for an RPi or MacBook Air)
1666
+ continue
1667
+
1668
+ result.append(m)
1669
+
1670
+ return result
1671
+
1672
+ # ─── Stage 2: Score ──────────────────────────────────────
1673
+
1674
+ def _score(self, model: ModelData, profile: dict[str, Any]) -> ScoredModel:
1675
+ """Compute weighted composite score for a model against a profile."""
1676
+
1677
+ # --- Benchmark scoring (up to 40 points) ---
1678
+ evidence = _benchmark_evidence(model.benchmark_scores, profile)
1679
+ bench_score_scaled = evidence["benchmark_lower_bound"]
1680
+
1681
+ # --- Capability scoring (up to 20 points) ---
1682
+ cap_weights = profile.get("capability_weights", {})
1683
+ cap_score = 0.0
1684
+ total_cap_weight = sum(cap_weights.values()) if cap_weights else 1.0
1685
+
1686
+ for cap_name, weight in cap_weights.items():
1687
+ # Look for the main capability tier (e.g., "coding", "reasoning")
1688
+ tier = model.capability_tiers.get(cap_name)
1689
+ if tier:
1690
+ tier_pts = _tier_points(tier)
1691
+ cap_score += tier_pts * (weight / total_cap_weight)
1692
+ else:
1693
+ # Check for sub-capabilities (e.g., "coding:debugging")
1694
+ sub_tiers = [
1695
+ t for cid, t in model.capability_tiers.items()
1696
+ if cid.startswith(f"{cap_name}:")
1697
+ ]
1698
+ if sub_tiers:
1699
+ # Average the sub-capability tiers
1700
+ best_tier = min(sub_tiers, key=_tier_rank)
1701
+ tier_pts = _tier_points(best_tier) * 0.7 # Discount vs explicit overall
1702
+ cap_score += tier_pts * (weight / total_cap_weight)
1703
+
1704
+ cap_score_scaled = cap_score * 2.0 # Scale to max ~20 points
1705
+
1706
+ # --- Cost efficiency scoring (up to profile's cost_weight * 100 points) ---
1707
+ cost_weight = profile.get("cost_weight", 0.10)
1708
+ cost_score = 0.0
1709
+ if model.cost_input is not None:
1710
+ if model.cost_input == 0:
1711
+ cost_score = 10.0 # Free is the best
1712
+ else:
1713
+ # Log scale with floor at $0.10 to keep free always on top.
1714
+ # $0.10 -> ~9pts, $1 -> ~7pts, $5 -> ~5.6pts, $15 -> ~4.6pts
1715
+ clamped = max(model.cost_input, 0.10)
1716
+ cost_score = max(0.0, min(9.5, 9.0 + 2.0 * math.log10(0.10 / clamped)))
1717
+ cost_score_scaled = cost_score * cost_weight * 10 # Up to ~10 points
1718
+
1719
+ # --- Context window scoring (log-scaled, up to profile weight) ---
1720
+ ctx_weight = profile.get("context_weight", 0.10)
1721
+ ctx_score = 0.0
1722
+ if model.context_window and model.context_window > 0:
1723
+ # log10(4K)=3.6, log10(32K)=4.5, log10(128K)=5.1, log10(1M)=6.0, log10(10M)=7.0
1724
+ ctx_score = min(10.0, max(0, (math.log10(model.context_window) - 3.5) * 3.0))
1725
+ ctx_score_scaled = ctx_score * ctx_weight * 10 # Up to ~10 points
1726
+
1727
+ # --- Type match bonus (up to 15 points) ---
1728
+ # Check primary type AND subtypes against preferred types
1729
+ preferred = profile.get("preferred_types", [])
1730
+ type_bonus = 0.0
1731
+ if model.model_type:
1732
+ # Collect all types this model claims (primary + subtypes)
1733
+ all_types = [model.model_type] + list(model.model_subtypes)
1734
+ best_idx = None
1735
+ for t in all_types:
1736
+ if t in preferred:
1737
+ idx = preferred.index(t)
1738
+ if best_idx is None or idx < best_idx:
1739
+ best_idx = idx
1740
+ if best_idx is not None:
1741
+ # First preferred type gets full bonus, descending
1742
+ type_bonus = max(5.0, 15.0 - best_idx * 3.0)
1743
+
1744
+ # --- Speed bonus (up to 10 points, only when hardware specified) ---
1745
+ speed_score = 0.0
1746
+ if model.estimated_tps is not None:
1747
+ # Log-scale: 1 tok/s = 0, 10 tok/s = 5, 100 tok/s = 10
1748
+ speed_score = min(10.0, max(0.0, math.log10(max(1.0, model.estimated_tps)) * 5.0))
1749
+
1750
+ # --- Composite score (0-100 scale) ---
1751
+ raw_total = (
1752
+ bench_score_scaled
1753
+ + cap_score_scaled
1754
+ + cost_score_scaled
1755
+ + ctx_score_scaled
1756
+ + type_bonus
1757
+ + speed_score
1758
+ )
1759
+ # Clamp to 0-100
1760
+ final_score = max(0.0, min(100.0, raw_total))
1761
+ upper_score = max(0.0, min(100.0, raw_total
1762
+ + evidence["benchmark_upper_bound"] - bench_score_scaled))
1763
+
1764
+ return ScoredModel(
1765
+ model_id=model.model_id,
1766
+ display_name=model.display_name,
1767
+ model_type=model.model_type,
1768
+ score=round(final_score, 2) if evidence["rank_status"] == "ranked" else None,
1769
+ score_lower_bound=round(final_score, 2),
1770
+ score_upper_bound=round(upper_score, 2),
1771
+ benchmark_score=round(bench_score_scaled, 2),
1772
+ capability_score=round(cap_score_scaled, 2),
1773
+ cost_score=round(cost_score_scaled, 2),
1774
+ context_score=round(ctx_score_scaled, 2),
1775
+ type_bonus=round(type_bonus, 2),
1776
+ speed_score=round(speed_score, 2),
1777
+ estimated_tps=model.estimated_tps,
1778
+ concurrent_instances=model.concurrent_instances,
1779
+ **evidence,
1780
+ arena_elo_overall=model.arena_elo_overall,
1781
+ total_parameters=model.total_parameters,
1782
+ context_window=model.context_window,
1783
+ cost_input=model.cost_input,
1784
+ cost_output=model.cost_output,
1785
+ open_weights=model.open_weights,
1786
+ provider=model.provider,
1787
+ status=model.status,
1788
+ )
1789
+
1790
+ # ─── Stage 4: Explain ────────────────────────────────────
1791
+
1792
+ def _explain(
1793
+ self,
1794
+ scored: list[ScoredModel],
1795
+ profile: dict[str, Any],
1796
+ use_case: str | None,
1797
+ ) -> None:
1798
+ """Generate human-readable reasons for each model's ranking. Mutates in place."""
1799
+ if not scored:
1800
+ return
1801
+
1802
+ for sm in scored:
1803
+ reasons = [
1804
+ f"Benchmark coverage: {sm.benchmark_coverage:.0%}; "
1805
+ f"composite bounds {sm.score_lower_bound:.2f}–{sm.score_upper_bound:.2f} "
1806
+ "(missing benchmarks, not statistical confidence)."
1807
+ ]
1808
+ if sm.rank_status == "unranked":
1809
+ sm.reasons = ["Unranked for insufficient benchmark evidence; not ranked low.", *reasons]
1810
+ continue
1811
+ reasons.append("Ordered by conservative lower bound, not an estimate of ability.")
1812
+
1813
+ # Benchmark highlights
1814
+ if sm.benchmark_contributions:
1815
+ # Find top contributing benchmarks
1816
+ sorted_benches = sorted(
1817
+ sm.benchmark_contributions.items(),
1818
+ key=lambda x: x[1],
1819
+ reverse=True,
1820
+ )
1821
+ for bench_id, contrib in sorted_benches[:3]:
1822
+ if contrib > 0.5:
1823
+ nice_name = bench_id.replace("_", " ").title()
1824
+ reasons.append(f"Strong {nice_name} performance")
1825
+
1826
+ # Type match
1827
+ if sm.type_bonus >= 12:
1828
+ reasons.append(f"Ideal model type for {use_case or 'general'}")
1829
+ elif sm.type_bonus >= 5:
1830
+ reasons.append(f"Good model type match")
1831
+
1832
+ # Capability highlights
1833
+ if sm.capability_score >= 15:
1834
+ reasons.append("Top-tier capabilities")
1835
+ elif sm.capability_score >= 8:
1836
+ reasons.append("Strong capabilities")
1837
+
1838
+ # Cost efficiency
1839
+ if sm.cost_score >= 8:
1840
+ if sm.cost_input == 0:
1841
+ reasons.append("Free tier available")
1842
+ else:
1843
+ reasons.append("Excellent cost efficiency")
1844
+ elif sm.cost_score >= 4:
1845
+ reasons.append("Good cost/quality ratio")
1846
+
1847
+ # Context window
1848
+ if sm.context_window:
1849
+ if sm.context_window >= 1_000_000:
1850
+ reasons.append(f"{sm.context_window // 1_000_000}M token context")
1851
+ elif sm.context_window >= 128_000:
1852
+ reasons.append(f"{sm.context_window // 1000}K token context")
1853
+
1854
+ # Arena ELO
1855
+ if sm.arena_elo_overall and sm.arena_elo_overall >= 1300:
1856
+ reasons.append(f"Arena ELO: {int(sm.arena_elo_overall)}")
1857
+
1858
+ # Speed estimate
1859
+ if sm.estimated_tps is not None:
1860
+ if sm.estimated_tps >= 50:
1861
+ reasons.append(f"~{sm.estimated_tps:.0f} tok/s (fast)")
1862
+ elif sm.estimated_tps >= 10:
1863
+ reasons.append(f"~{sm.estimated_tps:.0f} tok/s")
1864
+ elif sm.estimated_tps >= 1:
1865
+ reasons.append(f"~{sm.estimated_tps:.1f} tok/s (slow)")
1866
+
1867
+ # Open weights
1868
+ if sm.open_weights:
1869
+ reasons.append("Open weights")
1870
+
1871
+ # If no reasons generated, add a generic one
1872
+ if not reasons:
1873
+ if sm.score > 0:
1874
+ reasons.append("Matches basic criteria")
1875
+ else:
1876
+ reasons.append("Limited data available")
1877
+
1878
+ sm.reasons = reasons
1879
+
1880
+
1881
+ # ═══════════════════════════════════════════════════════════════
1882
+ # Helper functions
1883
+ # ═══════════════════════════════════════════════════════════════
1884
+
1885
+ def _normalize_benchmark(bench_id: str, raw_value: float) -> float:
1886
+ """Normalize a benchmark score to 0-100, higher always meaning better.
1887
+
1888
+ Direction is explicit rather than assumed. A lower-is-better metric such as
1889
+ word error rate is inverted here, so that the rest of the pipeline can treat
1890
+ every normalised score the same way.
1891
+ """
1892
+ range_info = BENCHMARK_RANGES.get(bench_id)
1893
+ if range_info is None:
1894
+ # Unknown benchmark: assume a 0-100 scale. This is a guess, and it is
1895
+ # wrong for any lower-is-better metric — add the benchmark to
1896
+ # BENCHMARK_RANGES and BENCHMARK_DIRECTIONS rather than relying on it.
1897
+ return max(0.0, min(100.0, raw_value))
1898
+
1899
+ low, high = range_info
1900
+ if high <= low:
1901
+ return 50.0 # Degenerate range
1902
+ normalized = ((raw_value - low) / (high - low)) * 100.0
1903
+ if BENCHMARK_DIRECTIONS.get(bench_id) == "lower_is_better":
1904
+ normalized = 100.0 - normalized
1905
+ return max(0.0, min(100.0, normalized))
1906
+
1907
+
1908
+ def _tier_points(tier: str) -> float:
1909
+ """Convert a tier string to point value."""
1910
+ return {
1911
+ "tier-1": 10.0,
1912
+ "tier-2": 6.0,
1913
+ "tier-3": 3.0,
1914
+ "n/a": 0.0,
1915
+ }.get(tier, 0.0)
1916
+
1917
+
1918
+ def _tier_rank(tier: str) -> int:
1919
+ """Numeric rank for tier comparison (lower = better)."""
1920
+ return {
1921
+ "tier-1": 1,
1922
+ "tier-2": 2,
1923
+ "tier-3": 3,
1924
+ "n/a": 99,
1925
+ }.get(tier, 50)
1926
+
1927
+
1928
+ def _safe_int(val) -> int | None:
1929
+ if val is None:
1930
+ return None
1931
+ try:
1932
+ return int(val)
1933
+ except (ValueError, TypeError):
1934
+ return None
1935
+
1936
+
1937
+ def _safe_float(val) -> float | None:
1938
+ if val is None:
1939
+ return None
1940
+ try:
1941
+ return float(val)
1942
+ except (ValueError, TypeError):
1943
+ return None