llm-relay 0.32.0 → 0.32.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/dispatch.js +20 -1
- package/dist/dispatch.js.map +1 -1
- package/dist/dynamic-pools.d.ts +1 -0
- package/dist/dynamic-pools.js +7 -1
- package/dist/dynamic-pools.js.map +1 -1
- package/docs/tier-data.json +17149 -14110
- package/package.json +1 -1
- package/scripts/sync-tiers.mjs +187 -5
package/package.json
CHANGED
package/scripts/sync-tiers.mjs
CHANGED
|
@@ -22,6 +22,7 @@
|
|
|
22
22
|
// Usage: node scripts/sync-tiers.mjs
|
|
23
23
|
import { writeFileSync, readFileSync, existsSync, mkdirSync } from "node:fs";
|
|
24
24
|
import { dirname, join } from "node:path";
|
|
25
|
+
import { homedir } from "node:os";
|
|
25
26
|
import { fileURLToPath } from "node:url";
|
|
26
27
|
import {
|
|
27
28
|
CAPABILITY_DIMENSIONS,
|
|
@@ -39,18 +40,63 @@ const ARENA_PARQUET =
|
|
|
39
40
|
"https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset/resolve/main/text_style_control/latest-00000-of-00001.parquet";
|
|
40
41
|
const AIDER_YML =
|
|
41
42
|
"https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml";
|
|
43
|
+
const AA_MODELS = "https://artificialanalysis.ai/api/v2/data/llms/models";
|
|
44
|
+
const AA_KEY_ENV = "ARTIFICIALANALYSIS_API_KEY";
|
|
42
45
|
|
|
43
46
|
// Columns we depend on — a rename here should FAIL the source, not silently drop data.
|
|
44
47
|
const BFCL_REQUIRED = ["Model", "Overall Acc"];
|
|
45
48
|
const BFCL_WANTED = ["Model", "Overall Acc", "Multi Turn Acc", "Irrelevance Detection"];
|
|
46
49
|
|
|
47
|
-
/**
|
|
50
|
+
/**
|
|
51
|
+
* Reasoning-effort qualifiers, as a CLOSED vocabulary — same reasoning as the per-provider alias
|
|
52
|
+
* list in src/authEnv.ts: a heuristic "does this trailing word look like an effort level" would
|
|
53
|
+
* eventually decide that the `-max` in `glm-5.2-max` (a SKU tier) is an effort setting and collapse
|
|
54
|
+
* two different models into one.
|
|
55
|
+
*/
|
|
56
|
+
const EFFORT_TOKENS = new Set([
|
|
57
|
+
"none", "minimal", "low", "medium", "high", "xhigh", "max",
|
|
58
|
+
"thinking", "reasoning", "no thinking", "non-thinking",
|
|
59
|
+
]);
|
|
60
|
+
/** Aider spells out thinking budgets: "(32k thinking)", "(32k thinking tokens)". */
|
|
61
|
+
const THINKING_BUDGET = /^\d+k thinking(?: tokens)?$/;
|
|
62
|
+
|
|
63
|
+
function canonicalEffort(inner) {
|
|
64
|
+
const s = inner.trim().toLowerCase().replace(/\s+/g, " ");
|
|
65
|
+
if (EFFORT_TOKENS.has(s)) return s.replace(/ /g, "-");
|
|
66
|
+
if (THINKING_BUDGET.test(s)) return s.replace(/ tokens$/, "").replace(/\s+/g, "-");
|
|
67
|
+
return null;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Strip BFCL mode suffixes ("(FC)", "(Prompt)", "(FC thinking)"), lowercase, and CANONICALIZE a
|
|
72
|
+
* trailing reasoning-effort qualifier so the sources' three notations produce ONE join key:
|
|
73
|
+
*
|
|
74
|
+
* aider "gpt-5 (high)" ─┐
|
|
75
|
+
* lmarena "gpt-5-high" ─┼─> "gpt-5-high"
|
|
76
|
+
* openrouter "openai/gpt-5-high" (via normId) ─┘
|
|
77
|
+
*
|
|
78
|
+
* Measured 2026-08-08 before this existed: 0 of 60 effort-qualified rows in the snapshot carried
|
|
79
|
+
* more than one source. Every one was a single-signal orphan, and none joined to the OpenRouter row
|
|
80
|
+
* holding that model's AA scores, context window and price — so multi-source consensus was being
|
|
81
|
+
* shattered into guesses, and `signal_count: 1` on every effort variant looked like thin publishing
|
|
82
|
+
* when it was a failed join.
|
|
83
|
+
*
|
|
84
|
+
* ⚠ This REWRITES notation, it never STRIPS. "gpt-5 (high)" becomes "gpt-5-high", never "gpt-5" —
|
|
85
|
+
* an effort variant must stay a separate row from its base model, which is exactly the borrowed-
|
|
86
|
+
* score bug the merge comment in main() warns about. A qualifier outside the vocabulary (a date,
|
|
87
|
+
* "(prev)") is left alone rather than guessed at.
|
|
88
|
+
*/
|
|
48
89
|
function normName(raw) {
|
|
49
|
-
|
|
90
|
+
const base = String(raw)
|
|
50
91
|
.replace(/\((?:FC|Prompt)[^)]*\)/gi, "")
|
|
51
92
|
.replace(/\s+/g, " ")
|
|
52
93
|
.trim()
|
|
53
94
|
.toLowerCase();
|
|
95
|
+
// Trailing parenthetical only: "o3 (high) + gpt-4.1" is a composite, not a variant of "o3".
|
|
96
|
+
const m = /^(.*?)\s*\(([^()]+)\)$/.exec(base);
|
|
97
|
+
if (!m) return base;
|
|
98
|
+
const effort = canonicalEffort(m[2]);
|
|
99
|
+
return effort ? `${m[1].trim()}-${effort}` : base;
|
|
54
100
|
}
|
|
55
101
|
|
|
56
102
|
/**
|
|
@@ -225,6 +271,112 @@ async function fetchAider() {
|
|
|
225
271
|
return [...best.values()].map(({ dirname, ...rest }) => ({ ...rest, aider_run: dirname }));
|
|
226
272
|
}
|
|
227
273
|
|
|
274
|
+
/**
|
|
275
|
+
* Read a key out of ~/.llm-relay/.env without overwriting the real environment — same contract as
|
|
276
|
+
* src/dotenv.ts, because the more explicit signal must win. This script is the only one that needs
|
|
277
|
+
* a credential, so it reads the file directly rather than growing a dependency for one lookup.
|
|
278
|
+
*/
|
|
279
|
+
function keyFromRelayEnv(name) {
|
|
280
|
+
if (process.env[name]) return process.env[name];
|
|
281
|
+
const file = join(homedir(), ".llm-relay", ".env");
|
|
282
|
+
if (!existsSync(file)) return null;
|
|
283
|
+
for (const line of readFileSync(file, "utf8").split("\n")) {
|
|
284
|
+
const kv = /^\s*(?:export\s+)?([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.*)$/.exec(line);
|
|
285
|
+
if (kv && kv[1] === name) return kv[2].trim().replace(/^["']|["']$/g, "") || null;
|
|
286
|
+
}
|
|
287
|
+
return null;
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
/**
|
|
291
|
+
* Artificial Analysis, FIRST-HAND. Its intelligence/coding/agentic indices already reach the
|
|
292
|
+
* snapshot second-hand through OpenRouter's per-model `benchmarks.artificial_analysis` block — but
|
|
293
|
+
* that block is keyed by OpenRouter's catalogue ids, which carry no reasoning-effort dimension
|
|
294
|
+
* (measured 2026-08-08: 154 of 400 OpenRouter models carry an AA block; 6 of those encode anything
|
|
295
|
+
* effort-like in the id, and only `openai/o3-mini-high` is genuinely an effort variant rather than
|
|
296
|
+
* a SKU name). AA publishes per-effort rows directly, so this is the source that can actually
|
|
297
|
+
* answer "is luna at xhigh stronger than terra at low".
|
|
298
|
+
*
|
|
299
|
+
* ⚠ Key-gated: the endpoint 401s without one, and there is no public mirror. Absent key ⇒ the
|
|
300
|
+
* source reports `configured: false` and contributes nothing, exactly like a dead endpoint — this
|
|
301
|
+
* must never be fatal, because `npm run sync:tiers` has to keep working for anyone who has not
|
|
302
|
+
* signed up. Get a free key at https://artificialanalysis.ai/ and put it in ~/.llm-relay/.env as
|
|
303
|
+
* ARTIFICIALANALYSIS_API_KEY.
|
|
304
|
+
*
|
|
305
|
+
* ⚠ The response schema is NOT publicly documented (the docs URL 404s), so the field mapping below
|
|
306
|
+
* is an ALIAS LIST in the style of `limitsFromRecord()` in src/catalog.ts rather than a hardcoded
|
|
307
|
+
* per-field path — and it THROWS if a payload arrives in which no model resolves a single score.
|
|
308
|
+
* That is the established contract here: schema drift inside a source is corruption and fails that
|
|
309
|
+
* source loudly, it does not silently drop data. If it throws on your first keyed run, the fix is
|
|
310
|
+
* to extend the alias lists, not to soften the check.
|
|
311
|
+
*/
|
|
312
|
+
const AA_ALIASES = {
|
|
313
|
+
aa_intelligence: ["intelligence_index", "artificial_analysis_intelligence_index", "intelligence"],
|
|
314
|
+
aa_coding: ["coding_index", "artificial_analysis_coding_index", "coding"],
|
|
315
|
+
aa_agentic: ["agentic_index", "artificial_analysis_agentic_index", "agentic"],
|
|
316
|
+
};
|
|
317
|
+
const AA_NAME_FIELDS = ["name", "model_name", "slug", "id"];
|
|
318
|
+
const AA_EFFORT_FIELDS = ["reasoning_effort", "effort", "variant", "reasoning_mode", "thinking"];
|
|
319
|
+
const AA_SCORE_CONTAINERS = ["evaluations", "benchmarks", "scores", "metrics"];
|
|
320
|
+
|
|
321
|
+
function aaPick(record, fields) {
|
|
322
|
+
for (const f of fields) {
|
|
323
|
+
const v = record?.[f];
|
|
324
|
+
if (typeof v === "string" && v.trim()) return v.trim();
|
|
325
|
+
}
|
|
326
|
+
return null;
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
function aaScore(record, aliases) {
|
|
330
|
+
const pools = [record, ...AA_SCORE_CONTAINERS.map((c) => record?.[c]).filter((c) => c && typeof c === "object")];
|
|
331
|
+
for (const pool of pools) {
|
|
332
|
+
for (const a of aliases) {
|
|
333
|
+
const v = pool[a];
|
|
334
|
+
if (Number.isFinite(v)) return Number(v);
|
|
335
|
+
if (typeof v === "string" && v.trim() !== "" && Number.isFinite(Number(v))) return Number(v);
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
return null;
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
async function fetchArtificialAnalysis() {
|
|
342
|
+
const key = keyFromRelayEnv(AA_KEY_ENV);
|
|
343
|
+
if (!key) return { rows: [], configured: false };
|
|
344
|
+
|
|
345
|
+
const res = await fetch(AA_MODELS, { headers: { "x-api-key": key } });
|
|
346
|
+
if (!res.ok) throw new Error(`Artificial Analysis fetch HTTP ${res.status}`);
|
|
347
|
+
const j = await res.json();
|
|
348
|
+
const list = Array.isArray(j) ? j : Array.isArray(j.data) ? j.data : null;
|
|
349
|
+
if (!list) throw new Error(`AA schema drift — no array payload. Top-level keys: ${Object.keys(j).join(" | ")}`);
|
|
350
|
+
if (list.length === 0) throw new Error("AA returned zero models");
|
|
351
|
+
|
|
352
|
+
const out = [];
|
|
353
|
+
for (const m of list) {
|
|
354
|
+
const rawName = aaPick(m, AA_NAME_FIELDS);
|
|
355
|
+
if (!rawName) continue;
|
|
356
|
+
// AA may express effort as its own field rather than baked into the name. Fold it into the
|
|
357
|
+
// name BEFORE normalizing so it lands on the same canonical key as the other sources.
|
|
358
|
+
const effort = canonicalEffort(aaPick(m, AA_EFFORT_FIELDS) ?? "");
|
|
359
|
+
const name = effort ? `${rawName} (${effort})` : rawName;
|
|
360
|
+
const row = { name, norm: normName(name) };
|
|
361
|
+
for (const [field, aliases] of Object.entries(AA_ALIASES)) {
|
|
362
|
+
const v = aaScore(m, aliases);
|
|
363
|
+
if (v != null) row[field] = v;
|
|
364
|
+
}
|
|
365
|
+
out.push(row);
|
|
366
|
+
}
|
|
367
|
+
if (!out.some((r) => r.aa_intelligence != null || r.aa_coding != null || r.aa_agentic != null)) {
|
|
368
|
+
const keys = Object.keys(list[0] ?? {}).join(" | ");
|
|
369
|
+
throw new Error(`AA schema drift — ${list.length} models but no recognized score field. Keys: ${keys}`);
|
|
370
|
+
}
|
|
371
|
+
// One row per (model, effort); keep the strongest run if AA repeats a key.
|
|
372
|
+
const best = new Map();
|
|
373
|
+
for (const m of out) {
|
|
374
|
+
const prev = best.get(m.norm);
|
|
375
|
+
if (!prev || (m.aa_intelligence ?? -1) > (prev.aa_intelligence ?? -1)) best.set(m.norm, m);
|
|
376
|
+
}
|
|
377
|
+
return { rows: [...best.values()], configured: true };
|
|
378
|
+
}
|
|
379
|
+
|
|
228
380
|
async function main() {
|
|
229
381
|
const warnings = [];
|
|
230
382
|
const sources = {};
|
|
@@ -241,15 +393,31 @@ async function main() {
|
|
|
241
393
|
}
|
|
242
394
|
};
|
|
243
395
|
|
|
396
|
+
// A key-gated source distinguishes "not configured" from "failed": an operator who never signed
|
|
397
|
+
// up should see a neutral note, not a warning that reads like an outage.
|
|
398
|
+
const runOptional = async (key, url, note, fn) => {
|
|
399
|
+
try {
|
|
400
|
+
const { rows, configured } = await fn();
|
|
401
|
+
sources[key] = { url, note, model_count: rows.length, ok: true, configured };
|
|
402
|
+
if (!configured) sources[key].skipped = `no ${AA_KEY_ENV} configured — source contributed nothing`;
|
|
403
|
+
return rows;
|
|
404
|
+
} catch (e) {
|
|
405
|
+
warnings.push(`${key} sync FAILED: ${e.message}`);
|
|
406
|
+
sources[key] = { url, note, model_count: 0, ok: false, configured: true, error: e.message };
|
|
407
|
+
return [];
|
|
408
|
+
}
|
|
409
|
+
};
|
|
410
|
+
|
|
244
411
|
// Independent: one dead source costs only its own columns.
|
|
245
|
-
const [openrouter, bfcl, arena, aider] = await Promise.all([
|
|
412
|
+
const [openrouter, bfcl, arena, aider, aa] = await Promise.all([
|
|
246
413
|
run("openrouter", OPENROUTER_MODELS, "AA intelligence/coding/agentic + design arena + context/pricing; EXACT ids", fetchOpenRouter),
|
|
247
414
|
run("bfcl", BFCL_CSV, "tool-use / function-calling accuracy", fetchBfcl),
|
|
248
415
|
run("lmarena", ARENA_PARQUET, "general capability", fetchArena),
|
|
249
416
|
run("aider", AIDER_YML, "polyglot edit benchmark + edit-format compliance", fetchAider),
|
|
417
|
+
runOptional("artificialanalysis", AA_MODELS, "AA intelligence/coding/agentic FIRST-HAND, per reasoning effort", fetchArtificialAnalysis),
|
|
250
418
|
]);
|
|
251
419
|
|
|
252
|
-
if (openrouter.length + bfcl.length + arena.length + aider.length === 0) {
|
|
420
|
+
if (openrouter.length + bfcl.length + arena.length + aider.length + aa.length === 0) {
|
|
253
421
|
throw new Error(`every source failed:\n ${warnings.join("\n ")}`);
|
|
254
422
|
}
|
|
255
423
|
|
|
@@ -270,6 +438,9 @@ async function main() {
|
|
|
270
438
|
absorb(bfcl, "bfcl");
|
|
271
439
|
absorb(arena, "lmarena");
|
|
272
440
|
absorb(aider, "aider");
|
|
441
|
+
// AA last of the AA-bearing sources on purpose: `absorb` is last-write-wins, and a figure read
|
|
442
|
+
// from AA directly outranks the same figure relayed through OpenRouter's catalogue metadata.
|
|
443
|
+
absorb(aa, "artificialanalysis");
|
|
273
444
|
const models = [...byNorm.values()];
|
|
274
445
|
|
|
275
446
|
const prev = existsSync(OUT) ? JSON.parse(readFileSync(OUT, "utf8")) : null;
|
|
@@ -310,9 +481,20 @@ async function main() {
|
|
|
310
481
|
|
|
311
482
|
console.log(`\nSynced ${models.length} models → ${OUT}`);
|
|
312
483
|
for (const [k, s] of Object.entries(sources)) {
|
|
313
|
-
|
|
484
|
+
const mark = s.configured === false ? "-" : s.ok ? "✓" : "✗";
|
|
485
|
+
const tail = s.configured === false ? ` — ${s.skipped}` : s.ok ? "" : ` — ${s.error}`;
|
|
486
|
+
console.log(` ${mark} ${k.padEnd(19)} ${String(s.model_count).padStart(4)} models${tail}`);
|
|
314
487
|
}
|
|
315
488
|
if (warnings.length) console.log(warnings.map((w) => ` ⚠ ${w}`).join("\n"));
|
|
489
|
+
|
|
490
|
+
// The effort-notation join is the thing most likely to regress silently, so it is REPORTED:
|
|
491
|
+
// a drop back toward zero multi-source means a source changed notation again.
|
|
492
|
+
const effort = models.filter((m) => canonicalEffort(String(m.norm).split("-").pop() ?? "") != null);
|
|
493
|
+
const joined = effort.filter((m) => m.sources.length > 1).length;
|
|
494
|
+
console.log(
|
|
495
|
+
`\nEffort-qualified rows: ${effort.length}, of which ${joined} carry >1 source ` +
|
|
496
|
+
`(was 0 of 60 before notation was canonicalized).`,
|
|
497
|
+
);
|
|
316
498
|
if (prev) console.log(` (previous snapshot: ${prev.synced_at}, ${prev.models?.length ?? "?"} models)`);
|
|
317
499
|
|
|
318
500
|
const multi = models.filter((m) => m.published_signal_count >= 3);
|