llm-relay 0.32.0 → 0.32.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "llm-relay",
3
- "version": "0.32.0",
3
+ "version": "0.32.2",
4
4
  "description": "Loopback bidirectional Anthropic/OpenAI API proxy with tool-call validation and multi-provider routing.",
5
5
  "type": "module",
6
6
  "engines": {
@@ -22,6 +22,7 @@
22
22
  // Usage: node scripts/sync-tiers.mjs
23
23
  import { writeFileSync, readFileSync, existsSync, mkdirSync } from "node:fs";
24
24
  import { dirname, join } from "node:path";
25
+ import { homedir } from "node:os";
25
26
  import { fileURLToPath } from "node:url";
26
27
  import {
27
28
  CAPABILITY_DIMENSIONS,
@@ -39,18 +40,63 @@ const ARENA_PARQUET =
39
40
  "https://huggingface.co/datasets/lmarena-ai/leaderboard-dataset/resolve/main/text_style_control/latest-00000-of-00001.parquet";
40
41
  const AIDER_YML =
41
42
  "https://raw.githubusercontent.com/Aider-AI/aider/main/aider/website/_data/polyglot_leaderboard.yml";
43
+ const AA_MODELS = "https://artificialanalysis.ai/api/v2/data/llms/models";
44
+ const AA_KEY_ENV = "ARTIFICIALANALYSIS_API_KEY";
42
45
 
43
46
  // Columns we depend on — a rename here should FAIL the source, not silently drop data.
44
47
  const BFCL_REQUIRED = ["Model", "Overall Acc"];
45
48
  const BFCL_WANTED = ["Model", "Overall Acc", "Multi Turn Acc", "Irrelevance Detection"];
46
49
 
47
- /** Strip BFCL mode suffixes ("(FC)", "(Prompt)", "(FC thinking)") + lowercase for joining. */
50
+ /**
51
+ * Reasoning-effort qualifiers, as a CLOSED vocabulary — same reasoning as the per-provider alias
52
+ * list in src/authEnv.ts: a heuristic "does this trailing word look like an effort level" would
53
+ * eventually decide that the `-max` in `glm-5.2-max` (a SKU tier) is an effort setting and collapse
54
+ * two different models into one.
55
+ */
56
+ const EFFORT_TOKENS = new Set([
57
+ "none", "minimal", "low", "medium", "high", "xhigh", "max",
58
+ "thinking", "reasoning", "no thinking", "non-thinking",
59
+ ]);
60
+ /** Aider spells out thinking budgets: "(32k thinking)", "(32k thinking tokens)". */
61
+ const THINKING_BUDGET = /^\d+k thinking(?: tokens)?$/;
62
+
63
+ function canonicalEffort(inner) {
64
+ const s = inner.trim().toLowerCase().replace(/\s+/g, " ");
65
+ if (EFFORT_TOKENS.has(s)) return s.replace(/ /g, "-");
66
+ if (THINKING_BUDGET.test(s)) return s.replace(/ tokens$/, "").replace(/\s+/g, "-");
67
+ return null;
68
+ }
69
+
70
+ /**
71
+ * Strip BFCL mode suffixes ("(FC)", "(Prompt)", "(FC thinking)"), lowercase, and CANONICALIZE a
72
+ * trailing reasoning-effort qualifier so the sources' three notations produce ONE join key:
73
+ *
74
+ * aider "gpt-5 (high)" ─┐
75
+ * lmarena "gpt-5-high" ─┼─> "gpt-5-high"
76
+ * openrouter "openai/gpt-5-high" (via normId) ─┘
77
+ *
78
+ * Measured 2026-08-08 before this existed: 0 of 60 effort-qualified rows in the snapshot carried
79
+ * more than one source. Every one was a single-signal orphan, and none joined to the OpenRouter row
80
+ * holding that model's AA scores, context window and price — so multi-source consensus was being
81
+ * shattered into guesses, and `signal_count: 1` on every effort variant looked like thin publishing
82
+ * when it was a failed join.
83
+ *
84
+ * ⚠ This REWRITES notation, it never STRIPS. "gpt-5 (high)" becomes "gpt-5-high", never "gpt-5" —
85
+ * an effort variant must stay a separate row from its base model, which is exactly the borrowed-
86
+ * score bug the merge comment in main() warns about. A qualifier outside the vocabulary (a date,
87
+ * "(prev)") is left alone rather than guessed at.
88
+ */
48
89
  function normName(raw) {
49
- return String(raw)
90
+ const base = String(raw)
50
91
  .replace(/\((?:FC|Prompt)[^)]*\)/gi, "")
51
92
  .replace(/\s+/g, " ")
52
93
  .trim()
53
94
  .toLowerCase();
95
+ // Trailing parenthetical only: "o3 (high) + gpt-4.1" is a composite, not a variant of "o3".
96
+ const m = /^(.*?)\s*\(([^()]+)\)$/.exec(base);
97
+ if (!m) return base;
98
+ const effort = canonicalEffort(m[2]);
99
+ return effort ? `${m[1].trim()}-${effort}` : base;
54
100
  }
55
101
 
56
102
  /**
@@ -225,6 +271,112 @@ async function fetchAider() {
225
271
  return [...best.values()].map(({ dirname, ...rest }) => ({ ...rest, aider_run: dirname }));
226
272
  }
227
273
 
274
+ /**
275
+ * Read a key out of ~/.llm-relay/.env without overwriting the real environment — same contract as
276
+ * src/dotenv.ts, because the more explicit signal must win. This script is the only one that needs
277
+ * a credential, so it reads the file directly rather than growing a dependency for one lookup.
278
+ */
279
+ function keyFromRelayEnv(name) {
280
+ if (process.env[name]) return process.env[name];
281
+ const file = join(homedir(), ".llm-relay", ".env");
282
+ if (!existsSync(file)) return null;
283
+ for (const line of readFileSync(file, "utf8").split("\n")) {
284
+ const kv = /^\s*(?:export\s+)?([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.*)$/.exec(line);
285
+ if (kv && kv[1] === name) return kv[2].trim().replace(/^["']|["']$/g, "") || null;
286
+ }
287
+ return null;
288
+ }
289
+
290
+ /**
291
+ * Artificial Analysis, FIRST-HAND. Its intelligence/coding/agentic indices already reach the
292
+ * snapshot second-hand through OpenRouter's per-model `benchmarks.artificial_analysis` block — but
293
+ * that block is keyed by OpenRouter's catalogue ids, which carry no reasoning-effort dimension
294
+ * (measured 2026-08-08: 154 of 400 OpenRouter models carry an AA block; 6 of those encode anything
295
+ * effort-like in the id, and only `openai/o3-mini-high` is genuinely an effort variant rather than
296
+ * a SKU name). AA publishes per-effort rows directly, so this is the source that can actually
297
+ * answer "is luna at xhigh stronger than terra at low".
298
+ *
299
+ * ⚠ Key-gated: the endpoint 401s without one, and there is no public mirror. Absent key ⇒ the
300
+ * source reports `configured: false` and contributes nothing, exactly like a dead endpoint — this
301
+ * must never be fatal, because `npm run sync:tiers` has to keep working for anyone who has not
302
+ * signed up. Get a free key at https://artificialanalysis.ai/ and put it in ~/.llm-relay/.env as
303
+ * ARTIFICIALANALYSIS_API_KEY.
304
+ *
305
+ * ⚠ The response schema is NOT publicly documented (the docs URL 404s), so the field mapping below
306
+ * is an ALIAS LIST in the style of `limitsFromRecord()` in src/catalog.ts rather than a hardcoded
307
+ * per-field path — and it THROWS if a payload arrives in which no model resolves a single score.
308
+ * That is the established contract here: schema drift inside a source is corruption and fails that
309
+ * source loudly, it does not silently drop data. If it throws on your first keyed run, the fix is
310
+ * to extend the alias lists, not to soften the check.
311
+ */
312
+ const AA_ALIASES = {
313
+ aa_intelligence: ["intelligence_index", "artificial_analysis_intelligence_index", "intelligence"],
314
+ aa_coding: ["coding_index", "artificial_analysis_coding_index", "coding"],
315
+ aa_agentic: ["agentic_index", "artificial_analysis_agentic_index", "agentic"],
316
+ };
317
+ const AA_NAME_FIELDS = ["name", "model_name", "slug", "id"];
318
+ const AA_EFFORT_FIELDS = ["reasoning_effort", "effort", "variant", "reasoning_mode", "thinking"];
319
+ const AA_SCORE_CONTAINERS = ["evaluations", "benchmarks", "scores", "metrics"];
320
+
321
+ function aaPick(record, fields) {
322
+ for (const f of fields) {
323
+ const v = record?.[f];
324
+ if (typeof v === "string" && v.trim()) return v.trim();
325
+ }
326
+ return null;
327
+ }
328
+
329
+ function aaScore(record, aliases) {
330
+ const pools = [record, ...AA_SCORE_CONTAINERS.map((c) => record?.[c]).filter((c) => c && typeof c === "object")];
331
+ for (const pool of pools) {
332
+ for (const a of aliases) {
333
+ const v = pool[a];
334
+ if (Number.isFinite(v)) return Number(v);
335
+ if (typeof v === "string" && v.trim() !== "" && Number.isFinite(Number(v))) return Number(v);
336
+ }
337
+ }
338
+ return null;
339
+ }
340
+
341
+ async function fetchArtificialAnalysis() {
342
+ const key = keyFromRelayEnv(AA_KEY_ENV);
343
+ if (!key) return { rows: [], configured: false };
344
+
345
+ const res = await fetch(AA_MODELS, { headers: { "x-api-key": key } });
346
+ if (!res.ok) throw new Error(`Artificial Analysis fetch HTTP ${res.status}`);
347
+ const j = await res.json();
348
+ const list = Array.isArray(j) ? j : Array.isArray(j.data) ? j.data : null;
349
+ if (!list) throw new Error(`AA schema drift — no array payload. Top-level keys: ${Object.keys(j).join(" | ")}`);
350
+ if (list.length === 0) throw new Error("AA returned zero models");
351
+
352
+ const out = [];
353
+ for (const m of list) {
354
+ const rawName = aaPick(m, AA_NAME_FIELDS);
355
+ if (!rawName) continue;
356
+ // AA may express effort as its own field rather than baked into the name. Fold it into the
357
+ // name BEFORE normalizing so it lands on the same canonical key as the other sources.
358
+ const effort = canonicalEffort(aaPick(m, AA_EFFORT_FIELDS) ?? "");
359
+ const name = effort ? `${rawName} (${effort})` : rawName;
360
+ const row = { name, norm: normName(name) };
361
+ for (const [field, aliases] of Object.entries(AA_ALIASES)) {
362
+ const v = aaScore(m, aliases);
363
+ if (v != null) row[field] = v;
364
+ }
365
+ out.push(row);
366
+ }
367
+ if (!out.some((r) => r.aa_intelligence != null || r.aa_coding != null || r.aa_agentic != null)) {
368
+ const keys = Object.keys(list[0] ?? {}).join(" | ");
369
+ throw new Error(`AA schema drift — ${list.length} models but no recognized score field. Keys: ${keys}`);
370
+ }
371
+ // One row per (model, effort); keep the strongest run if AA repeats a key.
372
+ const best = new Map();
373
+ for (const m of out) {
374
+ const prev = best.get(m.norm);
375
+ if (!prev || (m.aa_intelligence ?? -1) > (prev.aa_intelligence ?? -1)) best.set(m.norm, m);
376
+ }
377
+ return { rows: [...best.values()], configured: true };
378
+ }
379
+
228
380
  async function main() {
229
381
  const warnings = [];
230
382
  const sources = {};
@@ -241,15 +393,31 @@ async function main() {
241
393
  }
242
394
  };
243
395
 
396
+ // A key-gated source distinguishes "not configured" from "failed": an operator who never signed
397
+ // up should see a neutral note, not a warning that reads like an outage.
398
+ const runOptional = async (key, url, note, fn) => {
399
+ try {
400
+ const { rows, configured } = await fn();
401
+ sources[key] = { url, note, model_count: rows.length, ok: true, configured };
402
+ if (!configured) sources[key].skipped = `no ${AA_KEY_ENV} configured — source contributed nothing`;
403
+ return rows;
404
+ } catch (e) {
405
+ warnings.push(`${key} sync FAILED: ${e.message}`);
406
+ sources[key] = { url, note, model_count: 0, ok: false, configured: true, error: e.message };
407
+ return [];
408
+ }
409
+ };
410
+
244
411
  // Independent: one dead source costs only its own columns.
245
- const [openrouter, bfcl, arena, aider] = await Promise.all([
412
+ const [openrouter, bfcl, arena, aider, aa] = await Promise.all([
246
413
  run("openrouter", OPENROUTER_MODELS, "AA intelligence/coding/agentic + design arena + context/pricing; EXACT ids", fetchOpenRouter),
247
414
  run("bfcl", BFCL_CSV, "tool-use / function-calling accuracy", fetchBfcl),
248
415
  run("lmarena", ARENA_PARQUET, "general capability", fetchArena),
249
416
  run("aider", AIDER_YML, "polyglot edit benchmark + edit-format compliance", fetchAider),
417
+ runOptional("artificialanalysis", AA_MODELS, "AA intelligence/coding/agentic FIRST-HAND, per reasoning effort", fetchArtificialAnalysis),
250
418
  ]);
251
419
 
252
- if (openrouter.length + bfcl.length + arena.length + aider.length === 0) {
420
+ if (openrouter.length + bfcl.length + arena.length + aider.length + aa.length === 0) {
253
421
  throw new Error(`every source failed:\n ${warnings.join("\n ")}`);
254
422
  }
255
423
 
@@ -270,6 +438,9 @@ async function main() {
270
438
  absorb(bfcl, "bfcl");
271
439
  absorb(arena, "lmarena");
272
440
  absorb(aider, "aider");
441
+ // AA last of the AA-bearing sources on purpose: `absorb` is last-write-wins, and a figure read
442
+ // from AA directly outranks the same figure relayed through OpenRouter's catalogue metadata.
443
+ absorb(aa, "artificialanalysis");
273
444
  const models = [...byNorm.values()];
274
445
 
275
446
  const prev = existsSync(OUT) ? JSON.parse(readFileSync(OUT, "utf8")) : null;
@@ -310,9 +481,20 @@ async function main() {
310
481
 
311
482
  console.log(`\nSynced ${models.length} models → ${OUT}`);
312
483
  for (const [k, s] of Object.entries(sources)) {
313
- console.log(` ${s.ok ? "✓" : "✗"} ${k.padEnd(11)} ${String(s.model_count).padStart(4)} models${s.ok ? "" : ` — ${s.error}`}`);
484
+ const mark = s.configured === false ? "-" : s.ok ? "✓" : "✗";
485
+ const tail = s.configured === false ? ` — ${s.skipped}` : s.ok ? "" : ` — ${s.error}`;
486
+ console.log(` ${mark} ${k.padEnd(19)} ${String(s.model_count).padStart(4)} models${tail}`);
314
487
  }
315
488
  if (warnings.length) console.log(warnings.map((w) => ` ⚠ ${w}`).join("\n"));
489
+
490
+ // The effort-notation join is the thing most likely to regress silently, so it is REPORTED:
491
+ // a drop back toward zero multi-source means a source changed notation again.
492
+ const effort = models.filter((m) => canonicalEffort(String(m.norm).split("-").pop() ?? "") != null);
493
+ const joined = effort.filter((m) => m.sources.length > 1).length;
494
+ console.log(
495
+ `\nEffort-qualified rows: ${effort.length}, of which ${joined} carry >1 source ` +
496
+ `(was 0 of 60 before notation was canonicalized).`,
497
+ );
316
498
  if (prev) console.log(` (previous snapshot: ${prev.synced_at}, ${prev.models?.length ?? "?"} models)`);
317
499
 
318
500
  const multi = models.filter((m) => m.published_signal_count >= 3);