auto-model-router 0.30.3 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/.omp-plugin/marketplace.json +2 -2
  2. package/README.md +32 -2
  3. package/omp-extension/router-configure.ts +9 -7
  4. package/package.json +1 -1
  5. package/src/cli/config-cmd.ts +8 -7
  6. package/src/cli/explain.ts +10 -5
  7. package/src/cli/export.ts +6 -5
  8. package/src/cli/models.ts +10 -7
  9. package/src/cli/report.ts +6 -1
  10. package/src/cli/stats.ts +7 -7
  11. package/src/config/load.ts +10 -1
  12. package/src/config/types.ts +10 -1
  13. package/src/context/bridge.ts +7 -7
  14. package/src/context/index.ts +3 -3
  15. package/src/context/store.ts +39 -56
  16. package/src/context/types.ts +7 -6
  17. package/src/cost/blended.ts +28 -7
  18. package/src/cost/feedback.ts +33 -37
  19. package/src/cost/ledger-sql.ts +547 -0
  20. package/src/cost/ledger.ts +30 -459
  21. package/src/cost/report.ts +171 -129
  22. package/src/cost/retention.ts +10 -10
  23. package/src/cost/summary.ts +15 -10
  24. package/src/cost/types.ts +43 -62
  25. package/src/cost/views.ts +79 -49
  26. package/src/eval/calibrate.ts +47 -12
  27. package/src/eval/run.ts +18 -2
  28. package/src/lib.ts +6 -2
  29. package/src/router/candidates.ts +7 -15
  30. package/src/router/classify.ts +6 -4
  31. package/src/router/index.ts +95 -9
  32. package/src/router/select.ts +38 -21
  33. package/src/router/state.ts +90 -102
  34. package/src/router/types.ts +11 -5
  35. package/src/server/advise.ts +6 -4
  36. package/src/server/compaction-digest.ts +1 -1
  37. package/src/server/digest.ts +9 -10
  38. package/src/server/http.ts +109 -46
  39. package/src/server/providers.ts +18 -4
  40. package/src/server/turn.ts +32 -9
  41. package/src/tokens/estimate.ts +16 -6
  42. package/src/upstream/ollama-usage.ts +21 -11
  43. package/src/util/schema.ts +201 -0
  44. package/src/util/sql.ts +246 -0
  45. package/src/wire/anthropic/messages.ts +3 -4
  46. package/src/wire/openai/request.ts +1 -0
  47. package/src/wire/types.ts +7 -0
  48. package/test/anthropic-wire.test.ts +9 -9
  49. package/test/benchmark-feeds.test.ts +7 -7
  50. package/test/cache-control.test.ts +7 -7
  51. package/test/cache-estimate.test.ts +5 -5
  52. package/test/catalog-view.test.ts +4 -4
  53. package/test/catalog.test.ts +11 -11
  54. package/test/classify.test.ts +24 -24
  55. package/test/compaction.test.ts +20 -20
  56. package/test/config-wizard.test.ts +32 -32
  57. package/test/config.test.ts +10 -10
  58. package/test/connect-harnesses.test.ts +11 -11
  59. package/test/context-bridge.test.ts +40 -30
  60. package/test/context-prune.test.ts +43 -36
  61. package/test/context-query.test.ts +8 -8
  62. package/test/controls.test.ts +54 -27
  63. package/test/cost.test.ts +12 -12
  64. package/test/digest.test.ts +55 -44
  65. package/test/embed-lifecycle.test.ts +5 -5
  66. package/test/embed-logic.test.ts +26 -26
  67. package/test/escalate.test.ts +17 -17
  68. package/test/eval.test.ts +73 -16
  69. package/test/executable.test.ts +6 -6
  70. package/test/exploration.test.ts +19 -20
  71. package/test/failover.test.ts +22 -21
  72. package/test/fakes.ts +105 -0
  73. package/test/features.test.ts +21 -21
  74. package/test/harness-requests.test.ts +3 -3
  75. package/test/harness-switch.test.ts +5 -5
  76. package/test/hold-exploration.test.ts +13 -13
  77. package/test/hot-reload.test.ts +5 -5
  78. package/test/learned.test.ts +5 -5
  79. package/test/ledger-sql.test.ts +342 -0
  80. package/test/mcp-entry.test.ts +5 -5
  81. package/test/migrations.test.ts +28 -22
  82. package/test/models-yml.test.ts +18 -18
  83. package/test/ollama.test.ts +40 -34
  84. package/test/omp-credentials.test.ts +16 -16
  85. package/test/policy.test.ts +3 -3
  86. package/test/reconfigure.test.ts +4 -4
  87. package/test/redaction.test.ts +41 -35
  88. package/test/remote.test.ts +12 -12
  89. package/test/report-logic.test.ts +8 -8
  90. package/test/report.test.ts +95 -87
  91. package/test/retention.test.ts +79 -66
  92. package/test/schema.test.ts +123 -0
  93. package/test/scope.test.ts +8 -8
  94. package/test/select.test.ts +216 -257
  95. package/test/skills.test.ts +3 -3
  96. package/test/sql-shim.test.ts +154 -0
  97. package/test/state.test.ts +43 -36
  98. package/test/summary.test.ts +38 -27
  99. package/test/tier-plan.test.ts +45 -62
  100. package/test/toast-logic.test.ts +31 -31
  101. package/test/tokens.test.ts +95 -80
  102. package/test/trust-attribution.test.ts +217 -187
  103. package/test/trust-window.test.ts +37 -32
  104. package/test/turn.test.ts +55 -23
  105. package/test/upstreams.test.ts +13 -13
  106. package/test/views.test.ts +81 -59
  107. package/test/wire-request.test.ts +17 -17
  108. package/test/wire-responses.test.ts +4 -4
  109. package/tools/agentdox-e2e.ts +5 -2
  110. package/tools/export-benchmarks.ts +5 -5
  111. package/tools/ledger-parity.ts +266 -0
  112. package/tools/replay.ts +16 -8
@@ -2,7 +2,6 @@ import { describe, expect, test } from "bun:test";
2
2
 
3
3
  import { joinBenchmarks, normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
4
4
  import type { CatalogModel, CatalogSnapshot } from "../src/catalog/types.ts";
5
- import type { Ledger } from "../src/cost/types.ts";
6
5
  import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
7
6
  import { buildCandidates } from "../src/router/candidates.ts";
8
7
  import { extractFeatures } from "../src/router/features.ts";
@@ -48,7 +47,7 @@ function snapshot(list: CatalogModel[]): CatalogSnapshot {
48
47
  }
49
48
 
50
49
  describe("joinBenchmarks", () => {
51
- test("copies benchmarks onto key-scoped records matched by id", () => {
50
+ test("copies benchmarks onto key-scoped records matched by id", async () => {
52
51
  const keyScoped = [{ id: "a/one" }, { id: "b/two" }];
53
52
  const pub = [
54
53
  { id: "a/one", benchmarks: { artificial_analysis: { coding_index: 70 } } },
@@ -59,13 +58,13 @@ describe("joinBenchmarks", () => {
59
58
  expect(keyScoped[1]).not.toHaveProperty("benchmarks");
60
59
  });
61
60
 
62
- test("falls back to canonical_slug", () => {
61
+ test("falls back to canonical_slug", async () => {
63
62
  const keyScoped = [{ id: "vendor/model-preview", canonical_slug: "vendor/model" }];
64
63
  const pub = [{ id: "vendor/model", benchmarks: { artificial_analysis: { coding_index: 55 } } }];
65
64
  expect(joinBenchmarks(keyScoped, pub)).toBe(1);
66
65
  });
67
66
 
68
- test("strips a leading ~ from an alias id", () => {
67
+ test("strips a leading ~ from an alias id", async () => {
69
68
  const keyScoped = [{ id: "~vendor/model-latest" }];
70
69
  const pub = [{ id: "vendor/model-latest", benchmarks: { artificial_analysis: { coding_index: 60 } } }];
71
70
  expect(joinBenchmarks(keyScoped, pub)).toBe(1);
@@ -77,14 +76,14 @@ describe("joinBenchmarks", () => {
77
76
  return rec.benchmarks?.artificial_analysis?.coding_index;
78
77
  }
79
78
 
80
- test("never overwrites benchmarks that are already present", () => {
79
+ test("never overwrites benchmarks that are already present", async () => {
81
80
  const keyScoped: unknown[] = [{ id: "a/one", benchmarks: { artificial_analysis: { coding_index: 1 } } }];
82
81
  const pub: unknown[] = [{ id: "a/one", benchmarks: { artificial_analysis: { coding_index: 99 } } }];
83
82
  expect(joinBenchmarks(keyScoped, pub)).toBe(0);
84
83
  expect(codingOf(keyScoped[0])).toBe(1);
85
84
  });
86
85
 
87
- test("a real id beats an alias target for the same key", () => {
86
+ test("a real id beats an alias target for the same key", async () => {
88
87
  const keyScoped: unknown[] = [{ id: "vendor/model" }];
89
88
  const pub: unknown[] = [
90
89
  { id: "other/model", canonical_slug: "vendor/model", benchmarks: { artificial_analysis: { coding_index: 10 } } },
@@ -94,11 +93,11 @@ describe("joinBenchmarks", () => {
94
93
  expect(codingOf(keyScoped[0])).toBe(80);
95
94
  });
96
95
 
97
- test("tolerates junk records on both sides", () => {
96
+ test("tolerates junk records on both sides", async () => {
98
97
  expect(joinBenchmarks([null, 7, "x"], [null, { id: "a" }])).toBe(0);
99
98
  });
100
99
 
101
- test("normalizing a joined record yields a scored model", () => {
100
+ test("normalizing a joined record yields a scored model", async () => {
102
101
  const keyScoped: unknown[] = [raw("a/one", null, 1)];
103
102
  joinBenchmarks(keyScoped, [raw("a/one", 66, 1)]);
104
103
  const model = normalizeCatalogModel(keyScoped[0]);
@@ -107,7 +106,7 @@ describe("joinBenchmarks", () => {
107
106
  });
108
107
 
109
108
  describe("computeTierPlan", () => {
110
- test("bands ascend across tiers", () => {
109
+ test("bands ascend across tiers", async () => {
111
110
  const plan = computeTierPlan(
112
111
  models([
113
112
  ["a/1", 10, 0.1],
@@ -124,7 +123,7 @@ describe("computeTierPlan", () => {
124
123
  expect(f.hard).toBe(70);
125
124
  });
126
125
 
127
- test("an all-unscored catalog yields zero floors, never an imputed score", () => {
126
+ test("an all-unscored catalog yields zero floors, never an imputed score", async () => {
128
127
  const plan = computeTierPlan(
129
128
  models([
130
129
  ["a/1", null, 0.1],
@@ -136,7 +135,7 @@ describe("computeTierPlan", () => {
136
135
  expect(plan.scoredCount.coding).toBe(0);
137
136
  });
138
137
 
139
- test("every tier floor is met by at least one available model", () => {
138
+ test("every tier floor is met by at least one available model", async () => {
140
139
  const list = models([
141
140
  ["a/1", 12, 0.1],
142
141
  ["a/2", 44, 0.2],
@@ -151,7 +150,7 @@ describe("computeTierPlan", () => {
151
150
  }
152
151
  });
153
152
 
154
- test("excludes built-in denials from the ranking", () => {
153
+ test("excludes built-in denials from the ranking", async () => {
155
154
  // The batch entry is cheap and scored, but selection can never pick it,
156
155
  // so it must not drag the bands down.
157
156
  const plan = computeTierPlan(
@@ -166,12 +165,12 @@ describe("computeTierPlan", () => {
166
165
  expect(plan.floors.coding.trivial).toBe(70);
167
166
  });
168
167
 
169
- test("a single scored model puts that model in every tier", () => {
168
+ test("a single scored model puts that model in every tier", async () => {
170
169
  const plan = computeTierPlan(models([["a/1", 42, 0.1]]), BASE);
171
170
  for (const tier of TIER_ORDER) expect(plan.floors.coding[tier]).toBe(42);
172
171
  });
173
172
 
174
- test("scores each axis independently", () => {
173
+ test("scores each axis independently", async () => {
175
174
  const plan = computeTierPlan(models([["a/1", 30, 0.1]]), BASE);
176
175
  expect(plan.scoredCount.coding).toBe(1);
177
176
  expect(plan.scoredCount.agentic).toBe(1);
@@ -190,26 +189,26 @@ describe("effectiveQualityFloor", () => {
190
189
  BASE,
191
190
  );
192
191
 
193
- test("relaxes a floor the catalog cannot meet", () => {
192
+ test("relaxes a floor the catalog cannot meet", async () => {
194
193
  expect(effectiveQualityFloor(95, "hard", "coding", plan)).toBe(80);
195
194
  });
196
195
 
197
- test("never tightens a floor the catalog exceeds", () => {
196
+ test("never tightens a floor the catalog exceeds", async () => {
198
197
  expect(effectiveQualityFloor(10, "hard", "coding", plan)).toBe(10);
199
198
  });
200
199
 
201
- test("is a no-op when configured equals adaptive", () => {
200
+ test("is a no-op when configured equals adaptive", async () => {
202
201
  expect(effectiveQualityFloor(80, "hard", "coding", plan)).toBe(80);
203
202
  });
204
203
  });
205
204
 
206
205
  describe("tierPlanFor", () => {
207
- test("memoizes per snapshot object", () => {
206
+ test("memoizes per snapshot object", async () => {
208
207
  const snap = snapshot(models([["a/1", 50, 0.1]]));
209
208
  expect(tierPlanFor(snap, BASE)).toBe(tierPlanFor(snap, BASE));
210
209
  });
211
210
 
212
- test("a new snapshot recomputes, so a refresh tracks availability", () => {
211
+ test("a new snapshot recomputes, so a refresh tracks availability", async () => {
213
212
  const first = snapshot(models([["a/1", 50, 0.1]]));
214
213
  const second = snapshot(models([["a/1", 50, 0.1], ["a/2", 90, 0.1]]));
215
214
  expect(tierPlanFor(second, BASE)).not.toBe(tierPlanFor(first, BASE));
@@ -245,26 +244,25 @@ describe("adaptive floors in candidate selection", () => {
245
244
  tier: "hard",
246
245
  task: "coding",
247
246
  snapshot: lowCatalog,
248
- ledger: null,
249
247
  cfg,
250
248
  expectedCompletionTokens: 512,
251
249
  warmSlug: null,
252
250
  });
253
251
  }
254
252
 
255
- test("hard is empty with adaptive floors off", () => {
253
+ test("hard is empty with adaptive floors off", async () => {
256
254
  const { candidates } = build({ ...BASE, adaptiveTierFloors: false });
257
255
  expect(candidates).toHaveLength(0);
258
256
  });
259
257
 
260
- test("hard still selects the best available with adaptive floors on", () => {
258
+ test("hard still selects the best available with adaptive floors on", async () => {
261
259
  const { candidates } = build({ ...BASE, adaptiveTierFloors: true });
262
260
  expect(candidates.length).toBeGreaterThan(0);
263
261
  // The top band is the best-scoring model, not the cheapest.
264
262
  expect(candidates.some((c) => c.model.slug === "a/4")).toBe(true);
265
263
  });
266
264
 
267
- test("adaptive floors still order the tiers apart", () => {
265
+ test("adaptive floors still order the tiers apart", async () => {
268
266
  const cfg = { ...BASE, adaptiveTierFloors: true };
269
267
  const best = (tier: "trivial" | "hard"): number => {
270
268
  const { candidates } = buildCandidates({
@@ -273,7 +271,6 @@ describe("adaptive floors in candidate selection", () => {
273
271
  tier,
274
272
  task: "coding",
275
273
  snapshot: lowCatalog,
276
- ledger: null,
277
274
  cfg,
278
275
  expectedCompletionTokens: 512,
279
276
  warmSlug: null,
@@ -284,7 +281,7 @@ describe("adaptive floors in candidate selection", () => {
284
281
  expect(best("hard")).toBeGreaterThanOrEqual(best("trivial"));
285
282
  });
286
283
 
287
- test("excludeSlugs removes a model from the candidate set", () => {
284
+ test("excludeSlugs removes a model from the candidate set", async () => {
288
285
  const cfg = { ...BASE, adaptiveTierFloors: true };
289
286
  const all = build(cfg).candidates.map((c) => c.model.slug);
290
287
  const target = all[0];
@@ -295,7 +292,6 @@ describe("adaptive floors in candidate selection", () => {
295
292
  tier: "hard",
296
293
  task: "coding",
297
294
  snapshot: lowCatalog,
298
- ledger: null,
299
295
  cfg,
300
296
  expectedCompletionTokens: 512,
301
297
  warmSlug: null,
@@ -323,12 +319,12 @@ describe("adaptive price ceilings", () => {
323
319
  );
324
320
  const features = extractFeatures(req, 100);
325
321
 
326
- test("computeTierPlan derives per-tier price bands from the catalog", () => {
322
+ test("computeTierPlan derives per-tier price bands from the catalog", async () => {
327
323
  const plan = computeTierPlan(priced, BASE);
328
324
  expect(plan.priceCeilings).toEqual({ trivial: 1, simple: 2, moderate: 3, hard: 4 });
329
325
  });
330
326
 
331
- test("effectivePriceCeiling: band when on, tighter of config/band, config when off", () => {
327
+ test("effectivePriceCeiling: band when on, tighter of config/band, config when off", async () => {
332
328
  const plan = computeTierPlan(priced, BASE);
333
329
  expect(effectivePriceCeiling(undefined, "moderate", plan, true)).toBe(3); // band
334
330
  expect(effectivePriceCeiling(2, "moderate", plan, true)).toBe(2); // config tightens
@@ -337,7 +333,7 @@ describe("adaptive price ceilings", () => {
337
333
  expect(effectivePriceCeiling(undefined, "hard", plan, false)).toBeUndefined();
338
334
  });
339
335
 
340
- test("a model above the adaptive band is dropped in candidate selection", () => {
336
+ test("a model above the adaptive band is dropped in candidate selection", async () => {
341
337
  const snap = snapshot(priced);
342
338
  const run = (adaptivePriceCeilings: boolean) =>
343
339
  buildCandidates({
@@ -346,7 +342,6 @@ describe("adaptive price ceilings", () => {
346
342
  tier: "moderate",
347
343
  task: "coding",
348
344
  snapshot: snap,
349
- ledger: null,
350
345
  cfg: { ...BASE, adaptivePriceCeilings },
351
346
  expectedCompletionTokens: 512,
352
347
  warmSlug: null,
@@ -359,14 +354,14 @@ describe("adaptive price ceilings", () => {
359
354
  expect(on.rejected.some((r) => r.slug === "a/4" && r.reason === "over_price_ceiling")).toBe(true);
360
355
  });
361
356
 
362
- test("a price ceiling judges the BIASED price, so prepaid capacity is not thrown out on list", () => {
357
+ test("a price ceiling judges the BIASED price, so prepaid capacity is not thrown out on list", async () => {
363
358
  // A subscription upstream inherits its OpenRouter twin's list price ($4/Mtok here) and is
364
359
  // discounted by `costBias` because the capacity is already paid for. Judging the ceiling on
365
360
  // list threw it out before the bias was ever read, which made the bias entirely inert.
366
361
  const snap = snapshot(priced);
367
362
  const biased = { ...snap, providerBias: { [snap.models[0]!.provider]: 0.1 } };
368
363
  const run = (s: typeof snap) =>
369
- buildCandidates({ req, features, tier: "moderate", task: "coding", snapshot: s, ledger: null, cfg: { ...BASE, adaptivePriceCeilings: true }, expectedCompletionTokens: 512, warmSlug: null });
364
+ buildCandidates({ req, features, tier: "moderate", task: "coding", snapshot: s, cfg: { ...BASE, adaptivePriceCeilings: true }, expectedCompletionTokens: 512, warmSlug: null });
370
365
  // Unbiased: the band tightens moderate to $3 and a/4 is over it.
371
366
  expect(run(snap).rejected.some((r) => r.slug === "a/4" && r.reason === "over_price_ceiling")).toBe(true);
372
367
  // Biased ×0.1: $4 list is $0.40 to this deployment, so it clears the same ceiling.
@@ -375,7 +370,7 @@ describe("adaptive price ceilings", () => {
375
370
  expect(on.rejected.some((r) => r.slug === "a/4")).toBe(false);
376
371
  });
377
372
 
378
- test("a tool turn excludes tool-incapable models on the cheap tiers and ranks on agentic above them", () => {
373
+ test("a tool turn excludes tool-incapable models on the cheap tiers and ranks on agentic above them", async () => {
379
374
  // Two Ollama-priced models: the cheap one cannot drive a tool loop (agentic 1.4, as
380
375
  // ollama/gpt-oss:20b really scores), the dearer one can (agentic 51.2, glm-5.3-flash).
381
376
  const mk = (slug: string, price: number, intelligence: number, agentic: number) => ({
@@ -391,7 +386,7 @@ describe("adaptive price ceilings", () => {
391
386
  const snap = { models: [weak, capable], fetchedAtMs: Date.now(), keyScoped: false };
392
387
  const run = (tier: "trivial" | "moderate", min: number) =>
393
388
  buildCandidates({
394
- req, features, tier, task: "chat", snapshot: snap, ledger: null,
389
+ req, features, tier, task: "chat", snapshot: snap,
395
390
  cfg: { ...BASE, filters: { ...BASE.filters, minAgenticForToolTurns: min } },
396
391
  expectedCompletionTokens: 512, warmSlug: null,
397
392
  });
@@ -440,20 +435,19 @@ describe("quality normalization and capability floor (benchmark findings 4/6)",
440
435
  tier: "hard",
441
436
  task: "coding",
442
437
  snapshot: spread,
443
- ledger: null,
444
438
  cfg: { ...BASE, tiers: { ...BASE.tiers, hard: { ...BASE.tiers.hard, ...tierOverride } } },
445
439
  expectedCompletionTokens: 512,
446
440
  warmSlug: null,
447
441
  });
448
442
 
449
- test("raw scoring at the shipped exponent picks the cheapest ELIGIBLE model", () => {
443
+ test("raw scoring at the shipped exponent picks the cheapest ELIGIBLE model", async () => {
450
444
  const { candidates, rejected } = run({ qualityExponent: 3 });
451
445
  expect(candidates[0]?.model.slug).toBe("mid/2");
452
446
  // cheap/1 is under the hard floor of 72 and never competes.
453
447
  expect(rejected.some((r) => r.slug === "cheap/1" && r.reason === "below_quality_floor")).toBe(true);
454
448
  });
455
449
 
456
- test("normalization lets a single-digit exponent buy the best model, which raw cannot", () => {
450
+ test("normalization lets a single-digit exponent buy the best model, which raw cannot", async () => {
457
451
  // Raw at the same exponent still cannot reach it: that is the defect.
458
452
  expect(run({ qualityExponent: 12 }).candidates[0]?.model.slug).toBe("mid/2");
459
453
  // Normalised, the same 12 selects the top-quality model.
@@ -462,7 +456,7 @@ describe("quality normalization and capability floor (benchmark findings 4/6)",
462
456
  expect(normalised.candidates[0]?.reasons.some((r) => r.includes("quality normalised"))).toBe(true);
463
457
  });
464
458
 
465
- test("normalization is monotone in the exponent: higher never picks a weaker model", () => {
459
+ test("normalization is monotone in the exponent: higher never picks a weaker model", async () => {
466
460
  let lastQuality = 0;
467
461
  for (const qualityExponent of [1, 4, 8, 12, 20]) {
468
462
  const top = run({ qualityExponent, qualityNormalization: true }).candidates[0];
@@ -472,7 +466,7 @@ describe("quality normalization and capability floor (benchmark findings 4/6)",
472
466
  }
473
467
  });
474
468
 
475
- test("capability floor takes the best model inside the cap, ignoring the ratio", () => {
469
+ test("capability floor takes the best model inside the cap, ignoring the ratio", async () => {
476
470
  // mid/2 costs ~$0.0016 and good/3 ~$0.0049, so this cap admits both but
477
471
  // excludes best/4 (~$0.0082). The ranked winner is mid/2 (cheapest).
478
472
  const cap = 0.005;
@@ -485,14 +479,14 @@ describe("quality normalization and capability floor (benchmark findings 4/6)",
485
479
  expect(top?.reasons.some((r) => r.includes("capability floor"))).toBe(true);
486
480
  });
487
481
 
488
- test("capability floor is strictly an upgrade: an unaffordable cap changes nothing", () => {
482
+ test("capability floor is strictly an upgrade: an unaffordable cap changes nothing", async () => {
489
483
  const base = run({}).candidates.map((c) => c.model.slug);
490
484
  // A cap below every candidate's cost promotes nobody.
491
485
  const tiny = run({ capabilityFloorUsd: 1e-9 }).candidates.map((c) => c.model.slug);
492
486
  expect(tiny).toEqual(base);
493
487
  });
494
488
 
495
- test("both modes stay inert by default, so shipped behaviour is unchanged", () => {
489
+ test("both modes stay inert by default, so shipped behaviour is unchanged", async () => {
496
490
  const shipped = run({});
497
491
  expect(shipped.candidates[0]?.model.slug).toBe("mid/2");
498
492
  expect(shipped.candidates.every((c) => !c.reasons.some((r) => r.includes("normalised")))).toBe(true);
@@ -524,26 +518,26 @@ describe("thinness-gated relaxation (review 2026-09-05 §1)", () => {
524
518
  BASE,
525
519
  );
526
520
 
527
- test("the bands sit below the configured floors on a wide catalog", () => {
521
+ test("the bands sit below the configured floors on a wide catalog", async () => {
528
522
  // The premise the gate exists for: unconditional min() would relax here.
529
523
  expect(wide.floors.coding.moderate).toBeLessThan(60);
530
524
  expect(wide.floors.coding.hard).toBeLessThan(72);
531
525
  });
532
526
 
533
- test("a configured floor that three or more models meet stands as written", () => {
527
+ test("a configured floor that three or more models meet stands as written", async () => {
534
528
  expect(effectiveQualityFloor(60, "moderate", "coding", wide)).toBe(60); // 62,70,74,76,78 meet it
535
529
  expect(effectiveQualityFloor(72, "hard", "coding", wide)).toBe(72); // 74,76,78 meet it
536
530
  expect(effectiveQualityFloor(40, "simple", "coding", wide)).toBe(40);
537
531
  });
538
532
 
539
- test("a floor fewer than three models meet is relaxed to the band", () => {
533
+ test("a floor fewer than three models meet is relaxed to the band", async () => {
540
534
  // Only 76 and 78 clear 75: thin, so the hard band applies.
541
535
  expect(effectiveQualityFloor(75, "hard", "coding", wide)).toBe(Math.min(75, wide.floors.coding.hard));
542
536
  // Nothing clears 90: relaxed as before.
543
537
  expect(effectiveQualityFloor(90, "hard", "coding", wide)).toBe(wide.floors.coding.hard);
544
538
  });
545
539
 
546
- test("countAdmitted counts scores at or above the floor", () => {
540
+ test("countAdmitted counts scores at or above the floor", async () => {
547
541
  expect(countAdmitted([10, 20, 30, 40], 25)).toBe(2);
548
542
  expect(countAdmitted([10, 20, 30, 40], 40)).toBe(1);
549
543
  expect(countAdmitted([10, 20, 30, 40], 41)).toBe(0);
@@ -551,7 +545,7 @@ describe("thinness-gated relaxation (review 2026-09-05 §1)", () => {
551
545
  expect(countAdmitted([], 0)).toBe(0);
552
546
  });
553
547
 
554
- test("in candidate selection the wide catalog keeps weak models out of moderate", () => {
548
+ test("in candidate selection the wide catalog keeps weak models out of moderate", async () => {
555
549
  const req = parseChatRequest(
556
550
  {
557
551
  model: "auto",
@@ -576,7 +570,6 @@ describe("thinness-gated relaxation (review 2026-09-05 §1)", () => {
576
570
  tier: "moderate",
577
571
  task: "coding",
578
572
  snapshot: snap,
579
- ledger: null,
580
573
  cfg: { ...BASE, adaptiveTierFloors: true },
581
574
  expectedCompletionTokens: 512,
582
575
  warmSlug: null,
@@ -610,17 +603,6 @@ describe("escalation-cost term (review 2026-09-05 §2)", () => {
610
603
  slug === "cheap/flaky"
611
604
  ? { slug, attempts: 100, escalations: 4, errors: 0, successRate: 0.96, meanCostError: 0 }
612
605
  : { slug, attempts: 100, escalations: 0, errors: 4, successRate: 0.96, meanCostError: 0 };
613
- const ledger: Ledger = {
614
- record: () => {},
615
- conversationSpend: () => 0,
616
- spendSince: () => 0,
617
- blendedRate: () => null,
618
- trust: (slug) => trustOf(slug),
619
- allTrust: () => [],
620
- latency: () => null,
621
- tokenRatio: () => null,
622
- recentEntries: () => [],
623
- };
624
606
  function build(weight: number, usdPerPromptToken?: number) {
625
607
  return buildCandidates({
626
608
  req,
@@ -628,7 +610,8 @@ describe("escalation-cost term (review 2026-09-05 §2)", () => {
628
610
  tier: "trivial",
629
611
  task: "coding",
630
612
  snapshot: snap,
631
- ledger,
613
+ // Scoring reads trust out of the prefetched signals.
614
+ signals: new Map(snap.models.map((m) => [m.slug, { trust: trustOf(m.slug), latency: null }])),
632
615
  cfg: { ...BASE, filters: { ...BASE.filters, escalationCostWeight: weight } },
633
616
  expectedCompletionTokens: 512,
634
617
  warmSlug: null,
@@ -636,21 +619,21 @@ describe("escalation-cost term (review 2026-09-05 §2)", () => {
636
619
  });
637
620
  }
638
621
 
639
- test("with the term off, nothing separates them and the tie falls lexically to the flaky model", () => {
622
+ test("with the term off, nothing separates them and the tie falls lexically to the flaky model", async () => {
640
623
  const { candidates } = build(0, 1e-6);
641
624
  // Same success rate, same trust divisor: escalations are invisible.
642
625
  expect(candidates[0]!.model.slug).toBe("cheap/flaky");
643
626
  expect(candidates[0]!.reasons.some((r) => r.startsWith("escalation risk"))).toBe(false);
644
627
  });
645
628
 
646
- test("priced at what an escalated retry actually bills, the flaky model loses", () => {
629
+ test("priced at what an escalated retry actually bills, the flaky model loses", async () => {
647
630
  const { candidates } = build(1, 1e-6); // $1/Mtok of escalated-retry cost
648
631
  expect(candidates[0]!.model.slug).toBe("cheap/solid");
649
632
  const flaky = candidates.find((c) => c.model.slug === "cheap/flaky")!;
650
633
  expect(flaky.reasons.some((r) => r.startsWith("escalation risk"))).toBe(true);
651
634
  });
652
635
 
653
- test("inert until the ledger can measure the retry cost", () => {
636
+ test("inert until the ledger can measure the retry cost", async () => {
654
637
  const { candidates } = build(1);
655
638
  expect(candidates[0]!.model.slug).toBe("cheap/flaky");
656
639
  });
@@ -18,7 +18,7 @@ describe("resolveRouterUrl", () => {
18
18
  const resolve = (env: string | undefined, text: string | null): string =>
19
19
  resolveRouterUrl(env, text, parseYaml);
20
20
 
21
- test("the embed port file wins over AUTO_MODEL_ROUTER_PORT and the config", () => {
21
+ test("the embed port file wins over AUTO_MODEL_ROUTER_PORT and the config", async () => {
22
22
  // The embedded router binds a free OS-assigned port and writes it to the
23
23
  // port file; the toast must poll that actual address, not a stale config.
24
24
  expect(resolveRouterUrl(undefined, "server:\n port: 8788\n", parseYaml, "8812", 45678)).toBe(
@@ -26,57 +26,57 @@ describe("resolveRouterUrl", () => {
26
26
  );
27
27
  });
28
28
 
29
- test("AUTO_MODEL_ROUTER_URL still beats the embed port file", () => {
29
+ test("AUTO_MODEL_ROUTER_URL still beats the embed port file", async () => {
30
30
  expect(resolveRouterUrl("http://host:9999", "server:\n port: 8788\n", parseYaml, "8812", 45678)).toBe(
31
31
  "http://host:9999",
32
32
  );
33
33
  });
34
34
 
35
- test("an embed port file of null falls back to AUTO_MODEL_ROUTER_PORT", () => {
35
+ test("an embed port file of null falls back to AUTO_MODEL_ROUTER_PORT", async () => {
36
36
  expect(resolveRouterUrl(undefined, "server:\n port: 8788\n", parseYaml, "8812", null)).toBe("http://127.0.0.1:8812");
37
37
  });
38
38
 
39
- test("AUTO_MODEL_ROUTER_URL still beats AUTO_MODEL_ROUTER_PORT", () => {
39
+ test("AUTO_MODEL_ROUTER_URL still beats AUTO_MODEL_ROUTER_PORT", async () => {
40
40
  expect(resolveRouterUrl("http://host:9999", "server:\n port: 8788\n", parseYaml, "8812")).toBe("http://host:9999");
41
41
  });
42
42
 
43
- test("an invalid AUTO_MODEL_ROUTER_PORT falls back to the config port", () => {
43
+ test("an invalid AUTO_MODEL_ROUTER_PORT falls back to the config port", async () => {
44
44
  expect(resolveRouterUrl(undefined, "server:\n port: 8788\n", parseYaml, "notaport")).toBe("http://127.0.0.1:8788");
45
45
  expect(resolveRouterUrl(undefined, "server:\n port: 8788\n", parseYaml, "70000")).toBe("http://127.0.0.1:8788");
46
46
  });
47
47
 
48
- test("AUTO_MODEL_ROUTER_PORT with no config uses loopback", () => {
48
+ test("AUTO_MODEL_ROUTER_PORT with no config uses loopback", async () => {
49
49
  expect(resolveRouterUrl(undefined, null, parseYaml, "8812")).toBe("http://127.0.0.1:8812");
50
50
  });
51
51
 
52
- test("reads host and port from the router's own config", () => {
52
+ test("reads host and port from the router's own config", async () => {
53
53
  // The bug this prevents: defaulting to 8788 polls whatever else owns that
54
54
  // port once the router has been moved, and toasts silently never appear.
55
55
  expect(resolve(undefined, "server:\n host: 127.0.0.1\n port: 8788\n")).toBe("http://127.0.0.1:8788");
56
56
  });
57
57
 
58
- test("a port-only config keeps the loopback default host", () => {
58
+ test("a port-only config keeps the loopback default host", async () => {
59
59
  expect(resolve(undefined, "server:\n port: 8790\n")).toBe("http://127.0.0.1:8790");
60
60
  });
61
61
 
62
- test("a wildcard listen address becomes loopback", () => {
62
+ test("a wildcard listen address becomes loopback", async () => {
63
63
  expect(resolve(undefined, "server:\n host: 0.0.0.0\n port: 8788\n")).toBe("http://127.0.0.1:8788");
64
64
  expect(resolve(undefined, "server:\n host: '::'\n port: 8788\n")).toBe("http://127.0.0.1:8788");
65
65
  });
66
66
 
67
- test("falls back when there is no config, no server block, or junk", () => {
67
+ test("falls back when there is no config, no server block, or junk", async () => {
68
68
  expect(resolve(undefined, null)).toBe(DEFAULT_ROUTER_URL);
69
69
  expect(resolve(undefined, "")).toBe(DEFAULT_ROUTER_URL);
70
70
  expect(resolve(undefined, "logLevel: debug\n")).toBe(DEFAULT_ROUTER_URL);
71
71
  expect(resolve(undefined, "server: 5\n")).toBe(DEFAULT_ROUTER_URL);
72
72
  });
73
73
 
74
- test("ignores a non-integer or non-positive port", () => {
74
+ test("ignores a non-integer or non-positive port", async () => {
75
75
  expect(resolve(undefined, "server:\n port: 0\n")).toBe(DEFAULT_ROUTER_URL);
76
76
  expect(resolve(undefined, "server:\n port: notaport\n")).toBe(DEFAULT_ROUTER_URL);
77
77
  });
78
78
 
79
- test("an empty env override does not shadow the config", () => {
79
+ test("an empty env override does not shadow the config", async () => {
80
80
  expect(resolve("", "server:\n port: 8788\n")).toBe("http://127.0.0.1:8788");
81
81
  });
82
82
  });
@@ -95,12 +95,12 @@ function dec(partial: Partial<ToastDecision>): ToastDecision {
95
95
  }
96
96
 
97
97
  describe("selectToasts", () => {
98
- test("toasts nothing on the first tick (lastSeenId null)", () => {
98
+ test("toasts nothing on the first tick (lastSeenId null)", async () => {
99
99
  const entries = [dec({ id: "a" }), dec({ id: "b" })];
100
100
  expect(selectToasts(entries, null)).toEqual([]);
101
101
  });
102
102
 
103
- test("toasts only entries newer than the last-seen id, oldest first", () => {
103
+ test("toasts only entries newer than the last-seen id, oldest first", async () => {
104
104
  // newest-first order: d3 is newest, d1 oldest
105
105
  const entries = [dec({ id: "d3", slug: "x/c" }), dec({ id: "d2", slug: "x/b" }), dec({ id: "d1", slug: "x/a" })];
106
106
  const toasts = selectToasts(entries, "d1");
@@ -110,7 +110,7 @@ describe("selectToasts", () => {
110
110
  expect(toasts[1]?.model).toBe("x/c");
111
111
  });
112
112
 
113
- test("skips wasted (abandoned escalation) entries", () => {
113
+ test("skips wasted (abandoned escalation) entries", async () => {
114
114
  const entries = [dec({ id: "d2", slug: "served", wasted: false }), dec({ id: "d1", wasted: true })];
115
115
  // both newer than lastSeenId ""; only the non-wasted one toasts
116
116
  expect(selectToasts(entries, "")).toHaveLength(1);
@@ -120,12 +120,12 @@ describe("selectToasts", () => {
120
120
  expect(out[0]?.model).toBe("real");
121
121
  });
122
122
 
123
- test("empty input yields no toasts and null newest id", () => {
123
+ test("empty input yields no toasts and null newest id", async () => {
124
124
  expect(selectToasts([], "x")).toEqual([]);
125
125
  expect(newestId([])).toBeNull();
126
126
  });
127
127
 
128
- test("filters to the requesting harness when one is set", () => {
128
+ test("filters to the requesting harness when one is set", async () => {
129
129
  const entries = [
130
130
  dec({ id: "d3", slug: "mine", harnessId: "harness-a" }),
131
131
  dec({ id: "d2", slug: "other", harnessId: "harness-b" }),
@@ -137,7 +137,7 @@ describe("selectToasts", () => {
137
137
  expect(toasts[0]?.model).toBe("mine");
138
138
  });
139
139
 
140
- test("empty harness id toasts every harness", () => {
140
+ test("empty harness id toasts every harness", async () => {
141
141
  const entries = [
142
142
  dec({ id: "d2", slug: "a", harnessId: "harness-a" }),
143
143
  dec({ id: "d1", slug: "b", harnessId: "harness-b" }),
@@ -145,7 +145,7 @@ describe("selectToasts", () => {
145
145
  expect(selectToasts(entries, "", "")).toHaveLength(2);
146
146
  });
147
147
 
148
- test("filters to the requesting omp session when one is set", () => {
148
+ test("filters to the requesting omp session when one is set", async () => {
149
149
  // Two interactive omp sessions sharing one router's ledger: session-a's
150
150
  // toast must not surface session-b's decisions.
151
151
  const entries = [
@@ -158,7 +158,7 @@ describe("selectToasts", () => {
158
158
  expect(toasts[0]?.model).toBe("mine");
159
159
  });
160
160
 
161
- test("empty omp session id toasts every session", () => {
161
+ test("empty omp session id toasts every session", async () => {
162
162
  const entries = [
163
163
  dec({ id: "d2", slug: "a", ompSessionId: "sess-a" }),
164
164
  dec({ id: "d1", slug: "b", ompSessionId: "sess-b" }),
@@ -166,7 +166,7 @@ describe("selectToasts", () => {
166
166
  expect(selectToasts(entries, "", "", "")).toHaveLength(2);
167
167
  });
168
168
 
169
- test("harness and session filters compose", () => {
169
+ test("harness and session filters compose", async () => {
170
170
  const entries = [
171
171
  dec({ id: "d3", slug: "keep", harnessId: "h", ompSessionId: "sess-a" }),
172
172
  dec({ id: "d2", slug: "wrong-session", harnessId: "h", ompSessionId: "sess-b" }),
@@ -179,21 +179,21 @@ describe("selectToasts", () => {
179
179
  });
180
180
 
181
181
  describe("toToastText", () => {
182
- test("prefers servedSlug when present, else slug", () => {
182
+ test("prefers servedSlug when present, else slug", async () => {
183
183
  expect(toToastText(dec({ slug: "s/one", servedSlug: "s/real" }))).toContain("s/real");
184
184
  expect(toToastText(dec({ slug: "s/one", servedSlug: null }))).toContain("s/one");
185
185
  });
186
186
 
187
- test("includes cost when reported, omits otherwise", () => {
187
+ test("includes cost when reported, omits otherwise", async () => {
188
188
  expect(toToastText(dec({ reportedUsd: 0.5 }))).toContain("$0.50000");
189
189
  expect(toToastText(dec({ reportedUsd: null }))).not.toContain("$");
190
190
  });
191
191
 
192
- test("renders provider · model [tier]", () => {
192
+ test("renders provider · model [tier]", async () => {
193
193
  expect(toToastText(dec({ slug: "q/w", tier: "hard", reportedUsd: null }))).toBe("openrouter · q/w [hard]");
194
194
  });
195
195
 
196
- test("an Ollama slug is labelled with its provider and shown without the prefix", () => {
196
+ test("an Ollama slug is labelled with its provider and shown without the prefix", async () => {
197
197
  expect(toToastText(dec({ slug: "ollama/glm-5.3-flash", servedSlug: "ollama/glm-5.3-flash", tier: "moderate", reportedUsd: 0.0007 }))).toBe(
198
198
  "ollama · glm-5.3-flash [moderate] · $0.00070",
199
199
  );
@@ -219,7 +219,7 @@ describe("verbose toast", () => {
219
219
  ttftMs: 2100,
220
220
  });
221
221
 
222
- test("the headline keeps its shape and the body explains the choice", () => {
222
+ test("the headline keeps its shape and the body explains the choice", async () => {
223
223
  const lines = toToastText(FULL).split(String.fromCharCode(10));
224
224
  expect(lines[0]).toBe("ollama · gpt-oss:20b [trivial] · $0.00031");
225
225
  expect(lines[1]).toBe("why: policy: pinned to ollama/gpt-oss:20b");
@@ -227,12 +227,12 @@ describe("verbose toast", () => {
227
227
  expect(lines[3]).toBe("22.9k prompt · 12.8k compacted · 11 tools · tool continuation · attempt 2 · 2.1s to first token");
228
228
  });
229
229
 
230
- test("compact is the old single line", () => {
230
+ test("compact is the old single line", async () => {
231
231
  expect(toToastText(FULL, false)).toBe("ollama · gpt-oss:20b [trivial] · $0.00031");
232
232
  expect(toToastText(FULL, false).includes(String.fromCharCode(10))).toBe(false);
233
233
  });
234
234
 
235
- test("a surprise outranks ordinary ranking, and ranking shows when nothing surprised", () => {
235
+ test("a surprise outranks ordinary ranking, and ranking shows when nothing surprised", async () => {
236
236
  // "cheapest above the quality floor" is ordinary: a failover and a hold win.
237
237
  expect(whyReasons(["cheapest above the quality floor", "failover: x/y empty_completion; retrying a/b"])[0]).toContain("failover");
238
238
  expect(whyReasons(["cheapest above the quality floor", "held from the previous turn: switch margin not cleared"])[0]).toContain("held");
@@ -246,17 +246,17 @@ describe("verbose toast", () => {
246
246
  expect(whyReasons(["policy: pinned to ollama/gpt-oss:20b", "pinned to ollama/gpt-oss:20b by session override"])).toEqual(["policy: pinned to ollama/gpt-oss:20b"]);
247
247
  });
248
248
 
249
- test("an unreported cost falls back to the prediction, and thin decisions stay short", () => {
249
+ test("an unreported cost falls back to the prediction, and thin decisions stay short", async () => {
250
250
  expect(toToastText(dec({ reportedUsd: null, predictedUsd: 0.00042, reasons: [], features: null }))).toBe("openrouter · meta/muse-glimmer-30b [trivial] · ~$0.00042");
251
251
  expect(toToastText(dec({ reportedUsd: null, features: null }))).toBe("openrouter · meta/muse-glimmer-30b [trivial]");
252
252
  });
253
253
 
254
- test("facts skip what a reader does not need", () => {
254
+ test("facts skip what a reader does not need", async () => {
255
255
  expect(factsOf(dec({ features: { promptTokens: 0, toolCount: 0 }, attempt: 0, task: "coding" }))).toEqual([]);
256
256
  expect(factsOf(dec({ features: { promptTokens: 900 }, task: "vision" }))).toEqual(["900 prompt", "vision"]);
257
257
  });
258
258
 
259
- test("selectToasts renders compact when asked", () => {
259
+ test("selectToasts renders compact when asked", async () => {
260
260
  const entries = [dec({ id: "d2", slug: "x/b", reasons: ["failover: nope"] }), dec({ id: "d1" })];
261
261
  expect(selectToasts(entries, "d1", "", "", false)[0]?.text.includes(String.fromCharCode(10))).toBe(false);
262
262
  expect(selectToasts(entries, "d1")[0]?.text.includes(String.fromCharCode(10))).toBe(true);