auto-model-router 0.31.0 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/.omp-plugin/marketplace.json +2 -2
  2. package/README.md +32 -2
  3. package/omp-extension/router-configure.ts +9 -7
  4. package/package.json +1 -1
  5. package/src/cli/config-cmd.ts +8 -7
  6. package/src/cli/explain.ts +10 -5
  7. package/src/cli/export.ts +6 -5
  8. package/src/cli/models.ts +10 -7
  9. package/src/cli/report.ts +6 -1
  10. package/src/cli/stats.ts +7 -7
  11. package/src/config/load.ts +10 -1
  12. package/src/config/types.ts +10 -1
  13. package/src/context/bridge.ts +7 -7
  14. package/src/context/index.ts +3 -3
  15. package/src/context/store.ts +39 -56
  16. package/src/context/types.ts +7 -6
  17. package/src/cost/blended.ts +28 -7
  18. package/src/cost/feedback.ts +33 -37
  19. package/src/cost/ledger-sql.ts +547 -0
  20. package/src/cost/ledger.ts +30 -459
  21. package/src/cost/report.ts +171 -129
  22. package/src/cost/retention.ts +10 -10
  23. package/src/cost/summary.ts +15 -10
  24. package/src/cost/types.ts +43 -62
  25. package/src/cost/views.ts +79 -49
  26. package/src/lib.ts +6 -2
  27. package/src/router/candidates.ts +7 -15
  28. package/src/router/classify.ts +6 -4
  29. package/src/router/index.ts +95 -9
  30. package/src/router/select.ts +38 -21
  31. package/src/router/state.ts +90 -102
  32. package/src/router/types.ts +11 -5
  33. package/src/server/advise.ts +6 -4
  34. package/src/server/compaction-digest.ts +1 -1
  35. package/src/server/digest.ts +9 -10
  36. package/src/server/http.ts +101 -41
  37. package/src/server/providers.ts +18 -4
  38. package/src/server/turn.ts +32 -9
  39. package/src/tokens/estimate.ts +16 -6
  40. package/src/upstream/ollama-usage.ts +21 -11
  41. package/src/util/schema.ts +201 -0
  42. package/src/util/sql.ts +246 -0
  43. package/src/wire/anthropic/messages.ts +3 -4
  44. package/src/wire/openai/request.ts +1 -0
  45. package/src/wire/types.ts +7 -0
  46. package/test/anthropic-wire.test.ts +9 -9
  47. package/test/benchmark-feeds.test.ts +7 -7
  48. package/test/cache-control.test.ts +7 -7
  49. package/test/cache-estimate.test.ts +5 -5
  50. package/test/catalog-view.test.ts +4 -4
  51. package/test/catalog.test.ts +11 -11
  52. package/test/classify.test.ts +24 -24
  53. package/test/compaction.test.ts +20 -20
  54. package/test/config-wizard.test.ts +32 -32
  55. package/test/config.test.ts +10 -10
  56. package/test/connect-harnesses.test.ts +11 -11
  57. package/test/context-bridge.test.ts +40 -30
  58. package/test/context-prune.test.ts +43 -36
  59. package/test/context-query.test.ts +8 -8
  60. package/test/controls.test.ts +54 -27
  61. package/test/cost.test.ts +12 -12
  62. package/test/digest.test.ts +55 -44
  63. package/test/embed-lifecycle.test.ts +5 -5
  64. package/test/embed-logic.test.ts +26 -26
  65. package/test/escalate.test.ts +17 -17
  66. package/test/eval.test.ts +13 -13
  67. package/test/executable.test.ts +6 -6
  68. package/test/exploration.test.ts +19 -20
  69. package/test/failover.test.ts +22 -21
  70. package/test/fakes.ts +105 -0
  71. package/test/features.test.ts +21 -21
  72. package/test/harness-requests.test.ts +3 -3
  73. package/test/harness-switch.test.ts +5 -5
  74. package/test/hold-exploration.test.ts +13 -13
  75. package/test/hot-reload.test.ts +5 -5
  76. package/test/learned.test.ts +5 -5
  77. package/test/ledger-sql.test.ts +342 -0
  78. package/test/mcp-entry.test.ts +5 -5
  79. package/test/migrations.test.ts +28 -22
  80. package/test/models-yml.test.ts +18 -18
  81. package/test/ollama.test.ts +40 -34
  82. package/test/omp-credentials.test.ts +16 -16
  83. package/test/policy.test.ts +3 -3
  84. package/test/reconfigure.test.ts +4 -4
  85. package/test/redaction.test.ts +41 -35
  86. package/test/remote.test.ts +12 -12
  87. package/test/report-logic.test.ts +8 -8
  88. package/test/report.test.ts +95 -87
  89. package/test/retention.test.ts +79 -66
  90. package/test/schema.test.ts +123 -0
  91. package/test/scope.test.ts +8 -8
  92. package/test/select.test.ts +216 -257
  93. package/test/skills.test.ts +3 -3
  94. package/test/sql-shim.test.ts +154 -0
  95. package/test/state.test.ts +43 -36
  96. package/test/summary.test.ts +38 -27
  97. package/test/tier-plan.test.ts +45 -62
  98. package/test/toast-logic.test.ts +31 -31
  99. package/test/tokens.test.ts +95 -80
  100. package/test/trust-attribution.test.ts +217 -187
  101. package/test/trust-window.test.ts +37 -32
  102. package/test/turn.test.ts +55 -23
  103. package/test/upstreams.test.ts +13 -13
  104. package/test/views.test.ts +81 -59
  105. package/test/wire-request.test.ts +17 -17
  106. package/test/wire-responses.test.ts +4 -4
  107. package/tools/agentdox-e2e.ts +5 -2
  108. package/tools/export-benchmarks.ts +5 -5
  109. package/tools/ledger-parity.ts +266 -0
  110. package/tools/replay.ts +16 -8
@@ -1,11 +1,14 @@
1
1
  import { describe, expect, test } from "bun:test";
2
+ import { fakeLedger } from "./fakes.ts";
3
+ import type { ModelLatency } from "../src/cost/types.ts";
4
+ import { prefetchTurnReads } from "../src/router/index.ts";
5
+ import type { AsyncLedger } from "../src/cost/types.ts";
2
6
 
3
7
  import { normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
4
8
  import type { CatalogModel, CatalogSnapshot } from "../src/catalog/types.ts";
5
9
  import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
6
10
  import { loadConfig } from "../src/config/load.ts";
7
11
  import type { ProfileConfig, RouterConfig } from "../src/config/types.ts";
8
- import type { Ledger } from "../src/cost/types.ts";
9
12
  import { extractFeatures } from "../src/router/features.ts";
10
13
  import { scoreHeuristic } from "../src/router/classify.ts";
11
14
  import { latencyWeightFor } from "../src/router/candidates.ts";
@@ -74,13 +77,14 @@ function state(over: Partial<ConversationState> = {}): ConversationState {
74
77
  };
75
78
  }
76
79
 
77
- function run(opts: {
80
+
81
+ async function run(opts: {
78
82
  userText?: string;
79
83
  promptTokens?: number;
80
84
  cfg?: RouterConfig;
81
85
  st?: ConversationState;
82
86
  tier?: Tier;
83
- ledger?: Ledger | null;
87
+ ledger?: AsyncLedger | null;
84
88
  harnessId?: string;
85
89
  maxTokens?: number;
86
90
  }) {
@@ -90,23 +94,28 @@ function run(opts: {
90
94
  const features = extractFeatures(req, opts.promptTokens ?? 4000);
91
95
  const heuristic = scoreHeuristic(features, cfg);
92
96
  const classification = opts.tier === undefined ? heuristic : { ...heuristic, tier: opts.tier };
97
+ const finalReq = opts.harnessId === undefined ? req : { ...req, harnessId: opts.harnessId };
98
+ // `select` takes the ledger's answers as data; the fakes below are read
99
+ // through the same prefetch the router uses, so the tests exercise the real
100
+ // path rather than a second one.
101
+ const reads = await prefetchTurnReads(opts.ledger ?? null, finalReq, PROFILE, cfg, SNAPSHOT, classification.task);
93
102
  return select({
94
- req: opts.harnessId === undefined ? req : { ...req, harnessId: opts.harnessId },
103
+ req: finalReq,
95
104
  features,
96
105
  classification,
97
106
  profile: PROFILE,
98
107
  state: opts.st ?? state(),
99
108
  snapshot: SNAPSHOT,
100
- ledger: opts.ledger === undefined ? null : opts.ledger,
109
+ reads,
101
110
  cfg,
102
111
  nowMs: Date.now(),
103
112
  });
104
113
  }
105
114
 
106
115
  describe("hard exclusions", () => {
107
- test("never selects a meta-router, floating alias, batch endpoint, or cloaked model", () => {
116
+ test("never selects a meta-router, floating alias, batch endpoint, or cloaked model", async () => {
108
117
  for (const tier of ["trivial", "simple", "moderate", "hard"] as Tier[]) {
109
- const d = run({ tier });
118
+ const d = await run({ tier });
110
119
  expect(d.slug.startsWith("openrouter/")).toBe(false);
111
120
  expect(d.slug.startsWith("~")).toBe(false);
112
121
  expect(d.slug.endsWith(":batch")).toBe(false);
@@ -119,23 +128,23 @@ describe("hard exclusions", () => {
119
128
  }
120
129
  });
121
130
 
122
- test("only offers tool-capable models when the request offers tools", () => {
131
+ test("only offers tool-capable models when the request offers tools", async () => {
123
132
  for (const tier of ["trivial", "simple", "moderate", "hard"] as Tier[]) {
124
- const d = run({ tier });
133
+ const d = await run({ tier });
125
134
  for (const c of d.considered) expect(c.model.supportsTools).toBe(true);
126
135
  }
127
136
  });
128
137
 
129
- test("excludes free models by default", () => {
130
- const d = run({ tier: "trivial" });
138
+ test("excludes free models by default", async () => {
139
+ const d = await run({ tier: "trivial" });
131
140
  for (const c of d.considered) expect(c.model.isFree).toBe(false);
132
141
  });
133
142
  });
134
143
 
135
144
  describe("quality floor", () => {
136
- test("an unscored model never satisfies a tier with a floor above zero", () => {
145
+ test("an unscored model never satisfies a tier with a floor above zero", async () => {
137
146
  for (const tier of ["simple", "moderate", "hard"] as Tier[]) {
138
- const d = run({ tier });
147
+ const d = await run({ tier });
139
148
  for (const c of d.considered) {
140
149
  const q = c.model.quality;
141
150
  const unscored = q.coding === undefined && q.agentic === undefined && q.intelligence === undefined;
@@ -144,15 +153,15 @@ describe("quality floor", () => {
144
153
  }
145
154
  });
146
155
 
147
- test("unscored models are eligible in the trivial tier, whose floor is zero", () => {
148
- const d = run({ tier: "trivial" });
156
+ test("unscored models are eligible in the trivial tier, whose floor is zero", async () => {
157
+ const d = await run({ tier: "trivial" });
149
158
  expect(BASE.tiers.trivial.minQuality).toBe(0);
150
159
  expect(d.considered.length).toBeGreaterThan(0);
151
160
  });
152
161
 
153
- test("a higher tier selects a higher-quality model than a lower tier", () => {
154
- const cheap = run({ tier: "trivial" });
155
- const dear = run({ tier: "hard" });
162
+ test("a higher tier selects a higher-quality model than a lower tier", async () => {
163
+ const cheap = await run({ tier: "trivial" });
164
+ const dear = await run({ tier: "hard" });
156
165
  const cheapModel = MODELS.find((m) => m.slug === cheap.slug);
157
166
  const dearModel = MODELS.find((m) => m.slug === dear.slug);
158
167
  expect(cheapModel).toBeDefined();
@@ -162,17 +171,17 @@ describe("quality floor", () => {
162
171
  });
163
172
 
164
173
  describe("context window", () => {
165
- test("rejects models whose context cannot hold the prompt", () => {
174
+ test("rejects models whose context cannot hold the prompt", async () => {
166
175
  // Far larger than the small-context models in the catalog can take.
167
- const d = run({ tier: "trivial", promptTokens: 300_000 });
176
+ const d = await run({ tier: "trivial", promptTokens: 300_000 });
168
177
  expect(d.rejected.some((r) => r.reason === "context_too_small")).toBe(true);
169
178
  const chosen = MODELS.find((m) => m.slug === d.slug);
170
179
  expect(chosen).toBeDefined();
171
180
  expect(chosen?.contextLength ?? 0).toBeGreaterThan(300_000);
172
181
  });
173
182
 
174
- test("applies headroom so a token-estimate error cannot overflow the window", () => {
175
- const d = run({ tier: "trivial", promptTokens: 100_000 });
183
+ test("applies headroom so a token-estimate error cannot overflow the window", async () => {
184
+ const d = await run({ tier: "trivial", promptTokens: 100_000 });
176
185
  const chosen = MODELS.find((m) => m.slug === d.slug);
177
186
  expect(chosen?.contextLength ?? 0).toBeGreaterThanOrEqual(100_000 * BASE.filters.contextHeadroom);
178
187
  });
@@ -183,9 +192,9 @@ describe("cache-aware switching", () => {
183
192
  // choice against a better option, or the switch logic is never exercised.
184
193
  const warmSlug = "x-ai/grok-4.6";
185
194
 
186
- test("keeps the warm model when switching does not clear the margin", () => {
195
+ test("keeps the warm model when switching does not clear the margin", async () => {
187
196
  const cfg: RouterConfig = { ...BASE, hysteresis: { ...BASE.hysteresis, switchMargin: 1e6 } };
188
- const d = run({
197
+ const d = await run({
189
198
  tier: "hard",
190
199
  promptTokens: 80_000,
191
200
  cfg,
@@ -201,8 +210,8 @@ describe("cache-aware switching", () => {
201
210
  expect(d.sticky).toBe(true);
202
211
  });
203
212
 
204
- test("abandons a warm cache whose TTL has expired", () => {
205
- const d = run({
213
+ test("abandons a warm cache whose TTL has expired", async () => {
214
+ const d = await run({
206
215
  tier: "hard",
207
216
  promptTokens: 80_000,
208
217
  st: state({
@@ -219,7 +228,7 @@ describe("cache-aware switching", () => {
219
228
  });
220
229
 
221
230
  describe("budget guard", () => {
222
- test("downgrades when the cold forecast breaches the per-turn cap", () => {
231
+ test("downgrades when the cold forecast breaches the per-turn cap", async () => {
223
232
  // A hard-tier turn at this size forecasts ~$0.02 cold, while cheaper
224
233
  // tiers land well under a cent, so a $0.005 cap is breachable AND
225
234
  // satisfiable further down.
@@ -227,57 +236,57 @@ describe("budget guard", () => {
227
236
  ...BASE,
228
237
  budget: { ...BASE.budget, perTurnUsd: 0.005, onExceeded: "downgrade" },
229
238
  };
230
- const d = run({ tier: "hard", promptTokens: 50_000, cfg });
239
+ const d = await run({ tier: "hard", promptTokens: 50_000, cfg });
231
240
  expect(d.budgetDowngraded).toBe(true);
232
241
  expect(d.forecast.coldUsd).toBeLessThanOrEqual(0.005);
233
242
  });
234
243
 
235
- test("throws in downgrade mode when no candidate at any tier fits", () => {
244
+ test("throws in downgrade mode when no candidate at any tier fits", async () => {
236
245
  // Failing loudly beats silently spending past an impossible cap.
237
246
  const cfg: RouterConfig = {
238
247
  ...BASE,
239
248
  budget: { ...BASE.budget, perTurnUsd: 1e-9, onExceeded: "downgrade" },
240
249
  };
241
- expect(() => run({ tier: "hard", promptTokens: 50_000, cfg })).toThrow(BudgetExceededError);
250
+ await expect(run({ tier: "hard", promptTokens: 50_000, cfg })).rejects.toThrow(BudgetExceededError);
242
251
  });
243
252
 
244
- test("rejects outright when configured to", () => {
253
+ test("rejects outright when configured to", async () => {
245
254
  const cfg: RouterConfig = {
246
255
  ...BASE,
247
256
  budget: { ...BASE.budget, perTurnUsd: 1e-9, onExceeded: "reject" },
248
257
  };
249
- expect(() => run({ tier: "hard", promptTokens: 50_000, cfg })).toThrow(BudgetExceededError);
258
+ await expect(run({ tier: "hard", promptTokens: 50_000, cfg })).rejects.toThrow(BudgetExceededError);
250
259
  });
251
260
 
252
- test("a satisfiable budget does not downgrade", () => {
261
+ test("a satisfiable budget does not downgrade", async () => {
253
262
  const cfg: RouterConfig = { ...BASE, budget: { ...BASE.budget, perTurnUsd: 100, onExceeded: "reject" } };
254
- const d = run({ tier: "moderate", promptTokens: 5000, cfg });
263
+ const d = await run({ tier: "moderate", promptTokens: 5000, cfg });
255
264
  expect(d.budgetDowngraded).toBe(false);
256
265
  });
257
266
 
258
- test("scopes the daily budget to the requesting harness", () => {
267
+ test("scopes the daily budget to the requesting harness", async () => {
259
268
  // Harness A has already spent the whole daily cap; harness B has spent
260
269
  // nothing. A request from B must NOT be budget-blocked by A's spend.
261
270
  const spendByHarness: Record<string, number> = { "harness-a": 1.0 };
262
- const ledger: Ledger = {
263
- record: () => {},
264
- conversationSpend: () => 0,
265
- spendSince: (_sinceMs, harnessId) => (harnessId === undefined ? 1.0 : spendByHarness[harnessId] ?? 0),
266
- blendedRate: () => null,
267
- latency: () => null,
268
- trust: () => null,
269
- allTrust: () => [],
270
- tokenRatio: () => null,
271
- recentEntries: () => [],
272
- };
271
+ const ledger: AsyncLedger = fakeLedger({
272
+ record: async () => {},
273
+ conversationSpend: async () => 0,
274
+ spendSince: async (_sinceMs, harnessId) => (harnessId === undefined ? 1.0 : spendByHarness[harnessId] ?? 0),
275
+ blendedRate: async () => null,
276
+ latency: async () => null,
277
+ trust: async () => null,
278
+ allTrust: async () => [],
279
+ tokenRatio: async () => null,
280
+ recentEntries: async () => [],
281
+ });
273
282
  const cfg: RouterConfig = { ...BASE, budget: { ...BASE.budget, perDayUsd: 0.5, onExceeded: "reject" } };
274
283
 
275
284
  // Harness A is over its daily cap → rejected.
276
- expect(() => run({ tier: "hard", promptTokens: 50_000, cfg, ledger, harnessId: "harness-a" })).toThrow(
285
+ await expect(run({ tier: "hard", promptTokens: 50_000, cfg, ledger, harnessId: "harness-a" })).rejects.toThrow(
277
286
  BudgetExceededError,
278
287
  );
279
288
  // Harness B has spent nothing → not blocked by A's spend.
280
- const d = run({ tier: "hard", promptTokens: 50_000, cfg, ledger, harnessId: "harness-b" });
289
+ const d = await run({ tier: "hard", promptTokens: 50_000, cfg, ledger, harnessId: "harness-b" });
281
290
  expect(d.budgetDowngraded).toBe(false);
282
291
  });
283
292
  });
@@ -286,68 +295,48 @@ describe("per-harness trust scoping", () => {
286
295
  // When filters.trustScopedByHarness is on, trust is read from the requesting
287
296
  // harness's own ledger rows, so one harness's flaky-model demotion does not
288
297
  // leak into another's routing. Off (default), trust is shared.
289
- const untrustedLedger = (): Ledger => ({
290
- record: () => {},
291
- conversationSpend: () => 0,
292
- spendSince: () => 0,
293
- blendedRate: () => null,
294
- latency: () => null,
295
- trust: (_slug, harnessId) => {
296
- // Harness A has burned the model; harness B has never tried it.
297
- if (harnessId === "harness-a") {
298
- return { slug: "x", attempts: 40, escalations: 30, errors: 30, successRate: 0.1, meanCostError: 0.2 };
299
- }
300
- return null; // harness B / shared → unmeasured
301
- },
302
- allTrust: () => [],
303
- tokenRatio: () => null,
304
- recentEntries: () => [],
305
- });
306
-
307
- test("scoped trust passes the harness id into the ledger trust query", () => {
298
+ test("scoped trust passes the harness id into the ledger trust query", async () => {
308
299
  // The feature's contract is that the router's trust lookup is scoped to
309
- // the requesting harness when enabled. Assert the wiring directly rather
310
- // than via a post-rescue `rejected` reason, which tier-rescue relaxes.
300
+ // the requesting harness when enabled. The lookup is the prefetch, so
301
+ // assert the harness id arrives there.
311
302
  let queriedWith: string | undefined;
312
- const ledger: Ledger = {
313
- ...untrustedLedger(),
314
- trust: (_slug, harnessId) => {
303
+ const ledger: AsyncLedger = fakeLedger({
304
+ signals: async (slugs, harnessId) => {
315
305
  queriedWith = harnessId;
316
- return null;
306
+ return new Map(slugs.map((slug) => [slug, { trust: null, latency: null }]));
317
307
  },
318
- };
308
+ });
319
309
  const cfg: RouterConfig = {
320
310
  ...BASE,
321
311
  filters: { ...BASE.filters, trustScopedByHarness: true },
322
312
  };
323
- run({ tier: "simple", cfg, ledger, harnessId: "harness-a" });
313
+ await run({ tier: "simple", cfg, ledger, harnessId: "harness-a" });
324
314
  expect(queriedWith).toBe("harness-a");
325
315
  });
326
316
 
327
- test("shared trust (default) reads the whole ledger, not per-harness", () => {
317
+ test("shared trust (default) reads the whole ledger, not per-harness", async () => {
328
318
  // With scoping off, the trust lookup must NOT carry the harness id, so
329
319
  // harness A's flaky history is visible globally (shared reliability).
330
320
  let queriedWith: string | undefined;
331
- const ledger: Ledger = {
332
- ...untrustedLedger(),
333
- trust: (_slug, harnessId) => {
321
+ const ledger: AsyncLedger = fakeLedger({
322
+ signals: async (slugs, harnessId) => {
334
323
  queriedWith = harnessId;
335
- return null;
324
+ return new Map(slugs.map((slug) => [slug, { trust: null, latency: null }]));
336
325
  },
337
- };
326
+ });
338
327
  const cfg: RouterConfig = {
339
328
  ...BASE,
340
329
  filters: { ...BASE.filters, trustScopedByHarness: false },
341
330
  };
342
- run({ tier: "simple", cfg, ledger, harnessId: "harness-a" });
331
+ await run({ tier: "simple", cfg, ledger, harnessId: "harness-a" });
343
332
  // The trust lookup must NOT carry the harness id when scoping is off.
344
333
  expect(queriedWith).toBeUndefined();
345
334
  });
346
335
  });
347
336
 
348
337
  describe("decision shape", () => {
349
- test("clamps max tokens to the chosen model's published ceiling", () => {
350
- const d = run({ tier: "moderate" });
338
+ test("clamps max tokens to the chosen model's published ceiling", async () => {
339
+ const d = await run({ tier: "moderate" });
351
340
  const chosen = MODELS.find((m) => m.slug === d.slug);
352
341
  const ceiling = chosen?.maxCompletionTokens;
353
342
  if (ceiling !== undefined && d.maxTokens !== undefined) {
@@ -355,7 +344,7 @@ describe("decision shape", () => {
355
344
  }
356
345
  });
357
346
 
358
- test("a reasoning model gets the completion floor; a direct one keeps the caller's cap", () => {
347
+ test("a reasoning model gets the completion floor; a direct one keeps the caller's cap", async () => {
359
348
  const usable = (m: (typeof MODELS)[number]): boolean => m.supportsTools && m.contextLength >= 32_000 && (m.maxCompletionTokens ?? 100_000) >= 4096;
360
349
  const thinker = MODELS.find((m) => usable(m) && m.supportsReasoning);
361
350
  const direct = MODELS.find((m) => usable(m) && !m.supportsReasoning && !m.reasoningMandatory);
@@ -365,29 +354,29 @@ describe("decision shape", () => {
365
354
 
366
355
  // omp asks for a dozen tokens for a title; a reasoning model would spend them thinking
367
356
  // and return nothing, so the dispatch is raised.
368
- const raised = run({ tier: "trivial", cfg: withFloor(thinker!.slug, 512), maxTokens: 12 });
357
+ const raised = await run({ tier: "trivial", cfg: withFloor(thinker!.slug, 512), maxTokens: 12 });
369
358
  expect(raised.slug).toBe(thinker!.slug);
370
359
  expect(raised.maxTokens).toBe(512);
371
360
  expect(raised.reasons.some((r) => r.includes("reasons before it answers"))).toBe(true);
372
361
 
373
362
  // A model that answers directly is untouched: its cap is the caller's.
374
- const kept = run({ tier: "trivial", cfg: withFloor(direct!.slug, 512), maxTokens: 12 });
363
+ const kept = await run({ tier: "trivial", cfg: withFloor(direct!.slug, 512), maxTokens: 12 });
375
364
  expect(kept.slug).toBe(direct!.slug);
376
365
  expect(kept.maxTokens).toBe(12);
377
366
 
378
367
  // The floor never raises past what the caller already asked for, and 0 disables it.
379
- expect(run({ tier: "trivial", cfg: withFloor(thinker!.slug, 512), maxTokens: 4000 }).maxTokens).toBe(4000);
380
- expect(run({ tier: "trivial", cfg: withFloor(thinker!.slug, 0), maxTokens: 12 }).maxTokens).toBe(12);
368
+ expect((await run({ tier: "trivial", cfg: withFloor(thinker!.slug, 512), maxTokens: 4000 })).maxTokens).toBe(4000);
369
+ expect((await run({ tier: "trivial", cfg: withFloor(thinker!.slug, 0), maxTokens: 12 })).maxTokens).toBe(12);
381
370
  });
382
371
 
383
- test("plans a probe for cheap tiers and leaves the top tier unprobed", () => {
384
- expect(run({ tier: "trivial" }).probe.enabled).toBe(true);
372
+ test("plans a probe for cheap tiers and leaves the top tier unprobed", async () => {
373
+ expect((await run({ tier: "trivial" })).probe.enabled).toBe(true);
385
374
  // Nothing above `hard` to escalate into, so probing it would only add latency.
386
- expect(run({ tier: "hard" }).probe.enabled).toBe(false);
375
+ expect((await run({ tier: "hard" })).probe.enabled).toBe(false);
387
376
  });
388
377
 
389
- test("carries the session id, features, and a reasoning trail", () => {
390
- const d = run({ tier: "simple" });
378
+ test("carries the session id, features, and a reasoning trail", async () => {
379
+ const d = await run({ tier: "simple" });
391
380
  expect(d.sessionId.startsWith("omp-")).toBe(true);
392
381
  // `d.reasons` holds decision-level notes — a widening, a hysteresis hold — and is
393
382
  // legitimately empty when a tier serves the turn without incident. The trail that is
@@ -397,7 +386,7 @@ describe("decision shape", () => {
397
386
  expect(d.considered.length).toBeGreaterThan(0);
398
387
  });
399
388
 
400
- test("respects a profile that caps the tier", () => {
389
+ test("respects a profile that caps the tier", async () => {
401
390
  const req = request("redesign the whole architecture and explain the race condition root cause");
402
391
  const features = extractFeatures(req, 4000);
403
392
  const d = select({
@@ -407,7 +396,6 @@ describe("decision shape", () => {
407
396
  profile: { ...PROFILE, id: "auto-cheap", maxTier: "simple" },
408
397
  state: state(),
409
398
  snapshot: SNAPSHOT,
410
- ledger: null,
411
399
  cfg: BASE,
412
400
  nowMs: Date.now(),
413
401
  });
@@ -431,10 +419,11 @@ describe("tier rescue under a guardrail-constrained catalog", () => {
431
419
  keyScoped: true,
432
420
  };
433
421
 
434
- function runConstrained(ledger: Ledger | null = null) {
422
+ async function runConstrained(ledger: AsyncLedger | null = null) {
435
423
  const req = request("refactor the service layer and explain the cache coherence contract");
436
424
  const features = extractFeatures(req, 4000);
437
425
  const heuristic = scoreHeuristic(features, BASE);
426
+ const reads = await prefetchTurnReads(ledger, req, PROFILE, BASE, constrained, heuristic.task);
438
427
  return select({
439
428
  req,
440
429
  features,
@@ -442,47 +431,35 @@ describe("tier rescue under a guardrail-constrained catalog", () => {
442
431
  profile: PROFILE,
443
432
  state: state(),
444
433
  snapshot: constrained,
445
- ledger,
434
+ reads,
446
435
  cfg: BASE,
447
436
  nowMs: Date.now(),
448
437
  });
449
438
  }
450
439
 
451
440
  /** Every model is probed-and-failed: below the trust floor at every tier. */
452
- function untrustedLedger(): Ledger {
453
- return {
454
- record: () => {},
455
- conversationSpend: () => 0,
456
- spendSince: () => 0,
457
- blendedRate: () => null,
458
- latency: () => null,
459
- trust: (slug) => ({
460
- slug,
461
- attempts: 40,
462
- escalations: 30,
463
- errors: 30,
464
- successRate: 0.1,
465
- meanCostError: 0.2,
466
- }),
467
- allTrust: () => [],
468
- tokenRatio: () => null,
469
- recentEntries: () => [],
470
- };
441
+ function untrustedLedger(): AsyncLedger {
442
+ const burned = (slug: string) => ({ slug, attempts: 40, escalations: 30, errors: 30, successRate: 0.1, meanCostError: 0.2 });
443
+ return fakeLedger({
444
+ trust: async (slug) => burned(slug),
445
+ // Candidate scoring reads `signals`, which is what the prefetch fills.
446
+ signals: async (slugs) => new Map(slugs.map((slug) => [slug, { trust: burned(slug), latency: null }])),
447
+ });
471
448
  }
472
449
 
473
- test("rescues a model instead of throwing when no strict tier admits the catalog", () => {
474
- const d = runConstrained(untrustedLedger());
450
+ test("rescues a model instead of throwing when no strict tier admits the catalog", async () => {
451
+ const d = await runConstrained(untrustedLedger());
475
452
  // It must pick one of the available models, not throw `catalog exhausted`.
476
453
  expect(constrained.models.some((m) => m.slug === d.slug)).toBe(true);
477
454
  });
478
455
 
479
- test("records the rescue in the reasoning trail", () => {
480
- const d = runConstrained(untrustedLedger());
456
+ test("records the rescue in the reasoning trail", async () => {
457
+ const d = await runConstrained(untrustedLedger());
481
458
  expect(d.reasons.some((r) => r.startsWith("tier rescue:"))).toBe(true);
482
459
  });
483
460
 
484
- test("the rescue chooses the cheapest available model when quality is secondary", () => {
485
- const d = runConstrained(untrustedLedger());
461
+ test("the rescue chooses the cheapest available model when quality is secondary", async () => {
462
+ const d = await runConstrained(untrustedLedger());
486
463
  const chosen = MODELS.find((m) => m.slug === d.slug);
487
464
  expect(chosen).toBeDefined();
488
465
  // Price ceilings are relaxed first; the cheapest surviving model wins.
@@ -490,17 +467,17 @@ describe("tier rescue under a guardrail-constrained catalog", () => {
490
467
  expect(d.slug).toBe(cheapest.slug);
491
468
  });
492
469
 
493
- test("a guardrail that leaves every model below the trust bar is rescued by relaxing it", () => {
470
+ test("a guardrail that leaves every model below the trust bar is rescued by relaxing it", async () => {
494
471
  // Reproduces the real failure: a tiny guardrail catalog whose models are
495
472
  // all marked untrusted (probed and failed). The trust floor (minTrust 0.7
496
473
  // over minTrustSamples 12) excludes them at EVERY tier, so strict widening
497
474
  // finds nothing; the rescue relaxes trust and picks a model.
498
- const d = runConstrained(untrustedLedger());
475
+ const d = await runConstrained(untrustedLedger());
499
476
  expect(constrained.models.some((m) => m.slug === d.slug)).toBe(true);
500
477
  expect(d.reasons.some((r) => r.startsWith("tier rescue:"))).toBe(true);
501
478
  });
502
479
 
503
- test("still throws when the catalog is empty after relaxing all economic constraints", () => {
480
+ test("still throws when the catalog is empty after relaxing all economic constraints", async () => {
504
481
  const empty: CatalogSnapshot = { models: [], fetchedAtMs: Date.now(), keyScoped: true };
505
482
  const req = request("anything");
506
483
  const features = extractFeatures(req, 4000);
@@ -513,7 +490,6 @@ describe("tier rescue under a guardrail-constrained catalog", () => {
513
490
  profile: PROFILE,
514
491
  state: state(),
515
492
  snapshot: empty,
516
- ledger: null,
517
493
  cfg: BASE,
518
494
  nowMs: Date.now(),
519
495
  }),
@@ -522,7 +498,7 @@ describe("tier rescue under a guardrail-constrained catalog", () => {
522
498
  });
523
499
 
524
500
  describe("task-type routing", () => {
525
- test("a vision task only considers image-capable models", () => {
501
+ test("a vision task only considers image-capable models", async () => {
526
502
  // Force the vision task and a tier; every considered candidate must
527
503
  // support image input.
528
504
  const req = request("describe this image");
@@ -535,7 +511,6 @@ describe("task-type routing", () => {
535
511
  profile: PROFILE,
536
512
  state: state(),
537
513
  snapshot: SNAPSHOT,
538
- ledger: null,
539
514
  cfg: BASE,
540
515
  nowMs: Date.now(),
541
516
  });
@@ -543,7 +518,7 @@ describe("task-type routing", () => {
543
518
  for (const c of d.considered) expect(c.model.inputModalities.includes("image")).toBe(true);
544
519
  });
545
520
 
546
- test("the task config's quality floor overrides the tier floor when higher", () => {
521
+ test("the task config's quality floor overrides the tier floor when higher", async () => {
547
522
  // A coding task with a high minQuality must not admit models below it,
548
523
  // even in a tier whose own floor is lower.
549
524
  const cfg: RouterConfig = {
@@ -560,7 +535,6 @@ describe("task-type routing", () => {
560
535
  profile: PROFILE,
561
536
  state: state(),
562
537
  snapshot: SNAPSHOT,
563
- ledger: null,
564
538
  cfg,
565
539
  nowMs: Date.now(),
566
540
  });
@@ -576,23 +550,18 @@ describe("task-type routing", () => {
576
550
  });
577
551
 
578
552
  describe("latency scoring", () => {
579
- function ledgerWithLatency(bySlug: Record<string, { ttftMs: number; samples: number; tokensPerSec?: number }>): Ledger {
580
- return {
581
- record: () => {},
582
- conversationSpend: () => 0,
583
- spendSince: () => 0,
584
- blendedRate: () => null,
585
- trust: () => null,
586
- allTrust: () => [],
587
- latency: (slug) => {
588
- const v = bySlug[slug];
589
- // Default throughput is fast, so these cases isolate the TTFT axis
590
- // unless a test sets tokensPerSec explicitly.
591
- return v === undefined ? null : { slug, samples: v.samples, ttftMs: v.ttftMs, tokensPerSec: v.tokensPerSec ?? 1000 };
592
- },
593
- tokenRatio: () => null,
594
- recentEntries: () => [],
553
+ function ledgerWithLatency(bySlug: Record<string, { ttftMs: number; samples: number; tokensPerSec?: number }>): AsyncLedger {
554
+ const latencyOf = (slug: string): ModelLatency | null => {
555
+ const v = bySlug[slug];
556
+ // Default throughput is fast, so these cases isolate the TTFT axis
557
+ // unless a test sets tokensPerSec explicitly.
558
+ return v === undefined ? null : { slug, samples: v.samples, ttftMs: v.ttftMs, tokensPerSec: v.tokensPerSec ?? 1000 };
595
559
  };
560
+ return fakeLedger({
561
+ latency: async (slug) => latencyOf(slug),
562
+ // Scoring reads latency out of the prefetched signals.
563
+ signals: async (slugs) => new Map(slugs.map((slug) => [slug, { trust: null, latency: latencyOf(slug) }])),
564
+ });
596
565
  }
597
566
 
598
567
  const withWeight = (latencyWeight: number): RouterConfig => ({
@@ -600,30 +569,30 @@ describe("latency scoring", () => {
600
569
  filters: { ...BASE.filters, latencyWeight, latencyReferenceMs: 5000, latencyMinSamples: 20 },
601
570
  });
602
571
 
603
- test("penalises a chronically slow model out of the top slot", () => {
604
- const slow = run({ tier: "simple" }).slug;
572
+ test("penalises a chronically slow model out of the top slot", async () => {
573
+ const slow = (await run({ tier: "simple" })).slug;
605
574
  const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 50 } });
606
- const d = run({ tier: "simple", cfg: withWeight(2), ledger });
575
+ const d = await run({ tier: "simple", cfg: withWeight(2), ledger });
607
576
  expect(d.slug).not.toBe(slow);
608
577
  });
609
578
 
610
- test("latencyWeight 0 disables the penalty", () => {
611
- const slow = run({ tier: "simple" }).slug;
579
+ test("latencyWeight 0 disables the penalty", async () => {
580
+ const slow = (await run({ tier: "simple" })).slug;
612
581
  const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 50 } });
613
- expect(run({ tier: "simple", cfg: withWeight(0), ledger }).slug).toBe(slow);
582
+ expect((await run({ tier: "simple", cfg: withWeight(0), ledger })).slug).toBe(slow);
614
583
  });
615
584
 
616
- test("a model with too few samples is not penalised", () => {
617
- const slow = run({ tier: "simple" }).slug;
585
+ test("a model with too few samples is not penalised", async () => {
586
+ const slow = (await run({ tier: "simple" })).slug;
618
587
  const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 5 } });
619
- expect(run({ tier: "simple", cfg: withWeight(2), ledger }).slug).toBe(slow);
588
+ expect((await run({ tier: "simple", cfg: withWeight(2), ledger })).slug).toBe(slow);
620
589
  });
621
590
 
622
- test("penalises a model that starts fast but streams slowly", () => {
591
+ test("penalises a model that starts fast but streams slowly", async () => {
623
592
  // The case TTFT-only scoring misses: quick first token, slow body.
624
- const slow = run({ tier: "simple" }).slug;
593
+ const slow = (await run({ tier: "simple" })).slug;
625
594
  const ledger = ledgerWithLatency({ [slow]: { ttftMs: 1500, samples: 50, tokensPerSec: 12 } });
626
- const d = run({ tier: "simple", cfg: withWeight(2), ledger });
595
+ const d = await run({ tier: "simple", cfg: withWeight(2), ledger });
627
596
  expect(d.slug).not.toBe(slow);
628
597
  });
629
598
 
@@ -632,24 +601,24 @@ describe("latency scoring", () => {
632
601
  filters: { ...BASE.filters, latencyWeight, latencyReferenceMs: 5000, latencyMinSamples: 20, maxExpectedWaitMs },
633
602
  });
634
603
 
635
- test("ceiling hard-drops a proven-slow model the penalty cannot, even at weight 0", () => {
636
- const slow = run({ tier: "simple" }).slug;
604
+ test("ceiling hard-drops a proven-slow model the penalty cannot, even at weight 0", async () => {
605
+ const slow = (await run({ tier: "simple" })).slug;
637
606
  const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 50 } });
638
607
  // latencyWeight 0 → the multiplier is inert; only the hard ceiling can act.
639
- const d = run({ tier: "simple", cfg: withCeiling(20_000), ledger });
608
+ const d = await run({ tier: "simple", cfg: withCeiling(20_000), ledger });
640
609
  expect(d.slug).not.toBe(slow);
641
610
  });
642
611
 
643
- test("ceiling spares an under-sampled slow model (cold-start grace)", () => {
644
- const slow = run({ tier: "simple" }).slug;
612
+ test("ceiling spares an under-sampled slow model (cold-start grace)", async () => {
613
+ const slow = (await run({ tier: "simple" })).slug;
645
614
  const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 5 } });
646
- expect(run({ tier: "simple", cfg: withCeiling(20_000), ledger }).slug).toBe(slow);
615
+ expect((await run({ tier: "simple", cfg: withCeiling(20_000), ledger })).slug).toBe(slow);
647
616
  });
648
617
 
649
- test("ceiling unset ⇒ no latency gate (proven-slow model still wins on price)", () => {
650
- const slow = run({ tier: "simple" }).slug;
618
+ test("ceiling unset ⇒ no latency gate (proven-slow model still wins on price)", async () => {
619
+ const slow = (await run({ tier: "simple" })).slug;
651
620
  const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 50 } });
652
- expect(run({ tier: "simple", cfg: withWeight(0), ledger }).slug).toBe(slow);
621
+ expect((await run({ tier: "simple", cfg: withWeight(0), ledger })).slug).toBe(slow);
653
622
  });
654
623
  });
655
624
 
@@ -688,7 +657,7 @@ describe("context compaction", () => {
688
657
  );
689
658
  }
690
659
 
691
- test("an over-budget turn produces a compaction plan and records savings", () => {
660
+ test("an over-budget turn produces a compaction plan and records savings", async () => {
692
661
  const req = loopReq();
693
662
  const features = extractFeatures(req, 5_000); // over budgetTokens=1000
694
663
  const d = select({
@@ -698,7 +667,6 @@ describe("context compaction", () => {
698
667
  profile: PROFILE,
699
668
  state: state(),
700
669
  snapshot: SNAPSHOT,
701
- ledger: null,
702
670
  cfg: COMPACT_CFG,
703
671
  nowMs: Date.now(),
704
672
  });
@@ -707,7 +675,7 @@ describe("context compaction", () => {
707
675
  expect(d.reasons.some((r) => r.startsWith("compaction:"))).toBe(true);
708
676
  });
709
677
 
710
- test("a small turn is left untouched", () => {
678
+ test("a small turn is left untouched", async () => {
711
679
  const req = loopReq();
712
680
  const features = extractFeatures(req, 500); // under budgetTokens=1000
713
681
  const d = select({
@@ -717,7 +685,6 @@ describe("context compaction", () => {
717
685
  profile: PROFILE,
718
686
  state: state(),
719
687
  snapshot: SNAPSHOT,
720
- ledger: null,
721
688
  cfg: COMPACT_CFG,
722
689
  nowMs: Date.now(),
723
690
  });
@@ -725,7 +692,7 @@ describe("context compaction", () => {
725
692
  expect(d.promptTokensSaved).toBe(0);
726
693
  });
727
694
 
728
- test("a carried plan is re-applied even when the turn is now under budget", () => {
695
+ test("a carried plan is re-applied even when the turn is now under budget", async () => {
729
696
  // The prompt cache is a byte-prefix cache: dropping an edit that was
730
697
  // already dispatched rewrites history the upstream had cached, and
731
698
  // re-sends the tokens the edit saved. So a carried plan survives a turn
@@ -739,7 +706,6 @@ describe("context compaction", () => {
739
706
  profile: PROFILE,
740
707
  state: state(),
741
708
  snapshot: SNAPSHOT,
742
- ledger: null,
743
709
  cfg: COMPACT_CFG,
744
710
  nowMs: Date.now(),
745
711
  });
@@ -753,7 +719,6 @@ describe("context compaction", () => {
753
719
  profile: PROFILE,
754
720
  state: state({ compactionPlan: first.compactionPlan }),
755
721
  snapshot: SNAPSHOT,
756
- ledger: null,
757
722
  cfg: COMPACT_CFG,
758
723
  nowMs: Date.now(),
759
724
  });
@@ -761,7 +726,7 @@ describe("context compaction", () => {
761
726
  expect(second.promptTokensSaved).toBeGreaterThan(0);
762
727
  });
763
728
 
764
- test("floorRatio below 1 compacts strictly past the budget so the plan holds longer", () => {
729
+ test("floorRatio below 1 compacts strictly past the budget so the plan holds longer", async () => {
765
730
  // Each plan change rewrites cached prompt bytes, so compaction overshoots
766
731
  // deliberately: eliding more now buys byte-stable turns later.
767
732
  const req = parseChatRequest(
@@ -791,7 +756,6 @@ describe("context compaction", () => {
791
756
  profile: PROFILE,
792
757
  state: state(),
793
758
  snapshot: SNAPSHOT,
794
- ledger: null,
795
759
  cfg: { ...COMPACT_CFG, compaction: { ...COMPACT_CFG.compaction, floorRatio } },
796
760
  nowMs: Date.now(),
797
761
  });
@@ -830,10 +794,10 @@ describe("hysteresis.breakHoldOnMechanical", () => {
830
794
  function decide(req: NormRequest, breakHold: boolean) {
831
795
  const cfg: RouterConfig = { ...BASE, hysteresis: { ...BASE.hysteresis, breakHoldOnMechanical: breakHold } };
832
796
  const features = extractFeatures(req, 4_000);
833
- return { d: select({ req, features, classification: scoreHeuristic(features, cfg), profile: PROFILE, state: held, snapshot: SNAPSHOT, ledger: null, cfg, nowMs: Date.now() }), features };
797
+ return { d: select({ req, features, classification: scoreHeuristic(features, cfg), profile: PROFILE, state: held, snapshot: SNAPSHOT, cfg, nowMs: Date.now() }), features };
834
798
  }
835
799
 
836
- test("off by default, so a hold still pins the tier", () => {
800
+ test("off by default, so a hold still pins the tier", async () => {
837
801
  expect(DEFAULT_CONFIG.hysteresis.breakHoldOnMechanical).toBe(false); // SHIPPED default, not the live config.yml (machine-dependent)
838
802
  const { d, features } = decide(continuation(), false);
839
803
  expect(features.isToolResultContinuation).toBe(true);
@@ -841,14 +805,14 @@ describe("hysteresis.breakHoldOnMechanical", () => {
841
805
  expect(d.classification.source).toBe("sticky");
842
806
  });
843
807
 
844
- test("on, a mechanical continuation escapes the hold", () => {
808
+ test("on, a mechanical continuation escapes the hold", async () => {
845
809
  const { d } = decide(continuation(), true);
846
810
  expect(d.tier).not.toBe("hard");
847
811
  expect(d.classification.source).not.toBe("sticky");
848
812
  expect(d.reasons.some((r) => /hold hard broken/.test(r))).toBe(true);
849
813
  });
850
814
 
851
- test("a NON-mechanical turn still gets the hold, so flap protection survives", () => {
815
+ test("a NON-mechanical turn still gets the hold, so flap protection survives", async () => {
852
816
  // This is the case hysteresis exists for: fresh user work mid-conversation
853
817
  // must not bounce the model and cold-start its cache.
854
818
  const { d, features } = decide(request("now refactor the retry helper"), true);
@@ -857,7 +821,7 @@ describe("hysteresis.breakHoldOnMechanical", () => {
857
821
  expect(d.classification.source).toBe("sticky");
858
822
  });
859
823
 
860
- test("the downgrade clamp still applies, so quality steps rather than falls", () => {
824
+ test("the downgrade clamp still applies, so quality steps rather than falls", async () => {
861
825
  const cfg: RouterConfig = {
862
826
  ...BASE,
863
827
  hysteresis: { ...BASE.hysteresis, breakHoldOnMechanical: true, maxDowngradePerTurn: 1 },
@@ -872,7 +836,6 @@ describe("hysteresis.breakHoldOnMechanical", () => {
872
836
  profile: PROFILE,
873
837
  state: held,
874
838
  snapshot: SNAPSHOT,
875
- ledger: null,
876
839
  cfg,
877
840
  nowMs: Date.now(),
878
841
  });
@@ -928,25 +891,24 @@ describe("hysteresis.switchHorizonTurns (review 2026-09-05 §4)", () => {
928
891
  profile: PROFILE,
929
892
  state: state({ currentSlug: "test/warm-dear", currentTier: "moderate", cacheWarmSlug: "test/warm-dear", cacheWarmAtMs: Date.now(), lastPromptTokens: promptTokens }),
930
893
  snapshot: snap,
931
- ledger: null,
932
894
  cfg,
933
895
  nowMs: Date.now(),
934
896
  });
935
897
  }
936
898
 
937
- test("the ranked winner is the cheaper cold model", () => {
899
+ test("the ranked winner is the cheaper cold model", async () => {
938
900
  const d = decide(1);
939
901
  expect(d.considered[0]!.model.slug).toBe("test/winner");
940
902
  });
941
903
 
942
- test("a one-turn horizon keeps the dear model warm (the shipped behaviour)", () => {
904
+ test("a one-turn horizon keeps the dear model warm (the shipped behaviour)", async () => {
943
905
  const d = decide(1);
944
906
  expect(d.slug).toBe("test/warm-dear");
945
907
  expect(d.sticky).toBe(true);
946
908
  expect(d.reasons.some((r) => r.startsWith("cache: keeping warm test/warm-dear"))).toBe(true);
947
909
  });
948
910
 
949
- test("amortised over a run of turns, the switch is taken", () => {
911
+ test("amortised over a run of turns, the switch is taken", async () => {
950
912
  const d = decide(8);
951
913
  expect(d.slug).toBe("test/winner");
952
914
  expect(d.sticky).toBe(false);
@@ -990,17 +952,17 @@ describe("compaction.replanGrowthRatio (review 2026-09-05 §7)", () => {
990
952
  }
991
953
  function decide(req: NormRequest, cfg: RouterConfig, st: ConversationState, promptTokens: number) {
992
954
  const features = extractFeatures(req, promptTokens);
993
- return select({ req, features, classification: scoreHeuristic(features, cfg), profile: PROFILE, state: st, snapshot: SNAPSHOT, ledger: null, cfg, nowMs: Date.now() });
955
+ return select({ req, features, classification: scoreHeuristic(features, cfg), profile: PROFILE, state: st, snapshot: SNAPSHOT, cfg, nowMs: Date.now() });
994
956
  }
995
957
 
996
- test("the plan records the compacted size it was made at", () => {
958
+ test("the plan records the compacted size it was made at", async () => {
997
959
  const first = decide(turn(2), cfgWith(1), state(), 4_000);
998
960
  expect(first.compactionPlan.length).toBe(1);
999
961
  expect(first.compactionPlanTokens).toBe(4_000 - first.promptTokensSaved);
1000
962
  expect(first.compactionSavedBytes).toBeGreaterThan(0);
1001
963
  });
1002
964
 
1003
- test("at 1 (shipped) a newly eligible result is compacted on the very next turn", () => {
965
+ test("at 1 (shipped) a newly eligible result is compacted on the very next turn", async () => {
1004
966
  const first = decide(turn(2), cfgWith(1), state(), 4_000);
1005
967
  const carried = state({ compactionPlan: first.compactionPlan, compactionPlanTokens: first.compactionPlanTokens });
1006
968
  // One more round: the prompt grew ~25%, one more result aged out.
@@ -1009,7 +971,7 @@ describe("compaction.replanGrowthRatio (review 2026-09-05 §7)", () => {
1009
971
  expect(second.reasons.some((r) => r.includes("(1 carried, 1 new)"))).toBe(true);
1010
972
  });
1011
973
 
1012
- test("above 1, an existing plan holds until the compacted prompt has grown by the ratio", () => {
974
+ test("above 1, an existing plan holds until the compacted prompt has grown by the ratio", async () => {
1013
975
  const first = decide(turn(2), cfgWith(2), state(), 4_000);
1014
976
  const carried = state({ compactionPlan: first.compactionPlan, compactionPlanTokens: first.compactionPlanTokens });
1015
977
  // The raw prompt grew 25% and the COMPACTED prompt ~60% (the carried
@@ -1032,9 +994,11 @@ describe("hysteresis.confirmUpgradesBelowConfidence", () => {
1032
994
  // A low-confidence heuristic upgrade from a warm model waits one turn.
1033
995
  // Measured: 65 of 67 moderate→hard upgrades in a week bounced back within
1034
996
  // 3 turns, each paying a cold hard-tier read of a ~120k prompt.
1035
- const warmSlug = run({ tier: "moderate" }).slug;
1036
- function upgrade(opts: { confidence?: number; source?: "heuristic" | "escalation"; st?: Partial<ConversationState>; cfg?: RouterConfig; lastToolFailed?: boolean }) {
997
+ // Resolved per test: a describe body cannot await.
998
+ const warmSlugOf = async (): Promise<string> => (await run({ tier: "moderate" })).slug;
999
+ async function upgrade(opts: { confidence?: number; source?: "heuristic" | "escalation"; st?: Partial<ConversationState>; cfg?: RouterConfig; lastToolFailed?: boolean }) {
1037
1000
  const cfg = opts.cfg ?? BASE;
1001
+ const warmSlug = await warmSlugOf();
1038
1002
  const req = request("now rework the whole scheduler");
1039
1003
  const base = extractFeatures(req, 120_000);
1040
1004
  const features = opts.lastToolFailed === true ? { ...base, lastToolFailed: true } : base;
@@ -1046,47 +1010,47 @@ describe("hysteresis.confirmUpgradesBelowConfidence", () => {
1046
1010
  profile: PROFILE,
1047
1011
  state: state({ turn: 4, currentTier: "moderate", currentSlug: warmSlug, cacheWarmSlug: warmSlug, cacheWarmAtMs: Date.now(), lastPromptTokens: 110_000, ...opts.st }),
1048
1012
  snapshot: SNAPSHOT,
1049
- ledger: null,
1050
1013
  cfg,
1051
1014
  nowMs: Date.now(),
1052
1015
  });
1053
1016
  }
1054
1017
 
1055
- test("a low-confidence upgrade from a warm model is deferred to the held tier", () => {
1056
- const d = upgrade({});
1018
+ test("a low-confidence upgrade from a warm model is deferred to the held tier", async () => {
1019
+ const d = await upgrade({});
1057
1020
  expect(d.tier).toBe("moderate");
1058
1021
  expect(d.upgradeDeferred).toBe("hard");
1059
1022
  expect(d.reasons.some((r) => r.includes("upgrade moderate → hard deferred one turn"))).toBe(true);
1060
1023
  });
1061
1024
 
1062
- test("a second consecutive upgrade classification confirms it", () => {
1063
- const d = upgrade({ st: { upgradeDeferredTier: "hard" } });
1025
+ test("a second consecutive upgrade classification confirms it", async () => {
1026
+ const d = await upgrade({ st: { upgradeDeferredTier: "hard" } });
1064
1027
  expect(d.tier).toBe("hard");
1065
1028
  expect(d.upgradeDeferred).toBeNull();
1066
1029
  expect(d.reasons.some((r) => r.includes("upgrade moderate → hard confirmed"))).toBe(true);
1067
1030
  });
1068
1031
 
1069
- test("confident classifications, cold caches, escalations, failing tools and the off switch all upgrade at once", () => {
1070
- expect(upgrade({ confidence: 0.9 }).tier).toBe("hard");
1071
- expect(upgrade({ st: { cacheWarmAtMs: Date.now() - 3_600_000 } }).tier).toBe("hard");
1072
- expect(upgrade({ source: "escalation" }).tier).toBe("hard");
1073
- expect(upgrade({ lastToolFailed: true }).tier).toBe("hard");
1032
+ test("confident classifications, cold caches, escalations, failing tools and the off switch all upgrade at once", async () => {
1033
+ expect((await upgrade({ confidence: 0.9 })).tier).toBe("hard");
1034
+ expect((await upgrade({ st: { cacheWarmAtMs: Date.now() - 3_600_000 } })).tier).toBe("hard");
1035
+ expect((await upgrade({ source: "escalation" })).tier).toBe("hard");
1036
+ expect((await upgrade({ lastToolFailed: true })).tier).toBe("hard");
1074
1037
  const off: RouterConfig = { ...BASE, hysteresis: { ...BASE.hysteresis, confirmUpgradesBelowConfidence: 0 } };
1075
- expect(upgrade({ cfg: off }).tier).toBe("hard");
1076
- for (const d of [upgrade({ confidence: 0.9 }), upgrade({ source: "escalation" })]) expect(d.upgradeDeferred).toBeNull();
1038
+ expect((await upgrade({ cfg: off })).tier).toBe("hard");
1039
+ for (const d of [await upgrade({ confidence: 0.9 }), await upgrade({ source: "escalation" })]) expect(d.upgradeDeferred).toBeNull();
1077
1040
  });
1078
1041
 
1079
- test("a downgrade or a same-tier turn is never deferred", () => {
1080
- const d = run({ tier: "simple", st: state({ turn: 4, currentTier: "moderate", currentSlug: warmSlug, cacheWarmSlug: warmSlug, cacheWarmAtMs: Date.now() }) });
1042
+ test("a downgrade or a same-tier turn is never deferred", async () => {
1043
+ const warmSlug = await warmSlugOf();
1044
+ const d = await run({ tier: "simple", st: state({ turn: 4, currentTier: "moderate", currentSlug: warmSlug, cacheWarmSlug: warmSlug, cacheWarmAtMs: Date.now() }) });
1081
1045
  expect(d.upgradeDeferred).toBeNull();
1082
1046
  });
1083
1047
  });
1084
1048
 
1085
1049
  describe("recorded forecast is the expected price, not the cold worst case", () => {
1086
1050
  const warmSlug = "x-ai/grok-4.6";
1087
- test("a warm stay prices the previous prompt as cache reads; coldUsd keeps the cold figure", () => {
1051
+ test("a warm stay prices the previous prompt as cache reads; coldUsd keeps the cold figure", async () => {
1088
1052
  const cfg: RouterConfig = { ...BASE, hysteresis: { ...BASE.hysteresis, switchMargin: 1e6 } };
1089
- const d = run({
1053
+ const d = await run({
1090
1054
  tier: "hard",
1091
1055
  promptTokens: 80_000,
1092
1056
  cfg,
@@ -1099,8 +1063,8 @@ describe("recorded forecast is the expected price, not the cold worst case", ()
1099
1063
  expect(d.forecast.breakdown.cacheRead).toBeGreaterThan(0);
1100
1064
  });
1101
1065
 
1102
- test("a cold turn records the cold price", () => {
1103
- const d = run({ tier: "hard", promptTokens: 80_000 });
1066
+ test("a cold turn records the cold price", async () => {
1067
+ const d = await run({ tier: "hard", promptTokens: 80_000 });
1104
1068
  expect(d.forecast.assumedCacheHitRate).toBe(0);
1105
1069
  expect(d.forecast.expectedUsd).toBeLessThanOrEqual(d.forecast.coldUsd);
1106
1070
  });
@@ -1108,22 +1072,18 @@ describe("recorded forecast is the expected price, not the cold worst case", ()
1108
1072
 
1109
1073
  describe("cache reliability in the stay/switch comparison", () => {
1110
1074
  const warmSlug = "x-ai/grok-4.6";
1111
- function ledgerWithReliability(rate: number | null, samples = 50): Ledger {
1112
- return {
1113
- record: () => {},
1114
- conversationSpend: () => 0,
1115
- spendSince: () => 0,
1116
- blendedRate: () => null,
1117
- trust: () => null,
1118
- allTrust: () => [],
1119
- latency: () => null,
1120
- tokenRatio: () => null,
1121
- recentEntries: () => [],
1122
- cacheReliability: (slug) => (rate === null || slug !== warmSlug ? null : { slug, samples, hitRate: rate }),
1123
- };
1075
+ function ledgerWithReliability(rate: number | null, samples = 50): AsyncLedger {
1076
+ return fakeLedger({
1077
+ // Only the warm model has a measured rate; everything else is unmeasured,
1078
+ // which is what the stay/switch comparison treats as "assume reliable".
1079
+ cacheReliability: async (slugs) =>
1080
+ rate === null
1081
+ ? new Map()
1082
+ : new Map(slugs.filter((s) => s === warmSlug).map((slug) => [slug, { slug, samples, hitRate: rate }])),
1083
+ });
1124
1084
  }
1125
- const stayCostOf = (ledger: Ledger, cfg: RouterConfig = BASE): number => {
1126
- const d = run({
1085
+ const stayCostOf = async (ledger: AsyncLedger, cfg: RouterConfig = BASE): Promise<number> => {
1086
+ const d = await run({
1127
1087
  tier: "hard",
1128
1088
  promptTokens: 80_000,
1129
1089
  cfg,
@@ -1135,23 +1095,23 @@ describe("cache reliability in the stay/switch comparison", () => {
1135
1095
  return Number(m[1]);
1136
1096
  };
1137
1097
 
1138
- test("an unreliable cache prices staying at the fresh rate, a reliable one at the cached rate", () => {
1139
- const reliable = stayCostOf(ledgerWithReliability(1));
1140
- const flaky = stayCostOf(ledgerWithReliability(0));
1141
- const unknown = stayCostOf(ledgerWithReliability(null));
1098
+ test("an unreliable cache prices staying at the fresh rate, a reliable one at the cached rate", async () => {
1099
+ const reliable = await stayCostOf(ledgerWithReliability(1));
1100
+ const flaky = await stayCostOf(ledgerWithReliability(0));
1101
+ const unknown = await stayCostOf(ledgerWithReliability(null));
1142
1102
  expect(flaky).toBeGreaterThan(reliable);
1143
1103
  expect(unknown).toBeCloseTo(reliable, 6);
1144
1104
  });
1145
1105
 
1146
- test("too few samples, or the feature off, assume a reliable cache", () => {
1147
- const reliable = stayCostOf(ledgerWithReliability(1));
1148
- expect(stayCostOf(ledgerWithReliability(0, 3))).toBeCloseTo(reliable, 6);
1106
+ test("too few samples, or the feature off, assume a reliable cache", async () => {
1107
+ const reliable = await stayCostOf(ledgerWithReliability(1));
1108
+ expect(await stayCostOf(ledgerWithReliability(0, 3))).toBeCloseTo(reliable, 6);
1149
1109
  const off: RouterConfig = { ...BASE, filters: { ...BASE.filters, cacheReliabilityMinSamples: 0 } };
1150
- expect(stayCostOf(ledgerWithReliability(0), off)).toBeCloseTo(reliable, 6);
1110
+ expect(await stayCostOf(ledgerWithReliability(0), off)).toBeCloseTo(reliable, 6);
1151
1111
  });
1152
1112
 
1153
- test("the reason names the measured hit rate", () => {
1154
- const d = run({
1113
+ test("the reason names the measured hit rate", async () => {
1114
+ const d = await run({
1155
1115
  tier: "hard",
1156
1116
  promptTokens: 80_000,
1157
1117
  ledger: ledgerWithReliability(0.5, 40),
@@ -1162,9 +1122,9 @@ describe("cache reliability in the stay/switch comparison", () => {
1162
1122
  });
1163
1123
 
1164
1124
  describe("session pin (forceSlug)", () => {
1165
- test("a pinned catalog model wins over ranking and the warm model; an unknown pin is ignored with a reason", () => {
1166
- const warmSlug = run({ tier: "moderate" }).slug;
1167
- const pinSlug = run({ tier: "hard" }).slug; // a real, differently-ranked model
1125
+ test("a pinned catalog model wins over ranking and the warm model; an unknown pin is ignored with a reason", async () => {
1126
+ const warmSlug = (await run({ tier: "moderate" })).slug;
1127
+ const pinSlug = (await run({ tier: "hard" })).slug; // a real, differently-ranked model
1168
1128
  const req = request("tidy the retry helper");
1169
1129
  const features = extractFeatures(req, 50_000);
1170
1130
  const base = {
@@ -1174,7 +1134,6 @@ describe("session pin (forceSlug)", () => {
1174
1134
  profile: PROFILE,
1175
1135
  state: state({ currentSlug: warmSlug, currentTier: "moderate", cacheWarmSlug: warmSlug, cacheWarmAtMs: Date.now(), lastPromptTokens: 50_000 }),
1176
1136
  snapshot: SNAPSHOT,
1177
- ledger: null,
1178
1137
  cfg: { ...BASE, hysteresis: { ...BASE.hysteresis, switchMargin: 1e6 } },
1179
1138
  nowMs: Date.now(),
1180
1139
  };
@@ -1189,7 +1148,7 @@ describe("session pin (forceSlug)", () => {
1189
1148
  });
1190
1149
 
1191
1150
  describe("budget.perMonthUsd pacing", () => {
1192
- test("monthPace spreads what is left over the days left, today included", () => {
1151
+ test("monthPace spreads what is left over the days left, today included", async () => {
1193
1152
  const sep7 = Date.UTC(2026, 8, 7, 12);
1194
1153
  expect(monthStartMs(sep7)).toBe(Date.UTC(2026, 8, 1));
1195
1154
  const p = monthPace(sep7, 60, 30);
@@ -1199,28 +1158,28 @@ describe("budget.perMonthUsd pacing", () => {
1199
1158
  expect(monthPace(Date.UTC(2026, 8, 30, 12), 60, 0).daysLeft).toBe(1);
1200
1159
  });
1201
1160
 
1202
- test("a month running ahead of pace tightens the daily cap and says so", () => {
1203
- const ledger: Ledger = {
1204
- record: () => {},
1205
- conversationSpend: () => 0,
1206
- spendSince: (sinceMs) => (sinceMs <= monthStartMs(Date.now()) + 1 ? 59.99 : 0), // month-to-date $59.99, last 24h $0
1207
- blendedRate: () => null,
1208
- trust: () => null,
1209
- allTrust: () => [],
1210
- latency: () => null,
1211
- tokenRatio: () => null,
1212
- recentEntries: () => [],
1213
- };
1161
+ test("a month running ahead of pace tightens the daily cap and says so", async () => {
1162
+ const ledger: AsyncLedger = fakeLedger({
1163
+ record: async () => {},
1164
+ conversationSpend: async () => 0,
1165
+ spendSince: async (sinceMs) => (sinceMs <= monthStartMs(Date.now()) + 1 ? 59.99 : 0), // month-to-date $59.99, last 24h $0
1166
+ blendedRate: async () => null,
1167
+ trust: async () => null,
1168
+ allTrust: async () => [],
1169
+ latency: async () => null,
1170
+ tokenRatio: async () => null,
1171
+ recentEntries: async () => [],
1172
+ });
1214
1173
  const cfg: RouterConfig = { ...BASE, budget: { ...BASE.budget, perMonthUsd: 60, onExceeded: "reject" } };
1215
- expect(() => run({ tier: "hard", promptTokens: 50_000, cfg, ledger })).toThrow(/month pacing: \$59\.99 of \$60 spent/);
1174
+ await expect(run({ tier: "hard", promptTokens: 50_000, cfg, ledger })).rejects.toThrow(/month pacing: \$59\.99 of \$60 spent/);
1216
1175
  // Under pace: the cap is generous and nothing breaches.
1217
- const easy: Ledger = { ...ledger, spendSince: () => 1 };
1218
- expect(run({ tier: "hard", promptTokens: 50_000, cfg, ledger: easy }).budgetDowngraded).toBe(false);
1176
+ const easy: AsyncLedger = { ...ledger, spendSince: async () => 1 };
1177
+ expect((await run({ tier: "hard", promptTokens: 50_000, cfg, ledger: easy })).budgetDowngraded).toBe(false);
1219
1178
  });
1220
1179
  });
1221
1180
 
1222
1181
  describe("filters.latencyWeightContinuation", () => {
1223
- test("applies only to tool-result continuations, and only when set", () => {
1182
+ test("applies only to tool-result continuations, and only when set", async () => {
1224
1183
  const f = { ...BASE.filters, latencyWeight: 0.75 };
1225
1184
  expect(latencyWeightFor(f, false)).toBe(0.75);
1226
1185
  expect(latencyWeightFor(f, true)).toBe(0.75);