auto-model-router 0.30.3 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/.omp-plugin/marketplace.json +2 -2
  2. package/README.md +32 -2
  3. package/omp-extension/router-configure.ts +9 -7
  4. package/package.json +1 -1
  5. package/src/cli/config-cmd.ts +8 -7
  6. package/src/cli/explain.ts +10 -5
  7. package/src/cli/export.ts +6 -5
  8. package/src/cli/models.ts +10 -7
  9. package/src/cli/report.ts +6 -1
  10. package/src/cli/stats.ts +7 -7
  11. package/src/config/load.ts +10 -1
  12. package/src/config/types.ts +10 -1
  13. package/src/context/bridge.ts +7 -7
  14. package/src/context/index.ts +3 -3
  15. package/src/context/store.ts +39 -56
  16. package/src/context/types.ts +7 -6
  17. package/src/cost/blended.ts +28 -7
  18. package/src/cost/feedback.ts +33 -37
  19. package/src/cost/ledger-sql.ts +547 -0
  20. package/src/cost/ledger.ts +30 -459
  21. package/src/cost/report.ts +171 -129
  22. package/src/cost/retention.ts +10 -10
  23. package/src/cost/summary.ts +15 -10
  24. package/src/cost/types.ts +43 -62
  25. package/src/cost/views.ts +79 -49
  26. package/src/eval/calibrate.ts +47 -12
  27. package/src/eval/run.ts +18 -2
  28. package/src/lib.ts +6 -2
  29. package/src/router/candidates.ts +7 -15
  30. package/src/router/classify.ts +6 -4
  31. package/src/router/index.ts +95 -9
  32. package/src/router/select.ts +38 -21
  33. package/src/router/state.ts +90 -102
  34. package/src/router/types.ts +11 -5
  35. package/src/server/advise.ts +6 -4
  36. package/src/server/compaction-digest.ts +1 -1
  37. package/src/server/digest.ts +9 -10
  38. package/src/server/http.ts +109 -46
  39. package/src/server/providers.ts +18 -4
  40. package/src/server/turn.ts +32 -9
  41. package/src/tokens/estimate.ts +16 -6
  42. package/src/upstream/ollama-usage.ts +21 -11
  43. package/src/util/schema.ts +201 -0
  44. package/src/util/sql.ts +246 -0
  45. package/src/wire/anthropic/messages.ts +3 -4
  46. package/src/wire/openai/request.ts +1 -0
  47. package/src/wire/types.ts +7 -0
  48. package/test/anthropic-wire.test.ts +9 -9
  49. package/test/benchmark-feeds.test.ts +7 -7
  50. package/test/cache-control.test.ts +7 -7
  51. package/test/cache-estimate.test.ts +5 -5
  52. package/test/catalog-view.test.ts +4 -4
  53. package/test/catalog.test.ts +11 -11
  54. package/test/classify.test.ts +24 -24
  55. package/test/compaction.test.ts +20 -20
  56. package/test/config-wizard.test.ts +32 -32
  57. package/test/config.test.ts +10 -10
  58. package/test/connect-harnesses.test.ts +11 -11
  59. package/test/context-bridge.test.ts +40 -30
  60. package/test/context-prune.test.ts +43 -36
  61. package/test/context-query.test.ts +8 -8
  62. package/test/controls.test.ts +54 -27
  63. package/test/cost.test.ts +12 -12
  64. package/test/digest.test.ts +55 -44
  65. package/test/embed-lifecycle.test.ts +5 -5
  66. package/test/embed-logic.test.ts +26 -26
  67. package/test/escalate.test.ts +17 -17
  68. package/test/eval.test.ts +73 -16
  69. package/test/executable.test.ts +6 -6
  70. package/test/exploration.test.ts +19 -20
  71. package/test/failover.test.ts +22 -21
  72. package/test/fakes.ts +105 -0
  73. package/test/features.test.ts +21 -21
  74. package/test/harness-requests.test.ts +3 -3
  75. package/test/harness-switch.test.ts +5 -5
  76. package/test/hold-exploration.test.ts +13 -13
  77. package/test/hot-reload.test.ts +5 -5
  78. package/test/learned.test.ts +5 -5
  79. package/test/ledger-sql.test.ts +342 -0
  80. package/test/mcp-entry.test.ts +5 -5
  81. package/test/migrations.test.ts +28 -22
  82. package/test/models-yml.test.ts +18 -18
  83. package/test/ollama.test.ts +40 -34
  84. package/test/omp-credentials.test.ts +16 -16
  85. package/test/policy.test.ts +3 -3
  86. package/test/reconfigure.test.ts +4 -4
  87. package/test/redaction.test.ts +41 -35
  88. package/test/remote.test.ts +12 -12
  89. package/test/report-logic.test.ts +8 -8
  90. package/test/report.test.ts +95 -87
  91. package/test/retention.test.ts +79 -66
  92. package/test/schema.test.ts +123 -0
  93. package/test/scope.test.ts +8 -8
  94. package/test/select.test.ts +216 -257
  95. package/test/skills.test.ts +3 -3
  96. package/test/sql-shim.test.ts +154 -0
  97. package/test/state.test.ts +43 -36
  98. package/test/summary.test.ts +38 -27
  99. package/test/tier-plan.test.ts +45 -62
  100. package/test/toast-logic.test.ts +31 -31
  101. package/test/tokens.test.ts +95 -80
  102. package/test/trust-attribution.test.ts +217 -187
  103. package/test/trust-window.test.ts +37 -32
  104. package/test/turn.test.ts +55 -23
  105. package/test/upstreams.test.ts +13 -13
  106. package/test/views.test.ts +81 -59
  107. package/test/wire-request.test.ts +17 -17
  108. package/test/wire-responses.test.ts +4 -4
  109. package/tools/agentdox-e2e.ts +5 -2
  110. package/tools/export-benchmarks.ts +5 -5
  111. package/tools/ledger-parity.ts +266 -0
  112. package/tools/replay.ts +16 -8
@@ -57,14 +57,14 @@ function toolDelta(
57
57
  }
58
58
 
59
59
  describe("createProbe", () => {
60
- test("valid tool-call JSON commits", () => {
60
+ test("valid tool-call JSON commits", async () => {
61
61
  const p = createProbe(plan(), req(), ALL_TRIGGERS);
62
62
  expect(p.observe(toolDelta(0, { id: "c1", name: "read", argsDelta: '{"path":"a' }))).toBeNull();
63
63
  const verdict = p.observe(toolDelta(0, { argsDelta: '.ts"}' }));
64
64
  expect(verdict?.action).toBe("commit");
65
65
  });
66
66
 
67
- test("truncated tool-call JSON at stream end yields malformed_tool_args", () => {
67
+ test("truncated tool-call JSON at stream end yields malformed_tool_args", async () => {
68
68
  // Via the finish event.
69
69
  const p1 = createProbe(plan(), req(), ALL_TRIGGERS);
70
70
  p1.observe(toolDelta(0, { id: "c1", name: "read", argsDelta: '{"path":"a' }));
@@ -80,7 +80,7 @@ describe("createProbe", () => {
80
80
  expect(v2).toMatchObject({ signal: "malformed_tool_args" });
81
81
  });
82
82
 
83
- test("a tool call identical to the previous assistant call yields repeat_tool_call", () => {
83
+ test("a tool call identical to the previous assistant call yields repeat_tool_call", async () => {
84
84
  const history: NormMessage[] = [
85
85
  { role: "user", text: "read it", images: 0, textBytes: 8, toolCalls: [] },
86
86
  {
@@ -99,7 +99,7 @@ describe("createProbe", () => {
99
99
  expect(verdict).toMatchObject({ signal: "repeat_tool_call" });
100
100
  });
101
101
 
102
- test("a different tool call is not a repeat", () => {
102
+ test("a different tool call is not a repeat", async () => {
103
103
  const history: NormMessage[] = [
104
104
  {
105
105
  role: "assistant",
@@ -114,14 +114,14 @@ describe("createProbe", () => {
114
114
  expect(verdict?.action).toBe("commit");
115
115
  });
116
116
 
117
- test("a disabled plan commits on the first chunk", () => {
117
+ test("a disabled plan commits on the first chunk", async () => {
118
118
  const p = createProbe(plan({ enabled: false }), req(), ALL_TRIGGERS);
119
119
  const verdict = p.observe(text("anything at all"));
120
120
  expect(verdict?.action).toBe("commit");
121
121
  expect(p.held()).toHaveLength(1);
122
122
  });
123
123
 
124
- test("a signal absent from triggers never fires", () => {
124
+ test("a signal absent from triggers never fires", async () => {
125
125
  // Truncated args would be malformed_tool_args, but the trigger is off.
126
126
  const p1 = createProbe(plan(), req(), new Set(["refusal"]));
127
127
  p1.observe(toolDelta(0, { id: "c1", name: "read", argsDelta: '{"path":"a' }));
@@ -135,14 +135,14 @@ describe("createProbe", () => {
135
135
  expect(v2.action).toBe("commit");
136
136
  });
137
137
 
138
- test("refusal openers escalate as soon as text arrives", () => {
138
+ test("refusal openers escalate as soon as text arrives", async () => {
139
139
  const p = createProbe(plan(), req(), ALL_TRIGGERS);
140
140
  const verdict = p.observe(text("I'm sorry, but I can't help with that request."));
141
141
  expect(verdict?.action).toBe("escalate");
142
142
  expect(verdict).toMatchObject({ signal: "refusal" });
143
143
  });
144
144
 
145
- test("a forced tool choice answered with prose yields missing_expected_tool_call", () => {
145
+ test("a forced tool choice answered with prose yields missing_expected_tool_call", async () => {
146
146
  const tools: NormTool[] = [{ name: "read", description: "read a file", schemaBytes: 42 }];
147
147
  const p = createProbe(plan(), req([], { tools, forcedToolChoice: true }), ALL_TRIGGERS);
148
148
  p.observe(text("Sure, here is some prose instead."));
@@ -151,20 +151,20 @@ describe("createProbe", () => {
151
151
  expect(verdict).toMatchObject({ signal: "missing_expected_tool_call" });
152
152
  });
153
153
 
154
- test("enough held text commits", () => {
154
+ test("enough held text commits", async () => {
155
155
  const p = createProbe(plan({ maxTokens: 2 }), req(), ALL_TRIGGERS);
156
156
  const verdict = p.observe(text("this is well over eight characters"));
157
157
  expect(verdict?.action).toBe("commit");
158
158
  });
159
159
 
160
- test("an empty stop with nothing emitted yields empty_completion", () => {
160
+ test("an empty stop with nothing emitted yields empty_completion", async () => {
161
161
  const p = createProbe(plan(), req(), ALL_TRIGGERS);
162
162
  const verdict = p.observe(chunk([{ type: "finish", reason: "stop" }]));
163
163
  expect(verdict?.action).toBe("escalate");
164
164
  expect(verdict).toMatchObject({ signal: "empty_completion" });
165
165
  });
166
166
 
167
- test("a stalled stream escalates at the hold ceiling instead of committing silence", () => {
167
+ test("a stalled stream escalates at the hold ceiling instead of committing silence", async () => {
168
168
  let t = 0;
169
169
  const p = createProbe(plan({ maxHoldMs: 1_000 }), req(), ALL_TRIGGERS, () => t);
170
170
  expect(p.observe(chunk([]))).toBeNull();
@@ -174,7 +174,7 @@ describe("createProbe", () => {
174
174
  expect(verdict).toMatchObject({ signal: "empty_completion" });
175
175
  });
176
176
 
177
- test("the hold ceiling still commits when content has arrived", () => {
177
+ test("the hold ceiling still commits when content has arrived", async () => {
178
178
  let t = 0;
179
179
  const p = createProbe(plan({ maxTokens: 1_000, maxHoldMs: 1_000 }), req(), ALL_TRIGGERS, () => t);
180
180
  expect(p.observe(text("partial answer"))).toBeNull();
@@ -183,7 +183,7 @@ describe("createProbe", () => {
183
183
  expect(verdict?.action).toBe("commit");
184
184
  });
185
185
 
186
- test("a length finish on prose commits: that is the caller's max_tokens", () => {
186
+ test("a length finish on prose commits: that is the caller's max_tokens", async () => {
187
187
  // Escalating cannot fix it — the retry runs under the same cap and
188
188
  // truncates in the same place, so it would just bill twice.
189
189
  const p = createProbe(plan({ maxTokens: 1_000 }), req(), ALL_TRIGGERS);
@@ -192,7 +192,7 @@ describe("createProbe", () => {
192
192
  expect(verdict?.action).toBe("commit");
193
193
  });
194
194
 
195
- test("a length finish that truncated tool-call arguments still escalates", () => {
195
+ test("a length finish that truncated tool-call arguments still escalates", async () => {
196
196
  // Structurally unusable output: another model may emit a complete call.
197
197
  const p = createProbe(plan({ maxTokens: 1_000 }), req(), ALL_TRIGGERS);
198
198
  expect(p.observe(toolDelta(0, { id: "c1", name: "read", argsDelta: '{"path":"a' }))).toBeNull();
@@ -200,14 +200,14 @@ describe("createProbe", () => {
200
200
  expect(verdict?.action).toBe("escalate");
201
201
  });
202
202
 
203
- test("a length finish having produced nothing escalates as an empty completion", () => {
203
+ test("a length finish having produced nothing escalates as an empty completion", async () => {
204
204
  const p = createProbe(plan({ maxTokens: 1_000 }), req(), ALL_TRIGGERS);
205
205
  const verdict = p.observe(chunk([{ type: "finish", reason: "length" }]));
206
206
  expect(verdict?.action).toBe("escalate");
207
207
  if (verdict?.action === "escalate") expect(verdict.signal).toBe("empty_completion");
208
208
  });
209
209
 
210
- test("reasoning-only output counts as alive at the hold ceiling", () => {
210
+ test("reasoning-only output counts as alive at the hold ceiling", async () => {
211
211
  // A reasoning model that has emitted only reasoning tokens after the
212
212
  // ceiling is working normally; escalating would discard a healthy paid
213
213
  // generation.
@@ -219,7 +219,7 @@ describe("createProbe", () => {
219
219
  expect(verdict?.action).toBe("commit");
220
220
  });
221
221
 
222
- test("a stream that ENDS with only reasoning is still hollow", () => {
222
+ test("a stream that ENDS with only reasoning is still hollow", async () => {
223
223
  const p = createProbe(plan({ maxTokens: 1_000 }), req(), ALL_TRIGGERS);
224
224
  expect(p.observe(chunk([{ type: "reasoning", delta: "thinking" }]))).toBeNull();
225
225
  const verdict = p.verdictOnEnd();
package/test/eval.test.ts CHANGED
@@ -3,34 +3,35 @@ import { describe, expect, test } from "bun:test";
3
3
  import { normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
4
4
  import { applyFeedScores, loadLocalScores, saveLocalScores, type FeedScore } from "../src/catalog/benchmark-feeds.ts";
5
5
  import { answerScore, extractJson, isRefusalOrEmpty, jsonField, tokenCoverage } from "../src/eval/grade.ts";
6
- import { applyFit, fitAxis, fitCalibration, pickAnchors, toLocalFeedScores, MIN_ANCHORS } from "../src/eval/calibrate.ts";
6
+ import { applyFit, fitAxis, fitCalibration, hardRaw, pickAnchors, toLocalFeedScores, MIN_ANCHORS, MIN_R, PUBLISH_MIN_R } from "../src/eval/calibrate.ts";
7
7
  import { runEval, type EvalResult } from "../src/eval/run.ts";
8
8
  import { EVAL_TASKS } from "../src/eval/tasks.ts";
9
+ import type { QualityAxis } from "../src/config/types.ts";
9
10
  import { makeJudge, parseScore } from "../src/eval/judge.ts";
10
11
  import type { EvalTask, JudgedTask } from "../src/eval/tasks.ts";
11
12
  import { openDb } from "../src/util/sqlite.ts";
12
13
 
13
14
  describe("grade helpers", () => {
14
- test("answerScore matches whole reply, last line, or a standalone token", () => {
15
+ test("answerScore matches whole reply, last line, or a standalone token", async () => {
15
16
  expect(answerScore("9.9", "9.9")).toBe(1);
16
17
  expect(answerScore("The answer is 9.9", "9.9")).toBe(1);
17
18
  expect(answerScore("reasoning...\n9.9", "9.9")).toBe(1);
18
19
  expect(answerScore("19.99", "9.9")).toBe(0); // not a substring match
19
20
  expect(answerScore("", "9.9")).toBe(0);
20
21
  });
21
- test("tokenCoverage is the fraction of tokens present", () => {
22
+ test("tokenCoverage is the fraction of tokens present", async () => {
22
23
  expect(tokenCoverage("return a + b;", ["a + b"])).toBe(1);
23
24
  expect(tokenCoverage("n * 2", ["n", "*", "2"])).toBe(1);
24
25
  expect(tokenCoverage("n plus two", ["n", "*", "2"])).toBeCloseTo(1 / 3);
25
26
  });
26
- test("extractJson tolerates fences and prose; jsonField reads a key", () => {
27
+ test("extractJson tolerates fences and prose; jsonField reads a key", async () => {
27
28
  expect(extractJson('here: {"answer": 8} ok')).toEqual({ answer: 8 });
28
29
  expect(extractJson("```json\n[2,3,5]\n```")).toEqual([2, 3, 5]);
29
30
  expect(extractJson("no json here")).toBeUndefined();
30
31
  expect(jsonField({ tool: "read_file" }, "tool")).toBe("read_file");
31
32
  expect(jsonField([1, 2], "tool")).toBeUndefined();
32
33
  });
33
- test("isRefusalOrEmpty flags empties and refusals", () => {
34
+ test("isRefusalOrEmpty flags empties and refusals", async () => {
34
35
  expect(isRefusalOrEmpty("")).toBe(true);
35
36
  expect(isRefusalOrEmpty("I cannot help with that")).toBe(true);
36
37
  expect(isRefusalOrEmpty("sure, here")).toBe(false);
@@ -38,7 +39,7 @@ describe("grade helpers", () => {
38
39
  });
39
40
 
40
41
  describe("calibration", () => {
41
- test("fitAxis is OLS, needs MIN_ANCHORS points and some spread", () => {
42
+ test("fitAxis is OLS, needs MIN_ANCHORS points and some spread", async () => {
42
43
  const fit = fitAxis([
43
44
  { raw: 0.2, aa: 40 },
44
45
  { raw: 0.5, aa: 60 },
@@ -55,7 +56,7 @@ describe("calibration", () => {
55
56
  expect(fitAxis([{ raw: 0.8, aa: 40 }, { raw: 0.5, aa: 60 }, { raw: 0.2, aa: 80 }])).toBeNull();
56
57
  });
57
58
 
58
- test("pickAnchors spreads over the score range, skips the target and the unscored", () => {
59
+ test("pickAnchors spreads over the score range, skips the target and the unscored", async () => {
59
60
  const m = (slug: string, coding: number | undefined, supportsTools = true) => ({ slug, quality: coding === undefined ? {} : { coding }, supportsTools });
60
61
  const catalog = [m("a/10", 10), m("a/30", 30), m("a/50", 50), m("a/70", 70), m("a/90", 90), m("a/target", undefined), m("a/notools", 60, false)];
61
62
  const picked = pickAnchors(catalog, "a/target");
@@ -100,7 +101,7 @@ describe("calibration", () => {
100
101
  expect(many!.byComplexity.easy).toBeUndefined();
101
102
  });
102
103
 
103
- test("the suite spans complexities, and hard items are not all pinned at the ceiling", () => {
104
+ test("the suite spans complexities, and hard items are not all pinned at the ceiling", async () => {
104
105
  const bands = new Set(EVAL_TASKS.map((t) => t.complexity ?? "easy"));
105
106
  expect(bands.has("easy")).toBe(true);
106
107
  expect(bands.has("hard")).toBe(true);
@@ -127,12 +128,12 @@ describe("calibration", () => {
127
128
  expect(hard.find((t) => t.id === "coding/regex-backtrack")!.grade("XX\nab\nfalse\nx|y\na[b$]c")).toBe(1);
128
129
  });
129
130
 
130
- test("fitCalibration + toLocalFeedScores place a target on the AA scale", () => {
131
+ test("fitCalibration + toLocalFeedScores place a target on the AA scale", async () => {
131
132
  expect(MIN_ANCHORS).toBe(3);
132
133
  const anchors: EvalResult[] = [
133
- { slug: "a/one", axes: { coding: { sum: 0.2, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {} },
134
- { slug: "a/two", axes: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {} },
135
- { slug: "a/three", axes: { coding: { sum: 0.8, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {} },
134
+ { slug: "a/one", axes: { coding: { sum: 0.2, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {}, axesHard: { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } } },
135
+ { slug: "a/two", axes: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {}, axesHard: { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } } },
136
+ { slug: "a/three", axes: { coding: { sum: 0.8, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {}, axesHard: { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } } },
136
137
  ];
137
138
  const aaOf: Record<string, number> = { "a/one": 40, "a/two": 60, "a/three": 80 };
138
139
  const cal = fitCalibration(anchors, (slug, axis) => (axis === "coding" ? aaOf[slug] : undefined));
@@ -140,7 +141,7 @@ describe("calibration", () => {
140
141
  expect(cal.intelligence).toBeUndefined(); // no anchor data on that axis
141
142
 
142
143
  const targets: EvalResult[] = [
143
- { slug: "z/gap", axes: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0.9, n: 1 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {} },
144
+ { slug: "z/gap", axes: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0.9, n: 1 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {}, axesHard: { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } } },
144
145
  ];
145
146
  const local = toLocalFeedScores(targets, cal, (s) => s.slice(0, s.indexOf("/")));
146
147
  expect(local).toHaveLength(1);
@@ -148,6 +149,62 @@ describe("calibration", () => {
148
149
  expect(local[0]!.coding).toBeCloseTo(60, 5); // calibrated from raw 0.5
149
150
  expect(local[0]!.intelligence).toBeUndefined(); // axis had no fit, so not emitted
150
151
  });
152
+
153
+ test("calibrating on the hard band beats calibrating on everything", async () => {
154
+ // Three anchors published 20/50/80 apart. On the FULL suite they all score ~0.97
155
+ // because easy and moderate pin everyone at the ceiling; on the hard band alone they
156
+ // separate. Same models, same publishing, different x — and only one of them can fit.
157
+ const mk = (slug: string, full: number, hard: number): EvalResult => ({
158
+ slug,
159
+ axes: { coding: { sum: full, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } },
160
+ axesHard: { coding: { sum: hard, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } },
161
+ errors: 0,
162
+ repeats: 1,
163
+ spread: {},
164
+ byComplexity: {},
165
+ });
166
+ const anchors = [mk("a/low", 0.96, 0.2), mk("a/mid", 0.97, 0.5), mk("a/high", 0.98, 0.8)];
167
+ const aa: Record<string, number> = { "a/low": 20, "a/mid": 50, "a/high": 80 };
168
+ const published = (slug: string, axis: QualityAxis) => (axis === "coding" ? aa[slug] : undefined);
169
+ const target = mk("z/target", 0.97, 0.5);
170
+
171
+ const onHard = toLocalFeedScores([target], fitCalibration(anchors, published, hardRaw), () => "z", hardRaw);
172
+ // The hard band spans 0.2-0.8 against 20-80, so the fit is a real line: raw 0.5 ⇒ ~50.
173
+ expect(onHard[0]!.coding).toBeCloseTo(50, 0);
174
+
175
+ // On the full suite the anchors span 0.96-0.98: the same published spread compressed into
176
+ // a fiftieth of the range. That is the shallow slope that produced 22.8 for a model
177
+ // published at 39.5, so the fit must be refused rather than published.
178
+ const pooledFit = fitCalibration(anchors, published);
179
+ const pooled = toLocalFeedScores([target], pooledFit, () => "z");
180
+ expect(pooled).toEqual([]);
181
+ });
182
+
183
+ test("a weak fit is refused, not published", async () => {
184
+ // Points with a real but noisy relationship: computable (r >= MIN_R) yet not worth
185
+ // acting on. `r` and `n` used to be computed and then thrown away.
186
+ const noisy = [
187
+ { raw: 0.1, aa: 20 },
188
+ { raw: 0.5, aa: 70 },
189
+ { raw: 0.6, aa: 30 },
190
+ { raw: 0.9, aa: 60 },
191
+ ];
192
+ const fit = fitAxis(noisy)!;
193
+ expect(fit.r).toBeGreaterThanOrEqual(MIN_R);
194
+ expect(fit.r).toBeLessThan(PUBLISH_MIN_R);
195
+ const target: EvalResult = {
196
+ slug: "z/t",
197
+ axes: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } },
198
+ axesHard: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } },
199
+ errors: 0,
200
+ repeats: 1,
201
+ spread: {},
202
+ byComplexity: {},
203
+ };
204
+ expect(toLocalFeedScores([target], { coding: fit }, () => "z", hardRaw)).toEqual([]);
205
+ // A caller that deliberately lowers the bar still can, so the gate is policy not dogma.
206
+ expect(toLocalFeedScores([target], { coding: fit }, () => "z", hardRaw, MIN_R)[0]!.coding).toBeGreaterThan(0);
207
+ });
151
208
  });
152
209
 
153
210
  describe("runEval", () => {
@@ -195,7 +252,7 @@ describe("local source integration", () => {
195
252
  };
196
253
  }
197
254
 
198
- test("local fills only where no stronger source has the axis", () => {
255
+ test("local fills only where no stronger source has the axis", async () => {
199
256
  const catalog = [raw("z-ai/glm-5.3-flash")];
200
257
  const feeds: FeedScore[] = [
201
258
  { key: "glm-5-3-flash", creator: "z-ai", source: "artificial_analysis", coding: 61 },
@@ -209,7 +266,7 @@ describe("local source integration", () => {
209
266
  expect(result.sources.artificial_analysis).toBe(1);
210
267
  });
211
268
 
212
- test("saveLocalScores / loadLocalScores round-trip", () => {
269
+ test("saveLocalScores / loadLocalScores round-trip", async () => {
213
270
  const db = openDb(":memory:");
214
271
  const scores: FeedScore[] = [{ key: "muse-glimmer-30b", creator: "meta", source: "local", coding: 42, agentic: 39 }];
215
272
  saveLocalScores(db, scores, 123);
@@ -219,7 +276,7 @@ describe("local source integration", () => {
219
276
  });
220
277
 
221
278
  describe("llm judge", () => {
222
- test("parseScore takes the last standalone 0-10 and scales to 0-1", () => {
279
+ test("parseScore takes the last standalone 0-10 and scales to 0-1", async () => {
223
280
  expect(parseScore("8")).toBeCloseTo(0.8, 5);
224
281
  expect(parseScore("Score: 10/10")).toBeCloseTo(1, 5);
225
282
  expect(parseScore("I count 3 issues, so 7")).toBeCloseTo(0.7, 5); // last wins
@@ -10,7 +10,7 @@ import { parseRemoteRouter } from "../omp-extension/remote-logic.ts";
10
10
  const NL = String.fromCharCode(10);
11
11
 
12
12
  describe("the single-file member install", () => {
13
- test("the embedded package is the install: CLI, harness integrations, runtime deps; no tests, no caches", () => {
13
+ test("the embedded package is the install: CLI, harness integrations, runtime deps; no tests, no caches", async () => {
14
14
  const pkg = collectPackageFiles(process.cwd());
15
15
  expect(pkg.version).toBe((JSON.parse(readFileSync("package.json", "utf8")) as { version: string }).version);
16
16
  expect(pkg.files["src/index.ts"]).toContain("connect");
@@ -28,7 +28,7 @@ describe("the single-file member install", () => {
28
28
  expect(names.some((n) => /\.d\.ts$/.test(n) && n.startsWith("node_modules/"))).toBe(false);
29
29
  });
30
30
 
31
- test("targets and file names", () => {
31
+ test("targets and file names", async () => {
32
32
  expect(isExecutableTarget("linux-x64")).toBe(true);
33
33
  expect(isExecutableTarget("linux-x86")).toBe(false);
34
34
  expect(executableFileName("windows-x64")).toBe("auto-model-router-windows-x64.exe");
@@ -44,7 +44,7 @@ describe("the single-file member install", () => {
44
44
  expect(await readEmbeddedPackage()).toBeNull();
45
45
  });
46
46
 
47
- test("materializing writes the files once, keyed by content, under the router home", () => {
47
+ test("materializing writes the files once, keyed by content, under the router home", async () => {
48
48
  const home = mkdtempSync(join(tmpdir(), "amr-mat-"));
49
49
  try {
50
50
  const pkg = { version: "9.9.9", files: { "package.json": `{"version":"9.9.9"}${NL}`, "src/index.ts": `console.log(1)${NL}`, "omp-extension/x.ts": "export {}" } };
@@ -63,7 +63,7 @@ describe("the single-file member install", () => {
63
63
  }
64
64
  });
65
65
 
66
- test("connect from the executable: it is Claude Code's key helper, goes on PATH, and remote.json names it", () => {
66
+ test("connect from the executable: it is Claude Code's key helper, goes on PATH, and remote.json names it", async () => {
67
67
  const home = mkdtempSync(join(tmpdir(), "amr-exe-connect-"));
68
68
  const claude = join(home, ".claude");
69
69
  mkdirSync(claude, { recursive: true });
@@ -98,7 +98,7 @@ describe("the single-file member install", () => {
98
98
  expect(seen[0]?.url).toBe("https://team.example/setup/exchange");
99
99
  expect(JSON.parse(seen[0]?.body ?? "{}")).toEqual({ token: "amrs_t", device: "laptop" });
100
100
  const refused = (async () => Response.json({ error: "invalid_token" }, { status: 401 })) as unknown as typeof fetch;
101
- await expect(exchangeSetupToken("https://team.example", "amrs_old", "laptop", refused)).rejects.toThrow("refused");
101
+ (await expect(exchangeSetupToken("https://team.example", "amrs_old", "laptop", refused))).rejects.toThrow("refused");
102
102
  });
103
103
 
104
104
  test("the executable builds for this host and knows its version from the embedded package", async () => {
@@ -118,7 +118,7 @@ describe("the single-file member install", () => {
118
118
  }
119
119
  }, 120_000);
120
120
 
121
- test("the global handle is what marks a compiled process", () => {
121
+ test("the global handle is what marks a compiled process", async () => {
122
122
  const g = globalThis as Record<string, unknown>;
123
123
  g[EMBEDDED_GLOBAL] = { manifestPath: "/$bunfs/root/manifest.json" };
124
124
  try {
@@ -103,7 +103,6 @@ function run(opts: {
103
103
  profile: opts.profile ?? PROFILE,
104
104
  state: opts.st ?? state(),
105
105
  snapshot: SNAPSHOT,
106
- ledger: null,
107
106
  cfg,
108
107
  nowMs: NOW,
109
108
  ...(opts.excludeSlugs === undefined ? {} : { excludeSlugs: opts.excludeSlugs }),
@@ -113,14 +112,14 @@ function run(opts: {
113
112
  const ALWAYS = { simple: 1, moderate: 1, hard: 1 };
114
113
 
115
114
  describe("exploration is opt-in", () => {
116
- test("never fires under the shipped defaults", () => {
115
+ test("never fires under the shipped defaults", async () => {
117
116
  expect(BASE.exploration.enabled).toBe(false);
118
117
  for (const tier of ["simple", "moderate", "hard"] as Tier[]) {
119
118
  expect(run({ tier }).explored).toBeNull();
120
119
  }
121
120
  });
122
121
 
123
- test("enabled with no rates configured still never fires", () => {
122
+ test("enabled with no rates configured still never fires", async () => {
124
123
  const cfg = withExploration({ enabled: true, rates: {} });
125
124
  for (const tier of ["simple", "moderate", "hard"] as Tier[]) {
126
125
  expect(run({ tier, cfg }).explored).toBeNull();
@@ -129,20 +128,20 @@ describe("exploration is opt-in", () => {
129
128
  });
130
129
 
131
130
  describe("per-tier rates", () => {
132
- test("each tier is governed by its own rate, not one global one", () => {
131
+ test("each tier is governed by its own rate, not one global one", async () => {
133
132
  const cfg = withExploration({ enabled: true, rates: { simple: 0, moderate: 0, hard: 1 } });
134
133
  expect(run({ tier: "simple", cfg }).explored).toBeNull();
135
134
  expect(run({ tier: "moderate", cfg }).explored).toBeNull();
136
135
  expect(run({ tier: "hard", cfg }).explored).toEqual({ from: "hard", to: "moderate" });
137
136
  });
138
137
 
139
- test("a tier absent from the rates map is never explored", () => {
138
+ test("a tier absent from the rates map is never explored", async () => {
140
139
  const cfg = withExploration({ enabled: true, rates: { hard: 1 } });
141
140
  expect(run({ tier: "moderate", cfg }).explored).toBeNull();
142
141
  expect(run({ tier: "hard", cfg }).explored).not.toBeNull();
143
142
  });
144
143
 
145
- test("the shipped defaults weight expensive tiers far above cheap ones", () => {
144
+ test("the shipped defaults weight expensive tiers far above cheap ones", async () => {
146
145
  const r = BASE.exploration.rates;
147
146
  expect(r.hard ?? 0).toBeGreaterThan(r.simple ?? 0);
148
147
  expect(r.moderate ?? 0).toBeGreaterThan(r.simple ?? 0);
@@ -152,19 +151,19 @@ describe("per-tier rates", () => {
152
151
  describe("exploration drops exactly one tier", () => {
153
152
  const cfg = withExploration({ enabled: true, rates: ALWAYS });
154
153
 
155
- test("moderate explores down to simple", () => {
154
+ test("moderate explores down to simple", async () => {
156
155
  expect(run({ tier: "moderate", cfg }).explored).toEqual({ from: "moderate", to: "simple" });
157
156
  });
158
157
 
159
- test("hard explores down to moderate, never further", () => {
158
+ test("hard explores down to moderate, never further", async () => {
160
159
  expect(run({ tier: "hard", cfg }).explored).toEqual({ from: "hard", to: "moderate" });
161
160
  });
162
161
 
163
- test("trivial is the floor and cannot be explored below", () => {
162
+ test("trivial is the floor and cannot be explored below", async () => {
164
163
  expect(run({ tier: "trivial", cfg }).explored).toBeNull();
165
164
  });
166
165
 
167
- test("the decision trail says so out loud", () => {
166
+ test("the decision trail says so out loud", async () => {
168
167
  expect(run({ tier: "moderate", cfg }).reasons.some((r) => r.startsWith("exploration:"))).toBe(true);
169
168
  });
170
169
  });
@@ -172,11 +171,11 @@ describe("exploration drops exactly one tier", () => {
172
171
  describe("hysteresis holds are explored only once the cache is cold", () => {
173
172
  const cfg = withExploration({ enabled: true, rates: ALWAYS, stickyPolicy: "cold-cache" });
174
173
 
175
- test("a held tier with a WARM cache is left alone", () => {
174
+ test("a held tier with a WARM cache is left alone", async () => {
176
175
  expect(run({ tier: "simple", cfg, st: heldState("hard", "warm") }).explored).toBeNull();
177
176
  });
178
177
 
179
- test("a held tier with a COLD cache is explorable", () => {
178
+ test("a held tier with a COLD cache is explorable", async () => {
180
179
  // This is the population that carries most of the spend: turns that
181
180
  // reach hard by hold rather than by classification.
182
181
  expect(run({ tier: "simple", cfg, st: heldState("hard", "cold") }).explored).toEqual({
@@ -185,18 +184,18 @@ describe("hysteresis holds are explored only once the cache is cold", () => {
185
184
  });
186
185
  });
187
186
 
188
- test("the reason names the hold and the cache state, for later analysis", () => {
187
+ test("the reason names the hold and the cache state, for later analysis", async () => {
189
188
  const d = run({ tier: "simple", cfg, st: heldState("hard", "cold") });
190
189
  expect(d.reasons.some((r) => r.includes("held tier (cold cache)"))).toBe(true);
191
190
  });
192
191
 
193
- test("stickyPolicy never leaves holds alone entirely", () => {
192
+ test("stickyPolicy never leaves holds alone entirely", async () => {
194
193
  const off = withExploration({ enabled: true, rates: ALWAYS, stickyPolicy: "never" });
195
194
  expect(run({ tier: "simple", cfg: off, st: heldState("hard", "cold") }).explored).toBeNull();
196
195
  expect(run({ tier: "simple", cfg: off, st: heldState("hard", "warm") }).explored).toBeNull();
197
196
  });
198
197
 
199
- test("stickyPolicy always reaches held turns even with a live cache", () => {
198
+ test("stickyPolicy always reaches held turns even with a live cache", async () => {
200
199
  // The only setting that samples the population carrying most of the
201
200
  // spend, at the price of a forfeited cache read.
202
201
  const always = withExploration({ enabled: true, rates: ALWAYS, stickyPolicy: "always" });
@@ -215,16 +214,16 @@ describe("hysteresis holds are explored only once the cache is cold", () => {
215
214
  describe("exploration respects the remaining guards", () => {
216
215
  const cfg = withExploration({ enabled: true, rates: ALWAYS });
217
216
 
218
- test("never routes below the profile floor", () => {
217
+ test("never routes below the profile floor", async () => {
219
218
  const floored: ProfileConfig = { ...PROFILE, minTier: "moderate" };
220
219
  expect(run({ tier: "moderate", cfg, profile: floored }).explored).toBeNull();
221
220
  });
222
221
 
223
- test("skips forced escalations, which already proved the cheap tier failed", () => {
222
+ test("skips forced escalations, which already proved the cheap tier failed", async () => {
224
223
  expect(run({ tier: "moderate", cfg, source: "escalation" }).explored).toBeNull();
225
224
  });
226
225
 
227
- test("skips failover retries so a second confound is not introduced", () => {
226
+ test("skips failover retries so a second confound is not introduced", async () => {
228
227
  expect(run({ tier: "moderate", cfg, excludeSlugs: ["vendor/broken"] }).explored).toBeNull();
229
228
  });
230
229
  });
@@ -232,7 +231,7 @@ describe("exploration respects the remaining guards", () => {
232
231
  describe("exploration is deterministic", () => {
233
232
  const cfg = withExploration({ enabled: true, rates: { simple: 0.5, moderate: 0.5, hard: 0.5 } });
234
233
 
235
- test("the same turn always draws the same way, so explain can replay it", () => {
234
+ test("the same turn always draws the same way, so explain can replay it", async () => {
236
235
  for (const text of ["alpha task", "beta task", "gamma task"]) {
237
236
  const first = run({ userText: text, tier: "moderate", cfg });
238
237
  for (let i = 0; i < 5; i++) {
@@ -241,7 +240,7 @@ describe("exploration is deterministic", () => {
241
240
  }
242
241
  });
243
242
 
244
- test("the draw honours the configured rate across many turns", () => {
243
+ test("the draw honours the configured rate across many turns", async () => {
245
244
  const cfg25 = withExploration({ enabled: true, rates: { moderate: 0.25 } });
246
245
  let explored = 0;
247
246
  const N = 400;
@@ -1,8 +1,9 @@
1
1
  import { describe, expect, test } from "bun:test";
2
+ import { fakeLedger } from "./fakes.ts";
2
3
  import { createDisabledBridge } from "../src/context/bridge.ts";
3
4
  import type { CatalogSource } from "../src/catalog/types.ts";
4
5
  import type { EscalationConfig, RouterConfig } from "../src/config/types.ts";
5
- import { EMPTY_USAGE, type Ledger, type LedgerEntry, type UsageCounts } from "../src/cost/types.ts";
6
+ import { EMPTY_USAGE, type AsyncLedger, type LedgerEntry, type UsageCounts } from "../src/cost/types.ts";
6
7
  import type {
7
8
  ConversationState,
8
9
  ConversationStore,
@@ -245,29 +246,29 @@ function mkRouter(decisions: Decision[]): { router: Router; calls: RouteCall[] }
245
246
  return { router, calls };
246
247
  }
247
248
 
248
- function mkLedger(): { ledger: Ledger; entries: LedgerEntry[] } {
249
+ function mkLedger(): { ledger: AsyncLedger; entries: LedgerEntry[] } {
249
250
  const entries: LedgerEntry[] = [];
250
- const ledger: Ledger = {
251
- record: (e) => {
251
+ const ledger: AsyncLedger = fakeLedger({
252
+ record: async (e) => {
252
253
  entries.push(e);
253
254
  },
254
- conversationSpend: () => 0,
255
- spendSince: () => 0,
256
- blendedRate: () => null,
257
- latency: () => null,
258
- trust: () => null,
259
- allTrust: () => [],
260
- tokenRatio: () => null,
261
- recentEntries: () => [],
262
- };
255
+ conversationSpend: async () => 0,
256
+ spendSince: async () => 0,
257
+ blendedRate: async () => null,
258
+ latency: async () => null,
259
+ trust: async () => null,
260
+ allTrust: async () => [],
261
+ tokenRatio: async () => null,
262
+ recentEntries: async () => [],
263
+ });
263
264
  return { ledger, entries };
264
265
  }
265
266
 
266
267
  function mkConversations(): { store: ConversationStore; map: Map<string, ConversationState> } {
267
268
  const map = new Map<string, ConversationState>();
268
269
  const store: ConversationStore = {
269
- get: (k) => map.get(k) ?? null,
270
- load: (k) => {
270
+ get: async (k) => map.get(k) ?? null,
271
+ load: async (k) => {
271
272
  const existing = map.get(k);
272
273
  if (existing) return existing;
273
274
  const fresh: ConversationState = {
@@ -290,11 +291,11 @@ function mkConversations(): { store: ConversationStore; map: Map<string, Convers
290
291
  map.set(k, fresh);
291
292
  return fresh;
292
293
  },
293
- save: (s) => {
294
+ save: async (s) => {
294
295
  map.set(s.key, s);
295
296
  },
296
- accrue: () => {},
297
- prune: () => 0,
297
+ accrue: async () => {},
298
+ prune: async () => 0,
298
299
  };
299
300
  return { store, map };
300
301
  }
@@ -664,19 +665,19 @@ describe("same-tier failover", () => {
664
665
  });
665
666
 
666
667
  describe("400 classification (review 2026-09-05 follow-up)", () => {
667
- test("a 400 naming a model capability limit is retryable, so failover picks a sibling", () => {
668
+ test("a 400 naming a model capability limit is retryable, so failover picks a sibling", async () => {
668
669
  const e = classifyUpstreamStatus(400, { error: { message: "This model only supports single tool-calls at once!" } });
669
670
  expect(e.kind).toBe("invalid_request");
670
671
  expect(e.retryable).toBe(true);
671
672
  });
672
673
 
673
- test("a 400 for a malformed request stays non-retryable", () => {
674
+ test("a 400 for a malformed request stays non-retryable", async () => {
674
675
  const e = classifyUpstreamStatus(400, { error: { message: "messages[3].content: invalid type" } });
675
676
  expect(e.kind).toBe("invalid_request");
676
677
  expect(e.retryable).toBe(false);
677
678
  });
678
679
 
679
- test("a 400 for context overflow is still context_length", () => {
680
+ test("a 400 for context overflow is still context_length", async () => {
680
681
  const e = classifyUpstreamStatus(400, { error: { message: "This endpoint's maximum context length is 131072 tokens" } });
681
682
  expect(e.kind).toBe("context_length");
682
683
  expect(e.retryable).toBe(false);