auto-model-router 0.30.3 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +32 -2
- package/omp-extension/router-configure.ts +9 -7
- package/package.json +1 -1
- package/src/cli/config-cmd.ts +8 -7
- package/src/cli/explain.ts +10 -5
- package/src/cli/export.ts +6 -5
- package/src/cli/models.ts +10 -7
- package/src/cli/report.ts +6 -1
- package/src/cli/stats.ts +7 -7
- package/src/config/load.ts +10 -1
- package/src/config/types.ts +10 -1
- package/src/context/bridge.ts +7 -7
- package/src/context/index.ts +3 -3
- package/src/context/store.ts +39 -56
- package/src/context/types.ts +7 -6
- package/src/cost/blended.ts +28 -7
- package/src/cost/feedback.ts +33 -37
- package/src/cost/ledger-sql.ts +547 -0
- package/src/cost/ledger.ts +30 -459
- package/src/cost/report.ts +171 -129
- package/src/cost/retention.ts +10 -10
- package/src/cost/summary.ts +15 -10
- package/src/cost/types.ts +43 -62
- package/src/cost/views.ts +79 -49
- package/src/eval/calibrate.ts +47 -12
- package/src/eval/run.ts +18 -2
- package/src/lib.ts +6 -2
- package/src/router/candidates.ts +7 -15
- package/src/router/classify.ts +6 -4
- package/src/router/index.ts +95 -9
- package/src/router/select.ts +38 -21
- package/src/router/state.ts +90 -102
- package/src/router/types.ts +11 -5
- package/src/server/advise.ts +6 -4
- package/src/server/compaction-digest.ts +1 -1
- package/src/server/digest.ts +9 -10
- package/src/server/http.ts +109 -46
- package/src/server/providers.ts +18 -4
- package/src/server/turn.ts +32 -9
- package/src/tokens/estimate.ts +16 -6
- package/src/upstream/ollama-usage.ts +21 -11
- package/src/util/schema.ts +201 -0
- package/src/util/sql.ts +246 -0
- package/src/wire/anthropic/messages.ts +3 -4
- package/src/wire/openai/request.ts +1 -0
- package/src/wire/types.ts +7 -0
- package/test/anthropic-wire.test.ts +9 -9
- package/test/benchmark-feeds.test.ts +7 -7
- package/test/cache-control.test.ts +7 -7
- package/test/cache-estimate.test.ts +5 -5
- package/test/catalog-view.test.ts +4 -4
- package/test/catalog.test.ts +11 -11
- package/test/classify.test.ts +24 -24
- package/test/compaction.test.ts +20 -20
- package/test/config-wizard.test.ts +32 -32
- package/test/config.test.ts +10 -10
- package/test/connect-harnesses.test.ts +11 -11
- package/test/context-bridge.test.ts +40 -30
- package/test/context-prune.test.ts +43 -36
- package/test/context-query.test.ts +8 -8
- package/test/controls.test.ts +54 -27
- package/test/cost.test.ts +12 -12
- package/test/digest.test.ts +55 -44
- package/test/embed-lifecycle.test.ts +5 -5
- package/test/embed-logic.test.ts +26 -26
- package/test/escalate.test.ts +17 -17
- package/test/eval.test.ts +73 -16
- package/test/executable.test.ts +6 -6
- package/test/exploration.test.ts +19 -20
- package/test/failover.test.ts +22 -21
- package/test/fakes.ts +105 -0
- package/test/features.test.ts +21 -21
- package/test/harness-requests.test.ts +3 -3
- package/test/harness-switch.test.ts +5 -5
- package/test/hold-exploration.test.ts +13 -13
- package/test/hot-reload.test.ts +5 -5
- package/test/learned.test.ts +5 -5
- package/test/ledger-sql.test.ts +342 -0
- package/test/mcp-entry.test.ts +5 -5
- package/test/migrations.test.ts +28 -22
- package/test/models-yml.test.ts +18 -18
- package/test/ollama.test.ts +40 -34
- package/test/omp-credentials.test.ts +16 -16
- package/test/policy.test.ts +3 -3
- package/test/reconfigure.test.ts +4 -4
- package/test/redaction.test.ts +41 -35
- package/test/remote.test.ts +12 -12
- package/test/report-logic.test.ts +8 -8
- package/test/report.test.ts +95 -87
- package/test/retention.test.ts +79 -66
- package/test/schema.test.ts +123 -0
- package/test/scope.test.ts +8 -8
- package/test/select.test.ts +216 -257
- package/test/skills.test.ts +3 -3
- package/test/sql-shim.test.ts +154 -0
- package/test/state.test.ts +43 -36
- package/test/summary.test.ts +38 -27
- package/test/tier-plan.test.ts +45 -62
- package/test/toast-logic.test.ts +31 -31
- package/test/tokens.test.ts +95 -80
- package/test/trust-attribution.test.ts +217 -187
- package/test/trust-window.test.ts +37 -32
- package/test/turn.test.ts +55 -23
- package/test/upstreams.test.ts +13 -13
- package/test/views.test.ts +81 -59
- package/test/wire-request.test.ts +17 -17
- package/test/wire-responses.test.ts +4 -4
- package/tools/agentdox-e2e.ts +5 -2
- package/tools/export-benchmarks.ts +5 -5
- package/tools/ledger-parity.ts +266 -0
- package/tools/replay.ts +16 -8
package/test/escalate.test.ts
CHANGED
|
@@ -57,14 +57,14 @@ function toolDelta(
|
|
|
57
57
|
}
|
|
58
58
|
|
|
59
59
|
describe("createProbe", () => {
|
|
60
|
-
test("valid tool-call JSON commits", () => {
|
|
60
|
+
test("valid tool-call JSON commits", async () => {
|
|
61
61
|
const p = createProbe(plan(), req(), ALL_TRIGGERS);
|
|
62
62
|
expect(p.observe(toolDelta(0, { id: "c1", name: "read", argsDelta: '{"path":"a' }))).toBeNull();
|
|
63
63
|
const verdict = p.observe(toolDelta(0, { argsDelta: '.ts"}' }));
|
|
64
64
|
expect(verdict?.action).toBe("commit");
|
|
65
65
|
});
|
|
66
66
|
|
|
67
|
-
test("truncated tool-call JSON at stream end yields malformed_tool_args", () => {
|
|
67
|
+
test("truncated tool-call JSON at stream end yields malformed_tool_args", async () => {
|
|
68
68
|
// Via the finish event.
|
|
69
69
|
const p1 = createProbe(plan(), req(), ALL_TRIGGERS);
|
|
70
70
|
p1.observe(toolDelta(0, { id: "c1", name: "read", argsDelta: '{"path":"a' }));
|
|
@@ -80,7 +80,7 @@ describe("createProbe", () => {
|
|
|
80
80
|
expect(v2).toMatchObject({ signal: "malformed_tool_args" });
|
|
81
81
|
});
|
|
82
82
|
|
|
83
|
-
test("a tool call identical to the previous assistant call yields repeat_tool_call", () => {
|
|
83
|
+
test("a tool call identical to the previous assistant call yields repeat_tool_call", async () => {
|
|
84
84
|
const history: NormMessage[] = [
|
|
85
85
|
{ role: "user", text: "read it", images: 0, textBytes: 8, toolCalls: [] },
|
|
86
86
|
{
|
|
@@ -99,7 +99,7 @@ describe("createProbe", () => {
|
|
|
99
99
|
expect(verdict).toMatchObject({ signal: "repeat_tool_call" });
|
|
100
100
|
});
|
|
101
101
|
|
|
102
|
-
test("a different tool call is not a repeat", () => {
|
|
102
|
+
test("a different tool call is not a repeat", async () => {
|
|
103
103
|
const history: NormMessage[] = [
|
|
104
104
|
{
|
|
105
105
|
role: "assistant",
|
|
@@ -114,14 +114,14 @@ describe("createProbe", () => {
|
|
|
114
114
|
expect(verdict?.action).toBe("commit");
|
|
115
115
|
});
|
|
116
116
|
|
|
117
|
-
test("a disabled plan commits on the first chunk", () => {
|
|
117
|
+
test("a disabled plan commits on the first chunk", async () => {
|
|
118
118
|
const p = createProbe(plan({ enabled: false }), req(), ALL_TRIGGERS);
|
|
119
119
|
const verdict = p.observe(text("anything at all"));
|
|
120
120
|
expect(verdict?.action).toBe("commit");
|
|
121
121
|
expect(p.held()).toHaveLength(1);
|
|
122
122
|
});
|
|
123
123
|
|
|
124
|
-
test("a signal absent from triggers never fires", () => {
|
|
124
|
+
test("a signal absent from triggers never fires", async () => {
|
|
125
125
|
// Truncated args would be malformed_tool_args, but the trigger is off.
|
|
126
126
|
const p1 = createProbe(plan(), req(), new Set(["refusal"]));
|
|
127
127
|
p1.observe(toolDelta(0, { id: "c1", name: "read", argsDelta: '{"path":"a' }));
|
|
@@ -135,14 +135,14 @@ describe("createProbe", () => {
|
|
|
135
135
|
expect(v2.action).toBe("commit");
|
|
136
136
|
});
|
|
137
137
|
|
|
138
|
-
test("refusal openers escalate as soon as text arrives", () => {
|
|
138
|
+
test("refusal openers escalate as soon as text arrives", async () => {
|
|
139
139
|
const p = createProbe(plan(), req(), ALL_TRIGGERS);
|
|
140
140
|
const verdict = p.observe(text("I'm sorry, but I can't help with that request."));
|
|
141
141
|
expect(verdict?.action).toBe("escalate");
|
|
142
142
|
expect(verdict).toMatchObject({ signal: "refusal" });
|
|
143
143
|
});
|
|
144
144
|
|
|
145
|
-
test("a forced tool choice answered with prose yields missing_expected_tool_call", () => {
|
|
145
|
+
test("a forced tool choice answered with prose yields missing_expected_tool_call", async () => {
|
|
146
146
|
const tools: NormTool[] = [{ name: "read", description: "read a file", schemaBytes: 42 }];
|
|
147
147
|
const p = createProbe(plan(), req([], { tools, forcedToolChoice: true }), ALL_TRIGGERS);
|
|
148
148
|
p.observe(text("Sure, here is some prose instead."));
|
|
@@ -151,20 +151,20 @@ describe("createProbe", () => {
|
|
|
151
151
|
expect(verdict).toMatchObject({ signal: "missing_expected_tool_call" });
|
|
152
152
|
});
|
|
153
153
|
|
|
154
|
-
test("enough held text commits", () => {
|
|
154
|
+
test("enough held text commits", async () => {
|
|
155
155
|
const p = createProbe(plan({ maxTokens: 2 }), req(), ALL_TRIGGERS);
|
|
156
156
|
const verdict = p.observe(text("this is well over eight characters"));
|
|
157
157
|
expect(verdict?.action).toBe("commit");
|
|
158
158
|
});
|
|
159
159
|
|
|
160
|
-
test("an empty stop with nothing emitted yields empty_completion", () => {
|
|
160
|
+
test("an empty stop with nothing emitted yields empty_completion", async () => {
|
|
161
161
|
const p = createProbe(plan(), req(), ALL_TRIGGERS);
|
|
162
162
|
const verdict = p.observe(chunk([{ type: "finish", reason: "stop" }]));
|
|
163
163
|
expect(verdict?.action).toBe("escalate");
|
|
164
164
|
expect(verdict).toMatchObject({ signal: "empty_completion" });
|
|
165
165
|
});
|
|
166
166
|
|
|
167
|
-
test("a stalled stream escalates at the hold ceiling instead of committing silence", () => {
|
|
167
|
+
test("a stalled stream escalates at the hold ceiling instead of committing silence", async () => {
|
|
168
168
|
let t = 0;
|
|
169
169
|
const p = createProbe(plan({ maxHoldMs: 1_000 }), req(), ALL_TRIGGERS, () => t);
|
|
170
170
|
expect(p.observe(chunk([]))).toBeNull();
|
|
@@ -174,7 +174,7 @@ describe("createProbe", () => {
|
|
|
174
174
|
expect(verdict).toMatchObject({ signal: "empty_completion" });
|
|
175
175
|
});
|
|
176
176
|
|
|
177
|
-
test("the hold ceiling still commits when content has arrived", () => {
|
|
177
|
+
test("the hold ceiling still commits when content has arrived", async () => {
|
|
178
178
|
let t = 0;
|
|
179
179
|
const p = createProbe(plan({ maxTokens: 1_000, maxHoldMs: 1_000 }), req(), ALL_TRIGGERS, () => t);
|
|
180
180
|
expect(p.observe(text("partial answer"))).toBeNull();
|
|
@@ -183,7 +183,7 @@ describe("createProbe", () => {
|
|
|
183
183
|
expect(verdict?.action).toBe("commit");
|
|
184
184
|
});
|
|
185
185
|
|
|
186
|
-
test("a length finish on prose commits: that is the caller's max_tokens", () => {
|
|
186
|
+
test("a length finish on prose commits: that is the caller's max_tokens", async () => {
|
|
187
187
|
// Escalating cannot fix it — the retry runs under the same cap and
|
|
188
188
|
// truncates in the same place, so it would just bill twice.
|
|
189
189
|
const p = createProbe(plan({ maxTokens: 1_000 }), req(), ALL_TRIGGERS);
|
|
@@ -192,7 +192,7 @@ describe("createProbe", () => {
|
|
|
192
192
|
expect(verdict?.action).toBe("commit");
|
|
193
193
|
});
|
|
194
194
|
|
|
195
|
-
test("a length finish that truncated tool-call arguments still escalates", () => {
|
|
195
|
+
test("a length finish that truncated tool-call arguments still escalates", async () => {
|
|
196
196
|
// Structurally unusable output: another model may emit a complete call.
|
|
197
197
|
const p = createProbe(plan({ maxTokens: 1_000 }), req(), ALL_TRIGGERS);
|
|
198
198
|
expect(p.observe(toolDelta(0, { id: "c1", name: "read", argsDelta: '{"path":"a' }))).toBeNull();
|
|
@@ -200,14 +200,14 @@ describe("createProbe", () => {
|
|
|
200
200
|
expect(verdict?.action).toBe("escalate");
|
|
201
201
|
});
|
|
202
202
|
|
|
203
|
-
test("a length finish having produced nothing escalates as an empty completion", () => {
|
|
203
|
+
test("a length finish having produced nothing escalates as an empty completion", async () => {
|
|
204
204
|
const p = createProbe(plan({ maxTokens: 1_000 }), req(), ALL_TRIGGERS);
|
|
205
205
|
const verdict = p.observe(chunk([{ type: "finish", reason: "length" }]));
|
|
206
206
|
expect(verdict?.action).toBe("escalate");
|
|
207
207
|
if (verdict?.action === "escalate") expect(verdict.signal).toBe("empty_completion");
|
|
208
208
|
});
|
|
209
209
|
|
|
210
|
-
test("reasoning-only output counts as alive at the hold ceiling", () => {
|
|
210
|
+
test("reasoning-only output counts as alive at the hold ceiling", async () => {
|
|
211
211
|
// A reasoning model that has emitted only reasoning tokens after the
|
|
212
212
|
// ceiling is working normally; escalating would discard a healthy paid
|
|
213
213
|
// generation.
|
|
@@ -219,7 +219,7 @@ describe("createProbe", () => {
|
|
|
219
219
|
expect(verdict?.action).toBe("commit");
|
|
220
220
|
});
|
|
221
221
|
|
|
222
|
-
test("a stream that ENDS with only reasoning is still hollow", () => {
|
|
222
|
+
test("a stream that ENDS with only reasoning is still hollow", async () => {
|
|
223
223
|
const p = createProbe(plan({ maxTokens: 1_000 }), req(), ALL_TRIGGERS);
|
|
224
224
|
expect(p.observe(chunk([{ type: "reasoning", delta: "thinking" }]))).toBeNull();
|
|
225
225
|
const verdict = p.verdictOnEnd();
|
package/test/eval.test.ts
CHANGED
|
@@ -3,34 +3,35 @@ import { describe, expect, test } from "bun:test";
|
|
|
3
3
|
import { normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
|
|
4
4
|
import { applyFeedScores, loadLocalScores, saveLocalScores, type FeedScore } from "../src/catalog/benchmark-feeds.ts";
|
|
5
5
|
import { answerScore, extractJson, isRefusalOrEmpty, jsonField, tokenCoverage } from "../src/eval/grade.ts";
|
|
6
|
-
import { applyFit, fitAxis, fitCalibration, pickAnchors, toLocalFeedScores, MIN_ANCHORS } from "../src/eval/calibrate.ts";
|
|
6
|
+
import { applyFit, fitAxis, fitCalibration, hardRaw, pickAnchors, toLocalFeedScores, MIN_ANCHORS, MIN_R, PUBLISH_MIN_R } from "../src/eval/calibrate.ts";
|
|
7
7
|
import { runEval, type EvalResult } from "../src/eval/run.ts";
|
|
8
8
|
import { EVAL_TASKS } from "../src/eval/tasks.ts";
|
|
9
|
+
import type { QualityAxis } from "../src/config/types.ts";
|
|
9
10
|
import { makeJudge, parseScore } from "../src/eval/judge.ts";
|
|
10
11
|
import type { EvalTask, JudgedTask } from "../src/eval/tasks.ts";
|
|
11
12
|
import { openDb } from "../src/util/sqlite.ts";
|
|
12
13
|
|
|
13
14
|
describe("grade helpers", () => {
|
|
14
|
-
test("answerScore matches whole reply, last line, or a standalone token", () => {
|
|
15
|
+
test("answerScore matches whole reply, last line, or a standalone token", async () => {
|
|
15
16
|
expect(answerScore("9.9", "9.9")).toBe(1);
|
|
16
17
|
expect(answerScore("The answer is 9.9", "9.9")).toBe(1);
|
|
17
18
|
expect(answerScore("reasoning...\n9.9", "9.9")).toBe(1);
|
|
18
19
|
expect(answerScore("19.99", "9.9")).toBe(0); // not a substring match
|
|
19
20
|
expect(answerScore("", "9.9")).toBe(0);
|
|
20
21
|
});
|
|
21
|
-
test("tokenCoverage is the fraction of tokens present", () => {
|
|
22
|
+
test("tokenCoverage is the fraction of tokens present", async () => {
|
|
22
23
|
expect(tokenCoverage("return a + b;", ["a + b"])).toBe(1);
|
|
23
24
|
expect(tokenCoverage("n * 2", ["n", "*", "2"])).toBe(1);
|
|
24
25
|
expect(tokenCoverage("n plus two", ["n", "*", "2"])).toBeCloseTo(1 / 3);
|
|
25
26
|
});
|
|
26
|
-
test("extractJson tolerates fences and prose; jsonField reads a key", () => {
|
|
27
|
+
test("extractJson tolerates fences and prose; jsonField reads a key", async () => {
|
|
27
28
|
expect(extractJson('here: {"answer": 8} ok')).toEqual({ answer: 8 });
|
|
28
29
|
expect(extractJson("```json\n[2,3,5]\n```")).toEqual([2, 3, 5]);
|
|
29
30
|
expect(extractJson("no json here")).toBeUndefined();
|
|
30
31
|
expect(jsonField({ tool: "read_file" }, "tool")).toBe("read_file");
|
|
31
32
|
expect(jsonField([1, 2], "tool")).toBeUndefined();
|
|
32
33
|
});
|
|
33
|
-
test("isRefusalOrEmpty flags empties and refusals", () => {
|
|
34
|
+
test("isRefusalOrEmpty flags empties and refusals", async () => {
|
|
34
35
|
expect(isRefusalOrEmpty("")).toBe(true);
|
|
35
36
|
expect(isRefusalOrEmpty("I cannot help with that")).toBe(true);
|
|
36
37
|
expect(isRefusalOrEmpty("sure, here")).toBe(false);
|
|
@@ -38,7 +39,7 @@ describe("grade helpers", () => {
|
|
|
38
39
|
});
|
|
39
40
|
|
|
40
41
|
describe("calibration", () => {
|
|
41
|
-
test("fitAxis is OLS, needs MIN_ANCHORS points and some spread", () => {
|
|
42
|
+
test("fitAxis is OLS, needs MIN_ANCHORS points and some spread", async () => {
|
|
42
43
|
const fit = fitAxis([
|
|
43
44
|
{ raw: 0.2, aa: 40 },
|
|
44
45
|
{ raw: 0.5, aa: 60 },
|
|
@@ -55,7 +56,7 @@ describe("calibration", () => {
|
|
|
55
56
|
expect(fitAxis([{ raw: 0.8, aa: 40 }, { raw: 0.5, aa: 60 }, { raw: 0.2, aa: 80 }])).toBeNull();
|
|
56
57
|
});
|
|
57
58
|
|
|
58
|
-
test("pickAnchors spreads over the score range, skips the target and the unscored", () => {
|
|
59
|
+
test("pickAnchors spreads over the score range, skips the target and the unscored", async () => {
|
|
59
60
|
const m = (slug: string, coding: number | undefined, supportsTools = true) => ({ slug, quality: coding === undefined ? {} : { coding }, supportsTools });
|
|
60
61
|
const catalog = [m("a/10", 10), m("a/30", 30), m("a/50", 50), m("a/70", 70), m("a/90", 90), m("a/target", undefined), m("a/notools", 60, false)];
|
|
61
62
|
const picked = pickAnchors(catalog, "a/target");
|
|
@@ -100,7 +101,7 @@ describe("calibration", () => {
|
|
|
100
101
|
expect(many!.byComplexity.easy).toBeUndefined();
|
|
101
102
|
});
|
|
102
103
|
|
|
103
|
-
test("the suite spans complexities, and hard items are not all pinned at the ceiling", () => {
|
|
104
|
+
test("the suite spans complexities, and hard items are not all pinned at the ceiling", async () => {
|
|
104
105
|
const bands = new Set(EVAL_TASKS.map((t) => t.complexity ?? "easy"));
|
|
105
106
|
expect(bands.has("easy")).toBe(true);
|
|
106
107
|
expect(bands.has("hard")).toBe(true);
|
|
@@ -127,12 +128,12 @@ describe("calibration", () => {
|
|
|
127
128
|
expect(hard.find((t) => t.id === "coding/regex-backtrack")!.grade("XX\nab\nfalse\nx|y\na[b$]c")).toBe(1);
|
|
128
129
|
});
|
|
129
130
|
|
|
130
|
-
test("fitCalibration + toLocalFeedScores place a target on the AA scale", () => {
|
|
131
|
+
test("fitCalibration + toLocalFeedScores place a target on the AA scale", async () => {
|
|
131
132
|
expect(MIN_ANCHORS).toBe(3);
|
|
132
133
|
const anchors: EvalResult[] = [
|
|
133
|
-
{ slug: "a/one", axes: { coding: { sum: 0.2, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {} },
|
|
134
|
-
{ slug: "a/two", axes: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {} },
|
|
135
|
-
{ slug: "a/three", axes: { coding: { sum: 0.8, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {} },
|
|
134
|
+
{ slug: "a/one", axes: { coding: { sum: 0.2, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {}, axesHard: { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } } },
|
|
135
|
+
{ slug: "a/two", axes: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {}, axesHard: { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } } },
|
|
136
|
+
{ slug: "a/three", axes: { coding: { sum: 0.8, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {}, axesHard: { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } } },
|
|
136
137
|
];
|
|
137
138
|
const aaOf: Record<string, number> = { "a/one": 40, "a/two": 60, "a/three": 80 };
|
|
138
139
|
const cal = fitCalibration(anchors, (slug, axis) => (axis === "coding" ? aaOf[slug] : undefined));
|
|
@@ -140,7 +141,7 @@ describe("calibration", () => {
|
|
|
140
141
|
expect(cal.intelligence).toBeUndefined(); // no anchor data on that axis
|
|
141
142
|
|
|
142
143
|
const targets: EvalResult[] = [
|
|
143
|
-
{ slug: "z/gap", axes: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0.9, n: 1 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {} },
|
|
144
|
+
{ slug: "z/gap", axes: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0.9, n: 1 }, agentic: { sum: 0, n: 0 } }, errors: 0, repeats: 1, spread: {}, byComplexity: {}, axesHard: { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } } },
|
|
144
145
|
];
|
|
145
146
|
const local = toLocalFeedScores(targets, cal, (s) => s.slice(0, s.indexOf("/")));
|
|
146
147
|
expect(local).toHaveLength(1);
|
|
@@ -148,6 +149,62 @@ describe("calibration", () => {
|
|
|
148
149
|
expect(local[0]!.coding).toBeCloseTo(60, 5); // calibrated from raw 0.5
|
|
149
150
|
expect(local[0]!.intelligence).toBeUndefined(); // axis had no fit, so not emitted
|
|
150
151
|
});
|
|
152
|
+
|
|
153
|
+
test("calibrating on the hard band beats calibrating on everything", async () => {
|
|
154
|
+
// Three anchors published 20/50/80 apart. On the FULL suite they all score ~0.97
|
|
155
|
+
// because easy and moderate pin everyone at the ceiling; on the hard band alone they
|
|
156
|
+
// separate. Same models, same publishing, different x — and only one of them can fit.
|
|
157
|
+
const mk = (slug: string, full: number, hard: number): EvalResult => ({
|
|
158
|
+
slug,
|
|
159
|
+
axes: { coding: { sum: full, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } },
|
|
160
|
+
axesHard: { coding: { sum: hard, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } },
|
|
161
|
+
errors: 0,
|
|
162
|
+
repeats: 1,
|
|
163
|
+
spread: {},
|
|
164
|
+
byComplexity: {},
|
|
165
|
+
});
|
|
166
|
+
const anchors = [mk("a/low", 0.96, 0.2), mk("a/mid", 0.97, 0.5), mk("a/high", 0.98, 0.8)];
|
|
167
|
+
const aa: Record<string, number> = { "a/low": 20, "a/mid": 50, "a/high": 80 };
|
|
168
|
+
const published = (slug: string, axis: QualityAxis) => (axis === "coding" ? aa[slug] : undefined);
|
|
169
|
+
const target = mk("z/target", 0.97, 0.5);
|
|
170
|
+
|
|
171
|
+
const onHard = toLocalFeedScores([target], fitCalibration(anchors, published, hardRaw), () => "z", hardRaw);
|
|
172
|
+
// The hard band spans 0.2-0.8 against 20-80, so the fit is a real line: raw 0.5 ⇒ ~50.
|
|
173
|
+
expect(onHard[0]!.coding).toBeCloseTo(50, 0);
|
|
174
|
+
|
|
175
|
+
// On the full suite the anchors span 0.96-0.98: the same published spread compressed into
|
|
176
|
+
// a fiftieth of the range. That is the shallow slope that produced 22.8 for a model
|
|
177
|
+
// published at 39.5, so the fit must be refused rather than published.
|
|
178
|
+
const pooledFit = fitCalibration(anchors, published);
|
|
179
|
+
const pooled = toLocalFeedScores([target], pooledFit, () => "z");
|
|
180
|
+
expect(pooled).toEqual([]);
|
|
181
|
+
});
|
|
182
|
+
|
|
183
|
+
test("a weak fit is refused, not published", async () => {
|
|
184
|
+
// Points with a real but noisy relationship: computable (r >= MIN_R) yet not worth
|
|
185
|
+
// acting on. `r` and `n` used to be computed and then thrown away.
|
|
186
|
+
const noisy = [
|
|
187
|
+
{ raw: 0.1, aa: 20 },
|
|
188
|
+
{ raw: 0.5, aa: 70 },
|
|
189
|
+
{ raw: 0.6, aa: 30 },
|
|
190
|
+
{ raw: 0.9, aa: 60 },
|
|
191
|
+
];
|
|
192
|
+
const fit = fitAxis(noisy)!;
|
|
193
|
+
expect(fit.r).toBeGreaterThanOrEqual(MIN_R);
|
|
194
|
+
expect(fit.r).toBeLessThan(PUBLISH_MIN_R);
|
|
195
|
+
const target: EvalResult = {
|
|
196
|
+
slug: "z/t",
|
|
197
|
+
axes: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } },
|
|
198
|
+
axesHard: { coding: { sum: 0.5, n: 1 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } },
|
|
199
|
+
errors: 0,
|
|
200
|
+
repeats: 1,
|
|
201
|
+
spread: {},
|
|
202
|
+
byComplexity: {},
|
|
203
|
+
};
|
|
204
|
+
expect(toLocalFeedScores([target], { coding: fit }, () => "z", hardRaw)).toEqual([]);
|
|
205
|
+
// A caller that deliberately lowers the bar still can, so the gate is policy not dogma.
|
|
206
|
+
expect(toLocalFeedScores([target], { coding: fit }, () => "z", hardRaw, MIN_R)[0]!.coding).toBeGreaterThan(0);
|
|
207
|
+
});
|
|
151
208
|
});
|
|
152
209
|
|
|
153
210
|
describe("runEval", () => {
|
|
@@ -195,7 +252,7 @@ describe("local source integration", () => {
|
|
|
195
252
|
};
|
|
196
253
|
}
|
|
197
254
|
|
|
198
|
-
test("local fills only where no stronger source has the axis", () => {
|
|
255
|
+
test("local fills only where no stronger source has the axis", async () => {
|
|
199
256
|
const catalog = [raw("z-ai/glm-5.3-flash")];
|
|
200
257
|
const feeds: FeedScore[] = [
|
|
201
258
|
{ key: "glm-5-3-flash", creator: "z-ai", source: "artificial_analysis", coding: 61 },
|
|
@@ -209,7 +266,7 @@ describe("local source integration", () => {
|
|
|
209
266
|
expect(result.sources.artificial_analysis).toBe(1);
|
|
210
267
|
});
|
|
211
268
|
|
|
212
|
-
test("saveLocalScores / loadLocalScores round-trip", () => {
|
|
269
|
+
test("saveLocalScores / loadLocalScores round-trip", async () => {
|
|
213
270
|
const db = openDb(":memory:");
|
|
214
271
|
const scores: FeedScore[] = [{ key: "muse-glimmer-30b", creator: "meta", source: "local", coding: 42, agentic: 39 }];
|
|
215
272
|
saveLocalScores(db, scores, 123);
|
|
@@ -219,7 +276,7 @@ describe("local source integration", () => {
|
|
|
219
276
|
});
|
|
220
277
|
|
|
221
278
|
describe("llm judge", () => {
|
|
222
|
-
test("parseScore takes the last standalone 0-10 and scales to 0-1", () => {
|
|
279
|
+
test("parseScore takes the last standalone 0-10 and scales to 0-1", async () => {
|
|
223
280
|
expect(parseScore("8")).toBeCloseTo(0.8, 5);
|
|
224
281
|
expect(parseScore("Score: 10/10")).toBeCloseTo(1, 5);
|
|
225
282
|
expect(parseScore("I count 3 issues, so 7")).toBeCloseTo(0.7, 5); // last wins
|
package/test/executable.test.ts
CHANGED
|
@@ -10,7 +10,7 @@ import { parseRemoteRouter } from "../omp-extension/remote-logic.ts";
|
|
|
10
10
|
const NL = String.fromCharCode(10);
|
|
11
11
|
|
|
12
12
|
describe("the single-file member install", () => {
|
|
13
|
-
test("the embedded package is the install: CLI, harness integrations, runtime deps; no tests, no caches", () => {
|
|
13
|
+
test("the embedded package is the install: CLI, harness integrations, runtime deps; no tests, no caches", async () => {
|
|
14
14
|
const pkg = collectPackageFiles(process.cwd());
|
|
15
15
|
expect(pkg.version).toBe((JSON.parse(readFileSync("package.json", "utf8")) as { version: string }).version);
|
|
16
16
|
expect(pkg.files["src/index.ts"]).toContain("connect");
|
|
@@ -28,7 +28,7 @@ describe("the single-file member install", () => {
|
|
|
28
28
|
expect(names.some((n) => /\.d\.ts$/.test(n) && n.startsWith("node_modules/"))).toBe(false);
|
|
29
29
|
});
|
|
30
30
|
|
|
31
|
-
test("targets and file names", () => {
|
|
31
|
+
test("targets and file names", async () => {
|
|
32
32
|
expect(isExecutableTarget("linux-x64")).toBe(true);
|
|
33
33
|
expect(isExecutableTarget("linux-x86")).toBe(false);
|
|
34
34
|
expect(executableFileName("windows-x64")).toBe("auto-model-router-windows-x64.exe");
|
|
@@ -44,7 +44,7 @@ describe("the single-file member install", () => {
|
|
|
44
44
|
expect(await readEmbeddedPackage()).toBeNull();
|
|
45
45
|
});
|
|
46
46
|
|
|
47
|
-
test("materializing writes the files once, keyed by content, under the router home", () => {
|
|
47
|
+
test("materializing writes the files once, keyed by content, under the router home", async () => {
|
|
48
48
|
const home = mkdtempSync(join(tmpdir(), "amr-mat-"));
|
|
49
49
|
try {
|
|
50
50
|
const pkg = { version: "9.9.9", files: { "package.json": `{"version":"9.9.9"}${NL}`, "src/index.ts": `console.log(1)${NL}`, "omp-extension/x.ts": "export {}" } };
|
|
@@ -63,7 +63,7 @@ describe("the single-file member install", () => {
|
|
|
63
63
|
}
|
|
64
64
|
});
|
|
65
65
|
|
|
66
|
-
test("connect from the executable: it is Claude Code's key helper, goes on PATH, and remote.json names it", () => {
|
|
66
|
+
test("connect from the executable: it is Claude Code's key helper, goes on PATH, and remote.json names it", async () => {
|
|
67
67
|
const home = mkdtempSync(join(tmpdir(), "amr-exe-connect-"));
|
|
68
68
|
const claude = join(home, ".claude");
|
|
69
69
|
mkdirSync(claude, { recursive: true });
|
|
@@ -98,7 +98,7 @@ describe("the single-file member install", () => {
|
|
|
98
98
|
expect(seen[0]?.url).toBe("https://team.example/setup/exchange");
|
|
99
99
|
expect(JSON.parse(seen[0]?.body ?? "{}")).toEqual({ token: "amrs_t", device: "laptop" });
|
|
100
100
|
const refused = (async () => Response.json({ error: "invalid_token" }, { status: 401 })) as unknown as typeof fetch;
|
|
101
|
-
await expect(exchangeSetupToken("https://team.example", "amrs_old", "laptop", refused)).rejects.toThrow("refused");
|
|
101
|
+
(await expect(exchangeSetupToken("https://team.example", "amrs_old", "laptop", refused))).rejects.toThrow("refused");
|
|
102
102
|
});
|
|
103
103
|
|
|
104
104
|
test("the executable builds for this host and knows its version from the embedded package", async () => {
|
|
@@ -118,7 +118,7 @@ describe("the single-file member install", () => {
|
|
|
118
118
|
}
|
|
119
119
|
}, 120_000);
|
|
120
120
|
|
|
121
|
-
test("the global handle is what marks a compiled process", () => {
|
|
121
|
+
test("the global handle is what marks a compiled process", async () => {
|
|
122
122
|
const g = globalThis as Record<string, unknown>;
|
|
123
123
|
g[EMBEDDED_GLOBAL] = { manifestPath: "/$bunfs/root/manifest.json" };
|
|
124
124
|
try {
|
package/test/exploration.test.ts
CHANGED
|
@@ -103,7 +103,6 @@ function run(opts: {
|
|
|
103
103
|
profile: opts.profile ?? PROFILE,
|
|
104
104
|
state: opts.st ?? state(),
|
|
105
105
|
snapshot: SNAPSHOT,
|
|
106
|
-
ledger: null,
|
|
107
106
|
cfg,
|
|
108
107
|
nowMs: NOW,
|
|
109
108
|
...(opts.excludeSlugs === undefined ? {} : { excludeSlugs: opts.excludeSlugs }),
|
|
@@ -113,14 +112,14 @@ function run(opts: {
|
|
|
113
112
|
const ALWAYS = { simple: 1, moderate: 1, hard: 1 };
|
|
114
113
|
|
|
115
114
|
describe("exploration is opt-in", () => {
|
|
116
|
-
test("never fires under the shipped defaults", () => {
|
|
115
|
+
test("never fires under the shipped defaults", async () => {
|
|
117
116
|
expect(BASE.exploration.enabled).toBe(false);
|
|
118
117
|
for (const tier of ["simple", "moderate", "hard"] as Tier[]) {
|
|
119
118
|
expect(run({ tier }).explored).toBeNull();
|
|
120
119
|
}
|
|
121
120
|
});
|
|
122
121
|
|
|
123
|
-
test("enabled with no rates configured still never fires", () => {
|
|
122
|
+
test("enabled with no rates configured still never fires", async () => {
|
|
124
123
|
const cfg = withExploration({ enabled: true, rates: {} });
|
|
125
124
|
for (const tier of ["simple", "moderate", "hard"] as Tier[]) {
|
|
126
125
|
expect(run({ tier, cfg }).explored).toBeNull();
|
|
@@ -129,20 +128,20 @@ describe("exploration is opt-in", () => {
|
|
|
129
128
|
});
|
|
130
129
|
|
|
131
130
|
describe("per-tier rates", () => {
|
|
132
|
-
test("each tier is governed by its own rate, not one global one", () => {
|
|
131
|
+
test("each tier is governed by its own rate, not one global one", async () => {
|
|
133
132
|
const cfg = withExploration({ enabled: true, rates: { simple: 0, moderate: 0, hard: 1 } });
|
|
134
133
|
expect(run({ tier: "simple", cfg }).explored).toBeNull();
|
|
135
134
|
expect(run({ tier: "moderate", cfg }).explored).toBeNull();
|
|
136
135
|
expect(run({ tier: "hard", cfg }).explored).toEqual({ from: "hard", to: "moderate" });
|
|
137
136
|
});
|
|
138
137
|
|
|
139
|
-
test("a tier absent from the rates map is never explored", () => {
|
|
138
|
+
test("a tier absent from the rates map is never explored", async () => {
|
|
140
139
|
const cfg = withExploration({ enabled: true, rates: { hard: 1 } });
|
|
141
140
|
expect(run({ tier: "moderate", cfg }).explored).toBeNull();
|
|
142
141
|
expect(run({ tier: "hard", cfg }).explored).not.toBeNull();
|
|
143
142
|
});
|
|
144
143
|
|
|
145
|
-
test("the shipped defaults weight expensive tiers far above cheap ones", () => {
|
|
144
|
+
test("the shipped defaults weight expensive tiers far above cheap ones", async () => {
|
|
146
145
|
const r = BASE.exploration.rates;
|
|
147
146
|
expect(r.hard ?? 0).toBeGreaterThan(r.simple ?? 0);
|
|
148
147
|
expect(r.moderate ?? 0).toBeGreaterThan(r.simple ?? 0);
|
|
@@ -152,19 +151,19 @@ describe("per-tier rates", () => {
|
|
|
152
151
|
describe("exploration drops exactly one tier", () => {
|
|
153
152
|
const cfg = withExploration({ enabled: true, rates: ALWAYS });
|
|
154
153
|
|
|
155
|
-
test("moderate explores down to simple", () => {
|
|
154
|
+
test("moderate explores down to simple", async () => {
|
|
156
155
|
expect(run({ tier: "moderate", cfg }).explored).toEqual({ from: "moderate", to: "simple" });
|
|
157
156
|
});
|
|
158
157
|
|
|
159
|
-
test("hard explores down to moderate, never further", () => {
|
|
158
|
+
test("hard explores down to moderate, never further", async () => {
|
|
160
159
|
expect(run({ tier: "hard", cfg }).explored).toEqual({ from: "hard", to: "moderate" });
|
|
161
160
|
});
|
|
162
161
|
|
|
163
|
-
test("trivial is the floor and cannot be explored below", () => {
|
|
162
|
+
test("trivial is the floor and cannot be explored below", async () => {
|
|
164
163
|
expect(run({ tier: "trivial", cfg }).explored).toBeNull();
|
|
165
164
|
});
|
|
166
165
|
|
|
167
|
-
test("the decision trail says so out loud", () => {
|
|
166
|
+
test("the decision trail says so out loud", async () => {
|
|
168
167
|
expect(run({ tier: "moderate", cfg }).reasons.some((r) => r.startsWith("exploration:"))).toBe(true);
|
|
169
168
|
});
|
|
170
169
|
});
|
|
@@ -172,11 +171,11 @@ describe("exploration drops exactly one tier", () => {
|
|
|
172
171
|
describe("hysteresis holds are explored only once the cache is cold", () => {
|
|
173
172
|
const cfg = withExploration({ enabled: true, rates: ALWAYS, stickyPolicy: "cold-cache" });
|
|
174
173
|
|
|
175
|
-
test("a held tier with a WARM cache is left alone", () => {
|
|
174
|
+
test("a held tier with a WARM cache is left alone", async () => {
|
|
176
175
|
expect(run({ tier: "simple", cfg, st: heldState("hard", "warm") }).explored).toBeNull();
|
|
177
176
|
});
|
|
178
177
|
|
|
179
|
-
test("a held tier with a COLD cache is explorable", () => {
|
|
178
|
+
test("a held tier with a COLD cache is explorable", async () => {
|
|
180
179
|
// This is the population that carries most of the spend: turns that
|
|
181
180
|
// reach hard by hold rather than by classification.
|
|
182
181
|
expect(run({ tier: "simple", cfg, st: heldState("hard", "cold") }).explored).toEqual({
|
|
@@ -185,18 +184,18 @@ describe("hysteresis holds are explored only once the cache is cold", () => {
|
|
|
185
184
|
});
|
|
186
185
|
});
|
|
187
186
|
|
|
188
|
-
test("the reason names the hold and the cache state, for later analysis", () => {
|
|
187
|
+
test("the reason names the hold and the cache state, for later analysis", async () => {
|
|
189
188
|
const d = run({ tier: "simple", cfg, st: heldState("hard", "cold") });
|
|
190
189
|
expect(d.reasons.some((r) => r.includes("held tier (cold cache)"))).toBe(true);
|
|
191
190
|
});
|
|
192
191
|
|
|
193
|
-
test("stickyPolicy never leaves holds alone entirely", () => {
|
|
192
|
+
test("stickyPolicy never leaves holds alone entirely", async () => {
|
|
194
193
|
const off = withExploration({ enabled: true, rates: ALWAYS, stickyPolicy: "never" });
|
|
195
194
|
expect(run({ tier: "simple", cfg: off, st: heldState("hard", "cold") }).explored).toBeNull();
|
|
196
195
|
expect(run({ tier: "simple", cfg: off, st: heldState("hard", "warm") }).explored).toBeNull();
|
|
197
196
|
});
|
|
198
197
|
|
|
199
|
-
test("stickyPolicy always reaches held turns even with a live cache", () => {
|
|
198
|
+
test("stickyPolicy always reaches held turns even with a live cache", async () => {
|
|
200
199
|
// The only setting that samples the population carrying most of the
|
|
201
200
|
// spend, at the price of a forfeited cache read.
|
|
202
201
|
const always = withExploration({ enabled: true, rates: ALWAYS, stickyPolicy: "always" });
|
|
@@ -215,16 +214,16 @@ describe("hysteresis holds are explored only once the cache is cold", () => {
|
|
|
215
214
|
describe("exploration respects the remaining guards", () => {
|
|
216
215
|
const cfg = withExploration({ enabled: true, rates: ALWAYS });
|
|
217
216
|
|
|
218
|
-
test("never routes below the profile floor", () => {
|
|
217
|
+
test("never routes below the profile floor", async () => {
|
|
219
218
|
const floored: ProfileConfig = { ...PROFILE, minTier: "moderate" };
|
|
220
219
|
expect(run({ tier: "moderate", cfg, profile: floored }).explored).toBeNull();
|
|
221
220
|
});
|
|
222
221
|
|
|
223
|
-
test("skips forced escalations, which already proved the cheap tier failed", () => {
|
|
222
|
+
test("skips forced escalations, which already proved the cheap tier failed", async () => {
|
|
224
223
|
expect(run({ tier: "moderate", cfg, source: "escalation" }).explored).toBeNull();
|
|
225
224
|
});
|
|
226
225
|
|
|
227
|
-
test("skips failover retries so a second confound is not introduced", () => {
|
|
226
|
+
test("skips failover retries so a second confound is not introduced", async () => {
|
|
228
227
|
expect(run({ tier: "moderate", cfg, excludeSlugs: ["vendor/broken"] }).explored).toBeNull();
|
|
229
228
|
});
|
|
230
229
|
});
|
|
@@ -232,7 +231,7 @@ describe("exploration respects the remaining guards", () => {
|
|
|
232
231
|
describe("exploration is deterministic", () => {
|
|
233
232
|
const cfg = withExploration({ enabled: true, rates: { simple: 0.5, moderate: 0.5, hard: 0.5 } });
|
|
234
233
|
|
|
235
|
-
test("the same turn always draws the same way, so explain can replay it", () => {
|
|
234
|
+
test("the same turn always draws the same way, so explain can replay it", async () => {
|
|
236
235
|
for (const text of ["alpha task", "beta task", "gamma task"]) {
|
|
237
236
|
const first = run({ userText: text, tier: "moderate", cfg });
|
|
238
237
|
for (let i = 0; i < 5; i++) {
|
|
@@ -241,7 +240,7 @@ describe("exploration is deterministic", () => {
|
|
|
241
240
|
}
|
|
242
241
|
});
|
|
243
242
|
|
|
244
|
-
test("the draw honours the configured rate across many turns", () => {
|
|
243
|
+
test("the draw honours the configured rate across many turns", async () => {
|
|
245
244
|
const cfg25 = withExploration({ enabled: true, rates: { moderate: 0.25 } });
|
|
246
245
|
let explored = 0;
|
|
247
246
|
const N = 400;
|
package/test/failover.test.ts
CHANGED
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
import { describe, expect, test } from "bun:test";
|
|
2
|
+
import { fakeLedger } from "./fakes.ts";
|
|
2
3
|
import { createDisabledBridge } from "../src/context/bridge.ts";
|
|
3
4
|
import type { CatalogSource } from "../src/catalog/types.ts";
|
|
4
5
|
import type { EscalationConfig, RouterConfig } from "../src/config/types.ts";
|
|
5
|
-
import { EMPTY_USAGE, type
|
|
6
|
+
import { EMPTY_USAGE, type AsyncLedger, type LedgerEntry, type UsageCounts } from "../src/cost/types.ts";
|
|
6
7
|
import type {
|
|
7
8
|
ConversationState,
|
|
8
9
|
ConversationStore,
|
|
@@ -245,29 +246,29 @@ function mkRouter(decisions: Decision[]): { router: Router; calls: RouteCall[] }
|
|
|
245
246
|
return { router, calls };
|
|
246
247
|
}
|
|
247
248
|
|
|
248
|
-
function mkLedger(): { ledger:
|
|
249
|
+
function mkLedger(): { ledger: AsyncLedger; entries: LedgerEntry[] } {
|
|
249
250
|
const entries: LedgerEntry[] = [];
|
|
250
|
-
const ledger:
|
|
251
|
-
record: (e) => {
|
|
251
|
+
const ledger: AsyncLedger = fakeLedger({
|
|
252
|
+
record: async (e) => {
|
|
252
253
|
entries.push(e);
|
|
253
254
|
},
|
|
254
|
-
conversationSpend: () => 0,
|
|
255
|
-
spendSince: () => 0,
|
|
256
|
-
blendedRate: () => null,
|
|
257
|
-
latency: () => null,
|
|
258
|
-
trust: () => null,
|
|
259
|
-
allTrust: () => [],
|
|
260
|
-
tokenRatio: () => null,
|
|
261
|
-
recentEntries: () => [],
|
|
262
|
-
};
|
|
255
|
+
conversationSpend: async () => 0,
|
|
256
|
+
spendSince: async () => 0,
|
|
257
|
+
blendedRate: async () => null,
|
|
258
|
+
latency: async () => null,
|
|
259
|
+
trust: async () => null,
|
|
260
|
+
allTrust: async () => [],
|
|
261
|
+
tokenRatio: async () => null,
|
|
262
|
+
recentEntries: async () => [],
|
|
263
|
+
});
|
|
263
264
|
return { ledger, entries };
|
|
264
265
|
}
|
|
265
266
|
|
|
266
267
|
function mkConversations(): { store: ConversationStore; map: Map<string, ConversationState> } {
|
|
267
268
|
const map = new Map<string, ConversationState>();
|
|
268
269
|
const store: ConversationStore = {
|
|
269
|
-
get: (k) => map.get(k) ?? null,
|
|
270
|
-
load: (k) => {
|
|
270
|
+
get: async (k) => map.get(k) ?? null,
|
|
271
|
+
load: async (k) => {
|
|
271
272
|
const existing = map.get(k);
|
|
272
273
|
if (existing) return existing;
|
|
273
274
|
const fresh: ConversationState = {
|
|
@@ -290,11 +291,11 @@ function mkConversations(): { store: ConversationStore; map: Map<string, Convers
|
|
|
290
291
|
map.set(k, fresh);
|
|
291
292
|
return fresh;
|
|
292
293
|
},
|
|
293
|
-
save: (s) => {
|
|
294
|
+
save: async (s) => {
|
|
294
295
|
map.set(s.key, s);
|
|
295
296
|
},
|
|
296
|
-
accrue: () => {},
|
|
297
|
-
prune: () => 0,
|
|
297
|
+
accrue: async () => {},
|
|
298
|
+
prune: async () => 0,
|
|
298
299
|
};
|
|
299
300
|
return { store, map };
|
|
300
301
|
}
|
|
@@ -664,19 +665,19 @@ describe("same-tier failover", () => {
|
|
|
664
665
|
});
|
|
665
666
|
|
|
666
667
|
describe("400 classification (review 2026-09-05 follow-up)", () => {
|
|
667
|
-
test("a 400 naming a model capability limit is retryable, so failover picks a sibling", () => {
|
|
668
|
+
test("a 400 naming a model capability limit is retryable, so failover picks a sibling", async () => {
|
|
668
669
|
const e = classifyUpstreamStatus(400, { error: { message: "This model only supports single tool-calls at once!" } });
|
|
669
670
|
expect(e.kind).toBe("invalid_request");
|
|
670
671
|
expect(e.retryable).toBe(true);
|
|
671
672
|
});
|
|
672
673
|
|
|
673
|
-
test("a 400 for a malformed request stays non-retryable", () => {
|
|
674
|
+
test("a 400 for a malformed request stays non-retryable", async () => {
|
|
674
675
|
const e = classifyUpstreamStatus(400, { error: { message: "messages[3].content: invalid type" } });
|
|
675
676
|
expect(e.kind).toBe("invalid_request");
|
|
676
677
|
expect(e.retryable).toBe(false);
|
|
677
678
|
});
|
|
678
679
|
|
|
679
|
-
test("a 400 for context overflow is still context_length", () => {
|
|
680
|
+
test("a 400 for context overflow is still context_length", async () => {
|
|
680
681
|
const e = classifyUpstreamStatus(400, { error: { message: "This endpoint's maximum context length is 131072 tokens" } });
|
|
681
682
|
expect(e.kind).toBe("context_length");
|
|
682
683
|
expect(e.retryable).toBe(false);
|