auto-model-router 0.30.3 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +32 -2
- package/omp-extension/router-configure.ts +9 -7
- package/package.json +1 -1
- package/src/cli/config-cmd.ts +8 -7
- package/src/cli/explain.ts +10 -5
- package/src/cli/export.ts +6 -5
- package/src/cli/models.ts +10 -7
- package/src/cli/report.ts +6 -1
- package/src/cli/stats.ts +7 -7
- package/src/config/load.ts +10 -1
- package/src/config/types.ts +10 -1
- package/src/context/bridge.ts +7 -7
- package/src/context/index.ts +3 -3
- package/src/context/store.ts +39 -56
- package/src/context/types.ts +7 -6
- package/src/cost/blended.ts +28 -7
- package/src/cost/feedback.ts +33 -37
- package/src/cost/ledger-sql.ts +547 -0
- package/src/cost/ledger.ts +30 -459
- package/src/cost/report.ts +171 -129
- package/src/cost/retention.ts +10 -10
- package/src/cost/summary.ts +15 -10
- package/src/cost/types.ts +43 -62
- package/src/cost/views.ts +79 -49
- package/src/eval/calibrate.ts +47 -12
- package/src/eval/run.ts +18 -2
- package/src/lib.ts +6 -2
- package/src/router/candidates.ts +7 -15
- package/src/router/classify.ts +6 -4
- package/src/router/index.ts +95 -9
- package/src/router/select.ts +38 -21
- package/src/router/state.ts +90 -102
- package/src/router/types.ts +11 -5
- package/src/server/advise.ts +6 -4
- package/src/server/compaction-digest.ts +1 -1
- package/src/server/digest.ts +9 -10
- package/src/server/http.ts +109 -46
- package/src/server/providers.ts +18 -4
- package/src/server/turn.ts +32 -9
- package/src/tokens/estimate.ts +16 -6
- package/src/upstream/ollama-usage.ts +21 -11
- package/src/util/schema.ts +201 -0
- package/src/util/sql.ts +246 -0
- package/src/wire/anthropic/messages.ts +3 -4
- package/src/wire/openai/request.ts +1 -0
- package/src/wire/types.ts +7 -0
- package/test/anthropic-wire.test.ts +9 -9
- package/test/benchmark-feeds.test.ts +7 -7
- package/test/cache-control.test.ts +7 -7
- package/test/cache-estimate.test.ts +5 -5
- package/test/catalog-view.test.ts +4 -4
- package/test/catalog.test.ts +11 -11
- package/test/classify.test.ts +24 -24
- package/test/compaction.test.ts +20 -20
- package/test/config-wizard.test.ts +32 -32
- package/test/config.test.ts +10 -10
- package/test/connect-harnesses.test.ts +11 -11
- package/test/context-bridge.test.ts +40 -30
- package/test/context-prune.test.ts +43 -36
- package/test/context-query.test.ts +8 -8
- package/test/controls.test.ts +54 -27
- package/test/cost.test.ts +12 -12
- package/test/digest.test.ts +55 -44
- package/test/embed-lifecycle.test.ts +5 -5
- package/test/embed-logic.test.ts +26 -26
- package/test/escalate.test.ts +17 -17
- package/test/eval.test.ts +73 -16
- package/test/executable.test.ts +6 -6
- package/test/exploration.test.ts +19 -20
- package/test/failover.test.ts +22 -21
- package/test/fakes.ts +105 -0
- package/test/features.test.ts +21 -21
- package/test/harness-requests.test.ts +3 -3
- package/test/harness-switch.test.ts +5 -5
- package/test/hold-exploration.test.ts +13 -13
- package/test/hot-reload.test.ts +5 -5
- package/test/learned.test.ts +5 -5
- package/test/ledger-sql.test.ts +342 -0
- package/test/mcp-entry.test.ts +5 -5
- package/test/migrations.test.ts +28 -22
- package/test/models-yml.test.ts +18 -18
- package/test/ollama.test.ts +40 -34
- package/test/omp-credentials.test.ts +16 -16
- package/test/policy.test.ts +3 -3
- package/test/reconfigure.test.ts +4 -4
- package/test/redaction.test.ts +41 -35
- package/test/remote.test.ts +12 -12
- package/test/report-logic.test.ts +8 -8
- package/test/report.test.ts +95 -87
- package/test/retention.test.ts +79 -66
- package/test/schema.test.ts +123 -0
- package/test/scope.test.ts +8 -8
- package/test/select.test.ts +216 -257
- package/test/skills.test.ts +3 -3
- package/test/sql-shim.test.ts +154 -0
- package/test/state.test.ts +43 -36
- package/test/summary.test.ts +38 -27
- package/test/tier-plan.test.ts +45 -62
- package/test/toast-logic.test.ts +31 -31
- package/test/tokens.test.ts +95 -80
- package/test/trust-attribution.test.ts +217 -187
- package/test/trust-window.test.ts +37 -32
- package/test/turn.test.ts +55 -23
- package/test/upstreams.test.ts +13 -13
- package/test/views.test.ts +81 -59
- package/test/wire-request.test.ts +17 -17
- package/test/wire-responses.test.ts +4 -4
- package/tools/agentdox-e2e.ts +5 -2
- package/tools/export-benchmarks.ts +5 -5
- package/tools/ledger-parity.ts +266 -0
- package/tools/replay.ts +16 -8
|
@@ -23,7 +23,7 @@ function userBody(content: unknown): Record<string, unknown> {
|
|
|
23
23
|
}
|
|
24
24
|
|
|
25
25
|
describe("parseChatRequest normalization", () => {
|
|
26
|
-
test("string content and equivalent content-part array normalize to the same text", () => {
|
|
26
|
+
test("string content and equivalent content-part array normalize to the same text", async () => {
|
|
27
27
|
const fromString = parseChatRequest(userBody("hello world"), HEADERS);
|
|
28
28
|
const fromParts = parseChatRequest(
|
|
29
29
|
userBody([{ type: "text", text: "hello world" }]),
|
|
@@ -34,7 +34,7 @@ describe("parseChatRequest normalization", () => {
|
|
|
34
34
|
expect(fromParts.messages[0]!.images).toBe(0);
|
|
35
35
|
});
|
|
36
36
|
|
|
37
|
-
test("image parts are counted and flag hasImages", () => {
|
|
37
|
+
test("image parts are counted and flag hasImages", async () => {
|
|
38
38
|
const req = parseChatRequest(
|
|
39
39
|
userBody([
|
|
40
40
|
{ type: "text", text: "look at these" },
|
|
@@ -49,7 +49,7 @@ describe("parseChatRequest normalization", () => {
|
|
|
49
49
|
expect(parseChatRequest(userBody("plain"), HEADERS).hasImages).toBe(false);
|
|
50
50
|
});
|
|
51
51
|
|
|
52
|
-
test("provider prefix is stripped from model", () => {
|
|
52
|
+
test("provider prefix is stripped from model", async () => {
|
|
53
53
|
expect(parseChatRequest(userBody("hi"), HEADERS).requestedModel).toBe("auto");
|
|
54
54
|
const prefixed = parseChatRequest(
|
|
55
55
|
{ model: "auto-model-router/auto", messages: [{ role: "user", content: "hi" }] },
|
|
@@ -58,18 +58,18 @@ describe("parseChatRequest normalization", () => {
|
|
|
58
58
|
expect(prefixed.requestedModel).toBe("auto");
|
|
59
59
|
});
|
|
60
60
|
|
|
61
|
-
test("reads harness and omp session ids from headers, trimmed", () => {
|
|
61
|
+
test("reads harness and omp session ids from headers, trimmed", async () => {
|
|
62
62
|
const headers = new Headers({ "x-omp-harness": " prod-a ", "x-omp-session": " sess-1 " });
|
|
63
63
|
const req = parseChatRequest(userBody("hi"), headers);
|
|
64
64
|
expect(req.harnessId).toBe("prod-a");
|
|
65
65
|
expect(req.ompSessionId).toBe("sess-1");
|
|
66
66
|
});
|
|
67
67
|
|
|
68
|
-
test("omp session id defaults to empty when the header is absent", () => {
|
|
68
|
+
test("omp session id defaults to empty when the header is absent", async () => {
|
|
69
69
|
expect(parseChatRequest(userBody("hi"), HEADERS).ompSessionId).toBe("");
|
|
70
70
|
});
|
|
71
71
|
|
|
72
|
-
test("tool schemas, names, and descriptions contribute to promptBytes", () => {
|
|
72
|
+
test("tool schemas, names, and descriptions contribute to promptBytes", async () => {
|
|
73
73
|
const parameters = { type: "object", properties: { path: { type: "string" } } };
|
|
74
74
|
const withTools = parseChatRequest(
|
|
75
75
|
{
|
|
@@ -92,7 +92,7 @@ describe("parseChatRequest normalization", () => {
|
|
|
92
92
|
);
|
|
93
93
|
});
|
|
94
94
|
|
|
95
|
-
test("tool calls and tool results are carried into NormMessage", () => {
|
|
95
|
+
test("tool calls and tool results are carried into NormMessage", async () => {
|
|
96
96
|
const req = parseChatRequest(
|
|
97
97
|
{
|
|
98
98
|
model: "auto",
|
|
@@ -114,7 +114,7 @@ describe("parseChatRequest normalization", () => {
|
|
|
114
114
|
expect(req.messages[1]!.toolName).toBe("read");
|
|
115
115
|
});
|
|
116
116
|
|
|
117
|
-
test("forcedToolChoice only for objects and non-auto/none strings", () => {
|
|
117
|
+
test("forcedToolChoice only for objects and non-auto/none strings", async () => {
|
|
118
118
|
const base = userBody("hi");
|
|
119
119
|
expect(parseChatRequest(base, HEADERS).forcedToolChoice).toBe(false);
|
|
120
120
|
expect(parseChatRequest({ ...base, tool_choice: "auto" }, HEADERS).forcedToolChoice).toBe(false);
|
|
@@ -128,7 +128,7 @@ describe("parseChatRequest normalization", () => {
|
|
|
128
128
|
).toBe(true);
|
|
129
129
|
});
|
|
130
130
|
|
|
131
|
-
test("reasoning accepted from both spellings, omitted when absent", () => {
|
|
131
|
+
test("reasoning accepted from both spellings, omitted when absent", async () => {
|
|
132
132
|
expect(
|
|
133
133
|
parseChatRequest({ ...userBody("hi"), reasoning_effort: "high" }, HEADERS).reasoning,
|
|
134
134
|
).toBe("high");
|
|
@@ -142,7 +142,7 @@ describe("parseChatRequest normalization", () => {
|
|
|
142
142
|
expect("reasoning" in plain).toBe(false);
|
|
143
143
|
});
|
|
144
144
|
|
|
145
|
-
test("malformed input throws WireErrorException", () => {
|
|
145
|
+
test("malformed input throws WireErrorException", async () => {
|
|
146
146
|
expect(() => parseChatRequest(null, HEADERS)).toThrow(WireErrorException);
|
|
147
147
|
expect(() => parseChatRequest({ model: "auto", messages: [] }, HEADERS)).toThrow(WireErrorException);
|
|
148
148
|
expect(() => parseChatRequest({ messages: [{ role: "user", content: "hi" }] }, HEADERS)).toThrow(
|
|
@@ -163,7 +163,7 @@ describe("conversationKey", () => {
|
|
|
163
163
|
const system = { role: "system", content: "You are a coding agent." };
|
|
164
164
|
const first = { role: "user", content: "Fix the bug in main.ts" };
|
|
165
165
|
|
|
166
|
-
test("stable across later turns of the same conversation", () => {
|
|
166
|
+
test("stable across later turns of the same conversation", async () => {
|
|
167
167
|
const turn1 = parseChatRequest({ model: "auto", messages: [system, first] }, HEADERS);
|
|
168
168
|
const turn3 = parseChatRequest(
|
|
169
169
|
{
|
|
@@ -181,7 +181,7 @@ describe("conversationKey", () => {
|
|
|
181
181
|
expect(turn1.conversationKey).toMatch(/^[0-9a-f]{32}$/);
|
|
182
182
|
});
|
|
183
183
|
|
|
184
|
-
test("differs when the first non-system message differs", () => {
|
|
184
|
+
test("differs when the first non-system message differs", async () => {
|
|
185
185
|
const a = parseChatRequest({ model: "auto", messages: [system, first] }, HEADERS);
|
|
186
186
|
const b = parseChatRequest(
|
|
187
187
|
{ model: "auto", messages: [system, { role: "user", content: "Write a poem" }] },
|
|
@@ -192,7 +192,7 @@ describe("conversationKey", () => {
|
|
|
192
192
|
});
|
|
193
193
|
|
|
194
194
|
describe("renderUpstreamBody", () => {
|
|
195
|
-
test("two renders are independent and never mutate the original body", () => {
|
|
195
|
+
test("two renders are independent and never mutate the original body", async () => {
|
|
196
196
|
const original = {
|
|
197
197
|
model: "auto",
|
|
198
198
|
messages: [{ role: "user", content: "hi" }],
|
|
@@ -236,7 +236,7 @@ describe("renderUpstreamBody", () => {
|
|
|
236
236
|
expect(JSON.stringify(original)).toBe(before);
|
|
237
237
|
});
|
|
238
238
|
|
|
239
|
-
test("maxTokens lands on whichever spelling the client used", () => {
|
|
239
|
+
test("maxTokens lands on whichever spelling the client used", async () => {
|
|
240
240
|
const req = parseChatRequest(
|
|
241
241
|
{ model: "auto", messages: [{ role: "user", content: "hi" }], max_completion_tokens: 200 },
|
|
242
242
|
HEADERS,
|
|
@@ -246,13 +246,13 @@ describe("renderUpstreamBody", () => {
|
|
|
246
246
|
expect("max_tokens" in out).toBe(false);
|
|
247
247
|
});
|
|
248
248
|
|
|
249
|
-
test("reasoning off renders as { enabled: false }", () => {
|
|
249
|
+
test("reasoning off renders as { enabled: false }", async () => {
|
|
250
250
|
const req = parseChatRequest(userBody("hi"), HEADERS);
|
|
251
251
|
const out = req.renderUpstreamBody(mutations({ reasoning: "off" }));
|
|
252
252
|
expect(out.reasoning).toEqual({ enabled: false });
|
|
253
253
|
});
|
|
254
254
|
|
|
255
|
-
test("cache breakpoints land on named messages and promote string content to parts", () => {
|
|
255
|
+
test("cache breakpoints land on named messages and promote string content to parts", async () => {
|
|
256
256
|
const req = parseChatRequest(
|
|
257
257
|
{
|
|
258
258
|
model: "auto",
|
|
@@ -280,7 +280,7 @@ describe("renderUpstreamBody", () => {
|
|
|
280
280
|
]);
|
|
281
281
|
});
|
|
282
282
|
|
|
283
|
-
test("stripAssistantReasoning removes all three spellings from assistant messages only", () => {
|
|
283
|
+
test("stripAssistantReasoning removes all three spellings from assistant messages only", async () => {
|
|
284
284
|
const req = parseChatRequest(
|
|
285
285
|
{
|
|
286
286
|
model: "auto",
|
|
@@ -12,7 +12,7 @@ import type { StreamEvent, TurnSummary, UpstreamChunk } from "../src/wire/types.
|
|
|
12
12
|
const HEADERS = new Headers({ "X-Omp-Harness": "codex" });
|
|
13
13
|
|
|
14
14
|
describe("responsesToChatBody", () => {
|
|
15
|
-
test("instructions, messages, function calls and their outputs become chat messages", () => {
|
|
15
|
+
test("instructions, messages, function calls and their outputs become chat messages", async () => {
|
|
16
16
|
const chat = responsesToChatBody({
|
|
17
17
|
model: "auto",
|
|
18
18
|
instructions: "Be terse.",
|
|
@@ -59,7 +59,7 @@ describe("responsesToChatBody", () => {
|
|
|
59
59
|
for (const k of ["instructions", "input", "store", "include", "prompt_cache_key", "max_output_tokens"]) expect(k in chat).toBe(false);
|
|
60
60
|
});
|
|
61
61
|
|
|
62
|
-
test("a string input is one user message; images survive; previous_response_id is refused", () => {
|
|
62
|
+
test("a string input is one user message; images survive; previous_response_id is refused", async () => {
|
|
63
63
|
expect(responsesToChatBody({ model: "auto", input: "hi" }).messages).toEqual([{ role: "user", content: "hi" }]);
|
|
64
64
|
const withImage = responsesToChatBody({ model: "auto", input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "what is this" }, { type: "input_image", image_url: "data:image/png;base64,AAAA" }] }] });
|
|
65
65
|
expect(withImage.messages).toEqual([{ role: "user", content: [{ type: "text", text: "what is this" }, { type: "image_url", image_url: { url: "data:image/png;base64,AAAA" } }] }]);
|
|
@@ -67,7 +67,7 @@ describe("responsesToChatBody", () => {
|
|
|
67
67
|
expect(() => responsesToChatBody({ model: "auto", input: [] })).toThrow(WireErrorException);
|
|
68
68
|
});
|
|
69
69
|
|
|
70
|
-
test("Codex identity is read from the body when the headers carry none", () => {
|
|
70
|
+
test("Codex identity is read from the body when the headers carry none", async () => {
|
|
71
71
|
const meta = (agent: string) => JSON.stringify({ session_id: "t1", thread_id: "t1", agent_name: agent, turn_id: "u1" });
|
|
72
72
|
const body = { model: "auto", input: "x", prompt_cache_key: "t1", client_metadata: { thread_id: "t1", session_id: "t1", "x-codex-turn-metadata": meta("/root") } };
|
|
73
73
|
const main = parseResponsesRequest(body, HEADERS);
|
|
@@ -81,7 +81,7 @@ describe("responsesToChatBody", () => {
|
|
|
81
81
|
expect(identityHeadersFromBody({ model: "auto", input: "x", client_metadata: { "x-codex-turn-metadata": "not json" } }, HEADERS).get("x-omp-subagent")).toBeNull();
|
|
82
82
|
});
|
|
83
83
|
|
|
84
|
-
test("parseResponsesRequest yields a routed request tagged with the wire", () => {
|
|
84
|
+
test("parseResponsesRequest yields a routed request tagged with the wire", async () => {
|
|
85
85
|
const req = parseResponsesRequest({ model: "auto-model-router/auto-cheap", input: "hello", stream: false }, HEADERS);
|
|
86
86
|
expect(req.protocol).toBe("openai-responses");
|
|
87
87
|
expect(req.requestedModel).toBe("auto-cheap");
|
package/tools/agentdox-e2e.ts
CHANGED
|
@@ -14,7 +14,8 @@ import { createContextBridge } from "../src/context/bridge.ts";
|
|
|
14
14
|
import { createContextStore } from "../src/context/store.ts";
|
|
15
15
|
import type { ContextResolveInput } from "../src/context/types.ts";
|
|
16
16
|
import { createLogger } from "../src/util/log.ts";
|
|
17
|
-
import {
|
|
17
|
+
import { migrateStore } from "../src/util/schema.ts";
|
|
18
|
+
import { openSqlDb } from "../src/util/sql.ts";
|
|
18
19
|
|
|
19
20
|
const baseUrl = process.env.AGENTDOX_URL ?? "http://localhost:3003";
|
|
20
21
|
const token = process.env.AGENTDOX_TOKEN ?? "";
|
|
@@ -32,7 +33,9 @@ function check(name: string, ok: boolean, detail = ""): void {
|
|
|
32
33
|
}
|
|
33
34
|
|
|
34
35
|
const log = createLogger("warn");
|
|
35
|
-
|
|
36
|
+
// A temp file rather than `:memory:`: the shim opens one store per handle.
|
|
37
|
+
const db = openSqlDb(`${process.env.TMPDIR ?? "/tmp"}/agentdox-e2e-${Date.now()}.db`);
|
|
38
|
+
await migrateStore(db);
|
|
36
39
|
const client = createAgentDoxClient({ baseUrl, token, timeoutMs: 5_000, log });
|
|
37
40
|
const bridge = createContextBridge({
|
|
38
41
|
client,
|
|
@@ -22,9 +22,9 @@
|
|
|
22
22
|
import { existsSync, readFileSync, writeFileSync } from "node:fs";
|
|
23
23
|
import { join, resolve } from "node:path";
|
|
24
24
|
import { loadConfig } from "../src/config/load.ts";
|
|
25
|
-
import {
|
|
25
|
+
import { createSqlLedger } from "../src/cost/ledger-sql.ts";
|
|
26
26
|
import { computeStats } from "../src/server/http.ts";
|
|
27
|
-
import {
|
|
27
|
+
import { openSqlDb } from "../src/util/sql.ts";
|
|
28
28
|
|
|
29
29
|
const ROOT = resolve(import.meta.dir, "..");
|
|
30
30
|
const DATA_PATH = join(ROOT, "site", "data", "benchmarks.json");
|
|
@@ -56,9 +56,9 @@ if (!existsSync(dbPath)) {
|
|
|
56
56
|
process.exit(0);
|
|
57
57
|
}
|
|
58
58
|
|
|
59
|
-
const db =
|
|
59
|
+
const db = openSqlDb(dbPath);
|
|
60
60
|
try {
|
|
61
|
-
const stats = computeStats(
|
|
61
|
+
const stats = await computeStats(createSqlLedger(db, cfg, { findModel: () => null }), days === undefined ? {} : { windowDays: days });
|
|
62
62
|
const perTurnUsd = stats.requests > 0 ? stats.windowSpendUsd / stats.requests : 0;
|
|
63
63
|
data.ledgerSnapshot = {
|
|
64
64
|
generatedAt: new Date(stats.generatedAtMs).toISOString().slice(0, 10),
|
|
@@ -78,5 +78,5 @@ try {
|
|
|
78
78
|
writeFileSync(DATA_PATH, `${JSON.stringify(data, null, 2)}\n`, "utf8");
|
|
79
79
|
console.log(`ledgerSnapshot \u2190 ${stats.requests} turns from ${dbPath} (${stats.windowDays === null ? "all time" : `${stats.windowDays}d`})`);
|
|
80
80
|
} finally {
|
|
81
|
-
db.close();
|
|
81
|
+
await db.close();
|
|
82
82
|
}
|
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Parity: the one ledger must answer identically on both engines, reading the
|
|
3
|
+
* same production rows.
|
|
4
|
+
*
|
|
5
|
+
* The point is not that the SQL looks portable — it is that trust, latency,
|
|
6
|
+
* spend, blend, escalation cost and every report come out EQUAL, because
|
|
7
|
+
* routing decisions are made from those numbers and a front door may read a
|
|
8
|
+
* SQLite file while the router writes Postgres. Every silent bug found while
|
|
9
|
+
* porting (a `SUM()` read as a string skewing a trust score, a double-encoded
|
|
10
|
+
* JSON column nulling an escalation term, a nested `json_extract` path that
|
|
11
|
+
* SQLite tolerates and Postgres reads as NULL) was caught by comparing
|
|
12
|
+
* computed values rather than by a type error.
|
|
13
|
+
*
|
|
14
|
+
* Usage: bun tools/ledger-parity.ts <sqlite-path> [postgres-url]
|
|
15
|
+
*/
|
|
16
|
+
import { Database } from "bun:sqlite";
|
|
17
|
+
|
|
18
|
+
import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
|
|
19
|
+
import type { RouterConfig } from "../src/config/types.ts";
|
|
20
|
+
import { createSqlLedger } from "../src/cost/ledger-sql.ts";
|
|
21
|
+
import { migrateStore } from "../src/util/schema.ts";
|
|
22
|
+
import type { AsyncLedger } from "../src/cost/types.ts";
|
|
23
|
+
import { openSqlDb, type SqlDb } from "../src/util/sql.ts";
|
|
24
|
+
import { decisionEntries, exportRows, feedbackView, spendUsdSince } from "../src/cost/views.ts";
|
|
25
|
+
import { buildUsageReport } from "../src/cost/report.ts";
|
|
26
|
+
import { buildDailySummary, countTierChanges } from "../src/cost/summary.ts";
|
|
27
|
+
|
|
28
|
+
const sqlitePath = process.argv[2] ?? "/data/router/router.db";
|
|
29
|
+
const pgUrl = process.argv[3] ?? "";
|
|
30
|
+
|
|
31
|
+
// Windows wide open so every row counts, and the scoped query shapes exercised.
|
|
32
|
+
const cfg: RouterConfig = structuredClone(DEFAULT_CONFIG);
|
|
33
|
+
cfg.filters.trustWindowDays = 0;
|
|
34
|
+
cfg.filters.feedbackWeight = 1;
|
|
35
|
+
cfg.filters.cacheReliabilityMinSamples = 1;
|
|
36
|
+
cfg.filters.escalationCostWeight = 1;
|
|
37
|
+
cfg.filters.latencyMinSamples = 1;
|
|
38
|
+
|
|
39
|
+
// The source of real rows. Read-only: the harness copies out of it and
|
|
40
|
+
// never writes to a live ledger.
|
|
41
|
+
const db = new Database(sqlitePath, { readonly: true });
|
|
42
|
+
|
|
43
|
+
const slugs = (db.query("SELECT DISTINCT slug FROM ledger WHERE slug IS NOT NULL").all() as { slug: string }[]).map((r) => r.slug);
|
|
44
|
+
const harnesses = (db.query("SELECT DISTINCT harness_id FROM ledger WHERE harness_id <> '' LIMIT 2").all() as { harness_id: string }[]).map(
|
|
45
|
+
(r) => r.harness_id,
|
|
46
|
+
);
|
|
47
|
+
const sessions = (
|
|
48
|
+
db.query("SELECT DISTINCT omp_session_id FROM ledger WHERE omp_session_id <> '' LIMIT 2").all() as { omp_session_id: string }[]
|
|
49
|
+
).map((r) => r.omp_session_id);
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Key-order-insensitive comparison. A JSON column round-tripped through
|
|
53
|
+
* Postgres' jsonb comes back with its keys reordered — jsonb does not preserve
|
|
54
|
+
* input order — which is not a difference in the value.
|
|
55
|
+
*/
|
|
56
|
+
function stable(value: unknown): string {
|
|
57
|
+
return JSON.stringify(value, (_key, v: unknown) =>
|
|
58
|
+
v !== null && typeof v === "object" && !Array.isArray(v)
|
|
59
|
+
? Object.fromEntries(Object.entries(v as Record<string, unknown>).sort(([a], [b]) => a.localeCompare(b)))
|
|
60
|
+
: v,
|
|
61
|
+
);
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Structural comparison with a tolerance on every number, at any depth.
|
|
66
|
+
*
|
|
67
|
+
* Two engines summing the same rows in a different order land a few ULPs apart
|
|
68
|
+
* ($39.722225609861425 against ...56), which is arithmetic, not divergence. A
|
|
69
|
+
* strict compare on a nested total would report that as a mismatch and bury
|
|
70
|
+
* the real ones.
|
|
71
|
+
*/
|
|
72
|
+
function deepEqual(a: unknown, b: unknown, tol: number): boolean {
|
|
73
|
+
if (typeof a === "number" && typeof b === "number") return Math.abs(a - b) <= tol * Math.max(1, Math.abs(a));
|
|
74
|
+
if (a === null || b === null || typeof a !== "object" || typeof b !== "object") return stable(a) === stable(b);
|
|
75
|
+
if (Array.isArray(a) !== Array.isArray(b)) return false;
|
|
76
|
+
if (Array.isArray(a) && Array.isArray(b)) {
|
|
77
|
+
return a.length === b.length && a.every((v, i) => deepEqual(v, b[i], tol));
|
|
78
|
+
}
|
|
79
|
+
const ra = a as Record<string, unknown>;
|
|
80
|
+
const rb = b as Record<string, unknown>;
|
|
81
|
+
const keys = new Set([...Object.keys(ra), ...Object.keys(rb)]);
|
|
82
|
+
for (const key of keys) if (!deepEqual(ra[key], rb[key], tol)) return false;
|
|
83
|
+
return true;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
let checks = 0;
|
|
87
|
+
let bad = 0;
|
|
88
|
+
/** Which side is which in a mismatch report; set before each comparison pass. */
|
|
89
|
+
let labelA = "a";
|
|
90
|
+
let labelB = "b";
|
|
91
|
+
function eq(label: string, a: unknown, b: unknown, tol = 1e-9): void {
|
|
92
|
+
checks++;
|
|
93
|
+
const same = deepEqual(a, b, tol);
|
|
94
|
+
if (!same) {
|
|
95
|
+
bad++;
|
|
96
|
+
console.log(` MISMATCH ${label}\n ${labelA}=${JSON.stringify(a)}\n ${labelB}=${JSON.stringify(b)}`);
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/** Copies the real rows into whichever store the unified ledger will read. */
|
|
101
|
+
async function load(target: SqlDb): Promise<void> {
|
|
102
|
+
await migrateStore(target);
|
|
103
|
+
for (const table of ["ledger", "feedback", "token_calibration"]) {
|
|
104
|
+
await target.sql.unsafe(`DELETE FROM ${table}`);
|
|
105
|
+
}
|
|
106
|
+
// The json columns are TEXT in the source; Postgres wants objects, so they
|
|
107
|
+
// are parsed when the destination stores JSONB.
|
|
108
|
+
const jsonCols = new Set(["usage", "cost_breakdown", "features", "classifier_reasons", "reasons"]);
|
|
109
|
+
const copy = async (table: string): Promise<number> => {
|
|
110
|
+
const cols = (db.query(`PRAGMA table_info(${table})`).all() as { name: string }[]).map((c) => c.name);
|
|
111
|
+
const rows = db.query(`SELECT ${cols.join(", ")} FROM ${table}`).all() as Record<string, unknown>[];
|
|
112
|
+
for (let i = 0; i < rows.length; i += 500) {
|
|
113
|
+
const batch = rows.slice(i, i + 500).map((r) => {
|
|
114
|
+
const o: Record<string, unknown> = {};
|
|
115
|
+
for (const c of cols) {
|
|
116
|
+
const v = r[c] ?? null;
|
|
117
|
+
o[c] = target.dialect === "postgres" && jsonCols.has(c) && typeof v === "string" ? JSON.parse(v) : v;
|
|
118
|
+
}
|
|
119
|
+
return o;
|
|
120
|
+
});
|
|
121
|
+
await target.sql`INSERT INTO ${target.sql.unsafe(table)} ${target.sql(batch, ...cols)}`;
|
|
122
|
+
}
|
|
123
|
+
return rows.length;
|
|
124
|
+
};
|
|
125
|
+
const ledgerRows = await copy("ledger");
|
|
126
|
+
// tokenRatio reads this table, so without it the unified side reports null
|
|
127
|
+
// against a real calibrated ratio — another harness-shaped "mismatch".
|
|
128
|
+
await copy("token_calibration");
|
|
129
|
+
let feedbackRows = 0;
|
|
130
|
+
try {
|
|
131
|
+
feedbackRows = await copy("feedback");
|
|
132
|
+
} catch (err) {
|
|
133
|
+
// Never swallowed: a failed feedback copy makes the unified side read no
|
|
134
|
+
// verdicts and shows up as a trust-score "mismatch" that is really a
|
|
135
|
+
// harness bug. It cost a diagnosis once already.
|
|
136
|
+
console.log(` feedback copy FAILED: ${err instanceof Error ? err.message : String(err)}`);
|
|
137
|
+
}
|
|
138
|
+
console.log(` loaded ${ledgerRows} ledger rows, ${feedbackRows} feedback rows`);
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Every ported view, as one comparable value. The views moved to the shim in
|
|
143
|
+
* the same change as the ledger, so there is no synchronous version left to
|
|
144
|
+
* diff against — what matters is that the two ENGINES agree, since a front
|
|
145
|
+
* door may read a file while the router writes a database.
|
|
146
|
+
*/
|
|
147
|
+
async function viewsSnapshot(db: SqlDb): Promise<Record<string, unknown>> {
|
|
148
|
+
const harness = harnesses[0] ?? null;
|
|
149
|
+
// A FIXED instant: the windows are relative to "now", so two engines read a
|
|
150
|
+
// moment apart would legitimately disagree and hide a real difference.
|
|
151
|
+
const nowMs = 1_789_000_000_000;
|
|
152
|
+
const report = await buildUsageReport(db, { windowDays: 3650, nowMs });
|
|
153
|
+
const summary = await buildDailySummary(db, { nowMs });
|
|
154
|
+
return {
|
|
155
|
+
reportTotals: report.totals,
|
|
156
|
+
reportProviders: report.providers.map((r) => [r.key, r.dispatches, Number(r.spendUsd.toFixed(9)), Number(r.cacheHitRate.toFixed(9)), r.escalations, r.errors]),
|
|
157
|
+
reportModels: report.models.map((m) => [m.key, m.provider, m.dispatches, Number(m.spendUsd.toFixed(9)), m.tiers, m.feedback]),
|
|
158
|
+
reportTiers: report.tiers.map((r) => [r.key, r.dispatches, Number(r.spendUsd.toFixed(9))]),
|
|
159
|
+
reportDays: report.days.map((d) => [d.day, d.dispatches, Number(d.spendUsd.toFixed(9)), Number(d.cacheHitRate.toFixed(9))]),
|
|
160
|
+
reportAnatomy: report.anatomy,
|
|
161
|
+
summaryCurrent: summary.current,
|
|
162
|
+
summaryPrevious: summary.previous,
|
|
163
|
+
summaryTopModels: summary.topModels.map((m) => [m.slug, m.dispatches, Number(m.spendUsd.toFixed(9))]),
|
|
164
|
+
tierChanges: await countTierChanges(db, 0, ""),
|
|
165
|
+
spendAll: await spendUsdSince(db, 0, null),
|
|
166
|
+
spendHarness: harness === null ? 0 : await spendUsdSince(db, 0, [harness]),
|
|
167
|
+
spendNone: await spendUsdSince(db, 0, []),
|
|
168
|
+
exportRows: (await exportRows(db, 0, null)).map((r) => [r.day, r.harnessId, r.slug, r.scope, r.dispatches, r.promptTokens, r.cachedTokens, r.completionTokens, Number(r.spendUsd.toFixed(9)), r.escalations, r.errors]),
|
|
169
|
+
feedback: await feedbackView(db, 0, null),
|
|
170
|
+
decisions: (await decisionEntries(db, { sinceMs: 0, harness: null, limit: 25 })).map((e) => [e.id, e.slug, e.servedSlug, e.tier, e.wasted, e.feedback.length, e.costBreakdown?.total ?? null]),
|
|
171
|
+
};
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* Every signal the router reads off the ledger, as one comparable value.
|
|
176
|
+
*
|
|
177
|
+
* A FIXED instant throughout: `spendSince` and the spike windows are relative
|
|
178
|
+
* to "now", so two engines read a moment apart would legitimately disagree and
|
|
179
|
+
* hide a real difference.
|
|
180
|
+
*/
|
|
181
|
+
async function ledgerSnapshot(unified: AsyncLedger): Promise<Record<string, unknown>> {
|
|
182
|
+
const nowMs = 1_789_000_000_000;
|
|
183
|
+
const out: Record<string, unknown> = {};
|
|
184
|
+
|
|
185
|
+
for (const days of [0, 1, 7]) {
|
|
186
|
+
const since = days === 0 ? 0 : nowMs - days * 86_400_000;
|
|
187
|
+
out[`spendSince(${days}d)`] = await unified.spendSince(since);
|
|
188
|
+
for (const h of harnesses) out[`spendSince(${days}d,${h})`] = await unified.spendSince(since, h);
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
const esc = await unified.escalationCost(cfg.ledger.blendWindowDays);
|
|
192
|
+
out.escalationCost = esc === null ? null : [esc.samples, esc.usdPerPromptToken];
|
|
193
|
+
const blend = await unified.blendedRate(cfg.ledger.blendWindowDays);
|
|
194
|
+
out.blendedRate = blend === null ? null : [blend.sampleCount, blend.inputPerMtok, blend.outputPerMtok, blend.cacheReadPerMtok];
|
|
195
|
+
|
|
196
|
+
// The prefetch the turn path actually uses, global and per harness.
|
|
197
|
+
for (const harness of [undefined, ...harnesses]) {
|
|
198
|
+
const tag = harness ?? "global";
|
|
199
|
+
const signals = await unified.signals(slugs, harness);
|
|
200
|
+
for (const slug of slugs) {
|
|
201
|
+
const s = signals.get(slug);
|
|
202
|
+
const trust = s?.trust ?? null;
|
|
203
|
+
const latency = s?.latency ?? null;
|
|
204
|
+
out[`trust[${tag}][${slug}]`] = trust === null
|
|
205
|
+
? null
|
|
206
|
+
: [trust.attempts, trust.escalations, trust.errors, trust.successRate, trust.meanCostError];
|
|
207
|
+
out[`latency[${tag}][${slug}]`] = latency === null ? null : [latency.samples, latency.ttftMs, latency.tokensPerSec];
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
// Keyed by slug: row order is not part of the meaning here.
|
|
212
|
+
out.allTrust = Object.fromEntries((await unified.allTrust()).map((t) => [t.slug, [t.attempts, t.successRate]]));
|
|
213
|
+
const cache = await unified.cacheReliability(slugs);
|
|
214
|
+
out.cacheReliability = Object.fromEntries(slugs.map((s) => [s, cache.get(s) === undefined ? null : [cache.get(s)?.samples, cache.get(s)?.hitRate]]));
|
|
215
|
+
|
|
216
|
+
for (const prefix of ["ollama/", "deepseek/", "anthropic-subscription/"]) {
|
|
217
|
+
out[`providerSpendSince(${prefix})`] = await unified.providerSpendSince(prefix, 0);
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
// Entry round-trip: the JSON columns are where a dialect difference shows.
|
|
221
|
+
out.recentEntries = (await unified.recentEntries(25)).map((e) => [e.id, e.usage, e.reasons, e.costBreakdown ?? null, e.scope ?? null, e.wasted]);
|
|
222
|
+
|
|
223
|
+
for (const session of sessions) {
|
|
224
|
+
out[`latestForSession(${session})`] = (await unified.latestForSession(session))?.id ?? null;
|
|
225
|
+
out[`entriesForSession(${session})`] = (await unified.entriesForSession(session, 5)).map((e) => e.id);
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
out.softFailureSpikes = (await unified.softFailureSpikes(nowMs)).map((s) => [s.slug, s.recentRate, s.recentDispatches, s.baselineRate]);
|
|
229
|
+
for (const tokenizer of ["gpt", "claude", "qwen", "llama"]) {
|
|
230
|
+
out[`tokenRatio(${tokenizer})`] = await unified.tokenRatio(tokenizer);
|
|
231
|
+
}
|
|
232
|
+
return out;
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
const targets: { engine: string; url: string }[] = [
|
|
236
|
+
{ engine: "sqlite", url: `sqlite:///tmp/ledger-parity-${Date.now()}.db` },
|
|
237
|
+
...(pgUrl === "" ? [] : [{ engine: "postgres", url: pgUrl }]),
|
|
238
|
+
];
|
|
239
|
+
|
|
240
|
+
const snapshots = new Map<string, Record<string, unknown>>();
|
|
241
|
+
for (const target of targets) {
|
|
242
|
+
console.log(`\n=== loading ${target.engine} ===`);
|
|
243
|
+
const store = openSqlDb(target.url);
|
|
244
|
+
await load(store);
|
|
245
|
+
const ledger = createSqlLedger(store, cfg, { findModel: () => null });
|
|
246
|
+
snapshots.set(target.engine, { ...(await ledgerSnapshot(ledger)), ...(await viewsSnapshot(store)) });
|
|
247
|
+
await store.close();
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
// Every signal and every view, engine against engine. One engine alone still
|
|
251
|
+
// proves nothing was thrown: it loads the rows and computes the lot.
|
|
252
|
+
const [first, ...rest] = [...snapshots.keys()];
|
|
253
|
+
if (first !== undefined) {
|
|
254
|
+
const a = snapshots.get(first) as Record<string, unknown>;
|
|
255
|
+
labelA = first;
|
|
256
|
+
for (const other of rest) {
|
|
257
|
+
console.log(`\n=== ${other} vs ${first} ===`);
|
|
258
|
+
labelB = other;
|
|
259
|
+
const b = snapshots.get(other) as Record<string, unknown>;
|
|
260
|
+
for (const key of Object.keys(a)) eq(key, a[key], b[key], 1e-9);
|
|
261
|
+
}
|
|
262
|
+
if (rest.length === 0) console.log(`\n=== ${first} only: ${Object.keys(a).length} values computed, nothing to compare against ===`);
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
console.log(`\n${checks - bad}/${checks} checks equal${bad === 0 ? "" : `, ${bad} MISMATCHED`}`);
|
|
266
|
+
process.exit(bad === 0 ? 0 : 1);
|
package/tools/replay.ts
CHANGED
|
@@ -70,7 +70,9 @@ import type { CatalogModel, CatalogSnapshot } from "../src/catalog/types.ts";
|
|
|
70
70
|
import { loadConfig } from "../src/config/load.ts";
|
|
71
71
|
import type { RouterConfig } from "../src/config/types.ts";
|
|
72
72
|
import { computeCost } from "../src/cost/forecast.ts";
|
|
73
|
-
import {
|
|
73
|
+
import { createSqlLedger } from "../src/cost/ledger-sql.ts";
|
|
74
|
+
import { prefetchTurnReads } from "../src/router/index.ts";
|
|
75
|
+
import { openSqlDb } from "../src/util/sql.ts";
|
|
74
76
|
import type { UsageCounts } from "../src/cost/types.ts";
|
|
75
77
|
import { classifyTask, scoreHeuristic } from "../src/router/classify.ts";
|
|
76
78
|
import { resolveHoldTurns } from "../src/router/explore.ts";
|
|
@@ -324,7 +326,11 @@ function snapshotFor(cfg: RouterConfig): CatalogSnapshot {
|
|
|
324
326
|
const snapshotA = snapshotFor(cfgA);
|
|
325
327
|
const snapshotB = snapshotFor(cfgB);
|
|
326
328
|
const bySlug = new Map([...snapshotA.models, ...snapshotB.models].map((m) => [m.slug, m]));
|
|
327
|
-
|
|
329
|
+
// Replay reads through the engine-agnostic handle: `select` takes the ledger's
|
|
330
|
+
// answers as DATA now, so each replayed turn prefetches the same reads a live
|
|
331
|
+
// turn would. Read-only — nothing here writes.
|
|
332
|
+
const sqlDb = openSqlDb(dbPath);
|
|
333
|
+
const ledger = createSqlLedger(sqlDb, cfgA, { findModel: (slug: string) => bySlug.get(slug) ?? null });
|
|
328
334
|
|
|
329
335
|
const predicate = args.where === "" ? "" : ` AND (${args.where})`;
|
|
330
336
|
// Newest-first to honour --limit, then flipped to chronological so each row can
|
|
@@ -371,7 +377,7 @@ interface Outcome {
|
|
|
371
377
|
trail: VariantTrail;
|
|
372
378
|
}
|
|
373
379
|
|
|
374
|
-
function run(
|
|
380
|
+
async function run(
|
|
375
381
|
cfg: RouterConfig,
|
|
376
382
|
snapshot: CatalogSnapshot,
|
|
377
383
|
row: Row,
|
|
@@ -379,7 +385,7 @@ function run(
|
|
|
379
385
|
prior: PriorTurn | undefined,
|
|
380
386
|
trail: VariantTrail | undefined,
|
|
381
387
|
esc: EscalationContext,
|
|
382
|
-
): Outcome {
|
|
388
|
+
): Promise<Outcome> {
|
|
383
389
|
const f = featuresOf(row, usage.promptTokens);
|
|
384
390
|
const req = requestOf(row, f);
|
|
385
391
|
const state = stateOf(row, prior, trail, args.warmth);
|
|
@@ -399,14 +405,16 @@ function run(
|
|
|
399
405
|
} else {
|
|
400
406
|
classification = scoreHeuristic(f, cfg);
|
|
401
407
|
}
|
|
408
|
+
const profile = profileOf(cfg, row.requested_model);
|
|
409
|
+
const reads = await prefetchTurnReads(ledger, req, profile, cfg, snapshot, classification.task, row.created_at_ms);
|
|
402
410
|
const decision: Decision = select({
|
|
403
411
|
req,
|
|
404
412
|
features: f,
|
|
405
413
|
classification,
|
|
406
|
-
profile
|
|
414
|
+
profile,
|
|
407
415
|
state,
|
|
408
416
|
snapshot,
|
|
409
|
-
|
|
417
|
+
reads,
|
|
410
418
|
cfg,
|
|
411
419
|
// The row's own clock: cache warmth and hold windows are judged against
|
|
412
420
|
// when the turn happened, not against today. Passing Date.now() here made
|
|
@@ -500,8 +508,8 @@ for (const row of rows) {
|
|
|
500
508
|
excludeSlugs: wasted.map((w) => w.served_slug ?? w.slug),
|
|
501
509
|
};
|
|
502
510
|
if (esc.escalateFrom !== undefined) escalationsReplayed++;
|
|
503
|
-
const a = run(cfgA, snapshotA, row, u, prior, trailA.get(row.conversation_key), esc);
|
|
504
|
-
const b = run(cfgB, snapshotB, row, u, prior, trailB.get(row.conversation_key), esc);
|
|
511
|
+
const a = await run(cfgA, snapshotA, row, u, prior, trailA.get(row.conversation_key), esc);
|
|
512
|
+
const b = await run(cfgB, snapshotB, row, u, prior, trailB.get(row.conversation_key), esc);
|
|
505
513
|
trailA.set(row.conversation_key, a.trail);
|
|
506
514
|
trailB.set(row.conversation_key, b.trail);
|
|
507
515
|
|