auto-model-router 0.30.3 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/.omp-plugin/marketplace.json +2 -2
  2. package/README.md +32 -2
  3. package/omp-extension/router-configure.ts +9 -7
  4. package/package.json +1 -1
  5. package/src/cli/config-cmd.ts +8 -7
  6. package/src/cli/explain.ts +10 -5
  7. package/src/cli/export.ts +6 -5
  8. package/src/cli/models.ts +10 -7
  9. package/src/cli/report.ts +6 -1
  10. package/src/cli/stats.ts +7 -7
  11. package/src/config/load.ts +10 -1
  12. package/src/config/types.ts +10 -1
  13. package/src/context/bridge.ts +7 -7
  14. package/src/context/index.ts +3 -3
  15. package/src/context/store.ts +39 -56
  16. package/src/context/types.ts +7 -6
  17. package/src/cost/blended.ts +28 -7
  18. package/src/cost/feedback.ts +33 -37
  19. package/src/cost/ledger-sql.ts +547 -0
  20. package/src/cost/ledger.ts +30 -459
  21. package/src/cost/report.ts +171 -129
  22. package/src/cost/retention.ts +10 -10
  23. package/src/cost/summary.ts +15 -10
  24. package/src/cost/types.ts +43 -62
  25. package/src/cost/views.ts +79 -49
  26. package/src/eval/calibrate.ts +47 -12
  27. package/src/eval/run.ts +18 -2
  28. package/src/lib.ts +6 -2
  29. package/src/router/candidates.ts +7 -15
  30. package/src/router/classify.ts +6 -4
  31. package/src/router/index.ts +95 -9
  32. package/src/router/select.ts +38 -21
  33. package/src/router/state.ts +90 -102
  34. package/src/router/types.ts +11 -5
  35. package/src/server/advise.ts +6 -4
  36. package/src/server/compaction-digest.ts +1 -1
  37. package/src/server/digest.ts +9 -10
  38. package/src/server/http.ts +109 -46
  39. package/src/server/providers.ts +18 -4
  40. package/src/server/turn.ts +32 -9
  41. package/src/tokens/estimate.ts +16 -6
  42. package/src/upstream/ollama-usage.ts +21 -11
  43. package/src/util/schema.ts +201 -0
  44. package/src/util/sql.ts +246 -0
  45. package/src/wire/anthropic/messages.ts +3 -4
  46. package/src/wire/openai/request.ts +1 -0
  47. package/src/wire/types.ts +7 -0
  48. package/test/anthropic-wire.test.ts +9 -9
  49. package/test/benchmark-feeds.test.ts +7 -7
  50. package/test/cache-control.test.ts +7 -7
  51. package/test/cache-estimate.test.ts +5 -5
  52. package/test/catalog-view.test.ts +4 -4
  53. package/test/catalog.test.ts +11 -11
  54. package/test/classify.test.ts +24 -24
  55. package/test/compaction.test.ts +20 -20
  56. package/test/config-wizard.test.ts +32 -32
  57. package/test/config.test.ts +10 -10
  58. package/test/connect-harnesses.test.ts +11 -11
  59. package/test/context-bridge.test.ts +40 -30
  60. package/test/context-prune.test.ts +43 -36
  61. package/test/context-query.test.ts +8 -8
  62. package/test/controls.test.ts +54 -27
  63. package/test/cost.test.ts +12 -12
  64. package/test/digest.test.ts +55 -44
  65. package/test/embed-lifecycle.test.ts +5 -5
  66. package/test/embed-logic.test.ts +26 -26
  67. package/test/escalate.test.ts +17 -17
  68. package/test/eval.test.ts +73 -16
  69. package/test/executable.test.ts +6 -6
  70. package/test/exploration.test.ts +19 -20
  71. package/test/failover.test.ts +22 -21
  72. package/test/fakes.ts +105 -0
  73. package/test/features.test.ts +21 -21
  74. package/test/harness-requests.test.ts +3 -3
  75. package/test/harness-switch.test.ts +5 -5
  76. package/test/hold-exploration.test.ts +13 -13
  77. package/test/hot-reload.test.ts +5 -5
  78. package/test/learned.test.ts +5 -5
  79. package/test/ledger-sql.test.ts +342 -0
  80. package/test/mcp-entry.test.ts +5 -5
  81. package/test/migrations.test.ts +28 -22
  82. package/test/models-yml.test.ts +18 -18
  83. package/test/ollama.test.ts +40 -34
  84. package/test/omp-credentials.test.ts +16 -16
  85. package/test/policy.test.ts +3 -3
  86. package/test/reconfigure.test.ts +4 -4
  87. package/test/redaction.test.ts +41 -35
  88. package/test/remote.test.ts +12 -12
  89. package/test/report-logic.test.ts +8 -8
  90. package/test/report.test.ts +95 -87
  91. package/test/retention.test.ts +79 -66
  92. package/test/schema.test.ts +123 -0
  93. package/test/scope.test.ts +8 -8
  94. package/test/select.test.ts +216 -257
  95. package/test/skills.test.ts +3 -3
  96. package/test/sql-shim.test.ts +154 -0
  97. package/test/state.test.ts +43 -36
  98. package/test/summary.test.ts +38 -27
  99. package/test/tier-plan.test.ts +45 -62
  100. package/test/toast-logic.test.ts +31 -31
  101. package/test/tokens.test.ts +95 -80
  102. package/test/trust-attribution.test.ts +217 -187
  103. package/test/trust-window.test.ts +37 -32
  104. package/test/turn.test.ts +55 -23
  105. package/test/upstreams.test.ts +13 -13
  106. package/test/views.test.ts +81 -59
  107. package/test/wire-request.test.ts +17 -17
  108. package/test/wire-responses.test.ts +4 -4
  109. package/tools/agentdox-e2e.ts +5 -2
  110. package/tools/export-benchmarks.ts +5 -5
  111. package/tools/ledger-parity.ts +266 -0
  112. package/tools/replay.ts +16 -8
@@ -23,7 +23,7 @@ function userBody(content: unknown): Record<string, unknown> {
23
23
  }
24
24
 
25
25
  describe("parseChatRequest normalization", () => {
26
- test("string content and equivalent content-part array normalize to the same text", () => {
26
+ test("string content and equivalent content-part array normalize to the same text", async () => {
27
27
  const fromString = parseChatRequest(userBody("hello world"), HEADERS);
28
28
  const fromParts = parseChatRequest(
29
29
  userBody([{ type: "text", text: "hello world" }]),
@@ -34,7 +34,7 @@ describe("parseChatRequest normalization", () => {
34
34
  expect(fromParts.messages[0]!.images).toBe(0);
35
35
  });
36
36
 
37
- test("image parts are counted and flag hasImages", () => {
37
+ test("image parts are counted and flag hasImages", async () => {
38
38
  const req = parseChatRequest(
39
39
  userBody([
40
40
  { type: "text", text: "look at these" },
@@ -49,7 +49,7 @@ describe("parseChatRequest normalization", () => {
49
49
  expect(parseChatRequest(userBody("plain"), HEADERS).hasImages).toBe(false);
50
50
  });
51
51
 
52
- test("provider prefix is stripped from model", () => {
52
+ test("provider prefix is stripped from model", async () => {
53
53
  expect(parseChatRequest(userBody("hi"), HEADERS).requestedModel).toBe("auto");
54
54
  const prefixed = parseChatRequest(
55
55
  { model: "auto-model-router/auto", messages: [{ role: "user", content: "hi" }] },
@@ -58,18 +58,18 @@ describe("parseChatRequest normalization", () => {
58
58
  expect(prefixed.requestedModel).toBe("auto");
59
59
  });
60
60
 
61
- test("reads harness and omp session ids from headers, trimmed", () => {
61
+ test("reads harness and omp session ids from headers, trimmed", async () => {
62
62
  const headers = new Headers({ "x-omp-harness": " prod-a ", "x-omp-session": " sess-1 " });
63
63
  const req = parseChatRequest(userBody("hi"), headers);
64
64
  expect(req.harnessId).toBe("prod-a");
65
65
  expect(req.ompSessionId).toBe("sess-1");
66
66
  });
67
67
 
68
- test("omp session id defaults to empty when the header is absent", () => {
68
+ test("omp session id defaults to empty when the header is absent", async () => {
69
69
  expect(parseChatRequest(userBody("hi"), HEADERS).ompSessionId).toBe("");
70
70
  });
71
71
 
72
- test("tool schemas, names, and descriptions contribute to promptBytes", () => {
72
+ test("tool schemas, names, and descriptions contribute to promptBytes", async () => {
73
73
  const parameters = { type: "object", properties: { path: { type: "string" } } };
74
74
  const withTools = parseChatRequest(
75
75
  {
@@ -92,7 +92,7 @@ describe("parseChatRequest normalization", () => {
92
92
  );
93
93
  });
94
94
 
95
- test("tool calls and tool results are carried into NormMessage", () => {
95
+ test("tool calls and tool results are carried into NormMessage", async () => {
96
96
  const req = parseChatRequest(
97
97
  {
98
98
  model: "auto",
@@ -114,7 +114,7 @@ describe("parseChatRequest normalization", () => {
114
114
  expect(req.messages[1]!.toolName).toBe("read");
115
115
  });
116
116
 
117
- test("forcedToolChoice only for objects and non-auto/none strings", () => {
117
+ test("forcedToolChoice only for objects and non-auto/none strings", async () => {
118
118
  const base = userBody("hi");
119
119
  expect(parseChatRequest(base, HEADERS).forcedToolChoice).toBe(false);
120
120
  expect(parseChatRequest({ ...base, tool_choice: "auto" }, HEADERS).forcedToolChoice).toBe(false);
@@ -128,7 +128,7 @@ describe("parseChatRequest normalization", () => {
128
128
  ).toBe(true);
129
129
  });
130
130
 
131
- test("reasoning accepted from both spellings, omitted when absent", () => {
131
+ test("reasoning accepted from both spellings, omitted when absent", async () => {
132
132
  expect(
133
133
  parseChatRequest({ ...userBody("hi"), reasoning_effort: "high" }, HEADERS).reasoning,
134
134
  ).toBe("high");
@@ -142,7 +142,7 @@ describe("parseChatRequest normalization", () => {
142
142
  expect("reasoning" in plain).toBe(false);
143
143
  });
144
144
 
145
- test("malformed input throws WireErrorException", () => {
145
+ test("malformed input throws WireErrorException", async () => {
146
146
  expect(() => parseChatRequest(null, HEADERS)).toThrow(WireErrorException);
147
147
  expect(() => parseChatRequest({ model: "auto", messages: [] }, HEADERS)).toThrow(WireErrorException);
148
148
  expect(() => parseChatRequest({ messages: [{ role: "user", content: "hi" }] }, HEADERS)).toThrow(
@@ -163,7 +163,7 @@ describe("conversationKey", () => {
163
163
  const system = { role: "system", content: "You are a coding agent." };
164
164
  const first = { role: "user", content: "Fix the bug in main.ts" };
165
165
 
166
- test("stable across later turns of the same conversation", () => {
166
+ test("stable across later turns of the same conversation", async () => {
167
167
  const turn1 = parseChatRequest({ model: "auto", messages: [system, first] }, HEADERS);
168
168
  const turn3 = parseChatRequest(
169
169
  {
@@ -181,7 +181,7 @@ describe("conversationKey", () => {
181
181
  expect(turn1.conversationKey).toMatch(/^[0-9a-f]{32}$/);
182
182
  });
183
183
 
184
- test("differs when the first non-system message differs", () => {
184
+ test("differs when the first non-system message differs", async () => {
185
185
  const a = parseChatRequest({ model: "auto", messages: [system, first] }, HEADERS);
186
186
  const b = parseChatRequest(
187
187
  { model: "auto", messages: [system, { role: "user", content: "Write a poem" }] },
@@ -192,7 +192,7 @@ describe("conversationKey", () => {
192
192
  });
193
193
 
194
194
  describe("renderUpstreamBody", () => {
195
- test("two renders are independent and never mutate the original body", () => {
195
+ test("two renders are independent and never mutate the original body", async () => {
196
196
  const original = {
197
197
  model: "auto",
198
198
  messages: [{ role: "user", content: "hi" }],
@@ -236,7 +236,7 @@ describe("renderUpstreamBody", () => {
236
236
  expect(JSON.stringify(original)).toBe(before);
237
237
  });
238
238
 
239
- test("maxTokens lands on whichever spelling the client used", () => {
239
+ test("maxTokens lands on whichever spelling the client used", async () => {
240
240
  const req = parseChatRequest(
241
241
  { model: "auto", messages: [{ role: "user", content: "hi" }], max_completion_tokens: 200 },
242
242
  HEADERS,
@@ -246,13 +246,13 @@ describe("renderUpstreamBody", () => {
246
246
  expect("max_tokens" in out).toBe(false);
247
247
  });
248
248
 
249
- test("reasoning off renders as { enabled: false }", () => {
249
+ test("reasoning off renders as { enabled: false }", async () => {
250
250
  const req = parseChatRequest(userBody("hi"), HEADERS);
251
251
  const out = req.renderUpstreamBody(mutations({ reasoning: "off" }));
252
252
  expect(out.reasoning).toEqual({ enabled: false });
253
253
  });
254
254
 
255
- test("cache breakpoints land on named messages and promote string content to parts", () => {
255
+ test("cache breakpoints land on named messages and promote string content to parts", async () => {
256
256
  const req = parseChatRequest(
257
257
  {
258
258
  model: "auto",
@@ -280,7 +280,7 @@ describe("renderUpstreamBody", () => {
280
280
  ]);
281
281
  });
282
282
 
283
- test("stripAssistantReasoning removes all three spellings from assistant messages only", () => {
283
+ test("stripAssistantReasoning removes all three spellings from assistant messages only", async () => {
284
284
  const req = parseChatRequest(
285
285
  {
286
286
  model: "auto",
@@ -12,7 +12,7 @@ import type { StreamEvent, TurnSummary, UpstreamChunk } from "../src/wire/types.
12
12
  const HEADERS = new Headers({ "X-Omp-Harness": "codex" });
13
13
 
14
14
  describe("responsesToChatBody", () => {
15
- test("instructions, messages, function calls and their outputs become chat messages", () => {
15
+ test("instructions, messages, function calls and their outputs become chat messages", async () => {
16
16
  const chat = responsesToChatBody({
17
17
  model: "auto",
18
18
  instructions: "Be terse.",
@@ -59,7 +59,7 @@ describe("responsesToChatBody", () => {
59
59
  for (const k of ["instructions", "input", "store", "include", "prompt_cache_key", "max_output_tokens"]) expect(k in chat).toBe(false);
60
60
  });
61
61
 
62
- test("a string input is one user message; images survive; previous_response_id is refused", () => {
62
+ test("a string input is one user message; images survive; previous_response_id is refused", async () => {
63
63
  expect(responsesToChatBody({ model: "auto", input: "hi" }).messages).toEqual([{ role: "user", content: "hi" }]);
64
64
  const withImage = responsesToChatBody({ model: "auto", input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "what is this" }, { type: "input_image", image_url: "data:image/png;base64,AAAA" }] }] });
65
65
  expect(withImage.messages).toEqual([{ role: "user", content: [{ type: "text", text: "what is this" }, { type: "image_url", image_url: { url: "data:image/png;base64,AAAA" } }] }]);
@@ -67,7 +67,7 @@ describe("responsesToChatBody", () => {
67
67
  expect(() => responsesToChatBody({ model: "auto", input: [] })).toThrow(WireErrorException);
68
68
  });
69
69
 
70
- test("Codex identity is read from the body when the headers carry none", () => {
70
+ test("Codex identity is read from the body when the headers carry none", async () => {
71
71
  const meta = (agent: string) => JSON.stringify({ session_id: "t1", thread_id: "t1", agent_name: agent, turn_id: "u1" });
72
72
  const body = { model: "auto", input: "x", prompt_cache_key: "t1", client_metadata: { thread_id: "t1", session_id: "t1", "x-codex-turn-metadata": meta("/root") } };
73
73
  const main = parseResponsesRequest(body, HEADERS);
@@ -81,7 +81,7 @@ describe("responsesToChatBody", () => {
81
81
  expect(identityHeadersFromBody({ model: "auto", input: "x", client_metadata: { "x-codex-turn-metadata": "not json" } }, HEADERS).get("x-omp-subagent")).toBeNull();
82
82
  });
83
83
 
84
- test("parseResponsesRequest yields a routed request tagged with the wire", () => {
84
+ test("parseResponsesRequest yields a routed request tagged with the wire", async () => {
85
85
  const req = parseResponsesRequest({ model: "auto-model-router/auto-cheap", input: "hello", stream: false }, HEADERS);
86
86
  expect(req.protocol).toBe("openai-responses");
87
87
  expect(req.requestedModel).toBe("auto-cheap");
@@ -14,7 +14,8 @@ import { createContextBridge } from "../src/context/bridge.ts";
14
14
  import { createContextStore } from "../src/context/store.ts";
15
15
  import type { ContextResolveInput } from "../src/context/types.ts";
16
16
  import { createLogger } from "../src/util/log.ts";
17
- import { openDb } from "../src/util/sqlite.ts";
17
+ import { migrateStore } from "../src/util/schema.ts";
18
+ import { openSqlDb } from "../src/util/sql.ts";
18
19
 
19
20
  const baseUrl = process.env.AGENTDOX_URL ?? "http://localhost:3003";
20
21
  const token = process.env.AGENTDOX_TOKEN ?? "";
@@ -32,7 +33,9 @@ function check(name: string, ok: boolean, detail = ""): void {
32
33
  }
33
34
 
34
35
  const log = createLogger("warn");
35
- const db = openDb(":memory:");
36
+ // A temp file rather than `:memory:`: the shim opens one store per handle.
37
+ const db = openSqlDb(`${process.env.TMPDIR ?? "/tmp"}/agentdox-e2e-${Date.now()}.db`);
38
+ await migrateStore(db);
36
39
  const client = createAgentDoxClient({ baseUrl, token, timeoutMs: 5_000, log });
37
40
  const bridge = createContextBridge({
38
41
  client,
@@ -22,9 +22,9 @@
22
22
  import { existsSync, readFileSync, writeFileSync } from "node:fs";
23
23
  import { join, resolve } from "node:path";
24
24
  import { loadConfig } from "../src/config/load.ts";
25
- import { createLedger } from "../src/cost/ledger.ts";
25
+ import { createSqlLedger } from "../src/cost/ledger-sql.ts";
26
26
  import { computeStats } from "../src/server/http.ts";
27
- import { openDb } from "../src/util/sqlite.ts";
27
+ import { openSqlDb } from "../src/util/sql.ts";
28
28
 
29
29
  const ROOT = resolve(import.meta.dir, "..");
30
30
  const DATA_PATH = join(ROOT, "site", "data", "benchmarks.json");
@@ -56,9 +56,9 @@ if (!existsSync(dbPath)) {
56
56
  process.exit(0);
57
57
  }
58
58
 
59
- const db = openDb(dbPath);
59
+ const db = openSqlDb(dbPath);
60
60
  try {
61
- const stats = computeStats(createLedger(db, cfg), days === undefined ? {} : { windowDays: days });
61
+ const stats = await computeStats(createSqlLedger(db, cfg, { findModel: () => null }), days === undefined ? {} : { windowDays: days });
62
62
  const perTurnUsd = stats.requests > 0 ? stats.windowSpendUsd / stats.requests : 0;
63
63
  data.ledgerSnapshot = {
64
64
  generatedAt: new Date(stats.generatedAtMs).toISOString().slice(0, 10),
@@ -78,5 +78,5 @@ try {
78
78
  writeFileSync(DATA_PATH, `${JSON.stringify(data, null, 2)}\n`, "utf8");
79
79
  console.log(`ledgerSnapshot \u2190 ${stats.requests} turns from ${dbPath} (${stats.windowDays === null ? "all time" : `${stats.windowDays}d`})`);
80
80
  } finally {
81
- db.close();
81
+ await db.close();
82
82
  }
@@ -0,0 +1,266 @@
1
+ /**
2
+ * Parity: the one ledger must answer identically on both engines, reading the
3
+ * same production rows.
4
+ *
5
+ * The point is not that the SQL looks portable — it is that trust, latency,
6
+ * spend, blend, escalation cost and every report come out EQUAL, because
7
+ * routing decisions are made from those numbers and a front door may read a
8
+ * SQLite file while the router writes Postgres. Every silent bug found while
9
+ * porting (a `SUM()` read as a string skewing a trust score, a double-encoded
10
+ * JSON column nulling an escalation term, a nested `json_extract` path that
11
+ * SQLite tolerates and Postgres reads as NULL) was caught by comparing
12
+ * computed values rather than by a type error.
13
+ *
14
+ * Usage: bun tools/ledger-parity.ts <sqlite-path> [postgres-url]
15
+ */
16
+ import { Database } from "bun:sqlite";
17
+
18
+ import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
19
+ import type { RouterConfig } from "../src/config/types.ts";
20
+ import { createSqlLedger } from "../src/cost/ledger-sql.ts";
21
+ import { migrateStore } from "../src/util/schema.ts";
22
+ import type { AsyncLedger } from "../src/cost/types.ts";
23
+ import { openSqlDb, type SqlDb } from "../src/util/sql.ts";
24
+ import { decisionEntries, exportRows, feedbackView, spendUsdSince } from "../src/cost/views.ts";
25
+ import { buildUsageReport } from "../src/cost/report.ts";
26
+ import { buildDailySummary, countTierChanges } from "../src/cost/summary.ts";
27
+
28
+ const sqlitePath = process.argv[2] ?? "/data/router/router.db";
29
+ const pgUrl = process.argv[3] ?? "";
30
+
31
+ // Windows wide open so every row counts, and the scoped query shapes exercised.
32
+ const cfg: RouterConfig = structuredClone(DEFAULT_CONFIG);
33
+ cfg.filters.trustWindowDays = 0;
34
+ cfg.filters.feedbackWeight = 1;
35
+ cfg.filters.cacheReliabilityMinSamples = 1;
36
+ cfg.filters.escalationCostWeight = 1;
37
+ cfg.filters.latencyMinSamples = 1;
38
+
39
+ // The source of real rows. Read-only: the harness copies out of it and
40
+ // never writes to a live ledger.
41
+ const db = new Database(sqlitePath, { readonly: true });
42
+
43
+ const slugs = (db.query("SELECT DISTINCT slug FROM ledger WHERE slug IS NOT NULL").all() as { slug: string }[]).map((r) => r.slug);
44
+ const harnesses = (db.query("SELECT DISTINCT harness_id FROM ledger WHERE harness_id <> '' LIMIT 2").all() as { harness_id: string }[]).map(
45
+ (r) => r.harness_id,
46
+ );
47
+ const sessions = (
48
+ db.query("SELECT DISTINCT omp_session_id FROM ledger WHERE omp_session_id <> '' LIMIT 2").all() as { omp_session_id: string }[]
49
+ ).map((r) => r.omp_session_id);
50
+
51
+ /**
52
+ * Key-order-insensitive comparison. A JSON column round-tripped through
53
+ * Postgres' jsonb comes back with its keys reordered — jsonb does not preserve
54
+ * input order — which is not a difference in the value.
55
+ */
56
+ function stable(value: unknown): string {
57
+ return JSON.stringify(value, (_key, v: unknown) =>
58
+ v !== null && typeof v === "object" && !Array.isArray(v)
59
+ ? Object.fromEntries(Object.entries(v as Record<string, unknown>).sort(([a], [b]) => a.localeCompare(b)))
60
+ : v,
61
+ );
62
+ }
63
+
64
+ /**
65
+ * Structural comparison with a tolerance on every number, at any depth.
66
+ *
67
+ * Two engines summing the same rows in a different order land a few ULPs apart
68
+ * ($39.722225609861425 against ...56), which is arithmetic, not divergence. A
69
+ * strict compare on a nested total would report that as a mismatch and bury
70
+ * the real ones.
71
+ */
72
+ function deepEqual(a: unknown, b: unknown, tol: number): boolean {
73
+ if (typeof a === "number" && typeof b === "number") return Math.abs(a - b) <= tol * Math.max(1, Math.abs(a));
74
+ if (a === null || b === null || typeof a !== "object" || typeof b !== "object") return stable(a) === stable(b);
75
+ if (Array.isArray(a) !== Array.isArray(b)) return false;
76
+ if (Array.isArray(a) && Array.isArray(b)) {
77
+ return a.length === b.length && a.every((v, i) => deepEqual(v, b[i], tol));
78
+ }
79
+ const ra = a as Record<string, unknown>;
80
+ const rb = b as Record<string, unknown>;
81
+ const keys = new Set([...Object.keys(ra), ...Object.keys(rb)]);
82
+ for (const key of keys) if (!deepEqual(ra[key], rb[key], tol)) return false;
83
+ return true;
84
+ }
85
+
86
+ let checks = 0;
87
+ let bad = 0;
88
+ /** Which side is which in a mismatch report; set before each comparison pass. */
89
+ let labelA = "a";
90
+ let labelB = "b";
91
+ function eq(label: string, a: unknown, b: unknown, tol = 1e-9): void {
92
+ checks++;
93
+ const same = deepEqual(a, b, tol);
94
+ if (!same) {
95
+ bad++;
96
+ console.log(` MISMATCH ${label}\n ${labelA}=${JSON.stringify(a)}\n ${labelB}=${JSON.stringify(b)}`);
97
+ }
98
+ }
99
+
100
+ /** Copies the real rows into whichever store the unified ledger will read. */
101
+ async function load(target: SqlDb): Promise<void> {
102
+ await migrateStore(target);
103
+ for (const table of ["ledger", "feedback", "token_calibration"]) {
104
+ await target.sql.unsafe(`DELETE FROM ${table}`);
105
+ }
106
+ // The json columns are TEXT in the source; Postgres wants objects, so they
107
+ // are parsed when the destination stores JSONB.
108
+ const jsonCols = new Set(["usage", "cost_breakdown", "features", "classifier_reasons", "reasons"]);
109
+ const copy = async (table: string): Promise<number> => {
110
+ const cols = (db.query(`PRAGMA table_info(${table})`).all() as { name: string }[]).map((c) => c.name);
111
+ const rows = db.query(`SELECT ${cols.join(", ")} FROM ${table}`).all() as Record<string, unknown>[];
112
+ for (let i = 0; i < rows.length; i += 500) {
113
+ const batch = rows.slice(i, i + 500).map((r) => {
114
+ const o: Record<string, unknown> = {};
115
+ for (const c of cols) {
116
+ const v = r[c] ?? null;
117
+ o[c] = target.dialect === "postgres" && jsonCols.has(c) && typeof v === "string" ? JSON.parse(v) : v;
118
+ }
119
+ return o;
120
+ });
121
+ await target.sql`INSERT INTO ${target.sql.unsafe(table)} ${target.sql(batch, ...cols)}`;
122
+ }
123
+ return rows.length;
124
+ };
125
+ const ledgerRows = await copy("ledger");
126
+ // tokenRatio reads this table, so without it the unified side reports null
127
+ // against a real calibrated ratio — another harness-shaped "mismatch".
128
+ await copy("token_calibration");
129
+ let feedbackRows = 0;
130
+ try {
131
+ feedbackRows = await copy("feedback");
132
+ } catch (err) {
133
+ // Never swallowed: a failed feedback copy makes the unified side read no
134
+ // verdicts and shows up as a trust-score "mismatch" that is really a
135
+ // harness bug. It cost a diagnosis once already.
136
+ console.log(` feedback copy FAILED: ${err instanceof Error ? err.message : String(err)}`);
137
+ }
138
+ console.log(` loaded ${ledgerRows} ledger rows, ${feedbackRows} feedback rows`);
139
+ }
140
+
141
+ /**
142
+ * Every ported view, as one comparable value. The views moved to the shim in
143
+ * the same change as the ledger, so there is no synchronous version left to
144
+ * diff against — what matters is that the two ENGINES agree, since a front
145
+ * door may read a file while the router writes a database.
146
+ */
147
+ async function viewsSnapshot(db: SqlDb): Promise<Record<string, unknown>> {
148
+ const harness = harnesses[0] ?? null;
149
+ // A FIXED instant: the windows are relative to "now", so two engines read a
150
+ // moment apart would legitimately disagree and hide a real difference.
151
+ const nowMs = 1_789_000_000_000;
152
+ const report = await buildUsageReport(db, { windowDays: 3650, nowMs });
153
+ const summary = await buildDailySummary(db, { nowMs });
154
+ return {
155
+ reportTotals: report.totals,
156
+ reportProviders: report.providers.map((r) => [r.key, r.dispatches, Number(r.spendUsd.toFixed(9)), Number(r.cacheHitRate.toFixed(9)), r.escalations, r.errors]),
157
+ reportModels: report.models.map((m) => [m.key, m.provider, m.dispatches, Number(m.spendUsd.toFixed(9)), m.tiers, m.feedback]),
158
+ reportTiers: report.tiers.map((r) => [r.key, r.dispatches, Number(r.spendUsd.toFixed(9))]),
159
+ reportDays: report.days.map((d) => [d.day, d.dispatches, Number(d.spendUsd.toFixed(9)), Number(d.cacheHitRate.toFixed(9))]),
160
+ reportAnatomy: report.anatomy,
161
+ summaryCurrent: summary.current,
162
+ summaryPrevious: summary.previous,
163
+ summaryTopModels: summary.topModels.map((m) => [m.slug, m.dispatches, Number(m.spendUsd.toFixed(9))]),
164
+ tierChanges: await countTierChanges(db, 0, ""),
165
+ spendAll: await spendUsdSince(db, 0, null),
166
+ spendHarness: harness === null ? 0 : await spendUsdSince(db, 0, [harness]),
167
+ spendNone: await spendUsdSince(db, 0, []),
168
+ exportRows: (await exportRows(db, 0, null)).map((r) => [r.day, r.harnessId, r.slug, r.scope, r.dispatches, r.promptTokens, r.cachedTokens, r.completionTokens, Number(r.spendUsd.toFixed(9)), r.escalations, r.errors]),
169
+ feedback: await feedbackView(db, 0, null),
170
+ decisions: (await decisionEntries(db, { sinceMs: 0, harness: null, limit: 25 })).map((e) => [e.id, e.slug, e.servedSlug, e.tier, e.wasted, e.feedback.length, e.costBreakdown?.total ?? null]),
171
+ };
172
+ }
173
+
174
+ /**
175
+ * Every signal the router reads off the ledger, as one comparable value.
176
+ *
177
+ * A FIXED instant throughout: `spendSince` and the spike windows are relative
178
+ * to "now", so two engines read a moment apart would legitimately disagree and
179
+ * hide a real difference.
180
+ */
181
+ async function ledgerSnapshot(unified: AsyncLedger): Promise<Record<string, unknown>> {
182
+ const nowMs = 1_789_000_000_000;
183
+ const out: Record<string, unknown> = {};
184
+
185
+ for (const days of [0, 1, 7]) {
186
+ const since = days === 0 ? 0 : nowMs - days * 86_400_000;
187
+ out[`spendSince(${days}d)`] = await unified.spendSince(since);
188
+ for (const h of harnesses) out[`spendSince(${days}d,${h})`] = await unified.spendSince(since, h);
189
+ }
190
+
191
+ const esc = await unified.escalationCost(cfg.ledger.blendWindowDays);
192
+ out.escalationCost = esc === null ? null : [esc.samples, esc.usdPerPromptToken];
193
+ const blend = await unified.blendedRate(cfg.ledger.blendWindowDays);
194
+ out.blendedRate = blend === null ? null : [blend.sampleCount, blend.inputPerMtok, blend.outputPerMtok, blend.cacheReadPerMtok];
195
+
196
+ // The prefetch the turn path actually uses, global and per harness.
197
+ for (const harness of [undefined, ...harnesses]) {
198
+ const tag = harness ?? "global";
199
+ const signals = await unified.signals(slugs, harness);
200
+ for (const slug of slugs) {
201
+ const s = signals.get(slug);
202
+ const trust = s?.trust ?? null;
203
+ const latency = s?.latency ?? null;
204
+ out[`trust[${tag}][${slug}]`] = trust === null
205
+ ? null
206
+ : [trust.attempts, trust.escalations, trust.errors, trust.successRate, trust.meanCostError];
207
+ out[`latency[${tag}][${slug}]`] = latency === null ? null : [latency.samples, latency.ttftMs, latency.tokensPerSec];
208
+ }
209
+ }
210
+
211
+ // Keyed by slug: row order is not part of the meaning here.
212
+ out.allTrust = Object.fromEntries((await unified.allTrust()).map((t) => [t.slug, [t.attempts, t.successRate]]));
213
+ const cache = await unified.cacheReliability(slugs);
214
+ out.cacheReliability = Object.fromEntries(slugs.map((s) => [s, cache.get(s) === undefined ? null : [cache.get(s)?.samples, cache.get(s)?.hitRate]]));
215
+
216
+ for (const prefix of ["ollama/", "deepseek/", "anthropic-subscription/"]) {
217
+ out[`providerSpendSince(${prefix})`] = await unified.providerSpendSince(prefix, 0);
218
+ }
219
+
220
+ // Entry round-trip: the JSON columns are where a dialect difference shows.
221
+ out.recentEntries = (await unified.recentEntries(25)).map((e) => [e.id, e.usage, e.reasons, e.costBreakdown ?? null, e.scope ?? null, e.wasted]);
222
+
223
+ for (const session of sessions) {
224
+ out[`latestForSession(${session})`] = (await unified.latestForSession(session))?.id ?? null;
225
+ out[`entriesForSession(${session})`] = (await unified.entriesForSession(session, 5)).map((e) => e.id);
226
+ }
227
+
228
+ out.softFailureSpikes = (await unified.softFailureSpikes(nowMs)).map((s) => [s.slug, s.recentRate, s.recentDispatches, s.baselineRate]);
229
+ for (const tokenizer of ["gpt", "claude", "qwen", "llama"]) {
230
+ out[`tokenRatio(${tokenizer})`] = await unified.tokenRatio(tokenizer);
231
+ }
232
+ return out;
233
+ }
234
+
235
+ const targets: { engine: string; url: string }[] = [
236
+ { engine: "sqlite", url: `sqlite:///tmp/ledger-parity-${Date.now()}.db` },
237
+ ...(pgUrl === "" ? [] : [{ engine: "postgres", url: pgUrl }]),
238
+ ];
239
+
240
+ const snapshots = new Map<string, Record<string, unknown>>();
241
+ for (const target of targets) {
242
+ console.log(`\n=== loading ${target.engine} ===`);
243
+ const store = openSqlDb(target.url);
244
+ await load(store);
245
+ const ledger = createSqlLedger(store, cfg, { findModel: () => null });
246
+ snapshots.set(target.engine, { ...(await ledgerSnapshot(ledger)), ...(await viewsSnapshot(store)) });
247
+ await store.close();
248
+ }
249
+
250
+ // Every signal and every view, engine against engine. One engine alone still
251
+ // proves nothing was thrown: it loads the rows and computes the lot.
252
+ const [first, ...rest] = [...snapshots.keys()];
253
+ if (first !== undefined) {
254
+ const a = snapshots.get(first) as Record<string, unknown>;
255
+ labelA = first;
256
+ for (const other of rest) {
257
+ console.log(`\n=== ${other} vs ${first} ===`);
258
+ labelB = other;
259
+ const b = snapshots.get(other) as Record<string, unknown>;
260
+ for (const key of Object.keys(a)) eq(key, a[key], b[key], 1e-9);
261
+ }
262
+ if (rest.length === 0) console.log(`\n=== ${first} only: ${Object.keys(a).length} values computed, nothing to compare against ===`);
263
+ }
264
+
265
+ console.log(`\n${checks - bad}/${checks} checks equal${bad === 0 ? "" : `, ${bad} MISMATCHED`}`);
266
+ process.exit(bad === 0 ? 0 : 1);
package/tools/replay.ts CHANGED
@@ -70,7 +70,9 @@ import type { CatalogModel, CatalogSnapshot } from "../src/catalog/types.ts";
70
70
  import { loadConfig } from "../src/config/load.ts";
71
71
  import type { RouterConfig } from "../src/config/types.ts";
72
72
  import { computeCost } from "../src/cost/forecast.ts";
73
- import { createLedger } from "../src/cost/ledger.ts";
73
+ import { createSqlLedger } from "../src/cost/ledger-sql.ts";
74
+ import { prefetchTurnReads } from "../src/router/index.ts";
75
+ import { openSqlDb } from "../src/util/sql.ts";
74
76
  import type { UsageCounts } from "../src/cost/types.ts";
75
77
  import { classifyTask, scoreHeuristic } from "../src/router/classify.ts";
76
78
  import { resolveHoldTurns } from "../src/router/explore.ts";
@@ -324,7 +326,11 @@ function snapshotFor(cfg: RouterConfig): CatalogSnapshot {
324
326
  const snapshotA = snapshotFor(cfgA);
325
327
  const snapshotB = snapshotFor(cfgB);
326
328
  const bySlug = new Map([...snapshotA.models, ...snapshotB.models].map((m) => [m.slug, m]));
327
- const ledger = createLedger(db, cfgA);
329
+ // Replay reads through the engine-agnostic handle: `select` takes the ledger's
330
+ // answers as DATA now, so each replayed turn prefetches the same reads a live
331
+ // turn would. Read-only — nothing here writes.
332
+ const sqlDb = openSqlDb(dbPath);
333
+ const ledger = createSqlLedger(sqlDb, cfgA, { findModel: (slug: string) => bySlug.get(slug) ?? null });
328
334
 
329
335
  const predicate = args.where === "" ? "" : ` AND (${args.where})`;
330
336
  // Newest-first to honour --limit, then flipped to chronological so each row can
@@ -371,7 +377,7 @@ interface Outcome {
371
377
  trail: VariantTrail;
372
378
  }
373
379
 
374
- function run(
380
+ async function run(
375
381
  cfg: RouterConfig,
376
382
  snapshot: CatalogSnapshot,
377
383
  row: Row,
@@ -379,7 +385,7 @@ function run(
379
385
  prior: PriorTurn | undefined,
380
386
  trail: VariantTrail | undefined,
381
387
  esc: EscalationContext,
382
- ): Outcome {
388
+ ): Promise<Outcome> {
383
389
  const f = featuresOf(row, usage.promptTokens);
384
390
  const req = requestOf(row, f);
385
391
  const state = stateOf(row, prior, trail, args.warmth);
@@ -399,14 +405,16 @@ function run(
399
405
  } else {
400
406
  classification = scoreHeuristic(f, cfg);
401
407
  }
408
+ const profile = profileOf(cfg, row.requested_model);
409
+ const reads = await prefetchTurnReads(ledger, req, profile, cfg, snapshot, classification.task, row.created_at_ms);
402
410
  const decision: Decision = select({
403
411
  req,
404
412
  features: f,
405
413
  classification,
406
- profile: profileOf(cfg, row.requested_model),
414
+ profile,
407
415
  state,
408
416
  snapshot,
409
- ledger,
417
+ reads,
410
418
  cfg,
411
419
  // The row's own clock: cache warmth and hold windows are judged against
412
420
  // when the turn happened, not against today. Passing Date.now() here made
@@ -500,8 +508,8 @@ for (const row of rows) {
500
508
  excludeSlugs: wasted.map((w) => w.served_slug ?? w.slug),
501
509
  };
502
510
  if (esc.escalateFrom !== undefined) escalationsReplayed++;
503
- const a = run(cfgA, snapshotA, row, u, prior, trailA.get(row.conversation_key), esc);
504
- const b = run(cfgB, snapshotB, row, u, prior, trailB.get(row.conversation_key), esc);
511
+ const a = await run(cfgA, snapshotA, row, u, prior, trailA.get(row.conversation_key), esc);
512
+ const b = await run(cfgB, snapshotB, row, u, prior, trailB.get(row.conversation_key), esc);
505
513
  trailA.set(row.conversation_key, a.trail);
506
514
  trailB.set(row.conversation_key, b.trail);
507
515