auto-model-router 0.31.0 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/.omp-plugin/marketplace.json +2 -2
  2. package/README.md +32 -2
  3. package/omp-extension/router-configure.ts +9 -7
  4. package/package.json +1 -1
  5. package/src/cli/config-cmd.ts +8 -7
  6. package/src/cli/explain.ts +10 -5
  7. package/src/cli/export.ts +6 -5
  8. package/src/cli/models.ts +10 -7
  9. package/src/cli/report.ts +6 -1
  10. package/src/cli/stats.ts +7 -7
  11. package/src/config/load.ts +10 -1
  12. package/src/config/types.ts +10 -1
  13. package/src/context/bridge.ts +7 -7
  14. package/src/context/index.ts +3 -3
  15. package/src/context/store.ts +39 -56
  16. package/src/context/types.ts +7 -6
  17. package/src/cost/blended.ts +28 -7
  18. package/src/cost/feedback.ts +33 -37
  19. package/src/cost/ledger-sql.ts +547 -0
  20. package/src/cost/ledger.ts +30 -459
  21. package/src/cost/report.ts +171 -129
  22. package/src/cost/retention.ts +10 -10
  23. package/src/cost/summary.ts +15 -10
  24. package/src/cost/types.ts +43 -62
  25. package/src/cost/views.ts +79 -49
  26. package/src/lib.ts +6 -2
  27. package/src/router/candidates.ts +7 -15
  28. package/src/router/classify.ts +6 -4
  29. package/src/router/index.ts +95 -9
  30. package/src/router/select.ts +38 -21
  31. package/src/router/state.ts +90 -102
  32. package/src/router/types.ts +11 -5
  33. package/src/server/advise.ts +6 -4
  34. package/src/server/compaction-digest.ts +1 -1
  35. package/src/server/digest.ts +9 -10
  36. package/src/server/http.ts +101 -41
  37. package/src/server/providers.ts +18 -4
  38. package/src/server/turn.ts +32 -9
  39. package/src/tokens/estimate.ts +16 -6
  40. package/src/upstream/ollama-usage.ts +21 -11
  41. package/src/util/schema.ts +201 -0
  42. package/src/util/sql.ts +246 -0
  43. package/src/wire/anthropic/messages.ts +3 -4
  44. package/src/wire/openai/request.ts +1 -0
  45. package/src/wire/types.ts +7 -0
  46. package/test/anthropic-wire.test.ts +9 -9
  47. package/test/benchmark-feeds.test.ts +7 -7
  48. package/test/cache-control.test.ts +7 -7
  49. package/test/cache-estimate.test.ts +5 -5
  50. package/test/catalog-view.test.ts +4 -4
  51. package/test/catalog.test.ts +11 -11
  52. package/test/classify.test.ts +24 -24
  53. package/test/compaction.test.ts +20 -20
  54. package/test/config-wizard.test.ts +32 -32
  55. package/test/config.test.ts +10 -10
  56. package/test/connect-harnesses.test.ts +11 -11
  57. package/test/context-bridge.test.ts +40 -30
  58. package/test/context-prune.test.ts +43 -36
  59. package/test/context-query.test.ts +8 -8
  60. package/test/controls.test.ts +54 -27
  61. package/test/cost.test.ts +12 -12
  62. package/test/digest.test.ts +55 -44
  63. package/test/embed-lifecycle.test.ts +5 -5
  64. package/test/embed-logic.test.ts +26 -26
  65. package/test/escalate.test.ts +17 -17
  66. package/test/eval.test.ts +13 -13
  67. package/test/executable.test.ts +6 -6
  68. package/test/exploration.test.ts +19 -20
  69. package/test/failover.test.ts +22 -21
  70. package/test/fakes.ts +105 -0
  71. package/test/features.test.ts +21 -21
  72. package/test/harness-requests.test.ts +3 -3
  73. package/test/harness-switch.test.ts +5 -5
  74. package/test/hold-exploration.test.ts +13 -13
  75. package/test/hot-reload.test.ts +5 -5
  76. package/test/learned.test.ts +5 -5
  77. package/test/ledger-sql.test.ts +342 -0
  78. package/test/mcp-entry.test.ts +5 -5
  79. package/test/migrations.test.ts +28 -22
  80. package/test/models-yml.test.ts +18 -18
  81. package/test/ollama.test.ts +40 -34
  82. package/test/omp-credentials.test.ts +16 -16
  83. package/test/policy.test.ts +3 -3
  84. package/test/reconfigure.test.ts +4 -4
  85. package/test/redaction.test.ts +41 -35
  86. package/test/remote.test.ts +12 -12
  87. package/test/report-logic.test.ts +8 -8
  88. package/test/report.test.ts +95 -87
  89. package/test/retention.test.ts +79 -66
  90. package/test/schema.test.ts +123 -0
  91. package/test/scope.test.ts +8 -8
  92. package/test/select.test.ts +216 -257
  93. package/test/skills.test.ts +3 -3
  94. package/test/sql-shim.test.ts +154 -0
  95. package/test/state.test.ts +43 -36
  96. package/test/summary.test.ts +38 -27
  97. package/test/tier-plan.test.ts +45 -62
  98. package/test/toast-logic.test.ts +31 -31
  99. package/test/tokens.test.ts +95 -80
  100. package/test/trust-attribution.test.ts +217 -187
  101. package/test/trust-window.test.ts +37 -32
  102. package/test/turn.test.ts +55 -23
  103. package/test/upstreams.test.ts +13 -13
  104. package/test/views.test.ts +81 -59
  105. package/test/wire-request.test.ts +17 -17
  106. package/test/wire-responses.test.ts +4 -4
  107. package/tools/agentdox-e2e.ts +5 -2
  108. package/tools/export-benchmarks.ts +5 -5
  109. package/tools/ledger-parity.ts +266 -0
  110. package/tools/replay.ts +16 -8
@@ -0,0 +1,342 @@
1
+ /**
2
+ * The unified ledger, on every engine it claims to support.
3
+ *
4
+ * SQLite runs always (a temp file); Postgres runs when AMR_ROUTER_TEST_PG
5
+ * points at one, following the team edition's convention for store tests. The
6
+ * same assertions run against both, because the reason this implementation
7
+ * replaced two backends is that two implementations of one meaning drift
8
+ * silently — a `SUM()` read as a string skewed a trust score by three points,
9
+ * and a double-encoded JSON column made an escalation-cost term null, both
10
+ * without raising anything.
11
+ */
12
+ import { afterAll, beforeAll, beforeEach, describe, expect, test } from "bun:test";
13
+ import { tmpdir } from "node:os";
14
+ import { join } from "node:path";
15
+
16
+ import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
17
+ import type { RouterConfig } from "../src/config/types.ts";
18
+ import { createSqlLedger } from "../src/cost/ledger-sql.ts";
19
+ import { createConversationStore } from "../src/router/state.ts";
20
+ import { migrateStore } from "../src/util/schema.ts";
21
+ import type { AsyncLedger, LedgerEntry } from "../src/cost/types.ts";
22
+ import { openSqlDb, type SqlDb } from "../src/util/sql.ts";
23
+
24
+ const DAY_MS = 86_400_000;
25
+ const PG = process.env.AMR_ROUTER_TEST_PG;
26
+
27
+ const engines: { name: string; url: string }[] = [
28
+ { name: "sqlite", url: `sqlite://${join(tmpdir(), `ledger-sql-${process.pid}-${Date.now()}.db`)}` },
29
+ ...(PG === undefined ? [] : [{ name: "postgres", url: PG }]),
30
+ ];
31
+
32
+ function cfgWith(over: Partial<RouterConfig["filters"]> = {}): RouterConfig {
33
+ const cfg = structuredClone(DEFAULT_CONFIG);
34
+ cfg.filters = { ...cfg.filters, trustWindowDays: 0, feedbackWeight: 1, ...over };
35
+ return cfg;
36
+ }
37
+
38
+ function entry(over: Partial<LedgerEntry> & { id: string; slug: string }): LedgerEntry {
39
+ return {
40
+ createdAtMs: Date.now(),
41
+ conversationKey: `conv-${over.id}`,
42
+ // `LedgerEntry.sessionId` is a string and the column is NOT NULL: a null
43
+ // here only ever passed because the second bootstrap declared the column
44
+ // laxer than the nineteen shipped migrations do.
45
+ sessionId: `sess-${over.id}`,
46
+ turn: 1,
47
+ requestedModel: "auto",
48
+ harnessId: "",
49
+ ompSessionId: "",
50
+ servedSlug: over.slug,
51
+ tier: "simple",
52
+ classificationSource: "forced",
53
+ reasons: ["because"],
54
+ predictedUsd: 0,
55
+ reportedUsd: null,
56
+ usage: { promptTokens: 100, completionTokens: 10, cachedTokens: 0, cacheWriteTokens: 0, reasoningTokens: 0, images: 0 },
57
+ attempt: 0,
58
+ escalationSignal: null,
59
+ latencyMs: 100,
60
+ ttftMs: 50,
61
+ finishReason: "stop",
62
+ wasted: false,
63
+ upstreamGenerationId: null,
64
+ error: null,
65
+ features: null,
66
+ score: null,
67
+ confidence: null,
68
+ task: null,
69
+ classifierReasons: null,
70
+ exploredFrom: null,
71
+ holdArm: null,
72
+ promptTokensSaved: null,
73
+ ...over,
74
+ } as LedgerEntry;
75
+ }
76
+
77
+ for (const engine of engines) {
78
+ describe(`ledger on ${engine.name}`, () => {
79
+ // ONE handle per engine. A `SqlDb` is a connection pool, and opening one
80
+ // per test exhausts a default Postgres (`sorry, too many clients
81
+ // already`) — the same limit a host packed with many small tenants hits.
82
+ let db: SqlDb;
83
+
84
+ beforeAll(async () => {
85
+ db = openSqlDb(engine.url);
86
+ await migrateStore(db);
87
+ });
88
+
89
+ afterAll(async () => {
90
+ await db.close();
91
+ });
92
+
93
+ // Tables are emptied rather than dropped: these tests assert over global
94
+ // aggregates (allTrust, recentEntries, unscoped spend), so leftovers from
95
+ // a neighbour would couple them.
96
+ beforeEach(async () => {
97
+ for (const table of ["ledger", "feedback", "token_calibration", "ollama_meter_samples"]) {
98
+ await db.sql.unsafe(`DELETE FROM ${table}`);
99
+ }
100
+ });
101
+
102
+ const setup = async (cfg = cfgWith()): Promise<{ db: SqlDb; ledger: AsyncLedger }> => ({
103
+ db,
104
+ ledger: createSqlLedger(db, cfg, { findModel: () => null }),
105
+ });
106
+
107
+ test("a write is visible to a second handle on the same store, so a shared cap holds", async () => {
108
+ const cfg = cfgWith();
109
+ const { db, ledger: a } = await setup(cfg);
110
+ // A second ledger over the same store is what a second router replica
111
+ // is. With a per-process file, B would read 0 for A's spend.
112
+ const b = createSqlLedger(db, cfg, { findModel: () => null });
113
+
114
+ expect(await b.spendSince(0, "h1")).toBe(0);
115
+ await a.record(entry({ id: "a1", slug: "x/m", harnessId: "h1", predictedUsd: 4 }));
116
+ expect(await b.spendSince(0, "h1")).toBeCloseTo(4, 9);
117
+
118
+ await b.record(entry({ id: "b1", slug: "x/m", harnessId: "h1", predictedUsd: 3 }));
119
+ // $7 between them breaches a $5 cap that neither replica's own share
120
+ // would have reached.
121
+ expect(await a.spendSince(0, "h1")).toBeCloseTo(7, 9);
122
+
123
+ // Harness scoping still isolates one user's spend from another's.
124
+ await a.record(entry({ id: "a2", slug: "x/m", harnessId: "h2", predictedUsd: 50 }));
125
+ expect(await b.spendSince(0, "h1")).toBeCloseTo(7, 9);
126
+ expect(await b.spendSince(0)).toBeCloseTo(57, 9);
127
+ });
128
+
129
+ test("reported spend beats predicted, and a window excludes older rows", async () => {
130
+ const { ledger } = await setup(cfgWith());
131
+ await ledger.record(entry({ id: "w1", slug: "y/m", harnessId: "hw", predictedUsd: 1, reportedUsd: 9 }));
132
+ await ledger.record(entry({ id: "w2", slug: "y/m", harnessId: "hw", predictedUsd: 2, createdAtMs: Date.now() - 3 * DAY_MS }));
133
+ expect(await ledger.spendSince(0, "hw")).toBeCloseTo(11, 9);
134
+ expect(await ledger.spendSince(Date.now() - DAY_MS, "hw")).toBeCloseTo(9, 9);
135
+ expect(await ledger.conversationSpend("conv-w1")).toBeCloseTo(9, 9);
136
+ });
137
+
138
+ test("trust counts escalations and upstream errors but forgives client aborts", async () => {
139
+ const { ledger } = await setup(cfgWith());
140
+ const slug = "t/model";
141
+ for (let i = 0; i < 6; i++) await ledger.record(entry({ id: `t${i}`, slug, harnessId: "ht" }));
142
+ await ledger.record(entry({ id: "te1", slug, harnessId: "ht", escalationSignal: "probe_failed" }));
143
+ await ledger.record(entry({ id: "te2", slug, harnessId: "ht", error: "upstream_5xx: boom" }));
144
+ // An aborted turn is the client's doing, not the model's: errorKindOf
145
+ // recovers the kind and ATTRIBUTABLE_ERROR excludes that set
146
+ // ('aborted', 'auth', 'moderation', 'model_unavailable', 'quota').
147
+ await ledger.record(entry({ id: "te3", slug, harnessId: "ht", error: "request aborted" }));
148
+
149
+ const trust = await ledger.trust(slug, "ht");
150
+ expect(trust?.attempts).toBe(9);
151
+ expect(trust?.escalations).toBe(1);
152
+ expect(trust?.errors).toBe(1);
153
+ // Numbers, not Postgres' aggregate strings: toTrust does arithmetic on
154
+ // these, so a string skews the score instead of failing.
155
+ expect(typeof trust?.attempts).toBe("number");
156
+ expect(typeof trust?.successRate).toBe("number");
157
+ expect(trust?.successRate).toBeGreaterThan(0.5);
158
+ expect(trust?.successRate).toBeLessThan(1);
159
+
160
+ const all = await ledger.allTrust();
161
+ expect(all.find((t) => t.slug === slug)?.attempts).toBe(9);
162
+ });
163
+
164
+ test("signals batch trust and latency for a candidate set", async () => {
165
+ const { ledger } = await setup(cfgWith({ latencyMinSamples: 1 }));
166
+ await ledger.record(entry({ id: "s1", slug: "a/one", harnessId: "hs", ttftMs: 100, latencyMs: 1100 }));
167
+ await ledger.record(entry({ id: "s2", slug: "a/one", harnessId: "hs", ttftMs: 300, latencyMs: 1300 }));
168
+ // Errored rows carry no usable timing and must not drag the mean.
169
+ await ledger.record(entry({ id: "s3", slug: "a/one", harnessId: "hs", ttftMs: 9000, latencyMs: 9900, error: "upstream_5xx: x" }));
170
+ await ledger.record(entry({ id: "s4", slug: "b/two", harnessId: "hs", ttftMs: 40, latencyMs: 540 }));
171
+
172
+ const signals = await ledger.signals(["a/one", "b/two", "c/absent"], "hs");
173
+ expect(signals.get("a/one")?.latency?.samples).toBe(2);
174
+ expect(signals.get("a/one")?.latency?.ttftMs).toBeCloseTo(200, 6);
175
+ expect(signals.get("a/one")?.latency?.tokensPerSec).toBeGreaterThan(0);
176
+ expect(signals.get("b/two")?.trust?.attempts).toBe(1);
177
+ // A slug the ledger has never seen yields an entry with no signals,
178
+ // not a missing key: candidate scoring reads it either way.
179
+ expect(signals.get("c/absent")).toEqual({ trust: null, latency: null });
180
+ expect(await ledger.latency("b/two", "hs")).not.toBeNull();
181
+ });
182
+
183
+ test("escalation cost needs enough samples before it reports a rate", async () => {
184
+ const cfg = cfgWith();
185
+ const { db, ledger } = await setup(cfg);
186
+ for (let i = 0; i < 4; i++) {
187
+ await ledger.record(entry({ id: `e${i}`, slug: "e/model", harnessId: "he", attempt: 1, predictedUsd: 1 }));
188
+ }
189
+ expect(await ledger.escalationCost(30)).toBeNull();
190
+
191
+ for (let i = 4; i < 14; i++) {
192
+ await ledger.record(entry({ id: `e${i}`, slug: "e/model", harnessId: "he", attempt: 1, predictedUsd: 1 }));
193
+ }
194
+ // A fresh handle: the result is memoised per instance for a minute.
195
+ const fresh = createSqlLedger(db, cfg, { findModel: () => null });
196
+ const cost = await fresh.escalationCost(30);
197
+ expect(cost?.samples).toBe(14);
198
+ // 14 rows × $1 over 14 × 100 prompt tokens. Reads inside the usage
199
+ // JSON, which yields NULL when the column is stored double-encoded.
200
+ expect(cost?.usdPerPromptToken).toBeCloseTo(0.01, 9);
201
+ });
202
+
203
+ test("entries round-trip, including the JSON columns", async () => {
204
+ const { ledger } = await setup(cfgWith());
205
+ await ledger.record(
206
+ entry({
207
+ id: "r1",
208
+ slug: "r/model",
209
+ harnessId: "hr",
210
+ ompSessionId: "sess-1",
211
+ reasons: ["cheapest", "warm cache"],
212
+ classifierReasons: ["short prompt"],
213
+ usage: { promptTokens: 11, completionTokens: 22, cachedTokens: 3, cacheWriteTokens: 4, reasoningTokens: 5, images: 0 },
214
+ scope: "team/proj",
215
+ redactions: 0,
216
+ }),
217
+ );
218
+ const recent = await ledger.recentEntries(5);
219
+ const row = recent.find((e) => e.id === "r1");
220
+ expect(row?.reasons).toEqual(["cheapest", "warm cache"]);
221
+ expect(row?.classifierReasons).toEqual(["short prompt"]);
222
+ expect(row?.usage.promptTokens).toBe(11);
223
+ expect(row?.usage.cachedTokens).toBe(3);
224
+ expect(row?.scope).toBe("team/proj");
225
+ expect(row?.redactions).toBe(0);
226
+ expect(row?.wasted).toBe(false);
227
+
228
+ expect((await ledger.latestForSession("sess-1"))?.id).toBe("r1");
229
+ expect((await ledger.entriesForSession("sess-1", 3)).length).toBe(1);
230
+ // A wasted row is excluded from the session view.
231
+ await ledger.markWasted("r1");
232
+ expect(await ledger.latestForSession("sess-1")).toBeNull();
233
+ });
234
+
235
+ test("prune deletes past retention, keeps newer rows, and reports the oldest kept", async () => {
236
+ const { ledger } = await setup(cfgWith());
237
+ await ledger.record(entry({ id: "p_old", slug: "p/m", harnessId: "hp", predictedUsd: 5, createdAtMs: Date.now() - 10 * DAY_MS }));
238
+ await ledger.record(entry({ id: "p_new", slug: "p/m", harnessId: "hp", predictedUsd: 7 }));
239
+ expect(await ledger.spendSince(0, "hp")).toBeCloseTo(12, 9);
240
+
241
+ // null and 0 both mean "keep everything", and still report the age.
242
+ expect((await ledger.prune(null)).deleted).toBe(0);
243
+ expect((await ledger.prune(0)).oldestKeptMs).not.toBeNull();
244
+
245
+ const result = await ledger.prune(1);
246
+ expect(result.deleted).toBe(1);
247
+ expect(await ledger.spendSince(0, "hp")).toBeCloseTo(7, 9);
248
+ });
249
+
250
+ test("provider spend matches on the served slug's prefix", async () => {
251
+ const { ledger } = await setup(cfgWith());
252
+ await ledger.record(entry({ id: "pv1", slug: "ollama/glm", harnessId: "hv", predictedUsd: 2 }));
253
+ await ledger.record(entry({ id: "pv2", slug: "openrouter/glm", harnessId: "hv", predictedUsd: 5 }));
254
+ // Attribution follows served_slug, which is what actually billed.
255
+ await ledger.record(entry({ id: "pv3", slug: "auto", servedSlug: "ollama/other", harnessId: "hv", predictedUsd: 1 }));
256
+ expect(await ledger.providerSpendSince("ollama/", 0)).toBeCloseTo(3, 9);
257
+ });
258
+
259
+ test("soft-failure spikes need a rate, a floor of failures, and a worse-than-baseline ratio", async () => {
260
+ const { ledger } = await setup(cfgWith());
261
+ const now = Date.now();
262
+ const recentMs = 15 * 60_000;
263
+ const baselineMs = 6 * 60 * 60_000;
264
+ // Baseline: healthy. Recent: mostly failing. Enough of both to clear
265
+ // SPIKE_MIN_DISPATCHES and SPIKE_MIN_FAILURES.
266
+ for (let i = 0; i < 20; i++) {
267
+ await ledger.record(entry({ id: `sb${i}`, slug: "sp/model", createdAtMs: now - recentMs - 60_000 }));
268
+ }
269
+ for (let i = 0; i < 6; i++) {
270
+ await ledger.record(entry({ id: `sr${i}`, slug: "sp/model", createdAtMs: now - 60_000, error: "upstream_5xx: boom" }));
271
+ }
272
+ const spikes = await ledger.softFailureSpikes(now, recentMs, baselineMs);
273
+ const spike = spikes.find((s) => s.slug === "sp/model");
274
+ expect(spike?.recentFailures).toBe(6);
275
+ expect(spike?.recentRate).toBeCloseTo(1, 6);
276
+ expect(spike?.baselineRate).toBeCloseTo(0, 6);
277
+ });
278
+
279
+ test("token calibration accumulates and stays unreported until enough samples", async () => {
280
+ const { ledger } = await setup(cfgWith());
281
+ // No pending estimate was registered for these conversations, so
282
+ // nothing calibrates: the ratio must stay unknown rather than guess.
283
+ await ledger.record(entry({ id: "c1", slug: "cal/model" }));
284
+ expect(await ledger.tokenRatio("gpt")).toBeNull();
285
+ });
286
+
287
+ test("blended rate stays null until the window has enough priced samples", async () => {
288
+ const { ledger } = await setup(cfgWith());
289
+ // findModel returns null here, so no cost_breakdown is written and the
290
+ // blend has nothing to apportion — null, not a fabricated rate.
291
+ await ledger.record(entry({ id: "bl1", slug: "bl/model", reportedUsd: 1 }));
292
+ expect(await ledger.blendedRate(30)).toBeNull();
293
+ });
294
+
295
+ // Two replicas, one store. Everything below is about a SECOND handle
296
+ // seeing what the first one wrote, because that is what a cap and a warm
297
+ // cache depend on once the store stops being a local file.
298
+ test("a concurrent bootstrap does not lose the race", async () => {
299
+ // `CREATE TABLE IF NOT EXISTS` is not atomic on Postgres: two replicas
300
+ // booting together both pass the existence check, and the loser used
301
+ // to die on the unique index over pg_type. Measured: one of two
302
+ // replicas started at once exited with "duplicate key value violates
303
+ // unique constraint pg_type_typname_nsp_index".
304
+ const url = engine.name === "sqlite" ? `sqlite://${join(tmpdir(), `boot-race-${process.pid}-${Date.now()}.db`)}` : engine.url;
305
+ const a = openSqlDb(url);
306
+ const b = openSqlDb(url);
307
+ try {
308
+ await Promise.all([migrateStore(a), migrateStore(b)]);
309
+ // Both handles land on a usable store, not a half-created one.
310
+ for (const handle of [a, b]) await handle.sql.unsafe("SELECT COUNT(*) FROM ledger");
311
+ } finally {
312
+ await a.close();
313
+ await b.close();
314
+ }
315
+ });
316
+
317
+ test("a second handle reads the spend and the warm conversation the first one wrote", async () => {
318
+ // The fixture bootstraps the ledger's own tables; the conversation
319
+ // store is the other half of the shared state.
320
+ await migrateStore(db);
321
+ const writer = createSqlLedger(db, cfgWith(), { findModel: () => null });
322
+ const writerConvs = createConversationStore(db);
323
+ const now = Date.now();
324
+ await writer.record(entry({ id: "sh1", slug: "warm/model", harnessId: "u_a", reportedUsd: 0.25, createdAtMs: now }));
325
+ const state = await writerConvs.load("shared-conv");
326
+ state.currentSlug = "warm/model";
327
+ state.currentTier = "moderate";
328
+ state.cacheWarmSlug = "warm/model";
329
+ state.cacheWarmAtMs = now;
330
+ state.turn = 1;
331
+ await writerConvs.save(state);
332
+
333
+ // A replica that has never seen this conversation or this spend.
334
+ const reader = createSqlLedger(db, cfgWith(), { findModel: () => null });
335
+ const readerConvs = createConversationStore(db);
336
+ expect(await reader.spendSince(0)).toBeCloseTo(0.25, 9);
337
+ expect(await reader.spendSince(0, "u_a")).toBeCloseTo(0.25, 9);
338
+ const seen = await readerConvs.load("shared-conv");
339
+ expect([seen.turn, seen.currentSlug, seen.currentTier, seen.cacheWarmSlug]).toEqual([1, "warm/model", "moderate", "warm/model"]);
340
+ });
341
+ });
342
+ }
@@ -12,7 +12,7 @@ const servers = (p: string): Record<string, unknown> => (read(p).mcpServers as R
12
12
  const auth = (p: string): string => (servers(p)[MCP_SERVER_NAME] as { headers: { Authorization: string } }).headers.Authorization;
13
13
 
14
14
  describe("mergeMcpServers", () => {
15
- test("adds the team-context server next to the others and keeps every other key", () => {
15
+ test("adds the team-context server next to the others and keeps every other key", async () => {
16
16
  const before = JSON.stringify({ $schema: "https://x/mcp-schema.json", mcpServers: { agentdox: { type: "http", url: "http://localhost:3003/mcp" } }, other: 1 });
17
17
  const after = mergeMcpServers(before, "https://team.example/mcp", "amrt_k");
18
18
  expect(after).not.toBeNull();
@@ -26,12 +26,12 @@ describe("mergeMcpServers", () => {
26
26
  expect(after!.endsWith("\n")).toBe(true);
27
27
  });
28
28
 
29
- test("an empty or missing file becomes a fresh mcpServers document", () => {
29
+ test("an empty or missing file becomes a fresh mcpServers document", async () => {
30
30
  expect(JSON.parse(mergeMcpServers("", "https://t/mcp", "k")!)).toEqual({ mcpServers: { [MCP_SERVER_NAME]: { type: "http", url: "https://t/mcp", headers: { Authorization: "Bearer k" } } } });
31
31
  expect(JSON.parse(mergeMcpServers(" \n", "https://t/mcp", "k")!)).toHaveProperty("mcpServers");
32
32
  });
33
33
 
34
- test("is idempotent, removes on null, and leaves a file it cannot parse alone", () => {
34
+ test("is idempotent, removes on null, and leaves a file it cannot parse alone", async () => {
35
35
  const one = mergeMcpServers("", "https://t/mcp", "k")!;
36
36
  expect(mergeMcpServers(one, "https://t/mcp", "k")).toBeNull(); // unchanged
37
37
  expect(mergeMcpServers(one, "https://t/mcp", "k2")).not.toBeNull(); // a new key rewrites
@@ -54,7 +54,7 @@ describe("connect writes the team MCP endpoint", () => {
54
54
  return { home, agent, env };
55
55
  }
56
56
 
57
- test("into omp mcp.json and Claude Code ~/.claude.json, only for the harnesses it configured", () => {
57
+ test("into omp mcp.json and Claude Code ~/.claude.json, only for the harnesses it configured", async () => {
58
58
  const { home, agent, env } = fixture();
59
59
  try {
60
60
  // Pre-existing servers and unrelated settings survive.
@@ -95,7 +95,7 @@ describe("connect writes the team MCP endpoint", () => {
95
95
  }
96
96
  });
97
97
 
98
- test("a dry run reports the files without writing them; no mcp option leaves them alone", () => {
98
+ test("a dry run reports the files without writing them; no mcp option leaves them alone", async () => {
99
99
  const { home, agent, env } = fixture();
100
100
  try {
101
101
  const r = connectRemote({ url: "https://team.example", key: "amrt_k", userId: "u", name: "Ada", profile: false, dryRun: true, only: ["omp"], env, home, packageDir: "/pkg", mcp: { url: "https://team.example/mcp" }, platform: "linux", pathHas: () => false });
@@ -1,16 +1,18 @@
1
1
  import { describe, expect, test } from "bun:test";
2
- import { copyFileSync, mkdtempSync, readdirSync, rmSync } from "node:fs";
3
2
  import { tmpdir } from "node:os";
4
3
  import { join } from "node:path";
5
4
 
5
+ import { openDb } from "../src/util/sqlite.ts";
6
+ import { openSqlDb } from "../src/util/sql.ts";
7
+ import { copyFileSync, mkdtempSync, readdirSync, rmSync } from "node:fs";
8
+
6
9
  import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
7
10
  import { createFeedbackStore } from "../src/cost/feedback.ts";
8
- import { createLedger } from "../src/cost/ledger.ts";
11
+ import { createSqlLedger } from "../src/cost/ledger-sql.ts";
9
12
  import { buildUsageReport } from "../src/cost/report.ts";
10
13
  import { buildDailySummary, createKv } from "../src/cost/summary.ts";
11
14
  import { exportRows, spendUsdSince } from "../src/cost/views.ts";
12
15
  import { createConversationStore } from "../src/router/state.ts";
13
- import { openDb } from "../src/util/sqlite.ts";
14
16
 
15
17
  /**
16
18
  * Every ledger a past release wrote must open under the current bootstrap:
@@ -27,19 +29,22 @@ const files = readdirSync(FIXTURES).filter((f) => /^router-v\d+\.db$/.test(f)).s
27
29
  const CURRENT_VERSION = 19;
28
30
 
29
31
  describe("schema migrations from every shipped version", () => {
30
- test("fixtures exist for the versions that shipped", () => {
32
+ test("fixtures exist for the versions that shipped", async () => {
31
33
  expect(files.map((f) => Number(/\d+/.exec(f)![0]))).toEqual([4, 5, 10, 12, 13, 14, 16]);
32
34
  });
33
35
 
34
36
  for (const file of files) {
35
37
  const from = Number(/\d+/.exec(file)![0]);
36
- test(`v${from} → v${CURRENT_VERSION}: opens, migrates, keeps its rows, and every consumer runs`, () => {
38
+ test(`v${from} → v${CURRENT_VERSION}: opens, migrates, keeps its rows, and every consumer runs`, async () => {
37
39
  const dir = mkdtempSync(join(tmpdir(), "amr-migrate-"));
38
40
  const path = join(dir, "router.db");
39
41
  copyFileSync(join(FIXTURES, file), path);
40
42
  const cfg = structuredClone(DEFAULT_CONFIG);
41
43
  cfg.ledger.path = path;
44
+ // `openDb` is the migration path for a SQLite file: it applies the
45
+ // nineteen versions in order. The shim handle then reads the result.
42
46
  const db = openDb(path);
47
+ const sdb = openSqlDb(path);
43
48
  try {
44
49
  expect((db.query("PRAGMA user_version").get() as { user_version: number }).user_version).toBe(CURRENT_VERSION);
45
50
  // Every column the current code writes exists after migration.
@@ -51,27 +56,28 @@ describe("schema migrations from every shipped version", () => {
51
56
  const row = db.query("SELECT id, error, slug FROM ledger").get() as { id: string; error: string | null; slug: string } | null;
52
57
  expect(row).toEqual({ id: "fixture-id", error: "upstream_error: 502", slug: "fixture-slug" });
53
58
  // Every prepared statement compiles and every consumer runs on the migrated file.
54
- const ledger = createLedger(db, cfg);
55
- const conversations = createConversationStore(db);
56
- createFeedbackStore(db);
59
+ const ledger = createSqlLedger(sdb, cfg, { findModel: () => null });
60
+ const conversations = createConversationStore(sdb);
61
+ createFeedbackStore(sdb);
57
62
  createKv(db);
58
- expect(ledger.recentEntries(5)).toHaveLength(1);
59
- expect(ledger.trust("fixture-slug")).not.toBeNull();
60
- expect(ledger.softFailureSpikes?.()).toEqual([]);
61
- expect(ledger.latestForSession?.("nope")).toBeNull();
62
- expect(conversations.load("fixture-key").key).toBe("fixture-key");
63
- expect(buildUsageReport(db, { windowDays: 3650 }).totals.dispatches).toBe(1);
64
- expect(buildDailySummary(db, {}).current.dispatches).toBe(0);
63
+ expect(await ledger.recentEntries(5)).toHaveLength(1);
64
+ expect(await ledger.trust("fixture-slug")).not.toBeNull();
65
+ expect(await ledger.softFailureSpikes()).toEqual([]);
66
+ expect(await ledger.latestForSession("nope")).toBeNull();
67
+ expect((await conversations.load("fixture-key")).key).toBe("fixture-key");
68
+ expect((await buildUsageReport(sdb, { windowDays: 3650 })).totals.dispatches).toBe(1);
69
+ expect((await buildDailySummary(sdb, {})).current.dispatches).toBe(0);
65
70
  // v18: the fixture's row predates `scope`, so it exports under "" and no
66
71
  // context scope claims its spend.
67
- expect(exportRows(db, 0, null).map((r) => r.scope)).toEqual([""]);
68
- expect(spendUsdSince(db, 0, null, "acme.api")).toBe(0);
69
- expect(spendUsdSince(db, 0, null)).toBeGreaterThanOrEqual(0);
70
- expect(ledger.prune?.(0)?.deleted).toBe(0);
72
+ expect((await exportRows(sdb, 0, null)).map((r) => r.scope)).toEqual([""]);
73
+ expect(await spendUsdSince(sdb, 0, null, "acme.api")).toBe(0);
74
+ expect(await spendUsdSince(sdb, 0, null)).toBeGreaterThanOrEqual(0);
75
+ expect((await ledger.prune(0)).deleted).toBe(0);
71
76
  // v19: the fixture's row predates `redactions`, so nothing claims a
72
77
  // redaction happened on it and the report totals it as zero.
73
- expect(buildUsageReport(db, { windowDays: 3650 }).totals.redactions).toBe(0);
78
+ expect((await buildUsageReport(sdb, { windowDays: 3650 })).totals.redactions).toBe(0);
74
79
  } finally {
80
+ await sdb.close();
75
81
  db.close();
76
82
  try {
77
83
  rmSync(dir, { recursive: true, force: true });
@@ -82,8 +88,8 @@ describe("schema migrations from every shipped version", () => {
82
88
  });
83
89
  }
84
90
 
85
- test("a fresh database lands on the same version as a migrated one", () => {
86
- const db = openDb(":memory:");
91
+ test("a fresh database lands on the same version as a migrated one", async () => {
92
+ const db = openDb(join(tmpdir(), `t-migrations.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
87
93
  try {
88
94
  expect((db.query("PRAGMA user_version").get() as { user_version: number }).user_version).toBe(CURRENT_VERSION);
89
95
  // A fresh ledger has the scope column the bootstrap never spells out in CREATE TABLE.
@@ -49,7 +49,7 @@ function countOf(text: string, needle: string): number {
49
49
  }
50
50
 
51
51
  describe("renderProviderBlock", () => {
52
- test("emits costs in USD per MILLION tokens, not per token", () => {
52
+ test("emits costs in USD per MILLION tokens, not per token", async () => {
53
53
  const providers = providersOf(`providers:\n${BLOCK.split("\n").map((l) => (l === "" ? "" : ` ${l}`)).join("\n")}\n`);
54
54
  const router = providers["auto-model-router"];
55
55
  expect(typeof router).toBe("object");
@@ -60,17 +60,17 @@ describe("renderProviderBlock", () => {
60
60
  expect(BLOCK).toContain(`input: ${cfg.ledger.fallbackBlend.inputPerMtok}`);
61
61
  });
62
62
 
63
- test("declares one model per configured profile", () => {
63
+ test("declares one model per configured profile", async () => {
64
64
  for (const profile of cfg.profiles) expect(BLOCK).toContain(`id: ${profile.id}`);
65
65
  });
66
66
 
67
- test("advertises a keyless openai-compatible provider", () => {
67
+ test("advertises a keyless openai-compatible provider", async () => {
68
68
  expect(BLOCK).toContain("api: openai-completions");
69
69
  expect(BLOCK).toContain("auth: none");
70
70
  expect(BLOCK).toContain("/v1");
71
71
  });
72
72
 
73
- test("with the bridge on, the scope and origin headers name the variables the extension sets", () => {
73
+ test("with the bridge on, the scope and origin headers name the variables the extension sets", async () => {
74
74
  // The file is machine-wide, so neither can be a literal: omp resolves a
75
75
  // header value that names an env var per request (src/context/scope.ts).
76
76
  const bridged = renderProviderBlock(loadConfig({ overrides: { context: { enabled: true, baseUrl: "http://agentdox:3003", token: "t" } } }), null);
@@ -81,7 +81,7 @@ describe("renderProviderBlock", () => {
81
81
  });
82
82
 
83
83
  describe("spliceProviderBlock", () => {
84
- test("preserves every original line and comment", () => {
84
+ test("preserves every original line and comment", async () => {
85
85
  const result = spliceProviderBlock(EXISTING, BLOCK);
86
86
  expect(result.action).toBe("inserted");
87
87
  const after = result.text.split("\n");
@@ -92,13 +92,13 @@ describe("spliceProviderBlock", () => {
92
92
  expect(result.text).toContain("# model's architectural limit.");
93
93
  });
94
94
 
95
- test("leaves the pre-existing providers intact and adds ours", () => {
95
+ test("leaves the pre-existing providers intact and adds ours", async () => {
96
96
  const result = spliceProviderBlock(EXISTING, BLOCK);
97
97
  const providers = providersOf(result.text);
98
98
  expect(Object.keys(providers).sort()).toEqual(["auto-model-router", "fastflowlm", "ollama"]);
99
99
  });
100
100
 
101
- test("is idempotent: a second run replaces rather than duplicates", () => {
101
+ test("is idempotent: a second run replaces rather than duplicates", async () => {
102
102
  const once = spliceProviderBlock(EXISTING, BLOCK);
103
103
  const twice = spliceProviderBlock(once.text, BLOCK);
104
104
  expect(twice.action).toBe("replaced");
@@ -108,7 +108,7 @@ describe("spliceProviderBlock", () => {
108
108
  providersOf(twice.text);
109
109
  });
110
110
 
111
- test("refreshed cost figures replace the old ones in place", () => {
111
+ test("refreshed cost figures replace the old ones in place", async () => {
112
112
  const first = spliceProviderBlock(EXISTING, renderProviderBlock(cfg, null));
113
113
  const updated = spliceProviderBlock(
114
114
  first.text,
@@ -126,20 +126,20 @@ describe("spliceProviderBlock", () => {
126
126
  providersOf(updated.text);
127
127
  });
128
128
 
129
- test("adds a providers mapping when the file has none", () => {
129
+ test("adds a providers mapping when the file has none", async () => {
130
130
  const result = spliceProviderBlock("# just a comment\n", BLOCK);
131
131
  expect(result.text).toContain("# just a comment");
132
132
  expect(countOf(result.text, "providers:")).toBe(1);
133
133
  expect(Object.keys(providersOf(result.text))).toContain("auto-model-router");
134
134
  });
135
135
 
136
- test("creates a whole file from empty input", () => {
136
+ test("creates a whole file from empty input", async () => {
137
137
  const result = spliceProviderBlock("", BLOCK);
138
138
  expect(result.action).toBe("created");
139
139
  expect(Object.keys(providersOf(result.text))).toContain("auto-model-router");
140
140
  });
141
141
 
142
- test("survives a UTF-8 BOM without producing a duplicate providers key", () => {
142
+ test("survives a UTF-8 BOM without producing a duplicate providers key", async () => {
143
143
  // A BOM made the first line read as "\uFEFFproviders:", so the top-level
144
144
  // key was missed and a second one appended -- which makes omp discard the
145
145
  // entire file.
@@ -150,7 +150,7 @@ describe("spliceProviderBlock", () => {
150
150
  expect(Object.keys(providersOf(result.text)).sort()).toEqual(["auto-model-router", "fastflowlm", "ollama"]);
151
151
  });
152
152
 
153
- test("preserves CRLF line endings", () => {
153
+ test("preserves CRLF line endings", async () => {
154
154
  const crlf = EXISTING.replaceAll("\n", "\r\n");
155
155
  const result = spliceProviderBlock(crlf, BLOCK);
156
156
  expect(result.text).toContain("\r\n");
@@ -159,7 +159,7 @@ describe("spliceProviderBlock", () => {
159
159
  providersOf(result.text);
160
160
  });
161
161
 
162
- test("matches the existing child indentation", () => {
162
+ test("matches the existing child indentation", async () => {
163
163
  // EXISTING indents providers' children by four spaces; mixing widths
164
164
  // under one mapping is invalid YAML.
165
165
  const result = spliceProviderBlock(EXISTING, BLOCK);
@@ -168,7 +168,7 @@ describe("spliceProviderBlock", () => {
168
168
  expect(begin?.startsWith(" #")).toBe(true);
169
169
  });
170
170
 
171
- test("a two-space file gets two-space children", () => {
171
+ test("a two-space file gets two-space children", async () => {
172
172
  const twoSpace = "providers:\n ollama:\n auth: none\n";
173
173
  const result = spliceProviderBlock(twoSpace, BLOCK);
174
174
  const begin = result.text.split("\n").find((l) => l.includes(BEGIN_GUARD));
@@ -178,19 +178,19 @@ describe("spliceProviderBlock", () => {
178
178
  });
179
179
 
180
180
  describe("assertUsableModelsYaml", () => {
181
- test("accepts a correctly spliced result", () => {
181
+ test("accepts a correctly spliced result", async () => {
182
182
  expect(() => assertUsableModelsYaml(spliceProviderBlock(EXISTING, BLOCK).text)).not.toThrow();
183
183
  });
184
184
 
185
- test("rejects a duplicate providers key, which YAML treats as fatal", () => {
185
+ test("rejects a duplicate providers key, which YAML treats as fatal", async () => {
186
186
  expect(() => assertUsableModelsYaml("providers:\n a:\n auth: none\nproviders:\n b:\n auth: none\n")).toThrow();
187
187
  });
188
188
 
189
- test("rejects a file without our provider", () => {
189
+ test("rejects a file without our provider", async () => {
190
190
  expect(() => assertUsableModelsYaml("providers:\n ollama:\n auth: none\n")).toThrow();
191
191
  });
192
192
 
193
- test("rejects a file with no providers mapping", () => {
193
+ test("rejects a file with no providers mapping", async () => {
194
194
  expect(() => assertUsableModelsYaml("something: else\n")).toThrow();
195
195
  });
196
196
  });