auto-model-router 0.30.3 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/.omp-plugin/marketplace.json +2 -2
  2. package/README.md +32 -2
  3. package/omp-extension/router-configure.ts +9 -7
  4. package/package.json +1 -1
  5. package/src/cli/config-cmd.ts +8 -7
  6. package/src/cli/explain.ts +10 -5
  7. package/src/cli/export.ts +6 -5
  8. package/src/cli/models.ts +10 -7
  9. package/src/cli/report.ts +6 -1
  10. package/src/cli/stats.ts +7 -7
  11. package/src/config/load.ts +10 -1
  12. package/src/config/types.ts +10 -1
  13. package/src/context/bridge.ts +7 -7
  14. package/src/context/index.ts +3 -3
  15. package/src/context/store.ts +39 -56
  16. package/src/context/types.ts +7 -6
  17. package/src/cost/blended.ts +28 -7
  18. package/src/cost/feedback.ts +33 -37
  19. package/src/cost/ledger-sql.ts +547 -0
  20. package/src/cost/ledger.ts +30 -459
  21. package/src/cost/report.ts +171 -129
  22. package/src/cost/retention.ts +10 -10
  23. package/src/cost/summary.ts +15 -10
  24. package/src/cost/types.ts +43 -62
  25. package/src/cost/views.ts +79 -49
  26. package/src/eval/calibrate.ts +47 -12
  27. package/src/eval/run.ts +18 -2
  28. package/src/lib.ts +6 -2
  29. package/src/router/candidates.ts +7 -15
  30. package/src/router/classify.ts +6 -4
  31. package/src/router/index.ts +95 -9
  32. package/src/router/select.ts +38 -21
  33. package/src/router/state.ts +90 -102
  34. package/src/router/types.ts +11 -5
  35. package/src/server/advise.ts +6 -4
  36. package/src/server/compaction-digest.ts +1 -1
  37. package/src/server/digest.ts +9 -10
  38. package/src/server/http.ts +109 -46
  39. package/src/server/providers.ts +18 -4
  40. package/src/server/turn.ts +32 -9
  41. package/src/tokens/estimate.ts +16 -6
  42. package/src/upstream/ollama-usage.ts +21 -11
  43. package/src/util/schema.ts +201 -0
  44. package/src/util/sql.ts +246 -0
  45. package/src/wire/anthropic/messages.ts +3 -4
  46. package/src/wire/openai/request.ts +1 -0
  47. package/src/wire/types.ts +7 -0
  48. package/test/anthropic-wire.test.ts +9 -9
  49. package/test/benchmark-feeds.test.ts +7 -7
  50. package/test/cache-control.test.ts +7 -7
  51. package/test/cache-estimate.test.ts +5 -5
  52. package/test/catalog-view.test.ts +4 -4
  53. package/test/catalog.test.ts +11 -11
  54. package/test/classify.test.ts +24 -24
  55. package/test/compaction.test.ts +20 -20
  56. package/test/config-wizard.test.ts +32 -32
  57. package/test/config.test.ts +10 -10
  58. package/test/connect-harnesses.test.ts +11 -11
  59. package/test/context-bridge.test.ts +40 -30
  60. package/test/context-prune.test.ts +43 -36
  61. package/test/context-query.test.ts +8 -8
  62. package/test/controls.test.ts +54 -27
  63. package/test/cost.test.ts +12 -12
  64. package/test/digest.test.ts +55 -44
  65. package/test/embed-lifecycle.test.ts +5 -5
  66. package/test/embed-logic.test.ts +26 -26
  67. package/test/escalate.test.ts +17 -17
  68. package/test/eval.test.ts +73 -16
  69. package/test/executable.test.ts +6 -6
  70. package/test/exploration.test.ts +19 -20
  71. package/test/failover.test.ts +22 -21
  72. package/test/fakes.ts +105 -0
  73. package/test/features.test.ts +21 -21
  74. package/test/harness-requests.test.ts +3 -3
  75. package/test/harness-switch.test.ts +5 -5
  76. package/test/hold-exploration.test.ts +13 -13
  77. package/test/hot-reload.test.ts +5 -5
  78. package/test/learned.test.ts +5 -5
  79. package/test/ledger-sql.test.ts +342 -0
  80. package/test/mcp-entry.test.ts +5 -5
  81. package/test/migrations.test.ts +28 -22
  82. package/test/models-yml.test.ts +18 -18
  83. package/test/ollama.test.ts +40 -34
  84. package/test/omp-credentials.test.ts +16 -16
  85. package/test/policy.test.ts +3 -3
  86. package/test/reconfigure.test.ts +4 -4
  87. package/test/redaction.test.ts +41 -35
  88. package/test/remote.test.ts +12 -12
  89. package/test/report-logic.test.ts +8 -8
  90. package/test/report.test.ts +95 -87
  91. package/test/retention.test.ts +79 -66
  92. package/test/schema.test.ts +123 -0
  93. package/test/scope.test.ts +8 -8
  94. package/test/select.test.ts +216 -257
  95. package/test/skills.test.ts +3 -3
  96. package/test/sql-shim.test.ts +154 -0
  97. package/test/state.test.ts +43 -36
  98. package/test/summary.test.ts +38 -27
  99. package/test/tier-plan.test.ts +45 -62
  100. package/test/toast-logic.test.ts +31 -31
  101. package/test/tokens.test.ts +95 -80
  102. package/test/trust-attribution.test.ts +217 -187
  103. package/test/trust-window.test.ts +37 -32
  104. package/test/turn.test.ts +55 -23
  105. package/test/upstreams.test.ts +13 -13
  106. package/test/views.test.ts +81 -59
  107. package/test/wire-request.test.ts +17 -17
  108. package/test/wire-responses.test.ts +4 -4
  109. package/tools/agentdox-e2e.ts +5 -2
  110. package/tools/export-benchmarks.ts +5 -5
  111. package/tools/ledger-parity.ts +266 -0
  112. package/tools/replay.ts +16 -8
@@ -1,10 +1,14 @@
1
1
  import { describe, expect, test } from "bun:test";
2
+ import { tmpdir } from "node:os";
3
+ import { join } from "node:path";
4
+
5
+ import { migrateStore } from "../src/util/schema.ts";
6
+ import { num, openSqlDb } from "../src/util/sql.ts";
2
7
 
3
8
  import { loadConfig } from "../src/config/load.ts";
4
- import { createLedger } from "../src/cost/ledger.ts";
9
+ import { createSqlLedger } from "../src/cost/ledger-sql.ts";
5
10
  import { EMPTY_USAGE, type LedgerEntry } from "../src/cost/types.ts";
6
11
  import { adjustPendingEstimate, DEFAULT_BYTES_PER_TOKEN, estimatePromptTokens, estimateTokens } from "../src/tokens/estimate.ts";
7
- import { openDb } from "../src/util/sqlite.ts";
8
12
  import { parseChatRequest } from "../src/wire/openai/request.ts";
9
13
 
10
14
  const cfg = loadConfig({});
@@ -48,34 +52,35 @@ function entry(over: Partial<LedgerEntry>): LedgerEntry {
48
52
  }
49
53
 
50
54
  describe("estimateTokens", () => {
51
- test("uses the default ratio for an unknown tokenizer family", () => {
55
+ test("uses the default ratio for an unknown tokenizer family", async () => {
52
56
  expect(estimateTokens(3600, "no-such-tokenizer", null)).toBe(Math.ceil(3600 / DEFAULT_BYTES_PER_TOKEN));
53
57
  });
54
58
 
55
- test("scales linearly with byte count and never goes negative", () => {
59
+ test("scales linearly with byte count and never goes negative", async () => {
56
60
  expect(estimateTokens(0, "gpt", null)).toBe(0);
57
61
  const small = estimateTokens(1000, "gpt", null);
58
62
  const large = estimateTokens(10_000, "gpt", null);
59
63
  expect(large).toBeGreaterThan(small);
60
64
  });
61
65
 
62
- test("a code-dense family estimates more tokens for the same bytes", () => {
66
+ test("a code-dense family estimates more tokens for the same bytes", async () => {
63
67
  // BPE tokenizers emit more tokens per character on code than on prose,
64
68
  // so a lower bytes-per-token ratio must yield a higher token count.
65
69
  expect(estimateTokens(10_000, "deepseek", null)).toBeGreaterThan(estimateTokens(10_000, "gpt", null));
66
70
  });
67
71
 
68
- test("is case-insensitive about the tokenizer name", () => {
72
+ test("is case-insensitive about the tokenizer name", async () => {
69
73
  expect(estimateTokens(5000, "Claude", null)).toBe(estimateTokens(5000, "claude", null));
70
74
  });
71
75
  });
72
76
 
73
77
  describe("ledger calibration", () => {
74
- test("a calibrated ratio replaces the family default once enough samples land", () => {
75
- const db = openDb(":memory:");
78
+ test("a calibrated ratio replaces the family default once enough samples land", async () => {
79
+ const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
80
+ await migrateStore(db);
76
81
  try {
77
- const ledger = createLedger(db, cfg);
78
- expect(ledger.tokenRatio("claude")).toBeNull();
82
+ const ledger = createSqlLedger(db, cfg, { findModel: () => null });
83
+ expect(await ledger.tokenRatio("claude")).toBeNull();
79
84
 
80
85
  const req = parseChatRequest(
81
86
  { model: "auto", messages: [{ role: "user", content: "x".repeat(4000) }] },
@@ -86,8 +91,8 @@ describe("ledger calibration", () => {
86
91
  // bytes/token; the ledger must converge on the measurement.
87
92
  const observedTokens = Math.round(req.promptBytes / 2);
88
93
  for (let i = 0; i < 30; i++) {
89
- estimatePromptTokens(req, "claude", ledger);
90
- ledger.record(
94
+ estimatePromptTokens(req, "claude", null);
95
+ await ledger.record(
91
96
  entry({
92
97
  conversationKey: req.conversationKey,
93
98
  usage: { ...EMPTY_USAGE, promptTokens: observedTokens, completionTokens: 10 },
@@ -95,33 +100,35 @@ describe("ledger calibration", () => {
95
100
  );
96
101
  }
97
102
 
98
- const ratio = ledger.tokenRatio("claude");
103
+ const ratio = await ledger.tokenRatio("claude");
99
104
  expect(ratio).not.toBeNull();
100
105
  if (ratio === null) return;
101
106
  expect(ratio).toBeCloseTo(2, 1);
102
107
 
103
- // And the estimate now follows the measurement, not the default.
104
- const calibrated = estimateTokens(4000, "claude", ledger);
108
+ // And the estimate follows the measurement, not the family default.
109
+ const calibrated = estimateTokens(4000, "claude", ratio);
105
110
  expect(calibrated).toBeGreaterThan(estimateTokens(4000, "claude", null));
106
111
  } finally {
107
- db.close();
112
+ await db.close();
108
113
  }
109
114
  });
110
115
 
111
- test("an uncalibrated family still falls back to its default", () => {
112
- const db = openDb(":memory:");
116
+ test("an uncalibrated family still falls back to its default", async () => {
117
+ const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
118
+ await migrateStore(db);
113
119
  try {
114
- const ledger = createLedger(db, cfg);
115
- expect(ledger.tokenRatio("gemini")).toBeNull();
116
- expect(estimateTokens(3600, "gemini", ledger)).toBe(estimateTokens(3600, "gemini", null));
120
+ const ledger = createSqlLedger(db, cfg, { findModel: () => null });
121
+ expect(await ledger.tokenRatio("gemini")).toBeNull();
122
+ // An uncalibrated family reads null, which is exactly the default path.
123
+ expect(estimateTokens(3600, "gemini", await ledger.tokenRatio("gemini"))).toBe(estimateTokens(3600, "gemini", null));
117
124
  } finally {
118
- db.close();
125
+ await db.close();
119
126
  }
120
127
  });
121
128
  });
122
129
 
123
130
  describe("estimatePromptTokens", () => {
124
- test("counts tool schemas, not just message text", () => {
131
+ test("counts tool schemas, not just message text", async () => {
125
132
  const bare = parseChatRequest({ model: "auto", messages: [{ role: "user", content: "hi" }] }, new Headers());
126
133
  const withTools = parseChatRequest(
127
134
  {
@@ -143,7 +150,7 @@ describe("estimatePromptTokens", () => {
143
150
  expect(estimatePromptTokens(withTools, "gpt", null)).toBeGreaterThan(estimatePromptTokens(bare, "gpt", null));
144
151
  });
145
152
 
146
- test("charges a per-image allowance on top of text", () => {
153
+ test("charges a per-image allowance on top of text", async () => {
147
154
  const text = parseChatRequest(
148
155
  { model: "auto", messages: [{ role: "user", content: [{ type: "text", text: "describe" }] }] },
149
156
  new Headers(),
@@ -173,99 +180,103 @@ describe("calibration hygiene (review 2026-09-05 §8)", () => {
173
180
  return parseChatRequest({ model: "auto", messages: [{ role: "user", content: text }] }, new Headers());
174
181
  }
175
182
 
176
- test("a provider reporting impossible token counts never calibrates its family", () => {
177
- const db = openDb(":memory:");
183
+ test("a provider reporting impossible token counts never calibrates its family", async () => {
184
+ const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
185
+ await migrateStore(db);
178
186
  try {
179
- const ledger = createLedger(db, cfg);
187
+ const ledger = createSqlLedger(db, cfg, { findModel: () => null });
180
188
  const req = requestOf("x".repeat(10_000));
181
189
  for (let i = 0; i < 30; i++) {
182
- estimatePromptTokens(req, "qwen3", ledger);
190
+ estimatePromptTokens(req, "qwen3", null);
183
191
  // 0.4 bytes/token: ~8x what the bytes imply (seen live from one provider).
184
- ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 0.4) } }));
192
+ await ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 0.4) } }));
185
193
  }
186
- expect(ledger.tokenRatio("qwen3")).toBeNull();
194
+ expect(await ledger.tokenRatio("qwen3")).toBeNull();
187
195
  for (let i = 0; i < 30; i++) {
188
- estimatePromptTokens(req, "qwen3", ledger);
189
- ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 3.2) } }));
196
+ estimatePromptTokens(req, "qwen3", null);
197
+ await ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 3.2) } }));
190
198
  }
191
- expect(ledger.tokenRatio("qwen3")).toBeCloseTo(3.2, 1);
199
+ expect(await ledger.tokenRatio("qwen3")).toBeCloseTo(3.2, 1);
192
200
  } finally {
193
- db.close();
201
+ await db.close();
194
202
  }
195
203
  });
196
204
 
197
- test("adjustPendingEstimate calibrates against the dispatched bytes, not the raw request", () => {
198
- const db = openDb(":memory:");
205
+ test("adjustPendingEstimate calibrates against the dispatched bytes, not the raw request", async () => {
206
+ const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
207
+ await migrateStore(db);
199
208
  try {
200
- const ledger = createLedger(db, cfg);
209
+ const ledger = createSqlLedger(db, cfg, { findModel: () => null });
201
210
  const req = requestOf("y".repeat(10_000));
202
211
  for (let i = 0; i < 30; i++) {
203
- estimatePromptTokens(req, "grok", ledger);
212
+ estimatePromptTokens(req, "grok", null);
204
213
  // Compaction halved the prompt before dispatch; the upstream billed the half.
205
214
  adjustPendingEstimate(req.conversationKey, req.promptBytes / 2);
206
- ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 2 / 3.5) } }));
215
+ await ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 2 / 3.5) } }));
207
216
  }
208
217
  // Paired with the raw bytes this would have learned 7.0; the dispatched bytes give the true 3.5.
209
- expect(ledger.tokenRatio("grok")).toBeCloseTo(3.5, 1);
218
+ expect(await ledger.tokenRatio("grok")).toBeCloseTo(3.5, 1);
210
219
  } finally {
211
- db.close();
220
+ await db.close();
212
221
  }
213
222
  });
214
223
  });
215
224
 
216
225
  describe("ledger.escalationCost", () => {
217
- test("measures what escalated retries bill per prompt token, once enough exist", () => {
218
- const db = openDb(":memory:");
226
+ test("measures what escalated retries bill per prompt token, once enough exist", async () => {
227
+ const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
228
+ await migrateStore(db);
219
229
  try {
220
- const ledger = createLedger(db, cfg);
230
+ const ledger = createSqlLedger(db, cfg, { findModel: () => null });
221
231
  for (let i = 0; i < 9; i++) {
222
- ledger.record(entry({ attempt: 1, reportedUsd: 0.02, usage: { ...EMPTY_USAGE, promptTokens: 1_000 } }));
232
+ await ledger.record(entry({ attempt: 1, reportedUsd: 0.02, usage: { ...EMPTY_USAGE, promptTokens: 1_000 } }));
223
233
  }
224
- expect(ledger.escalationCost?.(7)).toBeNull(); // 9 < the sample floor
225
- ledger.record(entry({ attempt: 1, reportedUsd: 0.02, usage: { ...EMPTY_USAGE, promptTokens: 1_000 } }));
234
+ expect(await ledger.escalationCost(7)).toBeNull(); // 9 < the sample floor
235
+ await ledger.record(entry({ attempt: 1, reportedUsd: 0.02, usage: { ...EMPTY_USAGE, promptTokens: 1_000 } }));
226
236
  // Errored retries carry no usage and are excluded.
227
- ledger.record(entry({ attempt: 1, reportedUsd: null, error: "upstream_error: boom", usage: EMPTY_USAGE }));
237
+ await ledger.record(entry({ attempt: 1, reportedUsd: null, error: "upstream_error: boom", usage: EMPTY_USAGE }));
228
238
  // Memoised: a fresh ledger reads through.
229
- const fresh = createLedger(db, cfg);
230
- const cost = fresh.escalationCost?.(7);
239
+ const fresh = createSqlLedger(db, cfg, { findModel: () => null });
240
+ const cost = await fresh.escalationCost(7);
231
241
  expect(cost).not.toBeNull();
232
- expect(cost!.samples).toBe(10);
233
- expect(cost!.usdPerPromptToken).toBeCloseTo(0.02 / 1_000, 8);
242
+ expect(cost?.samples).toBe(10);
243
+ expect(cost?.usdPerPromptToken).toBeCloseTo(0.02 / 1_000, 8);
234
244
  } finally {
235
- db.close();
245
+ await db.close();
236
246
  }
237
247
  });
238
248
  });
239
249
 
240
250
 
241
251
  describe("ledger.softFailureSpikes", () => {
242
- test("flags a model whose last-hour failure rate is a spike against its own 7-day baseline", () => {
243
- const db = openDb(":memory:");
252
+ test("flags a model whose last-hour failure rate is a spike against its own 7-day baseline", async () => {
253
+ const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
254
+ await migrateStore(db);
244
255
  try {
245
- const ledger = createLedger(db, cfg);
256
+ const ledger = createSqlLedger(db, cfg, { findModel: () => null });
246
257
  const now = 1_800_000_000_000;
247
258
  const H = 3_600_000;
248
259
  // Baseline: 100 dispatches over the prior week at 5% soft failures.
249
260
  for (let i = 0; i < 100; i++) {
250
- ledger.record(entry({ createdAtMs: now - 2 * H - i * 60 * 60_000, escalationSignal: i % 20 === 0 ? "empty_completion" : null, wasted: i % 20 === 0 }));
261
+ await ledger.record(entry({ createdAtMs: now - 2 * H - i * 60 * 60_000, escalationSignal: i % 20 === 0 ? "empty_completion" : null, wasted: i % 20 === 0 }));
251
262
  }
252
263
  // Last hour: 10 dispatches, 4 soft failures (40%): a spike.
253
264
  for (let i = 0; i < 10; i++) {
254
- ledger.record(entry({ createdAtMs: now - 5 * 60_000 - i * 60_000, escalationSignal: i < 4 ? "repeat_tool_call" : null, wasted: i < 4 }));
265
+ await ledger.record(entry({ createdAtMs: now - 5 * 60_000 - i * 60_000, escalationSignal: i < 4 ? "repeat_tool_call" : null, wasted: i < 4 }));
255
266
  }
256
267
  // A second model with plenty of failures but a matching baseline is not spiking.
257
268
  for (let i = 0; i < 100; i++) {
258
- ledger.record(entry({ slug: "x/steady", servedSlug: "x/steady", createdAtMs: now - 2 * H - i * 60 * 60_000, error: i % 2 === 0 ? "upstream_error: 502" : null }));
269
+ await ledger.record(entry({ slug: "x/steady", servedSlug: "x/steady", createdAtMs: now - 2 * H - i * 60 * 60_000, error: i % 2 === 0 ? "upstream_error: 502" : null }));
259
270
  }
260
271
  for (let i = 0; i < 10; i++) {
261
- ledger.record(entry({ slug: "x/steady", servedSlug: "x/steady", createdAtMs: now - 5 * 60_000 - i * 60_000, error: i % 2 === 0 ? "upstream_error: 502" : null }));
272
+ await ledger.record(entry({ slug: "x/steady", servedSlug: "x/steady", createdAtMs: now - 5 * 60_000 - i * 60_000, error: i % 2 === 0 ? "upstream_error: 502" : null }));
262
273
  }
263
274
  // Aborted and quota errors are not attributable; digest rows are side calls.
264
275
  for (let i = 0; i < 10; i++) {
265
- ledger.record(entry({ slug: "x/aborted", servedSlug: "x/aborted", createdAtMs: now - 5 * 60_000 - i * 60_000, error: "aborted: client closed" }));
266
- ledger.record(entry({ slug: "x/digest", servedSlug: "x/digest", requestedModel: "digest", createdAtMs: now - 5 * 60_000 - i * 60_000, error: "upstream_error: 500" }));
276
+ await ledger.record(entry({ slug: "x/aborted", servedSlug: "x/aborted", createdAtMs: now - 5 * 60_000 - i * 60_000, error: "aborted: client closed" }));
277
+ await ledger.record(entry({ slug: "x/digest", servedSlug: "x/digest", requestedModel: "digest", createdAtMs: now - 5 * 60_000 - i * 60_000, error: "upstream_error: 500" }));
267
278
  }
268
- const spikes = ledger.softFailureSpikes?.(now) ?? [];
279
+ const spikes = await ledger.softFailureSpikes(now);
269
280
  expect(spikes.map((s) => s.slug)).toEqual(["openai/gpt-5-mini"]);
270
281
  const s = spikes[0]!;
271
282
  expect(s.recentDispatches).toBe(10);
@@ -275,35 +286,39 @@ describe("ledger.softFailureSpikes", () => {
275
286
  expect(s.baselineFailures).toBe(5);
276
287
  expect(s.baselineRate).toBeCloseTo(0.05, 6);
277
288
  // Too few recent dispatches: nothing spikes, however high the rate.
278
- expect(ledger.softFailureSpikes?.(now, 3 * 60_000)).toEqual([]);
289
+ expect(await ledger.softFailureSpikes(now, 3 * 60_000)).toEqual([]);
279
290
  } finally {
280
- db.close();
291
+ await db.close();
281
292
  }
282
293
  });
283
294
  });
284
295
 
285
296
  describe("ledger.prune and markWasted", () => {
286
- test("prune deletes rows past retention and 0 keeps everything; markWasted flips one row", () => {
287
- const db = openDb(":memory:");
297
+ test("prune deletes rows past retention and 0 keeps everything; markWasted flips one row", async () => {
298
+ const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
299
+ await migrateStore(db);
288
300
  try {
289
- const ledger = createLedger(db, cfg);
301
+ const ledger = createSqlLedger(db, cfg, { findModel: () => null });
290
302
  const now = 1_800_000_000_000;
291
303
  const DAY = 86_400_000;
292
- for (let i = 0; i < 5; i++) ledger.record(entry({ createdAtMs: now - i * 100 * DAY }));
293
- expect(ledger.prune?.(0, now)).toEqual({ deleted: 0, oldestKeptMs: now - 400 * DAY });
294
- expect(ledger.recentEntries(10)).toHaveLength(5);
295
- db.run("INSERT INTO ollama_meter_samples (at_ms, meter_usd, ledger_usd) VALUES (?, 1, 1), (?, 2, 2)", [now - 400 * DAY, now - DAY]);
296
- expect(ledger.prune?.(365, now)?.deleted).toBe(1); // only the 400-day-old row
297
- expect((db.query("SELECT COUNT(*) AS n FROM ollama_meter_samples").get() as { n: number }).n).toBe(1);
298
- expect(ledger.recentEntries(10)).toHaveLength(4);
299
- expect(ledger.prune?.(150, now)?.deleted).toBe(2); // 200 and 300 days old
300
- const left = ledger.recentEntries(10);
304
+ for (let i = 0; i < 5; i++) await ledger.record(entry({ createdAtMs: now - i * 100 * DAY }));
305
+ expect(await ledger.prune(0, now)).toEqual({ deleted: 0, oldestKeptMs: now - 400 * DAY });
306
+ expect(await ledger.recentEntries(10)).toHaveLength(5);
307
+ for (const atMs of [now - 400 * DAY, now - DAY]) {
308
+ await db.sql`INSERT INTO ollama_meter_samples (at_ms, meter_usd, ledger_usd) VALUES (${atMs}, 1, 1)`;
309
+ }
310
+ expect((await ledger.prune(365, now))?.deleted).toBe(1); // only the 400-day-old row
311
+ const samples = await db.one<{ n: unknown }>("SELECT COUNT(*) AS n FROM ollama_meter_samples");
312
+ expect(num(samples?.n)).toBe(1);
313
+ expect(await ledger.recentEntries(10)).toHaveLength(4);
314
+ expect((await ledger.prune(150, now))?.deleted).toBe(2); // 200 and 300 days old
315
+ const left = await ledger.recentEntries(10);
301
316
  expect(left).toHaveLength(2);
302
317
  expect(left.every((e) => e.wasted === false)).toBe(true);
303
- ledger.markWasted?.(left[0]!.id);
304
- expect(ledger.recentEntries(10).find((e) => e.id === left[0]!.id)?.wasted).toBe(true);
318
+ await ledger.markWasted(left[0]!.id);
319
+ expect((await ledger.recentEntries(10)).find((e) => e.id === left[0]!.id)?.wasted).toBe(true);
305
320
  } finally {
306
- db.close();
321
+ await db.close();
307
322
  }
308
323
  });
309
324
  });