auto-model-router 0.30.3 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +32 -2
- package/omp-extension/router-configure.ts +9 -7
- package/package.json +1 -1
- package/src/cli/config-cmd.ts +8 -7
- package/src/cli/explain.ts +10 -5
- package/src/cli/export.ts +6 -5
- package/src/cli/models.ts +10 -7
- package/src/cli/report.ts +6 -1
- package/src/cli/stats.ts +7 -7
- package/src/config/load.ts +10 -1
- package/src/config/types.ts +10 -1
- package/src/context/bridge.ts +7 -7
- package/src/context/index.ts +3 -3
- package/src/context/store.ts +39 -56
- package/src/context/types.ts +7 -6
- package/src/cost/blended.ts +28 -7
- package/src/cost/feedback.ts +33 -37
- package/src/cost/ledger-sql.ts +547 -0
- package/src/cost/ledger.ts +30 -459
- package/src/cost/report.ts +171 -129
- package/src/cost/retention.ts +10 -10
- package/src/cost/summary.ts +15 -10
- package/src/cost/types.ts +43 -62
- package/src/cost/views.ts +79 -49
- package/src/eval/calibrate.ts +47 -12
- package/src/eval/run.ts +18 -2
- package/src/lib.ts +6 -2
- package/src/router/candidates.ts +7 -15
- package/src/router/classify.ts +6 -4
- package/src/router/index.ts +95 -9
- package/src/router/select.ts +38 -21
- package/src/router/state.ts +90 -102
- package/src/router/types.ts +11 -5
- package/src/server/advise.ts +6 -4
- package/src/server/compaction-digest.ts +1 -1
- package/src/server/digest.ts +9 -10
- package/src/server/http.ts +109 -46
- package/src/server/providers.ts +18 -4
- package/src/server/turn.ts +32 -9
- package/src/tokens/estimate.ts +16 -6
- package/src/upstream/ollama-usage.ts +21 -11
- package/src/util/schema.ts +201 -0
- package/src/util/sql.ts +246 -0
- package/src/wire/anthropic/messages.ts +3 -4
- package/src/wire/openai/request.ts +1 -0
- package/src/wire/types.ts +7 -0
- package/test/anthropic-wire.test.ts +9 -9
- package/test/benchmark-feeds.test.ts +7 -7
- package/test/cache-control.test.ts +7 -7
- package/test/cache-estimate.test.ts +5 -5
- package/test/catalog-view.test.ts +4 -4
- package/test/catalog.test.ts +11 -11
- package/test/classify.test.ts +24 -24
- package/test/compaction.test.ts +20 -20
- package/test/config-wizard.test.ts +32 -32
- package/test/config.test.ts +10 -10
- package/test/connect-harnesses.test.ts +11 -11
- package/test/context-bridge.test.ts +40 -30
- package/test/context-prune.test.ts +43 -36
- package/test/context-query.test.ts +8 -8
- package/test/controls.test.ts +54 -27
- package/test/cost.test.ts +12 -12
- package/test/digest.test.ts +55 -44
- package/test/embed-lifecycle.test.ts +5 -5
- package/test/embed-logic.test.ts +26 -26
- package/test/escalate.test.ts +17 -17
- package/test/eval.test.ts +73 -16
- package/test/executable.test.ts +6 -6
- package/test/exploration.test.ts +19 -20
- package/test/failover.test.ts +22 -21
- package/test/fakes.ts +105 -0
- package/test/features.test.ts +21 -21
- package/test/harness-requests.test.ts +3 -3
- package/test/harness-switch.test.ts +5 -5
- package/test/hold-exploration.test.ts +13 -13
- package/test/hot-reload.test.ts +5 -5
- package/test/learned.test.ts +5 -5
- package/test/ledger-sql.test.ts +342 -0
- package/test/mcp-entry.test.ts +5 -5
- package/test/migrations.test.ts +28 -22
- package/test/models-yml.test.ts +18 -18
- package/test/ollama.test.ts +40 -34
- package/test/omp-credentials.test.ts +16 -16
- package/test/policy.test.ts +3 -3
- package/test/reconfigure.test.ts +4 -4
- package/test/redaction.test.ts +41 -35
- package/test/remote.test.ts +12 -12
- package/test/report-logic.test.ts +8 -8
- package/test/report.test.ts +95 -87
- package/test/retention.test.ts +79 -66
- package/test/schema.test.ts +123 -0
- package/test/scope.test.ts +8 -8
- package/test/select.test.ts +216 -257
- package/test/skills.test.ts +3 -3
- package/test/sql-shim.test.ts +154 -0
- package/test/state.test.ts +43 -36
- package/test/summary.test.ts +38 -27
- package/test/tier-plan.test.ts +45 -62
- package/test/toast-logic.test.ts +31 -31
- package/test/tokens.test.ts +95 -80
- package/test/trust-attribution.test.ts +217 -187
- package/test/trust-window.test.ts +37 -32
- package/test/turn.test.ts +55 -23
- package/test/upstreams.test.ts +13 -13
- package/test/views.test.ts +81 -59
- package/test/wire-request.test.ts +17 -17
- package/test/wire-responses.test.ts +4 -4
- package/tools/agentdox-e2e.ts +5 -2
- package/tools/export-benchmarks.ts +5 -5
- package/tools/ledger-parity.ts +266 -0
- package/tools/replay.ts +16 -8
package/test/tokens.test.ts
CHANGED
|
@@ -1,10 +1,14 @@
|
|
|
1
1
|
import { describe, expect, test } from "bun:test";
|
|
2
|
+
import { tmpdir } from "node:os";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
|
|
5
|
+
import { migrateStore } from "../src/util/schema.ts";
|
|
6
|
+
import { num, openSqlDb } from "../src/util/sql.ts";
|
|
2
7
|
|
|
3
8
|
import { loadConfig } from "../src/config/load.ts";
|
|
4
|
-
import {
|
|
9
|
+
import { createSqlLedger } from "../src/cost/ledger-sql.ts";
|
|
5
10
|
import { EMPTY_USAGE, type LedgerEntry } from "../src/cost/types.ts";
|
|
6
11
|
import { adjustPendingEstimate, DEFAULT_BYTES_PER_TOKEN, estimatePromptTokens, estimateTokens } from "../src/tokens/estimate.ts";
|
|
7
|
-
import { openDb } from "../src/util/sqlite.ts";
|
|
8
12
|
import { parseChatRequest } from "../src/wire/openai/request.ts";
|
|
9
13
|
|
|
10
14
|
const cfg = loadConfig({});
|
|
@@ -48,34 +52,35 @@ function entry(over: Partial<LedgerEntry>): LedgerEntry {
|
|
|
48
52
|
}
|
|
49
53
|
|
|
50
54
|
describe("estimateTokens", () => {
|
|
51
|
-
test("uses the default ratio for an unknown tokenizer family", () => {
|
|
55
|
+
test("uses the default ratio for an unknown tokenizer family", async () => {
|
|
52
56
|
expect(estimateTokens(3600, "no-such-tokenizer", null)).toBe(Math.ceil(3600 / DEFAULT_BYTES_PER_TOKEN));
|
|
53
57
|
});
|
|
54
58
|
|
|
55
|
-
test("scales linearly with byte count and never goes negative", () => {
|
|
59
|
+
test("scales linearly with byte count and never goes negative", async () => {
|
|
56
60
|
expect(estimateTokens(0, "gpt", null)).toBe(0);
|
|
57
61
|
const small = estimateTokens(1000, "gpt", null);
|
|
58
62
|
const large = estimateTokens(10_000, "gpt", null);
|
|
59
63
|
expect(large).toBeGreaterThan(small);
|
|
60
64
|
});
|
|
61
65
|
|
|
62
|
-
test("a code-dense family estimates more tokens for the same bytes", () => {
|
|
66
|
+
test("a code-dense family estimates more tokens for the same bytes", async () => {
|
|
63
67
|
// BPE tokenizers emit more tokens per character on code than on prose,
|
|
64
68
|
// so a lower bytes-per-token ratio must yield a higher token count.
|
|
65
69
|
expect(estimateTokens(10_000, "deepseek", null)).toBeGreaterThan(estimateTokens(10_000, "gpt", null));
|
|
66
70
|
});
|
|
67
71
|
|
|
68
|
-
test("is case-insensitive about the tokenizer name", () => {
|
|
72
|
+
test("is case-insensitive about the tokenizer name", async () => {
|
|
69
73
|
expect(estimateTokens(5000, "Claude", null)).toBe(estimateTokens(5000, "claude", null));
|
|
70
74
|
});
|
|
71
75
|
});
|
|
72
76
|
|
|
73
77
|
describe("ledger calibration", () => {
|
|
74
|
-
test("a calibrated ratio replaces the family default once enough samples land", () => {
|
|
75
|
-
const db =
|
|
78
|
+
test("a calibrated ratio replaces the family default once enough samples land", async () => {
|
|
79
|
+
const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
80
|
+
await migrateStore(db);
|
|
76
81
|
try {
|
|
77
|
-
const ledger =
|
|
78
|
-
expect(ledger.tokenRatio("claude")).toBeNull();
|
|
82
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
83
|
+
expect(await ledger.tokenRatio("claude")).toBeNull();
|
|
79
84
|
|
|
80
85
|
const req = parseChatRequest(
|
|
81
86
|
{ model: "auto", messages: [{ role: "user", content: "x".repeat(4000) }] },
|
|
@@ -86,8 +91,8 @@ describe("ledger calibration", () => {
|
|
|
86
91
|
// bytes/token; the ledger must converge on the measurement.
|
|
87
92
|
const observedTokens = Math.round(req.promptBytes / 2);
|
|
88
93
|
for (let i = 0; i < 30; i++) {
|
|
89
|
-
estimatePromptTokens(req, "claude",
|
|
90
|
-
ledger.record(
|
|
94
|
+
estimatePromptTokens(req, "claude", null);
|
|
95
|
+
await ledger.record(
|
|
91
96
|
entry({
|
|
92
97
|
conversationKey: req.conversationKey,
|
|
93
98
|
usage: { ...EMPTY_USAGE, promptTokens: observedTokens, completionTokens: 10 },
|
|
@@ -95,33 +100,35 @@ describe("ledger calibration", () => {
|
|
|
95
100
|
);
|
|
96
101
|
}
|
|
97
102
|
|
|
98
|
-
const ratio = ledger.tokenRatio("claude");
|
|
103
|
+
const ratio = await ledger.tokenRatio("claude");
|
|
99
104
|
expect(ratio).not.toBeNull();
|
|
100
105
|
if (ratio === null) return;
|
|
101
106
|
expect(ratio).toBeCloseTo(2, 1);
|
|
102
107
|
|
|
103
|
-
// And the estimate
|
|
104
|
-
const calibrated = estimateTokens(4000, "claude",
|
|
108
|
+
// And the estimate follows the measurement, not the family default.
|
|
109
|
+
const calibrated = estimateTokens(4000, "claude", ratio);
|
|
105
110
|
expect(calibrated).toBeGreaterThan(estimateTokens(4000, "claude", null));
|
|
106
111
|
} finally {
|
|
107
|
-
db.close();
|
|
112
|
+
await db.close();
|
|
108
113
|
}
|
|
109
114
|
});
|
|
110
115
|
|
|
111
|
-
test("an uncalibrated family still falls back to its default", () => {
|
|
112
|
-
const db =
|
|
116
|
+
test("an uncalibrated family still falls back to its default", async () => {
|
|
117
|
+
const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
118
|
+
await migrateStore(db);
|
|
113
119
|
try {
|
|
114
|
-
const ledger =
|
|
115
|
-
expect(ledger.tokenRatio("gemini")).toBeNull();
|
|
116
|
-
|
|
120
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
121
|
+
expect(await ledger.tokenRatio("gemini")).toBeNull();
|
|
122
|
+
// An uncalibrated family reads null, which is exactly the default path.
|
|
123
|
+
expect(estimateTokens(3600, "gemini", await ledger.tokenRatio("gemini"))).toBe(estimateTokens(3600, "gemini", null));
|
|
117
124
|
} finally {
|
|
118
|
-
db.close();
|
|
125
|
+
await db.close();
|
|
119
126
|
}
|
|
120
127
|
});
|
|
121
128
|
});
|
|
122
129
|
|
|
123
130
|
describe("estimatePromptTokens", () => {
|
|
124
|
-
test("counts tool schemas, not just message text", () => {
|
|
131
|
+
test("counts tool schemas, not just message text", async () => {
|
|
125
132
|
const bare = parseChatRequest({ model: "auto", messages: [{ role: "user", content: "hi" }] }, new Headers());
|
|
126
133
|
const withTools = parseChatRequest(
|
|
127
134
|
{
|
|
@@ -143,7 +150,7 @@ describe("estimatePromptTokens", () => {
|
|
|
143
150
|
expect(estimatePromptTokens(withTools, "gpt", null)).toBeGreaterThan(estimatePromptTokens(bare, "gpt", null));
|
|
144
151
|
});
|
|
145
152
|
|
|
146
|
-
test("charges a per-image allowance on top of text", () => {
|
|
153
|
+
test("charges a per-image allowance on top of text", async () => {
|
|
147
154
|
const text = parseChatRequest(
|
|
148
155
|
{ model: "auto", messages: [{ role: "user", content: [{ type: "text", text: "describe" }] }] },
|
|
149
156
|
new Headers(),
|
|
@@ -173,99 +180,103 @@ describe("calibration hygiene (review 2026-09-05 §8)", () => {
|
|
|
173
180
|
return parseChatRequest({ model: "auto", messages: [{ role: "user", content: text }] }, new Headers());
|
|
174
181
|
}
|
|
175
182
|
|
|
176
|
-
test("a provider reporting impossible token counts never calibrates its family", () => {
|
|
177
|
-
const db =
|
|
183
|
+
test("a provider reporting impossible token counts never calibrates its family", async () => {
|
|
184
|
+
const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
185
|
+
await migrateStore(db);
|
|
178
186
|
try {
|
|
179
|
-
const ledger =
|
|
187
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
180
188
|
const req = requestOf("x".repeat(10_000));
|
|
181
189
|
for (let i = 0; i < 30; i++) {
|
|
182
|
-
estimatePromptTokens(req, "qwen3",
|
|
190
|
+
estimatePromptTokens(req, "qwen3", null);
|
|
183
191
|
// 0.4 bytes/token: ~8x what the bytes imply (seen live from one provider).
|
|
184
|
-
ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 0.4) } }));
|
|
192
|
+
await ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 0.4) } }));
|
|
185
193
|
}
|
|
186
|
-
expect(ledger.tokenRatio("qwen3")).toBeNull();
|
|
194
|
+
expect(await ledger.tokenRatio("qwen3")).toBeNull();
|
|
187
195
|
for (let i = 0; i < 30; i++) {
|
|
188
|
-
estimatePromptTokens(req, "qwen3",
|
|
189
|
-
ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 3.2) } }));
|
|
196
|
+
estimatePromptTokens(req, "qwen3", null);
|
|
197
|
+
await ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 3.2) } }));
|
|
190
198
|
}
|
|
191
|
-
expect(ledger.tokenRatio("qwen3")).toBeCloseTo(3.2, 1);
|
|
199
|
+
expect(await ledger.tokenRatio("qwen3")).toBeCloseTo(3.2, 1);
|
|
192
200
|
} finally {
|
|
193
|
-
db.close();
|
|
201
|
+
await db.close();
|
|
194
202
|
}
|
|
195
203
|
});
|
|
196
204
|
|
|
197
|
-
test("adjustPendingEstimate calibrates against the dispatched bytes, not the raw request", () => {
|
|
198
|
-
const db =
|
|
205
|
+
test("adjustPendingEstimate calibrates against the dispatched bytes, not the raw request", async () => {
|
|
206
|
+
const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
207
|
+
await migrateStore(db);
|
|
199
208
|
try {
|
|
200
|
-
const ledger =
|
|
209
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
201
210
|
const req = requestOf("y".repeat(10_000));
|
|
202
211
|
for (let i = 0; i < 30; i++) {
|
|
203
|
-
estimatePromptTokens(req, "grok",
|
|
212
|
+
estimatePromptTokens(req, "grok", null);
|
|
204
213
|
// Compaction halved the prompt before dispatch; the upstream billed the half.
|
|
205
214
|
adjustPendingEstimate(req.conversationKey, req.promptBytes / 2);
|
|
206
|
-
ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 2 / 3.5) } }));
|
|
215
|
+
await ledger.record(entry({ conversationKey: req.conversationKey, usage: { ...EMPTY_USAGE, promptTokens: Math.round(req.promptBytes / 2 / 3.5) } }));
|
|
207
216
|
}
|
|
208
217
|
// Paired with the raw bytes this would have learned 7.0; the dispatched bytes give the true 3.5.
|
|
209
|
-
expect(ledger.tokenRatio("grok")).toBeCloseTo(3.5, 1);
|
|
218
|
+
expect(await ledger.tokenRatio("grok")).toBeCloseTo(3.5, 1);
|
|
210
219
|
} finally {
|
|
211
|
-
db.close();
|
|
220
|
+
await db.close();
|
|
212
221
|
}
|
|
213
222
|
});
|
|
214
223
|
});
|
|
215
224
|
|
|
216
225
|
describe("ledger.escalationCost", () => {
|
|
217
|
-
test("measures what escalated retries bill per prompt token, once enough exist", () => {
|
|
218
|
-
const db =
|
|
226
|
+
test("measures what escalated retries bill per prompt token, once enough exist", async () => {
|
|
227
|
+
const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
228
|
+
await migrateStore(db);
|
|
219
229
|
try {
|
|
220
|
-
const ledger =
|
|
230
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
221
231
|
for (let i = 0; i < 9; i++) {
|
|
222
|
-
ledger.record(entry({ attempt: 1, reportedUsd: 0.02, usage: { ...EMPTY_USAGE, promptTokens: 1_000 } }));
|
|
232
|
+
await ledger.record(entry({ attempt: 1, reportedUsd: 0.02, usage: { ...EMPTY_USAGE, promptTokens: 1_000 } }));
|
|
223
233
|
}
|
|
224
|
-
expect(ledger.escalationCost
|
|
225
|
-
ledger.record(entry({ attempt: 1, reportedUsd: 0.02, usage: { ...EMPTY_USAGE, promptTokens: 1_000 } }));
|
|
234
|
+
expect(await ledger.escalationCost(7)).toBeNull(); // 9 < the sample floor
|
|
235
|
+
await ledger.record(entry({ attempt: 1, reportedUsd: 0.02, usage: { ...EMPTY_USAGE, promptTokens: 1_000 } }));
|
|
226
236
|
// Errored retries carry no usage and are excluded.
|
|
227
|
-
ledger.record(entry({ attempt: 1, reportedUsd: null, error: "upstream_error: boom", usage: EMPTY_USAGE }));
|
|
237
|
+
await ledger.record(entry({ attempt: 1, reportedUsd: null, error: "upstream_error: boom", usage: EMPTY_USAGE }));
|
|
228
238
|
// Memoised: a fresh ledger reads through.
|
|
229
|
-
const fresh =
|
|
230
|
-
const cost = fresh.escalationCost
|
|
239
|
+
const fresh = createSqlLedger(db, cfg, { findModel: () => null });
|
|
240
|
+
const cost = await fresh.escalationCost(7);
|
|
231
241
|
expect(cost).not.toBeNull();
|
|
232
|
-
expect(cost
|
|
233
|
-
expect(cost
|
|
242
|
+
expect(cost?.samples).toBe(10);
|
|
243
|
+
expect(cost?.usdPerPromptToken).toBeCloseTo(0.02 / 1_000, 8);
|
|
234
244
|
} finally {
|
|
235
|
-
db.close();
|
|
245
|
+
await db.close();
|
|
236
246
|
}
|
|
237
247
|
});
|
|
238
248
|
});
|
|
239
249
|
|
|
240
250
|
|
|
241
251
|
describe("ledger.softFailureSpikes", () => {
|
|
242
|
-
test("flags a model whose last-hour failure rate is a spike against its own 7-day baseline", () => {
|
|
243
|
-
const db =
|
|
252
|
+
test("flags a model whose last-hour failure rate is a spike against its own 7-day baseline", async () => {
|
|
253
|
+
const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
254
|
+
await migrateStore(db);
|
|
244
255
|
try {
|
|
245
|
-
const ledger =
|
|
256
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
246
257
|
const now = 1_800_000_000_000;
|
|
247
258
|
const H = 3_600_000;
|
|
248
259
|
// Baseline: 100 dispatches over the prior week at 5% soft failures.
|
|
249
260
|
for (let i = 0; i < 100; i++) {
|
|
250
|
-
ledger.record(entry({ createdAtMs: now - 2 * H - i * 60 * 60_000, escalationSignal: i % 20 === 0 ? "empty_completion" : null, wasted: i % 20 === 0 }));
|
|
261
|
+
await ledger.record(entry({ createdAtMs: now - 2 * H - i * 60 * 60_000, escalationSignal: i % 20 === 0 ? "empty_completion" : null, wasted: i % 20 === 0 }));
|
|
251
262
|
}
|
|
252
263
|
// Last hour: 10 dispatches, 4 soft failures (40%): a spike.
|
|
253
264
|
for (let i = 0; i < 10; i++) {
|
|
254
|
-
ledger.record(entry({ createdAtMs: now - 5 * 60_000 - i * 60_000, escalationSignal: i < 4 ? "repeat_tool_call" : null, wasted: i < 4 }));
|
|
265
|
+
await ledger.record(entry({ createdAtMs: now - 5 * 60_000 - i * 60_000, escalationSignal: i < 4 ? "repeat_tool_call" : null, wasted: i < 4 }));
|
|
255
266
|
}
|
|
256
267
|
// A second model with plenty of failures but a matching baseline is not spiking.
|
|
257
268
|
for (let i = 0; i < 100; i++) {
|
|
258
|
-
ledger.record(entry({ slug: "x/steady", servedSlug: "x/steady", createdAtMs: now - 2 * H - i * 60 * 60_000, error: i % 2 === 0 ? "upstream_error: 502" : null }));
|
|
269
|
+
await ledger.record(entry({ slug: "x/steady", servedSlug: "x/steady", createdAtMs: now - 2 * H - i * 60 * 60_000, error: i % 2 === 0 ? "upstream_error: 502" : null }));
|
|
259
270
|
}
|
|
260
271
|
for (let i = 0; i < 10; i++) {
|
|
261
|
-
ledger.record(entry({ slug: "x/steady", servedSlug: "x/steady", createdAtMs: now - 5 * 60_000 - i * 60_000, error: i % 2 === 0 ? "upstream_error: 502" : null }));
|
|
272
|
+
await ledger.record(entry({ slug: "x/steady", servedSlug: "x/steady", createdAtMs: now - 5 * 60_000 - i * 60_000, error: i % 2 === 0 ? "upstream_error: 502" : null }));
|
|
262
273
|
}
|
|
263
274
|
// Aborted and quota errors are not attributable; digest rows are side calls.
|
|
264
275
|
for (let i = 0; i < 10; i++) {
|
|
265
|
-
ledger.record(entry({ slug: "x/aborted", servedSlug: "x/aborted", createdAtMs: now - 5 * 60_000 - i * 60_000, error: "aborted: client closed" }));
|
|
266
|
-
ledger.record(entry({ slug: "x/digest", servedSlug: "x/digest", requestedModel: "digest", createdAtMs: now - 5 * 60_000 - i * 60_000, error: "upstream_error: 500" }));
|
|
276
|
+
await ledger.record(entry({ slug: "x/aborted", servedSlug: "x/aborted", createdAtMs: now - 5 * 60_000 - i * 60_000, error: "aborted: client closed" }));
|
|
277
|
+
await ledger.record(entry({ slug: "x/digest", servedSlug: "x/digest", requestedModel: "digest", createdAtMs: now - 5 * 60_000 - i * 60_000, error: "upstream_error: 500" }));
|
|
267
278
|
}
|
|
268
|
-
const spikes = ledger.softFailureSpikes
|
|
279
|
+
const spikes = await ledger.softFailureSpikes(now);
|
|
269
280
|
expect(spikes.map((s) => s.slug)).toEqual(["openai/gpt-5-mini"]);
|
|
270
281
|
const s = spikes[0]!;
|
|
271
282
|
expect(s.recentDispatches).toBe(10);
|
|
@@ -275,35 +286,39 @@ describe("ledger.softFailureSpikes", () => {
|
|
|
275
286
|
expect(s.baselineFailures).toBe(5);
|
|
276
287
|
expect(s.baselineRate).toBeCloseTo(0.05, 6);
|
|
277
288
|
// Too few recent dispatches: nothing spikes, however high the rate.
|
|
278
|
-
expect(ledger.softFailureSpikes
|
|
289
|
+
expect(await ledger.softFailureSpikes(now, 3 * 60_000)).toEqual([]);
|
|
279
290
|
} finally {
|
|
280
|
-
db.close();
|
|
291
|
+
await db.close();
|
|
281
292
|
}
|
|
282
293
|
});
|
|
283
294
|
});
|
|
284
295
|
|
|
285
296
|
describe("ledger.prune and markWasted", () => {
|
|
286
|
-
test("prune deletes rows past retention and 0 keeps everything; markWasted flips one row", () => {
|
|
287
|
-
const db =
|
|
297
|
+
test("prune deletes rows past retention and 0 keeps everything; markWasted flips one row", async () => {
|
|
298
|
+
const db = openSqlDb(join(tmpdir(), `t-tokens.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
299
|
+
await migrateStore(db);
|
|
288
300
|
try {
|
|
289
|
-
const ledger =
|
|
301
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
290
302
|
const now = 1_800_000_000_000;
|
|
291
303
|
const DAY = 86_400_000;
|
|
292
|
-
for (let i = 0; i < 5; i++) ledger.record(entry({ createdAtMs: now - i * 100 * DAY }));
|
|
293
|
-
expect(ledger.prune
|
|
294
|
-
expect(ledger.recentEntries(10)).toHaveLength(5);
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
expect(ledger.
|
|
299
|
-
|
|
300
|
-
|
|
304
|
+
for (let i = 0; i < 5; i++) await ledger.record(entry({ createdAtMs: now - i * 100 * DAY }));
|
|
305
|
+
expect(await ledger.prune(0, now)).toEqual({ deleted: 0, oldestKeptMs: now - 400 * DAY });
|
|
306
|
+
expect(await ledger.recentEntries(10)).toHaveLength(5);
|
|
307
|
+
for (const atMs of [now - 400 * DAY, now - DAY]) {
|
|
308
|
+
await db.sql`INSERT INTO ollama_meter_samples (at_ms, meter_usd, ledger_usd) VALUES (${atMs}, 1, 1)`;
|
|
309
|
+
}
|
|
310
|
+
expect((await ledger.prune(365, now))?.deleted).toBe(1); // only the 400-day-old row
|
|
311
|
+
const samples = await db.one<{ n: unknown }>("SELECT COUNT(*) AS n FROM ollama_meter_samples");
|
|
312
|
+
expect(num(samples?.n)).toBe(1);
|
|
313
|
+
expect(await ledger.recentEntries(10)).toHaveLength(4);
|
|
314
|
+
expect((await ledger.prune(150, now))?.deleted).toBe(2); // 200 and 300 days old
|
|
315
|
+
const left = await ledger.recentEntries(10);
|
|
301
316
|
expect(left).toHaveLength(2);
|
|
302
317
|
expect(left.every((e) => e.wasted === false)).toBe(true);
|
|
303
|
-
ledger.markWasted
|
|
304
|
-
expect(ledger.recentEntries(10).find((e) => e.id === left[0]!.id)?.wasted).toBe(true);
|
|
318
|
+
await ledger.markWasted(left[0]!.id);
|
|
319
|
+
expect((await ledger.recentEntries(10)).find((e) => e.id === left[0]!.id)?.wasted).toBe(true);
|
|
305
320
|
} finally {
|
|
306
|
-
db.close();
|
|
321
|
+
await db.close();
|
|
307
322
|
}
|
|
308
323
|
});
|
|
309
324
|
});
|