auto-model-router 0.30.3 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +32 -2
- package/omp-extension/router-configure.ts +9 -7
- package/package.json +1 -1
- package/src/cli/config-cmd.ts +8 -7
- package/src/cli/explain.ts +10 -5
- package/src/cli/export.ts +6 -5
- package/src/cli/models.ts +10 -7
- package/src/cli/report.ts +6 -1
- package/src/cli/stats.ts +7 -7
- package/src/config/load.ts +10 -1
- package/src/config/types.ts +10 -1
- package/src/context/bridge.ts +7 -7
- package/src/context/index.ts +3 -3
- package/src/context/store.ts +39 -56
- package/src/context/types.ts +7 -6
- package/src/cost/blended.ts +28 -7
- package/src/cost/feedback.ts +33 -37
- package/src/cost/ledger-sql.ts +547 -0
- package/src/cost/ledger.ts +30 -459
- package/src/cost/report.ts +171 -129
- package/src/cost/retention.ts +10 -10
- package/src/cost/summary.ts +15 -10
- package/src/cost/types.ts +43 -62
- package/src/cost/views.ts +79 -49
- package/src/eval/calibrate.ts +47 -12
- package/src/eval/run.ts +18 -2
- package/src/lib.ts +6 -2
- package/src/router/candidates.ts +7 -15
- package/src/router/classify.ts +6 -4
- package/src/router/index.ts +95 -9
- package/src/router/select.ts +38 -21
- package/src/router/state.ts +90 -102
- package/src/router/types.ts +11 -5
- package/src/server/advise.ts +6 -4
- package/src/server/compaction-digest.ts +1 -1
- package/src/server/digest.ts +9 -10
- package/src/server/http.ts +109 -46
- package/src/server/providers.ts +18 -4
- package/src/server/turn.ts +32 -9
- package/src/tokens/estimate.ts +16 -6
- package/src/upstream/ollama-usage.ts +21 -11
- package/src/util/schema.ts +201 -0
- package/src/util/sql.ts +246 -0
- package/src/wire/anthropic/messages.ts +3 -4
- package/src/wire/openai/request.ts +1 -0
- package/src/wire/types.ts +7 -0
- package/test/anthropic-wire.test.ts +9 -9
- package/test/benchmark-feeds.test.ts +7 -7
- package/test/cache-control.test.ts +7 -7
- package/test/cache-estimate.test.ts +5 -5
- package/test/catalog-view.test.ts +4 -4
- package/test/catalog.test.ts +11 -11
- package/test/classify.test.ts +24 -24
- package/test/compaction.test.ts +20 -20
- package/test/config-wizard.test.ts +32 -32
- package/test/config.test.ts +10 -10
- package/test/connect-harnesses.test.ts +11 -11
- package/test/context-bridge.test.ts +40 -30
- package/test/context-prune.test.ts +43 -36
- package/test/context-query.test.ts +8 -8
- package/test/controls.test.ts +54 -27
- package/test/cost.test.ts +12 -12
- package/test/digest.test.ts +55 -44
- package/test/embed-lifecycle.test.ts +5 -5
- package/test/embed-logic.test.ts +26 -26
- package/test/escalate.test.ts +17 -17
- package/test/eval.test.ts +73 -16
- package/test/executable.test.ts +6 -6
- package/test/exploration.test.ts +19 -20
- package/test/failover.test.ts +22 -21
- package/test/fakes.ts +105 -0
- package/test/features.test.ts +21 -21
- package/test/harness-requests.test.ts +3 -3
- package/test/harness-switch.test.ts +5 -5
- package/test/hold-exploration.test.ts +13 -13
- package/test/hot-reload.test.ts +5 -5
- package/test/learned.test.ts +5 -5
- package/test/ledger-sql.test.ts +342 -0
- package/test/mcp-entry.test.ts +5 -5
- package/test/migrations.test.ts +28 -22
- package/test/models-yml.test.ts +18 -18
- package/test/ollama.test.ts +40 -34
- package/test/omp-credentials.test.ts +16 -16
- package/test/policy.test.ts +3 -3
- package/test/reconfigure.test.ts +4 -4
- package/test/redaction.test.ts +41 -35
- package/test/remote.test.ts +12 -12
- package/test/report-logic.test.ts +8 -8
- package/test/report.test.ts +95 -87
- package/test/retention.test.ts +79 -66
- package/test/schema.test.ts +123 -0
- package/test/scope.test.ts +8 -8
- package/test/select.test.ts +216 -257
- package/test/skills.test.ts +3 -3
- package/test/sql-shim.test.ts +154 -0
- package/test/state.test.ts +43 -36
- package/test/summary.test.ts +38 -27
- package/test/tier-plan.test.ts +45 -62
- package/test/toast-logic.test.ts +31 -31
- package/test/tokens.test.ts +95 -80
- package/test/trust-attribution.test.ts +217 -187
- package/test/trust-window.test.ts +37 -32
- package/test/turn.test.ts +55 -23
- package/test/upstreams.test.ts +13 -13
- package/test/views.test.ts +81 -59
- package/test/wire-request.test.ts +17 -17
- package/test/wire-responses.test.ts +4 -4
- package/tools/agentdox-e2e.ts +5 -2
- package/tools/export-benchmarks.ts +5 -5
- package/tools/ledger-parity.ts +266 -0
- package/tools/replay.ts +16 -8
|
@@ -1,9 +1,13 @@
|
|
|
1
1
|
import { describe, expect, test } from "bun:test";
|
|
2
|
+
import { tmpdir } from "node:os";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
|
|
5
|
+
import { migrateStore } from "../src/util/schema.ts";
|
|
6
|
+
import { openSqlDb } from "../src/util/sql.ts";
|
|
2
7
|
|
|
3
8
|
import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
|
|
4
|
-
import {
|
|
9
|
+
import { createSqlLedger } from "../src/cost/ledger-sql.ts";
|
|
5
10
|
import type { LedgerEntry } from "../src/cost/types.ts";
|
|
6
|
-
import { openDb } from "../src/util/sqlite.ts";
|
|
7
11
|
|
|
8
12
|
/**
|
|
9
13
|
* `filters.trustWindowDays` bounds the per-slug trust aggregate, which otherwise
|
|
@@ -66,69 +70,70 @@ function entry(over: Partial<LedgerEntry>): LedgerEntry {
|
|
|
66
70
|
}
|
|
67
71
|
|
|
68
72
|
/** Old rows: half of them failures. Recent rows: all clean. */
|
|
69
|
-
function seed(windowDays: number) {
|
|
73
|
+
async function seed(windowDays: number) {
|
|
70
74
|
const cfg = cfgWith(windowDays);
|
|
71
|
-
const db =
|
|
72
|
-
|
|
75
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-window.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
76
|
+
await migrateStore(db);
|
|
77
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
73
78
|
const now = Date.now();
|
|
74
79
|
for (let i = 0; i < 10; i++) {
|
|
75
|
-
ledger.record(
|
|
80
|
+
await ledger.record(
|
|
76
81
|
entry({
|
|
77
82
|
createdAtMs: now - 30 * DAY,
|
|
78
83
|
...(i % 2 === 0 ? { error: "server_error: boom", errorKind: "server_error" } : {}),
|
|
79
84
|
}),
|
|
80
85
|
);
|
|
81
86
|
}
|
|
82
|
-
for (let i = 0; i < 10; i++) ledger.record(entry({ createdAtMs: now - 1 * DAY }));
|
|
87
|
+
for (let i = 0; i < 10; i++) await ledger.record(entry({ createdAtMs: now - 1 * DAY }));
|
|
83
88
|
return { db, ledger, cfg };
|
|
84
89
|
}
|
|
85
90
|
|
|
86
91
|
describe("filters.trustWindowDays", () => {
|
|
87
|
-
test("0 means all-time: every row counts", () => {
|
|
88
|
-
const { db, ledger } = seed(0);
|
|
89
|
-
const trust = ledger.trust("vendor/model");
|
|
92
|
+
test("0 means all-time: every row counts", async () => {
|
|
93
|
+
const { db, ledger } = await seed(0);
|
|
94
|
+
const trust = await ledger.trust("vendor/model");
|
|
90
95
|
expect(trust?.attempts).toBe(20);
|
|
91
96
|
expect(trust?.errors).toBe(5);
|
|
92
|
-
db.close();
|
|
97
|
+
await db.close();
|
|
93
98
|
});
|
|
94
99
|
|
|
95
|
-
test("a window excludes rows older than it", () => {
|
|
96
|
-
const { db, ledger } = seed(7);
|
|
97
|
-
const trust = ledger.trust("vendor/model");
|
|
100
|
+
test("a window excludes rows older than it", async () => {
|
|
101
|
+
const { db, ledger } = await seed(7);
|
|
102
|
+
const trust = await ledger.trust("vendor/model");
|
|
98
103
|
// Only the 10 recent, clean rows remain.
|
|
99
104
|
expect(trust?.attempts).toBe(10);
|
|
100
105
|
expect(trust?.errors).toBe(0);
|
|
101
|
-
db.close();
|
|
106
|
+
await db.close();
|
|
102
107
|
});
|
|
103
108
|
|
|
104
|
-
test("the window moves the success rate, which is why it is opt-in", () => {
|
|
105
|
-
const all = seed(0);
|
|
106
|
-
const windowed = seed(7);
|
|
107
|
-
const allTrust = all.ledger.trust("vendor/model");
|
|
108
|
-
const winTrust = windowed.ledger.trust("vendor/model");
|
|
109
|
+
test("the window moves the success rate, which is why it is opt-in", async () => {
|
|
110
|
+
const all = await seed(0);
|
|
111
|
+
const windowed = await seed(7);
|
|
112
|
+
const allTrust = await all.ledger.trust("vendor/model");
|
|
113
|
+
const winTrust = await windowed.ledger.trust("vendor/model");
|
|
109
114
|
expect(allTrust?.successRate).toBeLessThan(winTrust?.successRate ?? 0);
|
|
110
|
-
all.db.close();
|
|
111
|
-
windowed.db.close();
|
|
115
|
+
await all.db.close();
|
|
116
|
+
await windowed.db.close();
|
|
112
117
|
});
|
|
113
118
|
|
|
114
|
-
test("is read per call, so a hot-reloaded edit takes effect immediately", () => {
|
|
115
|
-
const { db, ledger, cfg } = seed(0);
|
|
116
|
-
expect(ledger.trust("vendor/model")?.attempts).toBe(20);
|
|
119
|
+
test("is read per call, so a hot-reloaded edit takes effect immediately", async () => {
|
|
120
|
+
const { db, ledger, cfg } = await seed(0);
|
|
121
|
+
expect((await ledger.trust("vendor/model"))?.attempts).toBe(20);
|
|
117
122
|
// Hot reload mutates the shared config object in place.
|
|
118
123
|
cfg.filters.trustWindowDays = 7;
|
|
119
|
-
expect(ledger.trust("vendor/model")?.attempts).toBe(10);
|
|
120
|
-
db.close();
|
|
124
|
+
expect((await ledger.trust("vendor/model"))?.attempts).toBe(10);
|
|
125
|
+
await db.close();
|
|
121
126
|
});
|
|
122
127
|
|
|
123
|
-
test("allTrust honours the same window", () => {
|
|
124
|
-
const { db, ledger } = seed(7);
|
|
125
|
-
const rows = ledger.allTrust();
|
|
128
|
+
test("allTrust honours the same window", async () => {
|
|
129
|
+
const { db, ledger } = await seed(7);
|
|
130
|
+
const rows = await ledger.allTrust();
|
|
126
131
|
expect(rows).toHaveLength(1);
|
|
127
132
|
expect(rows[0]?.attempts).toBe(10);
|
|
128
|
-
db.close();
|
|
133
|
+
await db.close();
|
|
129
134
|
});
|
|
130
135
|
|
|
131
|
-
test("ships disabled, so the default install is unchanged", () => {
|
|
136
|
+
test("ships disabled, so the default install is unchanged", async () => {
|
|
132
137
|
// DEFAULT_CONFIG, not loadConfig: loadConfig reads the machine's real
|
|
133
138
|
// config.yml, which has broken this suite before.
|
|
134
139
|
expect(DEFAULT_CONFIG.filters.trustWindowDays).toBe(0);
|
package/test/turn.test.ts
CHANGED
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
import { describe, expect, test } from "bun:test";
|
|
2
|
+
import { fakeLedger } from "./fakes.ts";
|
|
2
3
|
import { createDisabledBridge } from "../src/context/bridge.ts";
|
|
3
4
|
import type { ContextBridge, ContextResolveInput, TurnRecord } from "../src/context/types.ts";
|
|
4
5
|
import type { CatalogModel, CatalogSource } from "../src/catalog/types.ts";
|
|
5
6
|
import type { EscalationConfig, RouterConfig } from "../src/config/types.ts";
|
|
6
|
-
import { EMPTY_USAGE, type
|
|
7
|
+
import { EMPTY_USAGE, type AsyncLedger, type LedgerEntry, type UsageCounts } from "../src/cost/types.ts";
|
|
7
8
|
import type {
|
|
8
9
|
ConversationState,
|
|
9
10
|
ConversationStore,
|
|
@@ -14,6 +15,7 @@ import type {
|
|
|
14
15
|
Tier,
|
|
15
16
|
} from "../src/router/types.ts";
|
|
16
17
|
import { runTurn } from "../src/server/turn.ts";
|
|
18
|
+
import { BudgetExceededError } from "../src/router/select.ts";
|
|
17
19
|
import { parseMessagesRequest } from "../src/wire/anthropic/messages.ts";
|
|
18
20
|
import { parseChatRequest } from "../src/wire/openai/request.ts";
|
|
19
21
|
import { UpstreamError, type DispatchOptions, type UpstreamClient } from "../src/upstream/types.ts";
|
|
@@ -241,21 +243,21 @@ function mkRouter(decisions: Decision[]): { router: Router; calls: { attempt: nu
|
|
|
241
243
|
return { router, calls };
|
|
242
244
|
}
|
|
243
245
|
|
|
244
|
-
function mkLedger(): { ledger:
|
|
246
|
+
function mkLedger(): { ledger: AsyncLedger; entries: LedgerEntry[] } {
|
|
245
247
|
const entries: LedgerEntry[] = [];
|
|
246
|
-
const ledger:
|
|
247
|
-
record: (e) => {
|
|
248
|
+
const ledger: AsyncLedger = fakeLedger({
|
|
249
|
+
record: async (e) => {
|
|
248
250
|
entries.push(e);
|
|
249
251
|
},
|
|
250
|
-
conversationSpend: () => 0,
|
|
251
|
-
spendSince: () => 0,
|
|
252
|
-
blendedRate: () => null,
|
|
253
|
-
latency: () => null,
|
|
254
|
-
trust: () => null,
|
|
255
|
-
allTrust: () => [],
|
|
256
|
-
tokenRatio: () => null,
|
|
257
|
-
recentEntries: () => [],
|
|
258
|
-
};
|
|
252
|
+
conversationSpend: async () => 0,
|
|
253
|
+
spendSince: async () => 0,
|
|
254
|
+
blendedRate: async () => null,
|
|
255
|
+
latency: async () => null,
|
|
256
|
+
trust: async () => null,
|
|
257
|
+
allTrust: async () => [],
|
|
258
|
+
tokenRatio: async () => null,
|
|
259
|
+
recentEntries: async () => [],
|
|
260
|
+
});
|
|
259
261
|
return { ledger, entries };
|
|
260
262
|
}
|
|
261
263
|
|
|
@@ -268,8 +270,8 @@ function mkConversations(): {
|
|
|
268
270
|
// Mirrors the real store: money accumulates here, NOT through `save`.
|
|
269
271
|
const accrued = new Map<string, { spentUsd: number; escalations: number }>();
|
|
270
272
|
const store: ConversationStore = {
|
|
271
|
-
get: (k) => map.get(k) ?? null,
|
|
272
|
-
load: (k) => {
|
|
273
|
+
get: async (k) => map.get(k) ?? null,
|
|
274
|
+
load: async (k) => {
|
|
273
275
|
const existing = map.get(k);
|
|
274
276
|
if (existing) return existing;
|
|
275
277
|
const fresh: ConversationState = {
|
|
@@ -292,16 +294,16 @@ function mkConversations(): {
|
|
|
292
294
|
map.set(k, fresh);
|
|
293
295
|
return fresh;
|
|
294
296
|
},
|
|
295
|
-
save: (s) => {
|
|
297
|
+
save: async (s) => {
|
|
296
298
|
map.set(s.key, s);
|
|
297
299
|
},
|
|
298
|
-
accrue: (k, d) => {
|
|
300
|
+
accrue: async (k, d) => {
|
|
299
301
|
const cur = accrued.get(k) ?? { spentUsd: 0, escalations: 0 };
|
|
300
302
|
cur.spentUsd += d.spentUsd ?? 0;
|
|
301
303
|
cur.escalations += d.escalations ?? 0;
|
|
302
304
|
accrued.set(k, cur);
|
|
303
305
|
},
|
|
304
|
-
prune: () => 0,
|
|
306
|
+
prune: async () => 0,
|
|
305
307
|
};
|
|
306
308
|
return { store, map, accrued };
|
|
307
309
|
}
|
|
@@ -480,6 +482,36 @@ describe("runTurn", () => {
|
|
|
480
482
|
expect(entries[0]!.error).toContain("auth");
|
|
481
483
|
});
|
|
482
484
|
|
|
485
|
+
test("a budget refusal answers 402 budget_exceeded, not a 500", async () => {
|
|
486
|
+
// `reject` mode is a refusal the router MEANS. Reported as a 500 it reads
|
|
487
|
+
// as "the router broke" — and a team front door relaying it sends a
|
|
488
|
+
// capped deployment looking for a crash. Anything else from `route()` is
|
|
489
|
+
// still an internal failure.
|
|
490
|
+
const refusing: Router = { route: () => Promise.reject(new BudgetExceededError("24h spend $0.10 > per-day budget $0.05")) };
|
|
491
|
+
const { upstream } = mkUpstream([{ kind: "chunks", chunks: [] }]);
|
|
492
|
+
const { ledger, entries } = mkLedger();
|
|
493
|
+
const { store } = mkConversations();
|
|
494
|
+
const { sink, errors, finishes } = mkSink();
|
|
495
|
+
|
|
496
|
+
await runTurn(mkReq(), sink, { config: mkConfig(), router: refusing, upstream, ledger, conversations: store, catalog, context: createDisabledBridge() }, new AbortController().signal);
|
|
497
|
+
|
|
498
|
+
expect(errors).toEqual([{ status: 402, code: "budget_exceeded", message: "24h spend $0.10 > per-day budget $0.05" }]);
|
|
499
|
+
expect(finishes).toHaveLength(0);
|
|
500
|
+
expect(entries).toHaveLength(0); // refused before dispatch: no turn to record
|
|
501
|
+
});
|
|
502
|
+
|
|
503
|
+
test("an unexpected routing failure is still a 500", async () => {
|
|
504
|
+
const broken: Router = { route: () => Promise.reject(new Error("catalog exhausted")) };
|
|
505
|
+
const { upstream } = mkUpstream([{ kind: "chunks", chunks: [] }]);
|
|
506
|
+
const { ledger } = mkLedger();
|
|
507
|
+
const { store } = mkConversations();
|
|
508
|
+
const { sink, errors } = mkSink();
|
|
509
|
+
|
|
510
|
+
await runTurn(mkReq(), sink, { config: mkConfig(), router: broken, upstream, ledger, conversations: store, catalog, context: createDisabledBridge() }, new AbortController().signal);
|
|
511
|
+
|
|
512
|
+
expect(errors).toEqual([{ status: 500, code: "router_error", message: "catalog exhausted" }]);
|
|
513
|
+
});
|
|
514
|
+
|
|
483
515
|
test("a 429 before commit fails over to a different model in the same tier", async () => {
|
|
484
516
|
const { router, calls } = mkRouter([
|
|
485
517
|
mkDecision("trivial", "cheap/model", { escalateTo: "simple" }),
|
|
@@ -627,7 +659,7 @@ describe("agentdox write-back sees the shape of the turn", () => {
|
|
|
627
659
|
records.push(rec);
|
|
628
660
|
},
|
|
629
661
|
flush: () => Promise.resolve(),
|
|
630
|
-
pruneBlocks: () => 0,
|
|
662
|
+
pruneBlocks: async () => 0,
|
|
631
663
|
close: () => {},
|
|
632
664
|
},
|
|
633
665
|
};
|
|
@@ -706,7 +738,7 @@ describe("agentdox injection sees the shape of the turn", () => {
|
|
|
706
738
|
},
|
|
707
739
|
recordTurn: () => {},
|
|
708
740
|
flush: () => Promise.resolve(),
|
|
709
|
-
pruneBlocks: () => 0,
|
|
741
|
+
pruneBlocks: async () => 0,
|
|
710
742
|
close: () => {},
|
|
711
743
|
},
|
|
712
744
|
};
|
|
@@ -736,7 +768,7 @@ describe("agentdox injection sees the shape of the turn", () => {
|
|
|
736
768
|
|
|
737
769
|
expect(errors).toHaveLength(0);
|
|
738
770
|
expect(resolves).toHaveLength(0);
|
|
739
|
-
expect(store.load(req.conversationKey).contextVersion).toBeNull();
|
|
771
|
+
expect((await store.load(req.conversationKey)).contextVersion).toBeNull();
|
|
740
772
|
});
|
|
741
773
|
|
|
742
774
|
test("an agent turn with tools is injected", async () => {
|
|
@@ -748,7 +780,7 @@ describe("agentdox injection sees the shape of the turn", () => {
|
|
|
748
780
|
|
|
749
781
|
expect(errors).toHaveLength(0);
|
|
750
782
|
expect(resolves).toHaveLength(1);
|
|
751
|
-
expect(store.load(req.conversationKey).contextVersion).toBe("v1");
|
|
783
|
+
expect((await store.load(req.conversationKey)).contextVersion).toBe("v1");
|
|
752
784
|
});
|
|
753
785
|
|
|
754
786
|
test("the layer headers and the harness id reach resolve, and the harness id reaches the record", async () => {
|
|
@@ -769,7 +801,7 @@ describe("agentdox injection sees the shape of the turn", () => {
|
|
|
769
801
|
records.push(rec);
|
|
770
802
|
},
|
|
771
803
|
flush: () => Promise.resolve(),
|
|
772
|
-
pruneBlocks: () => 0,
|
|
804
|
+
pruneBlocks: async () => 0,
|
|
773
805
|
close: () => {},
|
|
774
806
|
};
|
|
775
807
|
const req: NormRequest = { ...mkReq(), harnessId: "u_ada", agentdoxScope: "proj", agentdoxGroup: "group.g1", agentdoxPersonal: "proj.u.u_ada", tools: [AGENT_TOOL] };
|
package/test/upstreams.test.ts
CHANGED
|
@@ -48,7 +48,7 @@ function sse(frames: string[]): Response {
|
|
|
48
48
|
}
|
|
49
49
|
|
|
50
50
|
describe("upstreams config", () => {
|
|
51
|
-
test("an entry is accepted with defaults filled; a reserved or malformed id is refused", () => {
|
|
51
|
+
test("an entry is accepted with defaults filled; a reserved or malformed id is refused", async () => {
|
|
52
52
|
const ok = configInputSchema.safeParse({ upstreams: [{ id: "openai-direct", kind: "openai", baseUrl: "https://api.openai.com/v1", apiKey: "sk", models: [{ id: "gpt-4o", input: 2.5, output: 10 }] }] });
|
|
53
53
|
expect(ok.success).toBe(true);
|
|
54
54
|
const full = completeUpstreamEntry({ id: "vllm", kind: "openai", baseUrl: "http://vllm:8000/v1", models: [] });
|
|
@@ -61,7 +61,7 @@ describe("upstreams config", () => {
|
|
|
61
61
|
expect(configInputSchema.safeParse({ upstreams: [{ id: "x", kind: "bedrock", baseUrl: "https://x", models: [] }] }).success).toBe(false);
|
|
62
62
|
});
|
|
63
63
|
|
|
64
|
-
test("a live patch replaces the list and completes sparse entries", () => {
|
|
64
|
+
test("a live patch replaces the list and completes sparse entries", async () => {
|
|
65
65
|
const cfg = cfgWith([]);
|
|
66
66
|
const changed = applyConfigPatch(cfg, { upstreams: [{ id: "azure-eu", kind: "azure", baseUrl: "https://r.openai.azure.com", apiKey: "k", models: [{ id: "gpt-4o-deploy", input: 2.5, output: 10 }] }] } as never);
|
|
67
67
|
expect(changed).toEqual(["upstreams"]);
|
|
@@ -72,7 +72,7 @@ describe("upstreams config", () => {
|
|
|
72
72
|
describe("static catalog", () => {
|
|
73
73
|
const twins = [normalizeCatalogModel(orRaw("openai/gpt-4o", 70, true))!, normalizeCatalogModel(orRaw("anthropic/claude-sonnet-4", 80))!];
|
|
74
74
|
|
|
75
|
-
test("models are priced per token, namespaced by the entry id, and borrow the twin's scores and modalities", () => {
|
|
75
|
+
test("models are priced per token, namespaced by the entry id, and borrow the twin's scores and modalities", async () => {
|
|
76
76
|
const e = entry({ id: "openai-direct", kind: "openai", models: [{ id: "gpt-4o", input: 2.5, output: 10, cachedInput: 1.25 }, { id: "custom-ft", input: 3, output: 12, contextLength: 32_000, quality: { coding: 55 }, supportsTools: false }] });
|
|
77
77
|
const models = buildUpstreamModels(e, twins);
|
|
78
78
|
expect(models.map((m) => m.slug)).toEqual(["openai-direct/gpt-4o", "openai-direct/custom-ft"]);
|
|
@@ -91,7 +91,7 @@ describe("static catalog", () => {
|
|
|
91
91
|
expect(ft.inputModalities).toEqual(["text"]);
|
|
92
92
|
});
|
|
93
93
|
|
|
94
|
-
test("an explicit twin wins over the name match; an anthropic entry defaults to the Claude tokenizer", () => {
|
|
94
|
+
test("an explicit twin wins over the name match; an anthropic entry defaults to the Claude tokenizer", async () => {
|
|
95
95
|
const e = entry({ id: "anthropic-direct", kind: "anthropic", models: [{ id: "claude-sonnet-4-20250514", input: 3, output: 15, twin: "anthropic/claude-sonnet-4", cacheWrite: 3.75, cachedInput: 0.3 }, { id: "claude-unknown", input: 1, output: 5 }] });
|
|
96
96
|
const [sonnet, unknown] = buildUpstreamModels(e, twins);
|
|
97
97
|
expect(sonnet!.quality.coding).toBe(80);
|
|
@@ -100,7 +100,7 @@ describe("static catalog", () => {
|
|
|
100
100
|
expect(unknown!.tokenizer).toBe("Claude");
|
|
101
101
|
});
|
|
102
102
|
|
|
103
|
-
test("a subscription model with no price of its own inherits the twin's, so it never ranks as free", () => {
|
|
103
|
+
test("a subscription model with no price of its own inherits the twin's, so it never ranks as free", async () => {
|
|
104
104
|
// A Pro/Max subscription publishes no per-token rates, and pricing it at zero would beat
|
|
105
105
|
// every model in every tier outright. The twin sells the same weights, so it is the rate.
|
|
106
106
|
const e = entry({ id: "anthropic-subscription", kind: "anthropic", costBias: 0.1, models: [{ id: "claude-sonnet-4-20250514", twin: "anthropic/claude-sonnet-4" }] });
|
|
@@ -117,7 +117,7 @@ describe("static catalog", () => {
|
|
|
117
117
|
expect(buildUpstreamModels(free, twins)[0]!.price).toEqual({ prompt: 0, completion: 0 });
|
|
118
118
|
});
|
|
119
119
|
|
|
120
|
-
test("the source rebuilds only when the entries or the OpenRouter models change", () => {
|
|
120
|
+
test("the source rebuilds only when the entries or the OpenRouter models change", async () => {
|
|
121
121
|
const cfg = cfgWith([entry({ id: "vllm", kind: "openai" })]);
|
|
122
122
|
const src = createStaticCatalogSource(cfg);
|
|
123
123
|
const first = src.get(twins);
|
|
@@ -157,7 +157,7 @@ describe("dispatch by slug prefix", () => {
|
|
|
157
157
|
expect(namedUpstreamOf("openai-direct/gpt-4o", ["openai-direct"])).toBe("openai-direct");
|
|
158
158
|
});
|
|
159
159
|
|
|
160
|
-
test("the ledger names the provider of a slug from the known ids", () => {
|
|
160
|
+
test("the ledger names the provider of a slug from the known ids", async () => {
|
|
161
161
|
setKnownUpstreamIds(["azure-eu", "bad id"]);
|
|
162
162
|
expect(providerOfSlug("azure-eu/gpt-4o")).toBe("azure-eu");
|
|
163
163
|
expect(providerOfSlug("openai/gpt-4o")).toBe("openrouter");
|
|
@@ -168,7 +168,7 @@ describe("dispatch by slug prefix", () => {
|
|
|
168
168
|
});
|
|
169
169
|
|
|
170
170
|
describe("the OpenAI-compatible client", () => {
|
|
171
|
-
test("the body loses the router's OpenRouter dialect: prefix, cascade, session, cache markers; reasoning becomes reasoning_effort", () => {
|
|
171
|
+
test("the body loses the router's OpenRouter dialect: prefix, cascade, session, cache markers; reasoning becomes reasoning_effort", async () => {
|
|
172
172
|
const out = toCompatBody("vllm", { model: "vllm/llama", models: ["vllm/llama", "vllm/other"], session_id: "s", stream: true, reasoning: { effort: "xhigh" }, messages: [{ role: "system", content: [{ type: "text", text: "sys", cache_control: { type: "ephemeral" } }] }, { role: "user", content: "hi" }] });
|
|
173
173
|
expect(out.model).toBe("llama");
|
|
174
174
|
expect(out.models).toBeUndefined();
|
|
@@ -180,7 +180,7 @@ describe("the OpenAI-compatible client", () => {
|
|
|
180
180
|
expect(toCompatBody("vllm", { model: "vllm/llama", reasoning: { enabled: false } }).reasoning_effort).toBeUndefined();
|
|
181
181
|
});
|
|
182
182
|
|
|
183
|
-
test("endpoints: OpenAI bears a token; Azure names the deployment in the path and keys with api-key", () => {
|
|
183
|
+
test("endpoints: OpenAI bears a token; Azure names the deployment in the path and keys with api-key", async () => {
|
|
184
184
|
const oa = compatEndpoint(entry({ id: "openai-direct", kind: "openai", baseUrl: "https://api.openai.com/v1/", headers: { "x-org": "o" } }), "gpt-4o");
|
|
185
185
|
expect(oa.url).toBe("https://api.openai.com/v1/chat/completions");
|
|
186
186
|
expect(oa.headers).toMatchObject({ authorization: "Bearer sk-x", "x-org": "o" });
|
|
@@ -190,7 +190,7 @@ describe("the OpenAI-compatible client", () => {
|
|
|
190
190
|
expect(az.headers.authorization).toBeUndefined();
|
|
191
191
|
});
|
|
192
192
|
|
|
193
|
-
test("statuses: OpenAI's insufficient_quota 429 is the account, a plain 429 the moment; 400 context is final", () => {
|
|
193
|
+
test("statuses: OpenAI's insufficient_quota 429 is the account, a plain 429 the moment; 400 context is final", async () => {
|
|
194
194
|
expect(classifyCompatStatus("x", 429, { error: { code: "insufficient_quota", message: "You exceeded your current quota" } })).toMatchObject({ kind: "quota", retryable: true });
|
|
195
195
|
expect(classifyCompatStatus("x", 429, { error: { message: "Rate limit reached" } })).toMatchObject({ kind: "rate_limit", retryable: true });
|
|
196
196
|
expect(classifyCompatStatus("x", 400, { error: { message: "This model's maximum context length is 8192 tokens" } })).toMatchObject({ kind: "context_length", retryable: false });
|
|
@@ -246,7 +246,7 @@ describe("the OpenAI-compatible client", () => {
|
|
|
246
246
|
});
|
|
247
247
|
|
|
248
248
|
describe("the Anthropic client", () => {
|
|
249
|
-
test("the request: system blocks keep cache markers, turns alternate, tools and results map, thinking follows the effort", () => {
|
|
249
|
+
test("the request: system blocks keep cache markers, turns alternate, tools and results map, thinking follows the effort", async () => {
|
|
250
250
|
const body = {
|
|
251
251
|
model: "anthropic-direct/claude-sonnet-4",
|
|
252
252
|
stream: true,
|
|
@@ -286,7 +286,7 @@ describe("the Anthropic client", () => {
|
|
|
286
286
|
expect(plain.temperature).toBe(0.2);
|
|
287
287
|
});
|
|
288
288
|
|
|
289
|
-
test("the stream: message_start opens, text and tool blocks become chunks, message_delta closes with usage in the OpenAI convention", () => {
|
|
289
|
+
test("the stream: message_start opens, text and tool blocks become chunks, message_delta closes with usage in the OpenAI convention", async () => {
|
|
290
290
|
const t = createAnthropicTranslator("anthropic-direct/claude-sonnet-4");
|
|
291
291
|
const push = (event: string, data: Record<string, unknown>) => t.push({ event, data: JSON.stringify(data) });
|
|
292
292
|
const start = push("message_start", { type: "message_start", message: { id: "msg_1", model: "claude-sonnet-4-20250514", usage: { input_tokens: 100, cache_read_input_tokens: 40, cache_creation_input_tokens: 10 } } })!;
|
|
@@ -380,7 +380,7 @@ describe("the Anthropic client", () => {
|
|
|
380
380
|
};
|
|
381
381
|
const cfg = cfgWith([entry({ id: "sub", kind: "anthropic", auth: "oauth-bearer", baseUrl: "https://api.anthropic.com", apiKey: "sk-ant-oat01-x", models: [{ id: "claude-sonnet-5", input: 0, output: 0 }] })]);
|
|
382
382
|
const drain = async (messages: unknown[]): Promise<void> => {
|
|
383
|
-
const d = await createAnthropicClient(cfg, "sub", fetchImpl).dispatch({ body: { model: "sub/claude-sonnet-5", messages, max_tokens: 16 }, sessionId: "s", signal: new AbortController().signal });
|
|
383
|
+
const d = await (await createAnthropicClient(cfg, "sub", fetchImpl)).dispatch({ body: { model: "sub/claude-sonnet-5", messages, max_tokens: 16 }, sessionId: "s", signal: new AbortController().signal });
|
|
384
384
|
for await (const c of d.chunks) void c;
|
|
385
385
|
};
|
|
386
386
|
await drain([{ role: "user", content: "ping" }]);
|
package/test/views.test.ts
CHANGED
|
@@ -4,12 +4,12 @@ import { tmpdir } from "node:os";
|
|
|
4
4
|
import { join } from "node:path";
|
|
5
5
|
|
|
6
6
|
import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
|
|
7
|
-
import { createFeedbackStore } from "../src/cost/feedback.ts";
|
|
8
|
-
import { createLedger } from "../src/cost/ledger.ts";
|
|
9
7
|
import type { LedgerEntry } from "../src/cost/types.ts";
|
|
10
8
|
import { exportCsv, exportRows, feedbackView, harnessScopeParam, spendUsdSince } from "../src/cost/views.ts";
|
|
11
9
|
import { startServer, type StartedServer } from "../src/server/http.ts";
|
|
12
|
-
import {
|
|
10
|
+
import { jsonParam, openSqlDb, type SqlDb } from "../src/util/sql.ts";
|
|
11
|
+
import { createSqlLedger } from "../src/cost/ledger-sql.ts";
|
|
12
|
+
import { migrateStore } from "../src/util/schema.ts";
|
|
13
13
|
|
|
14
14
|
/**
|
|
15
15
|
* The ledger views a front door reads instead of the ledger file: spend over
|
|
@@ -58,48 +58,64 @@ function entry(over: Partial<LedgerEntry>): LedgerEntry {
|
|
|
58
58
|
} as LedgerEntry;
|
|
59
59
|
}
|
|
60
60
|
|
|
61
|
-
function seeded() {
|
|
61
|
+
async function seeded(): Promise<SqlDb> {
|
|
62
62
|
const cfg = structuredClone(DEFAULT_CONFIG);
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
const
|
|
67
|
-
ledger.
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
ledger
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
63
|
+
// A file, not :memory:, because the views read through their OWN handle on
|
|
64
|
+
// the same store — which is the arrangement in a real deployment, and what
|
|
65
|
+
// an in-memory database cannot represent.
|
|
66
|
+
const path = join(tmpdir(), `views-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`);
|
|
67
|
+
cfg.ledger.path = path;
|
|
68
|
+
const db = openSqlDb(path);
|
|
69
|
+
await migrateStore(db);
|
|
70
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
71
|
+
const feedback = {
|
|
72
|
+
record: async (
|
|
73
|
+
f: { ledgerId: string; ompSessionId: string; slug: string; tier: string; verdict: string; note: string },
|
|
74
|
+
atMs: number,
|
|
75
|
+
): Promise<void> => {
|
|
76
|
+
await db.sql`INSERT INTO feedback (id, ledger_id, omp_session_id, slug, tier, verdict, note, created_at_ms)
|
|
77
|
+
VALUES (${crypto.randomUUID()}, ${f.ledgerId}, ${f.ompSessionId}, ${f.slug}, ${f.tier}, ${f.verdict}, ${f.note}, ${atMs})`;
|
|
78
|
+
},
|
|
79
|
+
};
|
|
80
|
+
await ledger.record(entry({ id: "l1", harnessId: "u_ada", scope: "acme.api", slug: "anthropic/claude-sonnet-5", servedSlug: "anthropic/claude-sonnet-5", predictedUsd: 0.01, reportedUsd: 0.012 }));
|
|
81
|
+
await ledger.record(entry({ id: "l2", harnessId: "u_ada", scope: "acme.web", slug: "anthropic/claude-sonnet-5", servedSlug: null, predictedUsd: 0.01, reportedUsd: null, escalationSignal: "circular" }));
|
|
82
|
+
await ledger.record(entry({ id: "l3", harnessId: "u_bob", scope: "acme.api", slug: "ollama/glm-5.3-flash", servedSlug: "ollama/glm-5.3-flash", predictedUsd: 0.001, reportedUsd: 0.001, error: "boom" }));
|
|
83
|
+
await ledger.record(entry({ id: "l4", harnessId: "u_bob", requestedModel: "digest", slug: "ollama/glm-5.3-flash", servedSlug: "ollama/glm-5.3-flash", predictedUsd: 0.5, reportedUsd: 0.5 }));
|
|
84
|
+
await ledger.record(entry({ id: "l5", harnessId: "u_bob", createdAtMs: NOW - 40 * DAY, slug: "ollama/glm-5.3-flash", predictedUsd: 5, reportedUsd: 5 }));
|
|
85
|
+
await feedback.record({ ledgerId: "l1", ompSessionId: "s", slug: "anthropic/claude-sonnet-5", tier: "simple", verdict: "good", note: "" }, NOW - 1000);
|
|
86
|
+
await feedback.record({ ledgerId: "l2", ompSessionId: "s", slug: "anthropic/claude-sonnet-5", tier: "simple", verdict: "bad", note: "" }, NOW - 900);
|
|
87
|
+
await feedback.record({ ledgerId: "l3", ompSessionId: "s", slug: "ollama/glm-5.3-flash", tier: "simple", verdict: "bad", note: "looped" }, NOW - 800);
|
|
75
88
|
return db;
|
|
76
89
|
}
|
|
77
90
|
|
|
78
91
|
describe("ledger views", () => {
|
|
79
|
-
|
|
92
|
+
let db: SqlDb;
|
|
93
|
+
beforeAll(async () => {
|
|
94
|
+
db = await seeded();
|
|
95
|
+
});
|
|
80
96
|
const since = NOW - DAY;
|
|
81
97
|
|
|
82
|
-
test("spend over a harness set, everything, or nothing", () => {
|
|
83
|
-
expect(spendUsdSince(db, since, ["u_ada"])).toBeCloseTo(0.022, 6); // reported where present, predicted otherwise
|
|
84
|
-
expect(spendUsdSince(db, since, ["u_ada", "u_bob"])).toBeCloseTo(0.523, 6); // the digest row counts as spend
|
|
85
|
-
expect(spendUsdSince(db, since, null)).toBeCloseTo(0.523, 6);
|
|
86
|
-
expect(spendUsdSince(db, NOW - 60 * DAY, null)).toBeCloseTo(5.523, 6);
|
|
87
|
-
expect(spendUsdSince(db, since, [])).toBe(0);
|
|
98
|
+
test("spend over a harness set, everything, or nothing", async () => {
|
|
99
|
+
expect(await spendUsdSince(db, since, ["u_ada"])).toBeCloseTo(0.022, 6); // reported where present, predicted otherwise
|
|
100
|
+
expect(await spendUsdSince(db, since, ["u_ada", "u_bob"])).toBeCloseTo(0.523, 6); // the digest row counts as spend
|
|
101
|
+
expect(await spendUsdSince(db, since, null)).toBeCloseTo(0.523, 6);
|
|
102
|
+
expect(await spendUsdSince(db, NOW - 60 * DAY, null)).toBeCloseTo(5.523, 6);
|
|
103
|
+
expect(await spendUsdSince(db, since, [])).toBe(0);
|
|
88
104
|
});
|
|
89
105
|
|
|
90
|
-
test("spend narrowed to one context scope, so a front door charges a project", () => {
|
|
106
|
+
test("spend narrowed to one context scope, so a front door charges a project", async () => {
|
|
91
107
|
// l1 (0.012, u_ada) and l3 (0.001, u_bob) carried acme.api; l2 (0.01 predicted) carried acme.web.
|
|
92
|
-
expect(spendUsdSince(db, since, null, "acme.api")).toBeCloseTo(0.013, 6);
|
|
93
|
-
expect(spendUsdSince(db, since, null, "acme.web")).toBeCloseTo(0.01, 6);
|
|
94
|
-
expect(spendUsdSince(db, since, ["u_ada"], "acme.api")).toBeCloseTo(0.012, 6); // harness and scope compose
|
|
95
|
-
expect(spendUsdSince(db, since, ["u_bob"], "acme.web")).toBe(0);
|
|
96
|
-
expect(spendUsdSince(db, since, null, "nope")).toBe(0);
|
|
97
|
-
expect(spendUsdSince(db, since, null, "")).toBeCloseTo(0.523, 6); // no scope given: every turn, scoped or not
|
|
98
|
-
expect(spendUsdSince(db, since, [], "acme.api")).toBe(0);
|
|
108
|
+
expect(await spendUsdSince(db, since, null, "acme.api")).toBeCloseTo(0.013, 6);
|
|
109
|
+
expect(await spendUsdSince(db, since, null, "acme.web")).toBeCloseTo(0.01, 6);
|
|
110
|
+
expect(await spendUsdSince(db, since, ["u_ada"], "acme.api")).toBeCloseTo(0.012, 6); // harness and scope compose
|
|
111
|
+
expect(await spendUsdSince(db, since, ["u_bob"], "acme.web")).toBe(0);
|
|
112
|
+
expect(await spendUsdSince(db, since, null, "nope")).toBe(0);
|
|
113
|
+
expect(await spendUsdSince(db, since, null, "")).toBeCloseTo(0.523, 6); // no scope given: every turn, scoped or not
|
|
114
|
+
expect(await spendUsdSince(db, since, [], "acme.api")).toBe(0);
|
|
99
115
|
});
|
|
100
116
|
|
|
101
|
-
test("feedback by model with distinct judges, scoped by harness", () => {
|
|
102
|
-
const all = feedbackView(db, since, null);
|
|
117
|
+
test("feedback by model with distinct judges, scoped by harness", async () => {
|
|
118
|
+
const all = await feedbackView(db, since, null);
|
|
103
119
|
expect(all.byModel).toEqual([
|
|
104
120
|
{ slug: "anthropic/claude-sonnet-5", good: 1, bad: 1, judges: 1 },
|
|
105
121
|
{ slug: "ollama/glm-5.3-flash", good: 0, bad: 1, judges: 1 },
|
|
@@ -109,12 +125,12 @@ describe("ledger views", () => {
|
|
|
109
125
|
["u_ada", "bad", ""],
|
|
110
126
|
["u_ada", "good", ""],
|
|
111
127
|
]);
|
|
112
|
-
expect(feedbackView(db, since, ["u_bob"]).byModel).toEqual([{ slug: "ollama/glm-5.3-flash", good: 0, bad: 1, judges: 1 }]);
|
|
113
|
-
expect(feedbackView(db, since, []).recent).toEqual([]);
|
|
128
|
+
expect((await feedbackView(db, since, ["u_bob"])).byModel).toEqual([{ slug: "ollama/glm-5.3-flash", good: 0, bad: 1, judges: 1 }]);
|
|
129
|
+
expect((await feedbackView(db, since, [])).recent).toEqual([]);
|
|
114
130
|
});
|
|
115
131
|
|
|
116
|
-
test("export rows by day, harness, served model and scope; digest and old rows out; CSV quoting", () => {
|
|
117
|
-
const rows = exportRows(db, since, null);
|
|
132
|
+
test("export rows by day, harness, served model and scope; digest and old rows out; CSV quoting", async () => {
|
|
133
|
+
const rows = await exportRows(db, since, null);
|
|
118
134
|
// u_ada's two turns are one model on one day but two projects, so they no longer share a row.
|
|
119
135
|
expect(rows).toHaveLength(3);
|
|
120
136
|
expect(rows[0]).toMatchObject({ day: "2026-09-07", harnessId: "u_ada", slug: "anthropic/claude-sonnet-5", scope: "acme.api", provider: "openrouter", dispatches: 1, promptTokens: 1000, cachedTokens: 400, completionTokens: 50, escalations: 0, errors: 0 });
|
|
@@ -122,10 +138,10 @@ describe("ledger views", () => {
|
|
|
122
138
|
expect(rows[1]).toMatchObject({ harnessId: "u_ada", scope: "acme.web", dispatches: 1, escalations: 1 });
|
|
123
139
|
expect(rows[1]!.spendUsd).toBeCloseTo(0.01, 6);
|
|
124
140
|
expect(rows[2]).toMatchObject({ harnessId: "u_bob", scope: "acme.api", provider: "ollama", dispatches: 1, errors: 1 });
|
|
125
|
-
expect(exportRows(db, since, ["u_bob"])).toHaveLength(1);
|
|
126
|
-
expect(exportRows(db, since, [])).toEqual([]);
|
|
141
|
+
expect(await exportRows(db, since, ["u_bob"])).toHaveLength(1);
|
|
142
|
+
expect(await exportRows(db, since, [])).toEqual([]);
|
|
127
143
|
// A turn that carried no scope groups under "": what every row written before v18 does.
|
|
128
|
-
expect(exportRows(db, NOW - 60 * DAY, ["u_bob"]).map((r) => r.scope).sort()).toEqual(["", "acme.api"]);
|
|
144
|
+
expect((await exportRows(db, NOW - 60 * DAY, ["u_bob"])).map((r) => r.scope).sort()).toEqual(["", "acme.api"]);
|
|
129
145
|
const csv = exportCsv([{ ...rows[0]!, harnessId: 'ada, "L"' }]);
|
|
130
146
|
expect(csv.split("\n")[0]).toBe("day,harness,model,provider,dispatches,prompt_tokens,cached_tokens,completion_tokens,spend_usd,escalations,errors,scope");
|
|
131
147
|
expect(csv.split("\n")[1]).toBe('2026-09-07,"ada, ""L""",anthropic/claude-sonnet-5,openrouter,1,1000,400,50,0.012000,0,0,acme.api');
|
|
@@ -138,16 +154,20 @@ describe("ledger views", () => {
|
|
|
138
154
|
describe("view routes", () => {
|
|
139
155
|
let handle: StartedServer;
|
|
140
156
|
const dir = mkdtempSync(join(tmpdir(), "amr-views-"));
|
|
141
|
-
beforeAll(() => {
|
|
157
|
+
beforeAll(async () => {
|
|
142
158
|
const cfg = structuredClone(DEFAULT_CONFIG);
|
|
143
159
|
cfg.server.host = "127.0.0.1";
|
|
144
160
|
cfg.server.port = 0;
|
|
145
161
|
cfg.server.apiKey = "k";
|
|
146
162
|
cfg.ledger.path = join(dir, "router.db");
|
|
147
163
|
// Seed through the ledger on the same file before the server opens it.
|
|
148
|
-
const db =
|
|
149
|
-
|
|
150
|
-
db
|
|
164
|
+
const db = openSqlDb(cfg.ledger.path);
|
|
165
|
+
await migrateStore(db);
|
|
166
|
+
const seed = createSqlLedger(db, cfg, { findModel: () => null });
|
|
167
|
+
await seed.record(
|
|
168
|
+
entry({ id: "r1", createdAtMs: Date.now() - 1000, harnessId: "u_x", scope: "acme.api", predictedUsd: 0.2, reportedUsd: 0.25 }),
|
|
169
|
+
);
|
|
170
|
+
await db.close();
|
|
151
171
|
handle = startServer(cfg);
|
|
152
172
|
});
|
|
153
173
|
afterAll(async () => {
|
|
@@ -190,12 +210,15 @@ describe("view routes", () => {
|
|
|
190
210
|
});
|
|
191
211
|
|
|
192
212
|
describe("decision entries", () => {
|
|
193
|
-
|
|
213
|
+
let db: SqlDb;
|
|
214
|
+
beforeAll(async () => {
|
|
215
|
+
db = await seeded();
|
|
216
|
+
});
|
|
194
217
|
const since = NOW - DAY;
|
|
195
218
|
|
|
196
|
-
test("newest first over a harness set, with the verdicts given on each turn", () => {
|
|
219
|
+
test("newest first over a harness set, with the verdicts given on each turn", async () => {
|
|
197
220
|
const { decisionEntries } = require("../src/cost/views.ts") as typeof import("../src/cost/views.ts");
|
|
198
|
-
const ada = decisionEntries(db, { sinceMs: since, harness: ["u_ada"] });
|
|
221
|
+
const ada = await decisionEntries(db, { sinceMs: since, harness: ["u_ada"] });
|
|
199
222
|
expect(ada.map((e) => e.id)).toEqual(["l1", "l2"]); // same instant in the fixture; insertion order within it is stable
|
|
200
223
|
expect(ada.find((e) => e.id === "l1")?.feedback).toEqual([{ verdict: "good", note: "", createdAtMs: NOW - 1000 }]);
|
|
201
224
|
expect(ada.find((e) => e.id === "l2")?.escalationSignal).toBe("circular");
|
|
@@ -203,27 +226,26 @@ describe("decision entries", () => {
|
|
|
203
226
|
expect(ada.find((e) => e.id === "l1")?.scope).toBe("acme.api");
|
|
204
227
|
expect(ada.find((e) => e.id === "l2")?.scope).toBe("acme.web");
|
|
205
228
|
// Everyone, within the window: the 40-day-old row stays out; the digest row is a turn like any other.
|
|
206
|
-
expect(decisionEntries(db, { sinceMs: since, harness: null }).map((e) => e.id).sort()).toEqual(["l1", "l2", "l3", "l4"]);
|
|
207
|
-
expect(decisionEntries(db, { sinceMs: 0, harness: null }).length).toBe(5);
|
|
229
|
+
expect((await decisionEntries(db, { sinceMs: since, harness: null })).map((e) => e.id).sort()).toEqual(["l1", "l2", "l3", "l4"]);
|
|
230
|
+
expect((await decisionEntries(db, { sinceMs: 0, harness: null })).length).toBe(5);
|
|
208
231
|
// A model, a tier, nobody, and a cap.
|
|
209
|
-
expect(decisionEntries(db, { sinceMs: since, harness: null, slug: "ollama/glm-5.3-flash" }).map((e) => e.id).sort()).toEqual(["l3", "l4"]);
|
|
210
|
-
expect(decisionEntries(db, { sinceMs: since, harness: null, tier: "hard" })).toEqual([]);
|
|
211
|
-
expect(decisionEntries(db, { sinceMs: since, harness: [] })).toEqual([]);
|
|
212
|
-
expect(decisionEntries(db, { sinceMs: since, harness: null, limit: 1 }).length).toBe(1);
|
|
232
|
+
expect((await decisionEntries(db, { sinceMs: since, harness: null, slug: "ollama/glm-5.3-flash" })).map((e) => e.id).sort()).toEqual(["l3", "l4"]);
|
|
233
|
+
expect(await decisionEntries(db, { sinceMs: since, harness: null, tier: "hard" })).toEqual([]);
|
|
234
|
+
expect(await decisionEntries(db, { sinceMs: since, harness: [] })).toEqual([]);
|
|
235
|
+
expect((await decisionEntries(db, { sinceMs: since, harness: null, limit: 1 })).length).toBe(1);
|
|
213
236
|
// The error on l3 and its note ride along, so an explorer can show why a turn went wrong.
|
|
214
|
-
const bob = decisionEntries(db, { sinceMs: since, harness: ["u_bob"], slug: "ollama/glm-5.3-flash" });
|
|
237
|
+
const bob = await decisionEntries(db, { sinceMs: since, harness: ["u_bob"], slug: "ollama/glm-5.3-flash" });
|
|
215
238
|
expect(bob.find((e) => e.id === "l3")?.error).toBe("boom");
|
|
216
239
|
expect(bob.find((e) => e.id === "l3")?.feedback[0]?.note).toBe("looped");
|
|
217
240
|
});
|
|
218
241
|
|
|
219
|
-
test("the recorded cost split rides along; unpriced rows leave it absent", () => {
|
|
242
|
+
test("the recorded cost split rides along; unpriced rows leave it absent", async () => {
|
|
220
243
|
const { decisionEntries } = require("../src/cost/views.ts") as typeof import("../src/cost/views.ts");
|
|
221
244
|
// The fixture records before any catalog fetch, so every row stored NULL;
|
|
222
245
|
// one priced row stands in for a turn the router could price at record time.
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
const entries = decisionEntries(db, { sinceMs: since, harness: null });
|
|
246
|
+
const breakdown = { freshPrompt: 0.006, cacheRead: 0.004, cacheWrite: 0, completion: 0.002, reasoning: 0, images: 0, request: 0, total: 0.006, tierAtPromptTokens: 0 };
|
|
247
|
+
await db.sql`UPDATE ledger SET cost_breakdown = ${jsonParam(db, breakdown)} WHERE id = ${"l1"}`;
|
|
248
|
+
const entries = await decisionEntries(db, { sinceMs: since, harness: null });
|
|
227
249
|
const priced = entries.find((e) => e.id === "l1");
|
|
228
250
|
expect(priced?.costBreakdown?.total).toBeCloseTo(0.006, 6);
|
|
229
251
|
expect(priced?.costBreakdown?.cacheRead).toBeCloseTo(0.004, 6);
|