auto-model-router 0.30.3 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +32 -2
- package/omp-extension/router-configure.ts +9 -7
- package/package.json +1 -1
- package/src/cli/config-cmd.ts +8 -7
- package/src/cli/explain.ts +10 -5
- package/src/cli/export.ts +6 -5
- package/src/cli/models.ts +10 -7
- package/src/cli/report.ts +6 -1
- package/src/cli/stats.ts +7 -7
- package/src/config/load.ts +10 -1
- package/src/config/types.ts +10 -1
- package/src/context/bridge.ts +7 -7
- package/src/context/index.ts +3 -3
- package/src/context/store.ts +39 -56
- package/src/context/types.ts +7 -6
- package/src/cost/blended.ts +28 -7
- package/src/cost/feedback.ts +33 -37
- package/src/cost/ledger-sql.ts +547 -0
- package/src/cost/ledger.ts +30 -459
- package/src/cost/report.ts +171 -129
- package/src/cost/retention.ts +10 -10
- package/src/cost/summary.ts +15 -10
- package/src/cost/types.ts +43 -62
- package/src/cost/views.ts +79 -49
- package/src/eval/calibrate.ts +47 -12
- package/src/eval/run.ts +18 -2
- package/src/lib.ts +6 -2
- package/src/router/candidates.ts +7 -15
- package/src/router/classify.ts +6 -4
- package/src/router/index.ts +95 -9
- package/src/router/select.ts +38 -21
- package/src/router/state.ts +90 -102
- package/src/router/types.ts +11 -5
- package/src/server/advise.ts +6 -4
- package/src/server/compaction-digest.ts +1 -1
- package/src/server/digest.ts +9 -10
- package/src/server/http.ts +109 -46
- package/src/server/providers.ts +18 -4
- package/src/server/turn.ts +32 -9
- package/src/tokens/estimate.ts +16 -6
- package/src/upstream/ollama-usage.ts +21 -11
- package/src/util/schema.ts +201 -0
- package/src/util/sql.ts +246 -0
- package/src/wire/anthropic/messages.ts +3 -4
- package/src/wire/openai/request.ts +1 -0
- package/src/wire/types.ts +7 -0
- package/test/anthropic-wire.test.ts +9 -9
- package/test/benchmark-feeds.test.ts +7 -7
- package/test/cache-control.test.ts +7 -7
- package/test/cache-estimate.test.ts +5 -5
- package/test/catalog-view.test.ts +4 -4
- package/test/catalog.test.ts +11 -11
- package/test/classify.test.ts +24 -24
- package/test/compaction.test.ts +20 -20
- package/test/config-wizard.test.ts +32 -32
- package/test/config.test.ts +10 -10
- package/test/connect-harnesses.test.ts +11 -11
- package/test/context-bridge.test.ts +40 -30
- package/test/context-prune.test.ts +43 -36
- package/test/context-query.test.ts +8 -8
- package/test/controls.test.ts +54 -27
- package/test/cost.test.ts +12 -12
- package/test/digest.test.ts +55 -44
- package/test/embed-lifecycle.test.ts +5 -5
- package/test/embed-logic.test.ts +26 -26
- package/test/escalate.test.ts +17 -17
- package/test/eval.test.ts +73 -16
- package/test/executable.test.ts +6 -6
- package/test/exploration.test.ts +19 -20
- package/test/failover.test.ts +22 -21
- package/test/fakes.ts +105 -0
- package/test/features.test.ts +21 -21
- package/test/harness-requests.test.ts +3 -3
- package/test/harness-switch.test.ts +5 -5
- package/test/hold-exploration.test.ts +13 -13
- package/test/hot-reload.test.ts +5 -5
- package/test/learned.test.ts +5 -5
- package/test/ledger-sql.test.ts +342 -0
- package/test/mcp-entry.test.ts +5 -5
- package/test/migrations.test.ts +28 -22
- package/test/models-yml.test.ts +18 -18
- package/test/ollama.test.ts +40 -34
- package/test/omp-credentials.test.ts +16 -16
- package/test/policy.test.ts +3 -3
- package/test/reconfigure.test.ts +4 -4
- package/test/redaction.test.ts +41 -35
- package/test/remote.test.ts +12 -12
- package/test/report-logic.test.ts +8 -8
- package/test/report.test.ts +95 -87
- package/test/retention.test.ts +79 -66
- package/test/schema.test.ts +123 -0
- package/test/scope.test.ts +8 -8
- package/test/select.test.ts +216 -257
- package/test/skills.test.ts +3 -3
- package/test/sql-shim.test.ts +154 -0
- package/test/state.test.ts +43 -36
- package/test/summary.test.ts +38 -27
- package/test/tier-plan.test.ts +45 -62
- package/test/toast-logic.test.ts +31 -31
- package/test/tokens.test.ts +95 -80
- package/test/trust-attribution.test.ts +217 -187
- package/test/trust-window.test.ts +37 -32
- package/test/turn.test.ts +55 -23
- package/test/upstreams.test.ts +13 -13
- package/test/views.test.ts +81 -59
- package/test/wire-request.test.ts +17 -17
- package/test/wire-responses.test.ts +4 -4
- package/tools/agentdox-e2e.ts +5 -2
- package/tools/export-benchmarks.ts +5 -5
- package/tools/ledger-parity.ts +266 -0
- package/tools/replay.ts +16 -8
|
@@ -1,11 +1,17 @@
|
|
|
1
1
|
import { describe, expect, test } from "bun:test";
|
|
2
|
+
import { tmpdir } from "node:os";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
|
|
5
|
+
import { migrateStore } from "../src/util/schema.ts";
|
|
6
|
+
import { num, openSqlDb } from "../src/util/sql.ts";
|
|
7
|
+
import { openDb } from "../src/util/sqlite.ts";
|
|
2
8
|
import { createFeedbackStore } from "../src/cost/feedback.ts";
|
|
3
9
|
|
|
4
10
|
import { loadConfig } from "../src/config/load.ts";
|
|
5
|
-
import {
|
|
11
|
+
import { LATENCY_WINDOW_ROWS } from "../src/cost/ledger.ts";
|
|
12
|
+
import { createSqlLedger } from "../src/cost/ledger-sql.ts";
|
|
6
13
|
import { EMPTY_USAGE, type LedgerEntry } from "../src/cost/types.ts";
|
|
7
14
|
import { createConversationStore } from "../src/router/state.ts";
|
|
8
|
-
import { openDb } from "../src/util/sqlite.ts";
|
|
9
15
|
|
|
10
16
|
const cfg = loadConfig({});
|
|
11
17
|
|
|
@@ -54,131 +60,137 @@ function entry(over: Partial<LedgerEntry>): LedgerEntry {
|
|
|
54
60
|
* onto whichever models happened to avoid those conditions.
|
|
55
61
|
*/
|
|
56
62
|
describe("trust attribution", () => {
|
|
57
|
-
function trustAfter(errors: Array<string | null>): number {
|
|
58
|
-
const db =
|
|
63
|
+
async function trustAfter(errors: Array<string | null>): Promise<number> {
|
|
64
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
65
|
+
await migrateStore(db);
|
|
59
66
|
try {
|
|
60
|
-
const ledger =
|
|
61
|
-
for (const error of errors) ledger.record(entry({ error }));
|
|
62
|
-
const trust = ledger.trust("vendor/model");
|
|
67
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
68
|
+
for (const error of errors) await ledger.record(entry({ error }));
|
|
69
|
+
const trust = await ledger.trust("vendor/model");
|
|
63
70
|
expect(trust).not.toBeNull();
|
|
64
71
|
return trust?.successRate ?? 0;
|
|
65
72
|
} finally {
|
|
66
|
-
db.close();
|
|
73
|
+
await db.close();
|
|
67
74
|
}
|
|
68
75
|
}
|
|
69
76
|
|
|
70
|
-
|
|
77
|
+
// A clean run's rate, recomputed per test: a describe body cannot await.
|
|
78
|
+
const clean = async (): Promise<number> => await trustAfter([null, null, null, null]);
|
|
71
79
|
|
|
72
|
-
test("client aborts do not count against the model", () => {
|
|
73
|
-
expect(trustAfter([null, null, "request aborted", "request aborted"])).toBe(
|
|
80
|
+
test("client aborts do not count against the model", async () => {
|
|
81
|
+
expect(await trustAfter([null, null, "request aborted", "request aborted"])).toBe((await clean()));
|
|
74
82
|
});
|
|
75
83
|
|
|
76
|
-
test("account-level auth refusals do not count against the model", () => {
|
|
84
|
+
test("account-level auth refusals do not count against the model", async () => {
|
|
77
85
|
expect(
|
|
78
|
-
trustAfter([
|
|
86
|
+
await trustAfter([
|
|
79
87
|
null,
|
|
80
88
|
null,
|
|
81
89
|
"auth: No auth credentials found",
|
|
82
90
|
"auth: Insufficient credits",
|
|
83
91
|
]),
|
|
84
|
-
).toBe(
|
|
92
|
+
).toBe((await clean()));
|
|
85
93
|
});
|
|
86
94
|
|
|
87
|
-
test("provider moderation/policy blocks do not count against the model", () => {
|
|
95
|
+
test("provider moderation/policy blocks do not count against the model", async () => {
|
|
88
96
|
expect(
|
|
89
|
-
trustAfter([
|
|
97
|
+
await trustAfter([
|
|
90
98
|
null,
|
|
91
99
|
null,
|
|
92
100
|
"moderation: Request blocked: prompt injection patterns detected",
|
|
93
101
|
"moderation: This model requires 18+ age confirmation",
|
|
94
102
|
]),
|
|
95
|
-
).toBe(
|
|
103
|
+
).toBe((await clean()));
|
|
96
104
|
});
|
|
97
105
|
|
|
98
|
-
test("guardrail model_unavailable does not count against the model", () => {
|
|
106
|
+
test("guardrail model_unavailable does not count against the model", async () => {
|
|
99
107
|
expect(
|
|
100
|
-
trustAfter([null, null, "model_unavailable: No endpoints available matching your guardrail", null]),
|
|
108
|
+
await trustAfter([null, null, "model_unavailable: No endpoints available matching your guardrail", null]),
|
|
101
109
|
).toBeGreaterThan(0.7);
|
|
102
110
|
});
|
|
103
111
|
|
|
104
|
-
test("a genuine upstream error DOES count against the model", () => {
|
|
105
|
-
const withError = trustAfter([null, null, "upstream_error: Upstream request failed", "upstream_error: boom"]);
|
|
106
|
-
expect(withError).toBeLessThan(
|
|
112
|
+
test("a genuine upstream error DOES count against the model", async () => {
|
|
113
|
+
const withError = await trustAfter([null, null, "upstream_error: Upstream request failed", "upstream_error: boom"]);
|
|
114
|
+
expect(withError).toBeLessThan((await clean()));
|
|
107
115
|
});
|
|
108
116
|
|
|
109
|
-
test("an unclassifiable legacy error stays attributable", () => {
|
|
117
|
+
test("an unclassifiable legacy error stays attributable", async () => {
|
|
110
118
|
// No "<kind>: " prefix and not the known abort text: we cannot prove it
|
|
111
119
|
// was blameless, so it keeps counting (the stricter reading).
|
|
112
|
-
const withError = trustAfter([null, null, "something odd happened", "another"]);
|
|
113
|
-
expect(withError).toBeLessThan(
|
|
120
|
+
const withError = await trustAfter([null, null, "something odd happened", "another"]);
|
|
121
|
+
expect(withError).toBeLessThan((await clean()));
|
|
114
122
|
});
|
|
115
123
|
|
|
116
|
-
test("errors field still records the raw text regardless of attribution", () => {
|
|
117
|
-
const db =
|
|
124
|
+
test("errors field still records the raw text regardless of attribution", async () => {
|
|
125
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
126
|
+
await migrateStore(db);
|
|
118
127
|
try {
|
|
119
|
-
const ledger =
|
|
120
|
-
ledger.record(entry({ error: "request aborted" }));
|
|
121
|
-
const row = db.
|
|
122
|
-
error
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
expect(row
|
|
126
|
-
expect(row
|
|
128
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
129
|
+
await ledger.record(entry({ error: "request aborted" }));
|
|
130
|
+
const row = await db.one<{ error: string | null; error_kind: string | null }>(
|
|
131
|
+
"SELECT error, error_kind FROM ledger",
|
|
132
|
+
);
|
|
133
|
+
expect(row).not.toBeNull();
|
|
134
|
+
expect(row?.error).toBe("request aborted");
|
|
135
|
+
expect(row?.error_kind).toBe("aborted");
|
|
127
136
|
} finally {
|
|
128
|
-
db.close();
|
|
137
|
+
await db.close();
|
|
129
138
|
}
|
|
130
139
|
});
|
|
131
140
|
|
|
132
|
-
test("escalations still count as failures independently of errors", () => {
|
|
133
|
-
const db =
|
|
141
|
+
test("escalations still count as failures independently of errors", async () => {
|
|
142
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
143
|
+
await migrateStore(db);
|
|
134
144
|
try {
|
|
135
|
-
const ledger =
|
|
136
|
-
ledger.record(entry({ error: null }));
|
|
137
|
-
ledger.record(entry({ error: null }));
|
|
138
|
-
ledger.record(entry({ escalationSignal: "empty_completion" }));
|
|
139
|
-
const trust = ledger.trust("vendor/model");
|
|
145
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
146
|
+
await ledger.record(entry({ error: null }));
|
|
147
|
+
await ledger.record(entry({ error: null }));
|
|
148
|
+
await ledger.record(entry({ escalationSignal: "empty_completion" }));
|
|
149
|
+
const trust = await ledger.trust("vendor/model");
|
|
140
150
|
expect(trust?.escalations).toBe(1);
|
|
141
|
-
expect(trust?.successRate).toBeLessThan(
|
|
151
|
+
expect(trust?.successRate).toBeLessThan((await clean()));
|
|
142
152
|
} finally {
|
|
143
|
-
db.close();
|
|
153
|
+
await db.close();
|
|
144
154
|
}
|
|
145
155
|
});
|
|
146
156
|
|
|
147
|
-
test("aborted rows are still counted as attempts", () => {
|
|
148
|
-
const db =
|
|
157
|
+
test("aborted rows are still counted as attempts", async () => {
|
|
158
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
159
|
+
await migrateStore(db);
|
|
149
160
|
try {
|
|
150
|
-
const ledger =
|
|
151
|
-
ledger.record(entry({ error: "request aborted" }));
|
|
152
|
-
ledger.record(entry({ error: null }));
|
|
153
|
-
expect(ledger.trust("vendor/model")?.attempts).toBe(2);
|
|
161
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
162
|
+
await ledger.record(entry({ error: "request aborted" }));
|
|
163
|
+
await ledger.record(entry({ error: null }));
|
|
164
|
+
expect((await ledger.trust("vendor/model"))?.attempts).toBe(2);
|
|
154
165
|
// ...but not as errors.
|
|
155
|
-
expect(ledger.trust("vendor/model")?.errors).toBe(0);
|
|
166
|
+
expect((await ledger.trust("vendor/model"))?.errors).toBe(0);
|
|
156
167
|
} finally {
|
|
157
|
-
db.close();
|
|
168
|
+
await db.close();
|
|
158
169
|
}
|
|
159
170
|
});
|
|
160
171
|
});
|
|
161
172
|
|
|
162
173
|
describe("latency signal", () => {
|
|
163
|
-
function latencyOf(rows: Array<Partial<LedgerEntry>>): { samples: number; ttftMs: number; tokensPerSec: number } | null {
|
|
164
|
-
const db =
|
|
174
|
+
async function latencyOf(rows: Array<Partial<LedgerEntry>>): Promise<{ samples: number; ttftMs: number; tokensPerSec: number } | null> {
|
|
175
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
176
|
+
await migrateStore(db);
|
|
165
177
|
try {
|
|
166
|
-
const ledger =
|
|
167
|
-
for (const r of rows) ledger.record(entry(r));
|
|
168
|
-
const l = ledger.latency("vendor/model");
|
|
178
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
179
|
+
for (const r of rows) await ledger.record(entry(r));
|
|
180
|
+
const l = await ledger.latency("vendor/model");
|
|
169
181
|
return l === null ? null : { samples: l.samples, ttftMs: l.ttftMs, tokensPerSec: l.tokensPerSec };
|
|
170
182
|
} finally {
|
|
171
|
-
db.close();
|
|
183
|
+
await db.close();
|
|
172
184
|
}
|
|
173
185
|
}
|
|
174
186
|
|
|
175
|
-
test("averages TTFT over streamed, non-errored turns", () => {
|
|
176
|
-
expect(latencyOf([{ ttftMs: 50 }, { ttftMs: 100 }, { ttftMs: 150 }])).toEqual({ samples: 3, ttftMs: 100, tokensPerSec: 0 });
|
|
187
|
+
test("averages TTFT over streamed, non-errored turns", async () => {
|
|
188
|
+
expect(await latencyOf([{ ttftMs: 50 }, { ttftMs: 100 }, { ttftMs: 150 }])).toEqual({ samples: 3, ttftMs: 100, tokensPerSec: 0 });
|
|
177
189
|
});
|
|
178
190
|
|
|
179
|
-
test("excludes errored, aborted, and non-streamed (null TTFT) rows", () => {
|
|
191
|
+
test("excludes errored, aborted, and non-streamed (null TTFT) rows", async () => {
|
|
180
192
|
expect(
|
|
181
|
-
latencyOf([
|
|
193
|
+
await latencyOf([
|
|
182
194
|
{ ttftMs: 100 },
|
|
183
195
|
{ ttftMs: 9999, error: "upstream_error: boom" },
|
|
184
196
|
{ ttftMs: 9999, error: "request aborted" },
|
|
@@ -187,12 +199,12 @@ describe("latency signal", () => {
|
|
|
187
199
|
).toEqual({ samples: 1, ttftMs: 100, tokensPerSec: 0 });
|
|
188
200
|
});
|
|
189
201
|
|
|
190
|
-
test("null when no streamed sample exists", () => {
|
|
191
|
-
expect(latencyOf([{ ttftMs: null }, { ttftMs: 0 }])).toBeNull();
|
|
202
|
+
test("null when no streamed sample exists", async () => {
|
|
203
|
+
expect(await latencyOf([{ ttftMs: null }, { ttftMs: 0 }])).toBeNull();
|
|
192
204
|
});
|
|
193
205
|
|
|
194
|
-
test("throughput is aggregate completion tokens per post-TTFT second", () => {
|
|
195
|
-
const l = latencyOf([
|
|
206
|
+
test("throughput is aggregate completion tokens per post-TTFT second", async () => {
|
|
207
|
+
const l = await latencyOf([
|
|
196
208
|
{ ttftMs: 1000, latencyMs: 3000, usage: { ...EMPTY_USAGE, completionTokens: 200 } },
|
|
197
209
|
{ ttftMs: 1000, latencyMs: 3000, usage: { ...EMPTY_USAGE, completionTokens: 200 } },
|
|
198
210
|
]);
|
|
@@ -201,7 +213,7 @@ describe("latency signal", () => {
|
|
|
201
213
|
expect(l?.samples).toBe(2);
|
|
202
214
|
});
|
|
203
215
|
|
|
204
|
-
test("throughput and ttft track a recent window, not the lifetime average", () => {
|
|
216
|
+
test("throughput and ttft track a recent window, not the lifetime average", async () => {
|
|
205
217
|
// Old rows are fast; the recent window is slow. Latency must reflect the
|
|
206
218
|
// recent (slow) behaviour so a degraded model is penalised, not masked by
|
|
207
219
|
// its history. Lifetime blend here would be ~57 tok/s; the window is 10.
|
|
@@ -211,7 +223,7 @@ describe("latency signal", () => {
|
|
|
211
223
|
rows.push({ createdAtMs: t++, ttftMs: 200, latencyMs: 1200, usage: { ...EMPTY_USAGE, completionTokens: 1000 } });
|
|
212
224
|
for (let i = 0; i < LATENCY_WINDOW_ROWS; i++)
|
|
213
225
|
rows.push({ createdAtMs: t++, ttftMs: 4000, latencyMs: 14000, usage: { ...EMPTY_USAGE, completionTokens: 100 } });
|
|
214
|
-
const l = latencyOf(rows);
|
|
226
|
+
const l = await latencyOf(rows);
|
|
215
227
|
expect(l?.samples).toBe(LATENCY_WINDOW_ROWS);
|
|
216
228
|
expect(l?.tokensPerSec).toBeCloseTo(10, 0);
|
|
217
229
|
expect(l?.ttftMs).toBeCloseTo(4000, 5);
|
|
@@ -219,20 +231,19 @@ describe("latency signal", () => {
|
|
|
219
231
|
});
|
|
220
232
|
|
|
221
233
|
describe("v4 migration", () => {
|
|
222
|
-
test("backfills error_kind from stored error text", () => {
|
|
223
|
-
const db =
|
|
234
|
+
test("backfills error_kind from stored error text", async () => {
|
|
235
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
236
|
+
await migrateStore(db);
|
|
224
237
|
try {
|
|
225
|
-
const ledger =
|
|
226
|
-
ledger.record(entry({ error: "request aborted" }));
|
|
227
|
-
ledger.record(entry({ error: "auth: nope" }));
|
|
228
|
-
ledger.record(entry({ error: "moderation: Request blocked: prompt injection" }));
|
|
229
|
-
ledger.record(entry({ error: "model_unavailable: guardrail" }));
|
|
230
|
-
ledger.record(entry({ error: "upstream_error: boom" }));
|
|
231
|
-
ledger.record(entry({ error: null }));
|
|
232
|
-
|
|
233
|
-
const rows = db
|
|
234
|
-
.query("SELECT error, error_kind FROM ledger ORDER BY rowid")
|
|
235
|
-
.all() as Array<{ error: string | null; error_kind: string | null }>;
|
|
238
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
239
|
+
await ledger.record(entry({ error: "request aborted" }));
|
|
240
|
+
await ledger.record(entry({ error: "auth: nope" }));
|
|
241
|
+
await ledger.record(entry({ error: "moderation: Request blocked: prompt injection" }));
|
|
242
|
+
await ledger.record(entry({ error: "model_unavailable: guardrail" }));
|
|
243
|
+
await ledger.record(entry({ error: "upstream_error: boom" }));
|
|
244
|
+
await ledger.record(entry({ error: null }));
|
|
245
|
+
|
|
246
|
+
const rows = (await db.query<unknown>("SELECT error, error_kind FROM ledger ORDER BY created_at_ms")) as Array<{ error: string | null; error_kind: string | null }>;
|
|
236
247
|
expect(rows.map((r) => r.error_kind)).toEqual([
|
|
237
248
|
"aborted",
|
|
238
249
|
"auth",
|
|
@@ -242,43 +253,49 @@ describe("v4 migration", () => {
|
|
|
242
253
|
null,
|
|
243
254
|
]);
|
|
244
255
|
} finally {
|
|
245
|
-
db.close();
|
|
256
|
+
await db.close();
|
|
246
257
|
}
|
|
247
258
|
});
|
|
248
259
|
|
|
249
|
-
test("a deferred upgrade tier survives a save/load round trip", () => {
|
|
250
|
-
const db =
|
|
260
|
+
test("a deferred upgrade tier survives a save/load round trip", async () => {
|
|
261
|
+
const db = openSqlDb(join(tmpdir(), `trust-${process.pid}-${Date.now()}.db`));
|
|
262
|
+
await migrateStore(db);
|
|
251
263
|
const store = createConversationStore(db);
|
|
252
|
-
const st = store.load("conv-defer");
|
|
264
|
+
const st = await store.load("conv-defer");
|
|
253
265
|
st.upgradeDeferredTier = "hard";
|
|
254
|
-
store.save(st);
|
|
255
|
-
expect(store.load("conv-defer").upgradeDeferredTier).toBe("hard");
|
|
266
|
+
await store.save(st);
|
|
267
|
+
expect((await store.load("conv-defer")).upgradeDeferredTier).toBe("hard");
|
|
256
268
|
st.upgradeDeferredTier = null;
|
|
257
|
-
store.save(st);
|
|
258
|
-
expect(store.load("conv-defer").upgradeDeferredTier).toBeNull();
|
|
259
|
-
db.close();
|
|
269
|
+
await store.save(st);
|
|
270
|
+
expect((await store.load("conv-defer")).upgradeDeferredTier).toBeNull();
|
|
271
|
+
await db.close();
|
|
260
272
|
});
|
|
261
273
|
|
|
262
274
|
test("schema is at user_version 19", () => {
|
|
263
|
-
|
|
275
|
+
// `PRAGMA user_version` is SQLite's own migration marker, so this reads
|
|
276
|
+
// through the bootstrap handle rather than the engine-agnostic one.
|
|
277
|
+
const path = join(tmpdir(), `uv-${process.pid}-${Date.now()}.db`);
|
|
278
|
+
const db = openDb(path);
|
|
264
279
|
try {
|
|
265
|
-
|
|
266
|
-
|
|
280
|
+
// Our own pragma against our own file; the shape is fixed by SQLite.
|
|
281
|
+
const row = db.query("PRAGMA user_version").get() as { user_version: number } | null;
|
|
282
|
+
expect(row?.user_version).toBe(19);
|
|
267
283
|
} finally {
|
|
268
284
|
db.close();
|
|
269
285
|
}
|
|
270
286
|
});
|
|
271
287
|
|
|
272
|
-
test("persists omp_session_id and returns it via recentEntries", () => {
|
|
273
|
-
const db =
|
|
288
|
+
test("persists omp_session_id and returns it via recentEntries", async () => {
|
|
289
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
290
|
+
await migrateStore(db);
|
|
274
291
|
try {
|
|
275
|
-
const ledger =
|
|
276
|
-
ledger.record(entry({ ompSessionId: "sess-a" }));
|
|
277
|
-
ledger.record(entry({ ompSessionId: "" }));
|
|
278
|
-
const got = ledger.recentEntries(10).map((e) => e.ompSessionId).sort();
|
|
292
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
293
|
+
await ledger.record(entry({ ompSessionId: "sess-a" }));
|
|
294
|
+
await ledger.record(entry({ ompSessionId: "" }));
|
|
295
|
+
const got = (await ledger.recentEntries(10)).map((e) => e.ompSessionId).sort();
|
|
279
296
|
expect(got).toEqual(["", "sess-a"]);
|
|
280
297
|
} finally {
|
|
281
|
-
db.close();
|
|
298
|
+
await db.close();
|
|
282
299
|
}
|
|
283
300
|
});
|
|
284
301
|
});
|
|
@@ -291,11 +308,12 @@ describe("v6 classifier instrumentation", () => {
|
|
|
291
308
|
complexityKeywords: ["race", "debug"],
|
|
292
309
|
};
|
|
293
310
|
|
|
294
|
-
test("round-trips the feature vector and classifier outputs", () => {
|
|
295
|
-
const db =
|
|
311
|
+
test("round-trips the feature vector and classifier outputs", async () => {
|
|
312
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
313
|
+
await migrateStore(db);
|
|
296
314
|
try {
|
|
297
|
-
const ledger =
|
|
298
|
-
ledger.record(
|
|
315
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
316
|
+
await ledger.record(
|
|
299
317
|
entry({
|
|
300
318
|
features: FEATURES,
|
|
301
319
|
score: 0.42,
|
|
@@ -305,67 +323,70 @@ describe("v6 classifier instrumentation", () => {
|
|
|
305
323
|
}),
|
|
306
324
|
);
|
|
307
325
|
|
|
308
|
-
const got = ledger.recentEntries(1)[0];
|
|
326
|
+
const got = (await ledger.recentEntries(1))[0];
|
|
309
327
|
expect(got?.features).toEqual(FEATURES);
|
|
310
328
|
expect(got?.score).toBe(0.42);
|
|
311
329
|
expect(got?.confidence).toBe(0.75);
|
|
312
330
|
expect(got?.task).toBe("coding");
|
|
313
331
|
expect(got?.classifierReasons).toEqual(["-0.28 tool-result continuation"]);
|
|
314
332
|
} finally {
|
|
315
|
-
db.close();
|
|
333
|
+
await db.close();
|
|
316
334
|
}
|
|
317
335
|
});
|
|
318
336
|
|
|
319
|
-
test("an uninstrumented row reads back as null, not as invented data", () => {
|
|
320
|
-
const db =
|
|
337
|
+
test("an uninstrumented row reads back as null, not as invented data", async () => {
|
|
338
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
339
|
+
await migrateStore(db);
|
|
321
340
|
try {
|
|
322
|
-
const ledger =
|
|
323
|
-
ledger.record(entry({}));
|
|
324
|
-
const got = ledger.recentEntries(1)[0];
|
|
341
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
342
|
+
await ledger.record(entry({}));
|
|
343
|
+
const got = (await ledger.recentEntries(1))[0];
|
|
325
344
|
expect(got?.features).toBeNull();
|
|
326
345
|
expect(got?.score).toBeNull();
|
|
327
346
|
expect(got?.confidence).toBeNull();
|
|
328
347
|
expect(got?.task).toBeNull();
|
|
329
348
|
expect(got?.classifierReasons).toBeNull();
|
|
330
349
|
} finally {
|
|
331
|
-
db.close();
|
|
350
|
+
await db.close();
|
|
332
351
|
}
|
|
333
352
|
});
|
|
334
353
|
|
|
335
|
-
test("records which tier exploration dropped from, and NULL otherwise", () => {
|
|
336
|
-
const db =
|
|
354
|
+
test("records which tier exploration dropped from, and NULL otherwise", async () => {
|
|
355
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
356
|
+
await migrateStore(db);
|
|
337
357
|
try {
|
|
338
|
-
const ledger =
|
|
339
|
-
ledger.record(entry({ tier: "simple", exploredFrom: "moderate" }));
|
|
340
|
-
ledger.record(entry({ tier: "moderate" }));
|
|
358
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
359
|
+
await ledger.record(entry({ tier: "simple", exploredFrom: "moderate" }));
|
|
360
|
+
await ledger.record(entry({ tier: "moderate" }));
|
|
341
361
|
|
|
342
|
-
const got = ledger.recentEntries(10);
|
|
362
|
+
const got = await ledger.recentEntries(10);
|
|
343
363
|
expect(got.map((e) => e.exploredFrom).sort()).toEqual(["moderate", null] as unknown as string[]);
|
|
344
364
|
|
|
345
365
|
// The counterfactual query this whole column exists to make possible:
|
|
346
366
|
// of the turns we deliberately under-routed, how many had to escalate?
|
|
347
|
-
const counted = db
|
|
348
|
-
|
|
349
|
-
.get() as { n: number };
|
|
350
|
-
expect(counted.n).toBe(1);
|
|
367
|
+
const counted = await db.one<{ n: unknown }>("SELECT COUNT(*) n FROM ledger WHERE explored_from IS NOT NULL");
|
|
368
|
+
expect(num(counted?.n)).toBe(1);
|
|
351
369
|
} finally {
|
|
352
|
-
db.close();
|
|
370
|
+
await db.close();
|
|
353
371
|
}
|
|
354
372
|
});
|
|
355
|
-
test("features land in the column as queryable JSON", () => {
|
|
356
|
-
const db =
|
|
373
|
+
test("features land in the column as queryable JSON", async () => {
|
|
374
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
375
|
+
await migrateStore(db);
|
|
357
376
|
try {
|
|
358
|
-
const ledger =
|
|
359
|
-
ledger.record(entry({ features: FEATURES, score: 0.9, confidence: 0.1, task: "vision" }));
|
|
377
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
378
|
+
await ledger.record(entry({ features: FEATURES, score: 0.9, confidence: 0.1, task: "vision" }));
|
|
360
379
|
// SQLite json_extract proves the blob is real JSON, not a stringified object.
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
expect(row
|
|
380
|
+
// The blob is real JSON, not a stringified object: read a member back
|
|
381
|
+
// through whichever accessor the engine uses.
|
|
382
|
+
const row = await db.one<{ depth: unknown; score: unknown; task: string }>(
|
|
383
|
+
`SELECT ${db.jsonNum("features", "toolLoopDepth")} AS depth, score, task FROM ledger`,
|
|
384
|
+
);
|
|
385
|
+
expect(num(row?.depth)).toBe(3);
|
|
386
|
+
expect(num(row?.score)).toBe(0.9);
|
|
387
|
+
expect(row?.task).toBe("vision");
|
|
367
388
|
} finally {
|
|
368
|
-
db.close();
|
|
389
|
+
await db.close();
|
|
369
390
|
}
|
|
370
391
|
});
|
|
371
392
|
});
|
|
@@ -373,17 +394,18 @@ describe("v6 classifier instrumentation", () => {
|
|
|
373
394
|
describe("cache reliability signal", () => {
|
|
374
395
|
// Observed hit rate when a warm cache was expected: the previous kept turn
|
|
375
396
|
// of the conversation was on the same model within the warm TTL.
|
|
376
|
-
function seed(rows: Array<Partial<LedgerEntry>>) {
|
|
377
|
-
const db =
|
|
378
|
-
|
|
379
|
-
|
|
397
|
+
async function seed(rows: Array<Partial<LedgerEntry>>) {
|
|
398
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
399
|
+
await migrateStore(db);
|
|
400
|
+
const ledger = createSqlLedger(db, cfg, { findModel: () => null });
|
|
401
|
+
for (const r of rows) await ledger.record(entry(r));
|
|
380
402
|
return { db, ledger };
|
|
381
403
|
}
|
|
382
404
|
const t0 = Date.UTC(2026, 8, 7, 12);
|
|
383
405
|
const usage = (prompt: number, cached: number, estimated = false) => ({ ...EMPTY_USAGE, promptTokens: prompt, cachedTokens: cached, ...(estimated ? { cachedEstimated: true } : {}) });
|
|
384
406
|
|
|
385
|
-
test("a model that hits when warm scores 1; one that misses scores 0; the first turn never counts", () => {
|
|
386
|
-
const { db, ledger } = seed([
|
|
407
|
+
test("a model that hits when warm scores 1; one that misses scores 0; the first turn never counts", async () => {
|
|
408
|
+
const { db, ledger } = await seed([
|
|
387
409
|
{ conversationKey: "a", slug: "good/m", servedSlug: "good/m", createdAtMs: t0, usage: usage(50_000, 0) }, // first turn: no expectation
|
|
388
410
|
{ conversationKey: "a", slug: "good/m", servedSlug: "good/m", createdAtMs: t0 + 60_000, usage: usage(60_000, 50_000) },
|
|
389
411
|
{ conversationKey: "a", slug: "good/m", servedSlug: "good/m", createdAtMs: t0 + 120_000, usage: usage(70_000, 60_000) },
|
|
@@ -391,112 +413,120 @@ describe("cache reliability signal", () => {
|
|
|
391
413
|
{ conversationKey: "b", slug: "flaky/m", servedSlug: "flaky/m", createdAtMs: t0 + 60_000, usage: usage(60_000, 0) },
|
|
392
414
|
{ conversationKey: "b", slug: "flaky/m", servedSlug: "flaky/m", createdAtMs: t0 + 120_000, usage: usage(70_000, 30_000) },
|
|
393
415
|
]);
|
|
394
|
-
expect(ledger.cacheReliability
|
|
395
|
-
const flaky = ledger.cacheReliability
|
|
416
|
+
expect((await ledger.cacheReliability(["good/m"])).get("good/m")).toEqual({ slug: "good/m", samples: 2, hitRate: 1 });
|
|
417
|
+
const flaky = (await ledger.cacheReliability(["flaky/m"])).get("flaky/m");
|
|
396
418
|
expect(flaky?.samples).toBe(2);
|
|
397
419
|
expect(flaky?.hitRate).toBeCloseTo(0.25, 6); // (0 + 30k/60k) / 2
|
|
398
|
-
expect(ledger.cacheReliability
|
|
399
|
-
db.close();
|
|
420
|
+
expect((await ledger.cacheReliability(["never/m"])).has("never/m")).toBe(false);
|
|
421
|
+
await db.close();
|
|
400
422
|
});
|
|
401
423
|
|
|
402
|
-
test("a switch, an idle gap past the TTL, or a router-estimated count is not a warm-expected sample", () => {
|
|
403
|
-
const { db, ledger } = seed([
|
|
424
|
+
test("a switch, an idle gap past the TTL, or a router-estimated count is not a warm-expected sample", async () => {
|
|
425
|
+
const { db, ledger } = await seed([
|
|
404
426
|
{ conversationKey: "a", slug: "x/m", servedSlug: "x/m", createdAtMs: t0, usage: usage(50_000, 0) },
|
|
405
427
|
{ conversationKey: "a", slug: "y/m", servedSlug: "y/m", createdAtMs: t0 + 60_000, usage: usage(60_000, 0) }, // switch
|
|
406
428
|
{ conversationKey: "a", slug: "y/m", servedSlug: "y/m", createdAtMs: t0 + 60_000 + cfg.hysteresis.cacheWarmTtlMs + 1, usage: usage(70_000, 0) }, // gap
|
|
407
429
|
{ conversationKey: "a", slug: "y/m", servedSlug: "y/m", createdAtMs: t0 + 60_000 + cfg.hysteresis.cacheWarmTtlMs + 2, usage: usage(80_000, 70_000, true) }, // estimated
|
|
408
430
|
]);
|
|
409
|
-
expect(ledger.cacheReliability
|
|
410
|
-
expect(ledger.cacheReliability
|
|
411
|
-
db.close();
|
|
431
|
+
expect((await ledger.cacheReliability(["x/m"])).has("x/m")).toBe(false);
|
|
432
|
+
expect((await ledger.cacheReliability(["y/m"])).has("y/m")).toBe(false);
|
|
433
|
+
await db.close();
|
|
412
434
|
});
|
|
413
435
|
});
|
|
414
436
|
|
|
415
437
|
describe("feedback in trust", () => {
|
|
416
438
|
// A user verdict counts as filters.feedbackWeight attempts of that outcome.
|
|
417
|
-
function trustWith(weight: number, verdicts: Array<"good" | "bad">): { rate: number; good: number; bad: number } {
|
|
418
|
-
const db =
|
|
439
|
+
async function trustWith(weight: number, verdicts: Array<"good" | "bad">): Promise<{ rate: number; good: number; bad: number }> {
|
|
440
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
441
|
+
await migrateStore(db);
|
|
419
442
|
try {
|
|
420
443
|
const c = structuredClone(cfg);
|
|
421
444
|
c.filters.feedbackWeight = weight;
|
|
422
|
-
const ledger =
|
|
445
|
+
const ledger = createSqlLedger(db, c, { findModel: () => null });
|
|
423
446
|
const fb = createFeedbackStore(db);
|
|
424
447
|
let last = "";
|
|
425
448
|
for (let i = 0; i < 10; i++) {
|
|
426
449
|
const e = entry({ error: null });
|
|
427
450
|
last = e.id;
|
|
428
|
-
ledger.record(e);
|
|
451
|
+
await ledger.record(e);
|
|
452
|
+
}
|
|
453
|
+
for (const v of verdicts) {
|
|
454
|
+
await fb.record({ ledgerId: last, ompSessionId: "s", slug: "vendor/model", tier: "simple", verdict: v, note: "" });
|
|
429
455
|
}
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
return { rate:
|
|
456
|
+
const trust = await ledger.trust("vendor/model");
|
|
457
|
+
expect(trust).not.toBeNull();
|
|
458
|
+
return { rate: trust?.successRate ?? 0, good: trust?.feedbackGood ?? -1, bad: trust?.feedbackBad ?? -1 };
|
|
433
459
|
} finally {
|
|
434
|
-
db.close();
|
|
460
|
+
await db.close();
|
|
435
461
|
}
|
|
436
462
|
}
|
|
437
463
|
|
|
438
|
-
test("weight 0 records verdicts without moving the rate", () => {
|
|
439
|
-
const base = trustWith(0, []);
|
|
464
|
+
test("weight 0 records verdicts without moving the rate", async () => {
|
|
465
|
+
const base = await trustWith(0, []);
|
|
440
466
|
expect(base.rate).toBeCloseTo(11 / 12, 6); // (10 - 0 + 1) / (10 + 2)
|
|
441
|
-
expect(trustWith(0, ["bad", "bad"]).rate).toBeCloseTo(base.rate, 6);
|
|
467
|
+
expect((await trustWith(0, ["bad", "bad"])).rate).toBeCloseTo(base.rate, 6);
|
|
442
468
|
});
|
|
443
469
|
|
|
444
|
-
test("a bad verdict counts as `weight` failures, a good one as `weight` successes", () => {
|
|
470
|
+
test("a bad verdict counts as `weight` failures, a good one as `weight` successes", async () => {
|
|
445
471
|
// 10 clean attempts + one bad verdict at weight 3: attempts 13, failures 3.
|
|
446
|
-
const bad = trustWith(3, ["bad"]);
|
|
472
|
+
const bad = await trustWith(3, ["bad"]);
|
|
447
473
|
expect(bad.rate).toBeCloseTo((13 - 3 + 1) / (13 + 2), 6);
|
|
448
474
|
expect(bad.bad).toBe(1);
|
|
449
|
-
const good = trustWith(3, ["good"]);
|
|
475
|
+
const good = await trustWith(3, ["good"]);
|
|
450
476
|
expect(good.rate).toBeCloseTo((13 - 0 + 1) / (13 + 2), 6);
|
|
451
477
|
expect(good.good).toBe(1);
|
|
452
478
|
// allTrust and signals agree with trust().
|
|
453
|
-
const db =
|
|
479
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
480
|
+
await migrateStore(db);
|
|
454
481
|
const c = structuredClone(cfg);
|
|
455
482
|
c.filters.feedbackWeight = 3;
|
|
456
|
-
const ledger =
|
|
483
|
+
const ledger = createSqlLedger(db, c, { findModel: () => null });
|
|
457
484
|
const fb = createFeedbackStore(db);
|
|
458
485
|
const e = entry({ error: null });
|
|
459
|
-
ledger.record(e);
|
|
460
|
-
fb.record({ ledgerId: e.id, ompSessionId: "s", slug: "vendor/model", tier: "simple", verdict: "bad", note: "" });
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
486
|
+
await ledger.record(e);
|
|
487
|
+
await fb.record({ ledgerId: e.id, ompSessionId: "s", slug: "vendor/model", tier: "simple", verdict: "bad", note: "" });
|
|
488
|
+
// One reference rate: allTrust, trust and signals must agree on it.
|
|
489
|
+
const rate = (await ledger.trust("vendor/model"))?.successRate ?? 0;
|
|
490
|
+
expect((await ledger.allTrust())[0]?.successRate).toBeCloseTo(rate, 9);
|
|
491
|
+
expect((await ledger.signals(["vendor/model"])).get("vendor/model")?.trust?.successRate).toBeCloseTo(rate, 9);
|
|
492
|
+
await db.close();
|
|
464
493
|
});
|
|
465
494
|
});
|
|
466
495
|
|
|
467
496
|
describe("task-scoped feedback (filters.feedbackByTask)", () => {
|
|
468
|
-
test("a verdict counts only for its task type; an untasked verdict counts everywhere", () => {
|
|
469
|
-
const db =
|
|
497
|
+
test("a verdict counts only for its task type; an untasked verdict counts everywhere", async () => {
|
|
498
|
+
const db = openSqlDb(join(tmpdir(), `t-trust-attribution.test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2)}.db`));
|
|
499
|
+
await migrateStore(db);
|
|
470
500
|
try {
|
|
471
501
|
const c = structuredClone(cfg);
|
|
472
502
|
c.filters.feedbackWeight = 3;
|
|
473
503
|
c.filters.feedbackByTask = true;
|
|
474
|
-
const ledger =
|
|
504
|
+
const ledger = createSqlLedger(db, c, { findModel: () => null });
|
|
475
505
|
const fb = createFeedbackStore(db);
|
|
476
|
-
for (let i = 0; i < 10; i++) ledger.record(entry({ error: null, task: "coding" }));
|
|
506
|
+
for (let i = 0; i < 10; i++) await ledger.record(entry({ error: null, task: "coding" }));
|
|
477
507
|
const prose = entry({ error: null, task: "documentation" });
|
|
478
|
-
ledger.record(prose);
|
|
508
|
+
await ledger.record(prose);
|
|
479
509
|
const untasked = entry({ error: null, task: null });
|
|
480
|
-
ledger.record(untasked);
|
|
481
|
-
fb.record({ ledgerId: prose.id, ompSessionId: "s", slug: "vendor/model", tier: "simple", verdict: "bad", note: "" });
|
|
510
|
+
await ledger.record(untasked);
|
|
511
|
+
await fb.record({ ledgerId: prose.id, ompSessionId: "s", slug: "vendor/model", tier: "simple", verdict: "bad", note: "" });
|
|
482
512
|
// 12 clean attempts, weight 3, one bad verdict on a documentation turn.
|
|
483
513
|
const pooled = (12 - 0 + 1) / (12 + 2);
|
|
484
514
|
const withBad = (15 - 3 + 1) / (15 + 2);
|
|
485
|
-
expect(ledger.trust("vendor/model", undefined, "coding")
|
|
486
|
-
expect(ledger.trust("vendor/model", undefined, "documentation")
|
|
515
|
+
expect((await ledger.trust("vendor/model", undefined, "coding"))?.successRate).toBeCloseTo(pooled, 9);
|
|
516
|
+
expect((await ledger.trust("vendor/model", undefined, "documentation"))?.successRate).toBeCloseTo(withBad, 9);
|
|
487
517
|
// No task given (allTrust, reports): pooled behaviour, the verdict counts.
|
|
488
|
-
expect(ledger.trust("vendor/model")
|
|
489
|
-
expect(ledger.allTrust()[0]
|
|
518
|
+
expect((await ledger.trust("vendor/model"))?.successRate).toBeCloseTo(withBad, 9);
|
|
519
|
+
expect((await ledger.allTrust())[0]?.successRate).toBeCloseTo(withBad, 9);
|
|
490
520
|
// signals() honours the task the same way.
|
|
491
|
-
expect(ledger.signals
|
|
521
|
+
expect((await ledger.signals(["vendor/model"], undefined, "coding")).get("vendor/model")!.trust!.successRate).toBeCloseTo(pooled, 9);
|
|
492
522
|
// A verdict on a turn that recorded no task counts for every task.
|
|
493
|
-
fb.record({ ledgerId: untasked.id, ompSessionId: "s", slug: "vendor/model", tier: "simple", verdict: "bad", note: "" });
|
|
494
|
-
expect(ledger.trust("vendor/model", undefined, "coding")
|
|
523
|
+
await fb.record({ ledgerId: untasked.id, ompSessionId: "s", slug: "vendor/model", tier: "simple", verdict: "bad", note: "" });
|
|
524
|
+
expect((await ledger.trust("vendor/model", undefined, "coding"))?.successRate).toBeCloseTo(withBad, 9);
|
|
495
525
|
// Off: task is ignored and every verdict pools.
|
|
496
526
|
c.filters.feedbackByTask = false;
|
|
497
|
-
expect(ledger.trust("vendor/model", undefined, "coding")
|
|
527
|
+
expect((await ledger.trust("vendor/model", undefined, "coding"))?.successRate).toBeCloseTo((18 - 6 + 1) / (18 + 2), 9);
|
|
498
528
|
} finally {
|
|
499
|
-
db.close();
|
|
529
|
+
await db.close();
|
|
500
530
|
}
|
|
501
531
|
});
|
|
502
532
|
});
|