auto-model-router 0.20.0 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +126 -3
- package/docs/data-governance.md +210 -0
- package/package.json +1 -1
- package/src/catalog/benchmark-feeds.ts +19 -1
- package/src/cli/config-wizard.ts +11 -1
- package/src/config/defaults.ts +7 -1
- package/src/config/hot-reload.ts +3 -2
- package/src/config/redaction.ts +265 -0
- package/src/config/schema.ts +25 -1
- package/src/config/types.ts +52 -5
- package/src/cost/ledger.ts +39 -5
- package/src/cost/report.ts +19 -0
- package/src/cost/retention.ts +58 -0
- package/src/cost/types.ts +26 -2
- package/src/lib.ts +8 -2
- package/src/server/http.ts +58 -17
- package/src/server/redact.ts +104 -0
- package/src/server/turn.ts +21 -0
- package/src/util/sqlite.ts +23 -1
- package/test/benchmark-feeds.test.ts +35 -0
- package/test/config-wizard.test.ts +1 -1
- package/test/failover.test.ts +1 -0
- package/test/migrations.test.ts +6 -3
- package/test/reconfigure.test.ts +53 -0
- package/test/redaction.test.ts +300 -0
- package/test/report-hub.test.ts +2 -0
- package/test/report.test.ts +2 -0
- package/test/retention.test.ts +242 -0
- package/test/tokens.test.ts +3 -3
- package/test/trust-attribution.test.ts +2 -2
- package/test/turn.test.ts +107 -0
package/src/server/http.ts
CHANGED
|
@@ -5,9 +5,12 @@ import { createProviders } from "./providers.ts";
|
|
|
5
5
|
import { createBridgeFromConfig } from "../context/index.ts";
|
|
6
6
|
import { createFeedbackStore, type Verdict } from "../cost/feedback.ts";
|
|
7
7
|
import { createLedger } from "../cost/ledger.ts";
|
|
8
|
+
import { createRetentionRunner } from "../cost/retention.ts";
|
|
9
|
+
import { redactionRulesFor } from "../config/redaction.ts";
|
|
8
10
|
import { createSessionOverrides } from "./overrides.ts";
|
|
9
11
|
import { catalogView } from "./catalog-view.ts";
|
|
10
12
|
import { buildUpstreamModels } from "../catalog/static-catalog.ts";
|
|
13
|
+
import { invalidateFeedCache } from "../catalog/benchmark-feeds.ts";
|
|
11
14
|
import { applyRequestPolicy, resolveProfile } from "../router/index.ts";
|
|
12
15
|
import { parsePolicyHeader } from "../wire/openai/request.ts";
|
|
13
16
|
import { createDigester } from "./digest.ts";
|
|
@@ -265,6 +268,18 @@ export function startServer(cfg: RouterConfig): StartedServer {
|
|
|
265
268
|
},
|
|
266
269
|
);
|
|
267
270
|
|
|
271
|
+
// Compile the redaction rules before the listener exists: a rule that does
|
|
272
|
+
// not load is a hole in the guard, and an operator who configured redaction
|
|
273
|
+
// must not get a router that started and forwarded anyway. Programmatic
|
|
274
|
+
// overrides (an embedder's) never pass through the config schema, so this is
|
|
275
|
+
// the only place their rules are checked.
|
|
276
|
+
const redactionRules = redactionRulesFor(cfg.redaction);
|
|
277
|
+
if (redactionRules.length > 0) {
|
|
278
|
+
// Names only. The patterns describe the secrets and the matches are the
|
|
279
|
+
// secrets; neither belongs in a log line.
|
|
280
|
+
log.info("redaction enabled", { rules: redactionRules.map((r) => r.name).join(","), scanTools: cfg.redaction.scanTools });
|
|
281
|
+
}
|
|
282
|
+
|
|
268
283
|
if (context.enabled) {
|
|
269
284
|
log.info("agentdox context bridge enabled", {
|
|
270
285
|
url: cfg.context.baseUrl,
|
|
@@ -296,8 +311,24 @@ export function startServer(cfg: RouterConfig): StartedServer {
|
|
|
296
311
|
log.warn("initial catalog fetch failed", { error: err instanceof Error ? err.message : String(err) });
|
|
297
312
|
});
|
|
298
313
|
|
|
299
|
-
//
|
|
300
|
-
//
|
|
314
|
+
// Ledger retention. The runner owns the once-an-hour floor, so the minute
|
|
315
|
+
// timer below, the boot run and `POST /v1/router/prune` cannot between them
|
|
316
|
+
// run a whole-ledger delete more often than that. It reads the window live,
|
|
317
|
+
// so a hot reload that lowers it applies on the next tick.
|
|
318
|
+
const retention = createRetentionRunner({ ledger, retentionDays: () => cfg.ledger.retentionDays });
|
|
319
|
+
const retain = (): void => {
|
|
320
|
+
try {
|
|
321
|
+
const result = retention.maybeRun();
|
|
322
|
+
if (result !== null && result.deleted > 0) {
|
|
323
|
+
log.info("pruned ledger rows past retention", { deleted: result.deleted, retentionDays: cfg.ledger.retentionDays });
|
|
324
|
+
}
|
|
325
|
+
} catch (err) {
|
|
326
|
+
log.warn("ledger retention prune failed", { error: err instanceof Error ? err.message : String(err) });
|
|
327
|
+
}
|
|
328
|
+
};
|
|
329
|
+
|
|
330
|
+
// One housekeeping timer for all three tables. `unref`'d so it never holds
|
|
331
|
+
// the process open.
|
|
301
332
|
const pruneTimer = setInterval(() => {
|
|
302
333
|
try {
|
|
303
334
|
const dropped = conversations.prune(cfg.ledger.conversationTtlMs);
|
|
@@ -315,21 +346,12 @@ export function startServer(cfg: RouterConfig): StartedServer {
|
|
|
315
346
|
} catch (err) {
|
|
316
347
|
log.warn("context block prune failed", { error: err instanceof Error ? err.message : String(err) });
|
|
317
348
|
}
|
|
349
|
+
retain();
|
|
318
350
|
}, 60_000);
|
|
319
351
|
pruneTimer.unref();
|
|
320
352
|
|
|
321
|
-
//
|
|
322
|
-
//
|
|
323
|
-
const retain = (): void => {
|
|
324
|
-
try {
|
|
325
|
-
const dropped = ledger.prune?.(cfg.ledger.retentionDays) ?? 0;
|
|
326
|
-
if (dropped > 0) log.info("pruned ledger rows past retention", { dropped, retentionDays: cfg.ledger.retentionDays });
|
|
327
|
-
} catch (err) {
|
|
328
|
-
log.warn("ledger retention prune failed", { error: err instanceof Error ? err.message : String(err) });
|
|
329
|
-
}
|
|
330
|
-
};
|
|
331
|
-
const retentionTimer = setInterval(retain, 3_600_000);
|
|
332
|
-
retentionTimer.unref();
|
|
353
|
+
// Once shortly after boot, so a lowered window takes effect without waiting
|
|
354
|
+
// out an hour; the runner's own floor governs everything after that.
|
|
333
355
|
setTimeout(retain, 5_000).unref();
|
|
334
356
|
|
|
335
357
|
// Periodically refetch the (key-scoped) catalog in the background so
|
|
@@ -665,6 +687,15 @@ export function startServer(cfg: RouterConfig): StartedServer {
|
|
|
665
687
|
}),
|
|
666
688
|
);
|
|
667
689
|
}
|
|
690
|
+
if (req.method === "POST" && url.pathname === "/v1/router/prune") {
|
|
691
|
+
// A front door of the team edition holds a READ-ONLY handle on the
|
|
692
|
+
// ledger file by design, so this route is the only way it can act
|
|
693
|
+
// on its own retention policy — and the once-an-hour floor is the
|
|
694
|
+
// runner's, not the caller's, so calling it in a loop is harmless.
|
|
695
|
+
const result = retention.runNow();
|
|
696
|
+
if (result.deleted > 0) log.info("pruned ledger rows past retention", { deleted: result.deleted, retentionDays: cfg.ledger.retentionDays });
|
|
697
|
+
return json({ ...result, retentionDays: cfg.ledger.retentionDays });
|
|
698
|
+
}
|
|
668
699
|
if (req.method === "POST" && url.pathname === "/v1/router/feedback") {
|
|
669
700
|
// A user verdict on the newest routed turn of an omp session.
|
|
670
701
|
const body = (await req.json().catch(() => null)) as Record<string, unknown> | null;
|
|
@@ -769,15 +800,25 @@ export function startServer(cfg: RouterConfig): StartedServer {
|
|
|
769
800
|
defaultScope: cfg.context.defaultScope === "" ? "(per-request header only)" : cfg.context.defaultScope,
|
|
770
801
|
});
|
|
771
802
|
}
|
|
772
|
-
|
|
803
|
+
const benchmarksChanged = touched(changed, "benchmarks");
|
|
804
|
+
if (!touched(changed, "openrouter", "ollama") && !benchmarksChanged) return false;
|
|
805
|
+
if (benchmarksChanged) {
|
|
806
|
+
// The feed cache keeps its own ~daily TTL, so a refresh alone would rebuild
|
|
807
|
+
// the catalog from yesterday's feeds and the new Artificial Analysis key
|
|
808
|
+
// would do nothing until it expired. Age the row out so the refresh below
|
|
809
|
+
// re-fetches; the payload stays put, so a fetch that fails leaves the
|
|
810
|
+
// scores already serving in place. Only a benchmarks change does this —
|
|
811
|
+
// every other reconfigure keeps the cadence the cache is there for.
|
|
812
|
+
invalidateFeedCache(db);
|
|
813
|
+
}
|
|
773
814
|
// A key change makes the catalog key-scoped (or not), and enabling Ollama adds
|
|
774
815
|
// its models: the snapshot is rebuilt before the next turn ranks. Started, not
|
|
775
816
|
// awaited — the caller is a settings save, not a network client, and the
|
|
776
817
|
// previous snapshot serves turns until the new one lands.
|
|
777
818
|
void catalog
|
|
778
819
|
.refresh()
|
|
779
|
-
.then((snap) => log.info("catalog refreshed after
|
|
780
|
-
.catch((err: unknown) => log.warn("catalog refresh after
|
|
820
|
+
.then((snap) => log.info("catalog refreshed after a live config change", { models: snap.models.length, ollama: cfg.ollama.enabled, benchmarks: benchmarksChanged }))
|
|
821
|
+
.catch((err: unknown) => log.warn("catalog refresh after a live config change failed; the previous snapshot stands", { error: err instanceof Error ? err.message : String(err) }));
|
|
781
822
|
return true;
|
|
782
823
|
}
|
|
783
824
|
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Redaction: strings an operator forbids from leaving the process.
|
|
3
|
+
*
|
|
4
|
+
* An operator with a compliance obligation needs two things a router can
|
|
5
|
+
* actually give: certainty that certain shapes of text never reach a provider,
|
|
6
|
+
* and evidence that the guard ran. This module is the first half; the ledger's
|
|
7
|
+
* `redactions` count is the second. Neither ever records WHAT was matched — a
|
|
8
|
+
* redaction log that quotes the secret is just a second copy of the secret.
|
|
9
|
+
*
|
|
10
|
+
* `runTurn` applies this to the rendered upstream body, which is the
|
|
11
|
+
* chat-completions shape every front end normalises to and every provider
|
|
12
|
+
* client renders from, so the guard sits between the router and ALL of them:
|
|
13
|
+
* a new upstream cannot bypass it by construction.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import type { CompiledRedactionRule } from "../config/redaction.ts";
|
|
17
|
+
|
|
18
|
+
/** Applies every rule to one string. Returns the text and how many matches were replaced. */
|
|
19
|
+
export function redactText(text: string, rules: readonly CompiledRedactionRule[]): { text: string; count: number } {
|
|
20
|
+
let out = text;
|
|
21
|
+
let count = 0;
|
|
22
|
+
for (const rule of rules) {
|
|
23
|
+
// `replace` with a global regex resets and advances `lastIndex` itself,
|
|
24
|
+
// including over a zero-width position — which compilation refuses anyway.
|
|
25
|
+
out = out.replace(rule.regex, () => {
|
|
26
|
+
count++;
|
|
27
|
+
return rule.replacement;
|
|
28
|
+
});
|
|
29
|
+
}
|
|
30
|
+
return { text: out, count };
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
type Rec = Record<string, unknown>;
|
|
34
|
+
const isRec = (v: unknown): v is Rec => v !== null && typeof v === "object" && !Array.isArray(v);
|
|
35
|
+
|
|
36
|
+
/** Redacts a message's `content`, whether it is a string or an array of parts. Returns matches replaced. */
|
|
37
|
+
function redactContent(message: Rec, rules: readonly CompiledRedactionRule[]): number {
|
|
38
|
+
const content = message.content;
|
|
39
|
+
if (typeof content === "string") {
|
|
40
|
+
const r = redactText(content, rules);
|
|
41
|
+
if (r.count > 0) message.content = r.text;
|
|
42
|
+
return r.count;
|
|
43
|
+
}
|
|
44
|
+
if (!Array.isArray(content)) return 0;
|
|
45
|
+
let count = 0;
|
|
46
|
+
for (const part of content) {
|
|
47
|
+
// Text parts only: an `image_url` part carries a data URI, which no
|
|
48
|
+
// redaction rule can meaningfully read and every rule would be slow over.
|
|
49
|
+
if (!isRec(part) || part.type !== "text" || typeof part.text !== "string") continue;
|
|
50
|
+
const r = redactText(part.text, rules);
|
|
51
|
+
if (r.count > 0) part.text = r.text;
|
|
52
|
+
count += r.count;
|
|
53
|
+
}
|
|
54
|
+
return count;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Redacts the rendered upstream body in place and returns how many matches
|
|
59
|
+
* were replaced.
|
|
60
|
+
*
|
|
61
|
+
* The body is the chat-completions shape: every front end (chat completions,
|
|
62
|
+
* Responses, Anthropic Messages) translates into it before parsing, and every
|
|
63
|
+
* upstream client renders its own protocol FROM it, so this one pass covers
|
|
64
|
+
* every wire in and every provider out.
|
|
65
|
+
*
|
|
66
|
+
* What is scanned:
|
|
67
|
+
* - the text content of every message that is not a tool result — the system
|
|
68
|
+
* prompt (including the injected agentdox block), the user's words, the
|
|
69
|
+
* assistant's replay;
|
|
70
|
+
* - with `scanTools`, tool-call arguments and tool-result content as well.
|
|
71
|
+
* Off by default because tool results are where the bytes are: a turn's
|
|
72
|
+
* prompt is mostly file content, so scanning them is most of the cost — and
|
|
73
|
+
* also, for an operator who cares about a secret in a file the agent read,
|
|
74
|
+
* most of the point.
|
|
75
|
+
*
|
|
76
|
+
* Tool NAMES, ids and the tool schemas are left alone: they are the harness's
|
|
77
|
+
* own vocabulary, not conversation content, and rewriting one breaks the
|
|
78
|
+
* call/result pairing the model needs.
|
|
79
|
+
*/
|
|
80
|
+
export function redactUpstreamBody(
|
|
81
|
+
body: Rec,
|
|
82
|
+
rules: readonly CompiledRedactionRule[],
|
|
83
|
+
opts: { scanTools: boolean },
|
|
84
|
+
): number {
|
|
85
|
+
if (rules.length === 0) return 0;
|
|
86
|
+
const messages = body.messages;
|
|
87
|
+
if (!Array.isArray(messages)) return 0;
|
|
88
|
+
let count = 0;
|
|
89
|
+
for (const message of messages) {
|
|
90
|
+
if (!isRec(message)) continue;
|
|
91
|
+
const isToolResult = message.role === "tool";
|
|
92
|
+
if (!isToolResult || opts.scanTools) count += redactContent(message, rules);
|
|
93
|
+
if (!opts.scanTools || !Array.isArray(message.tool_calls)) continue;
|
|
94
|
+
for (const call of message.tool_calls) {
|
|
95
|
+
if (!isRec(call) || !isRec(call.function)) continue;
|
|
96
|
+
const args = call.function.arguments;
|
|
97
|
+
if (typeof args !== "string") continue;
|
|
98
|
+
const r = redactText(args, rules);
|
|
99
|
+
if (r.count > 0) call.function.arguments = r.text;
|
|
100
|
+
count += r.count;
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
return count;
|
|
104
|
+
}
|
package/src/server/turn.ts
CHANGED
|
@@ -29,6 +29,8 @@ import {
|
|
|
29
29
|
import { UpstreamError, type Dispatch, type UpstreamClient } from "../upstream/types.ts";
|
|
30
30
|
import type { SessionOverrides } from "./overrides.ts";
|
|
31
31
|
import { digestCompactionEdits, type CompactionDigester } from "./compaction-digest.ts";
|
|
32
|
+
import { redactUpstreamBody } from "./redact.ts";
|
|
33
|
+
import { redactionRulesFor } from "../config/redaction.ts";
|
|
32
34
|
import { createLogger } from "../util/log.ts";
|
|
33
35
|
import type { NormRequest, ResponseSink, TurnSummary, UpstreamChunk } from "../wire/types.ts";
|
|
34
36
|
|
|
@@ -165,6 +167,10 @@ export async function runTurn(
|
|
|
165
167
|
let pendingDecision: Decision | null = null;
|
|
166
168
|
// Digests made this turn, by edit; a retry re-plans and must not pay twice.
|
|
167
169
|
const digestMemo = new Map<string, string>();
|
|
170
|
+
// Compiled once per turn (and memoised across turns on the rules' own text),
|
|
171
|
+
// never per attempt: a retry renders the same request again and would
|
|
172
|
+
// otherwise recompile every pattern.
|
|
173
|
+
const redactionRules = redactionRulesFor(config.redaction);
|
|
168
174
|
|
|
169
175
|
for (let attempt = 0; attempt < maxAttempts; attempt++) {
|
|
170
176
|
// Client disconnected before anything was dispatched: spend nothing.
|
|
@@ -261,6 +267,18 @@ export async function runTurn(
|
|
|
261
267
|
...(decision.compactionPlan.length > 0 ? { compactionPlan: decision.compactionPlan } : {}),
|
|
262
268
|
});
|
|
263
269
|
|
|
270
|
+
// Redaction, immediately after the body is rendered and before ANYTHING
|
|
271
|
+
// can dispatch it. This is the only choke point that covers every path:
|
|
272
|
+
// all three front ends normalise to this shape, and every upstream client
|
|
273
|
+
// renders its own protocol from it, so a provider added later is covered
|
|
274
|
+
// without being told. The count is recorded on the ledger row below; the
|
|
275
|
+
// matched text is never logged, at any level.
|
|
276
|
+
let redactions = 0;
|
|
277
|
+
if (redactionRules.length > 0) {
|
|
278
|
+
redactions = redactUpstreamBody(body, redactionRules, { scanTools: config.redaction.scanTools });
|
|
279
|
+
if (redactions > 0) log.debug("redacted outgoing request", { matches: redactions, rules: redactionRules.length });
|
|
280
|
+
}
|
|
281
|
+
|
|
264
282
|
// Calibrate the token estimate against the bytes that actually go out —
|
|
265
283
|
// after compaction shrank the prompt and the context block was appended
|
|
266
284
|
// — not the raw request the estimate was taken from.
|
|
@@ -333,6 +351,9 @@ export async function runTurn(
|
|
|
333
351
|
upstreamGenerationId: generationId,
|
|
334
352
|
error: fields.error,
|
|
335
353
|
promptTokensSaved: decision.promptTokensSaved,
|
|
354
|
+
// Evidence that the guard ran, as a count and nothing more. Absent
|
|
355
|
+
// while redaction is off, so an untouched row stays NULL.
|
|
356
|
+
...(redactionRules.length === 0 ? {} : { redactions }),
|
|
336
357
|
// The ledger prices OpenRouter slugs from its own cached payload; any
|
|
337
358
|
// other provider's model exists only in the live catalog.
|
|
338
359
|
...(priceModel === undefined ? {} : { priceModel }),
|
package/src/util/sqlite.ts
CHANGED
|
@@ -18,7 +18,7 @@ import { mkdirSync } from "node:fs";
|
|
|
18
18
|
import { dirname } from "node:path";
|
|
19
19
|
|
|
20
20
|
/** Bump when a migration is added; guarded below so reopening never regresses it. */
|
|
21
|
-
const USER_VERSION =
|
|
21
|
+
const USER_VERSION = 19;
|
|
22
22
|
|
|
23
23
|
const MIGRATIONS = `
|
|
24
24
|
CREATE TABLE IF NOT EXISTS catalog_cache (
|
|
@@ -296,6 +296,16 @@ const MIGRATE_V18 = `
|
|
|
296
296
|
ALTER TABLE ledger ADD COLUMN scope TEXT;
|
|
297
297
|
`;
|
|
298
298
|
|
|
299
|
+
// v19: ledger records how many strings redaction removed from the turn's
|
|
300
|
+
// outgoing request (see `redaction` in the config). A COUNT and nothing else:
|
|
301
|
+
// the whole point of redaction is that the matched text does not exist outside
|
|
302
|
+
// the client, so the evidence that it ran must not reintroduce it. NULL on
|
|
303
|
+
// every row written before this and on every turn with redaction off; 0 means
|
|
304
|
+
// the rules ran and matched nothing.
|
|
305
|
+
const MIGRATE_V19 = `
|
|
306
|
+
ALTER TABLE ledger ADD COLUMN redactions INTEGER;
|
|
307
|
+
`;
|
|
308
|
+
|
|
299
309
|
// v9: benchmark_cache holds the external benchmark feeds (Artificial Analysis,
|
|
300
310
|
// BenchLM) that backfill quality scores OpenRouter leaves unpublished. It is a
|
|
301
311
|
// whole new table, created idempotently by the MIGRATIONS block above, so there
|
|
@@ -320,6 +330,17 @@ export function openDb(path: string): Database {
|
|
|
320
330
|
// ":memory:" has no parent directory to create.
|
|
321
331
|
if (path !== ":memory:") mkdirSync(dirname(path), { recursive: true });
|
|
322
332
|
const db = new Database(path);
|
|
333
|
+
// Incremental auto-vacuum so retention can hand freed pages back to the
|
|
334
|
+
// filesystem (`PRAGMA incremental_vacuum` after a prune). SQLite only honours
|
|
335
|
+
// a change of vacuum mode on an empty database or across a full VACUUM, so
|
|
336
|
+
// this takes effect for NEW ledgers; an existing one keeps its mode and
|
|
337
|
+
// simply reuses freed pages instead of releasing them. Must precede the
|
|
338
|
+
// journal-mode change, and is best-effort: never fail an open over it.
|
|
339
|
+
try {
|
|
340
|
+
db.exec("PRAGMA auto_vacuum = INCREMENTAL");
|
|
341
|
+
} catch {
|
|
342
|
+
/* an existing database in another mode; freed pages are reused instead */
|
|
343
|
+
}
|
|
323
344
|
// WAL + NORMAL: single-writer local service; favours read latency on the turn hot path.
|
|
324
345
|
db.exec("PRAGMA journal_mode = WAL");
|
|
325
346
|
db.exec("PRAGMA synchronous = NORMAL");
|
|
@@ -338,6 +359,7 @@ export function openDb(path: string): Database {
|
|
|
338
359
|
if (!ledgerCols.some((c) => c.name === "hold_arm")) db.exec(MIGRATE_V8);
|
|
339
360
|
if (!ledgerCols.some((c) => c.name === "prompt_tokens_saved")) db.exec(MIGRATE_V12);
|
|
340
361
|
if (!ledgerCols.some((c) => c.name === "scope")) db.exec(MIGRATE_V18);
|
|
362
|
+
if (!ledgerCols.some((c) => c.name === "redactions")) db.exec(MIGRATE_V19);
|
|
341
363
|
const convCols = db.query("PRAGMA table_info(conversations)").all() as { name: string }[];
|
|
342
364
|
if (!convCols.some((c) => c.name === "context_version")) db.exec(MIGRATE_V11);
|
|
343
365
|
if (!convCols.some((c) => c.name === "compaction_plan")) db.exec(MIGRATE_V13);
|
|
@@ -4,6 +4,7 @@ import { normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
|
|
|
4
4
|
import {
|
|
5
5
|
applyFeedScores,
|
|
6
6
|
fetchBenchlmScores,
|
|
7
|
+
invalidateFeedCache,
|
|
7
8
|
normalizeModelKey,
|
|
8
9
|
parseAaModels,
|
|
9
10
|
parseBenchlmModels,
|
|
@@ -198,6 +199,40 @@ describe("refreshFeedScores", () => {
|
|
|
198
199
|
db.close();
|
|
199
200
|
});
|
|
200
201
|
|
|
202
|
+
test("invalidateFeedCache re-fetches inside the TTL, and a failed forced fetch keeps the scores", async () => {
|
|
203
|
+
const db = openDb(":memory:");
|
|
204
|
+
let calls = 0;
|
|
205
|
+
let coding = 58;
|
|
206
|
+
const fakeFetch: FetchLike = async () => {
|
|
207
|
+
calls += 1;
|
|
208
|
+
return Response.json({
|
|
209
|
+
models: coding === 0 ? [] : [{ model: "MiniMax M3", creator: "MiniMax", evidenceStatus: "supported", categoryScores: { coding } }],
|
|
210
|
+
});
|
|
211
|
+
};
|
|
212
|
+
|
|
213
|
+
const cfg = cfgWith({ enabled: true, artificialAnalysisApiKey: "", benchlm: true, refreshMs: 1_000_000 });
|
|
214
|
+
await refreshFeedScores(cfg, db, { fetchImpl: fakeFetch, now: 1000 });
|
|
215
|
+
await refreshFeedScores(cfg, db, { fetchImpl: fakeFetch, now: 2000 });
|
|
216
|
+
expect(calls).toBe(1); // deep inside the TTL
|
|
217
|
+
|
|
218
|
+
// An Artificial Analysis key arriving cannot wait out the day still left on
|
|
219
|
+
// the cache; invalidating is what makes the next refresh actually fetch.
|
|
220
|
+
invalidateFeedCache(db);
|
|
221
|
+
coding = 71;
|
|
222
|
+
const forced = await refreshFeedScores(cfg, db, { fetchImpl: fakeFetch, now: 3000 });
|
|
223
|
+
expect(calls).toBe(2);
|
|
224
|
+
expect(forced[0]).toMatchObject({ key: "minimax-m3", coding: 71 });
|
|
225
|
+
|
|
226
|
+
// And the forced fetch is still best-effort: a feed that answers with nothing
|
|
227
|
+
// leaves the scores already serving in place rather than emptying them.
|
|
228
|
+
invalidateFeedCache(db);
|
|
229
|
+
coding = 0;
|
|
230
|
+
const after = await refreshFeedScores(cfg, db, { fetchImpl: fakeFetch, now: 4000 });
|
|
231
|
+
expect(calls).toBe(3);
|
|
232
|
+
expect(after[0]).toMatchObject({ key: "minimax-m3", coding: 71 });
|
|
233
|
+
db.close();
|
|
234
|
+
});
|
|
235
|
+
|
|
201
236
|
test("falls back to the stale cache when a refresh returns nothing", async () => {
|
|
202
237
|
const db = openDb(":memory:");
|
|
203
238
|
const seed = [{ key: "minimax-m3", creator: "minimax", coding: 58, source: "benchlm" }];
|
|
@@ -142,7 +142,7 @@ describe("validateField", () => {
|
|
|
142
142
|
|
|
143
143
|
describe("WIZARD_SECTIONS coverage", () => {
|
|
144
144
|
/** Leaves that are edited as whole records/arrays rather than fields. */
|
|
145
|
-
const RECORD_PATHS = new Set(["ollama.prices", "ollama.twins", "digest.toolAliases", "profiles", "upstreams"]);
|
|
145
|
+
const RECORD_PATHS = new Set(["ollama.prices", "ollama.twins", "digest.toolAliases", "profiles", "upstreams", "redaction.rules"]);
|
|
146
146
|
|
|
147
147
|
function leaves(obj: unknown, prefix = ""): string[] {
|
|
148
148
|
if (typeof obj !== "object" || obj === null || Array.isArray(obj)) return [prefix];
|
package/test/failover.test.ts
CHANGED
|
@@ -83,6 +83,7 @@ function mkConfig(escalation: Partial<EscalationConfig> = {}): RouterConfig {
|
|
|
83
83
|
digest: { enabled: false, minBytes: 12_000, maxBytes: 400_000, tools: ["read"], fromTier: "moderate", tier: "simple", model: "", maxOutputTokens: 700, maxCostUsd: 0.02, timeoutMs: 25_000, toolAliases: {} },
|
|
84
84
|
profiles: [],
|
|
85
85
|
ledger: { path: ":memory:", blendWindowDays: 7, blendMinSamples: 20, fallbackBlend: { inputPerMtok: 1, outputPerMtok: 4 }, conversationTtlMs: 86_400_000 , retentionDays: 0,},
|
|
86
|
+
redaction: { enabled: false, rules: [], scanTools: false },
|
|
86
87
|
adaptiveTierFloors: true,
|
|
87
88
|
adaptivePriceCeilings: false,
|
|
88
89
|
logLevel: "silent",
|
package/test/migrations.test.ts
CHANGED
|
@@ -24,7 +24,7 @@ import { openDb } from "../src/util/sqlite.ts";
|
|
|
24
24
|
|
|
25
25
|
const FIXTURES = join(import.meta.dir, "fixtures", "migrations");
|
|
26
26
|
const files = readdirSync(FIXTURES).filter((f) => /^router-v\d+\.db$/.test(f)).sort((a, b) => Number(/\d+/.exec(a)![0]) - Number(/\d+/.exec(b)![0]));
|
|
27
|
-
const CURRENT_VERSION =
|
|
27
|
+
const CURRENT_VERSION = 19;
|
|
28
28
|
|
|
29
29
|
describe("schema migrations from every shipped version", () => {
|
|
30
30
|
test("fixtures exist for the versions that shipped", () => {
|
|
@@ -44,7 +44,7 @@ describe("schema migrations from every shipped version", () => {
|
|
|
44
44
|
expect((db.query("PRAGMA user_version").get() as { user_version: number }).user_version).toBe(CURRENT_VERSION);
|
|
45
45
|
// Every column the current code writes exists after migration.
|
|
46
46
|
const ledgerCols = new Set((db.query("PRAGMA table_info(ledger)").all() as { name: string }[]).map((c) => c.name));
|
|
47
|
-
for (const c of ["harness_id", "error_kind", "omp_session_id", "features", "explored_from", "hold_arm", "prompt_tokens_saved", "scope"]) expect(ledgerCols.has(c)).toBe(true);
|
|
47
|
+
for (const c of ["harness_id", "error_kind", "omp_session_id", "features", "explored_from", "hold_arm", "prompt_tokens_saved", "scope", "redactions"]) expect(ledgerCols.has(c)).toBe(true);
|
|
48
48
|
const convCols = new Set((db.query("PRAGMA table_info(conversations)").all() as { name: string }[]).map((c) => c.name));
|
|
49
49
|
for (const c of ["context_version", "compaction_plan", "compaction_plan_tokens", "upgrade_deferred_tier"]) expect(convCols.has(c)).toBe(true);
|
|
50
50
|
// The fixture's ledger row survived the ALTERs with its values.
|
|
@@ -67,7 +67,10 @@ describe("schema migrations from every shipped version", () => {
|
|
|
67
67
|
expect(exportRows(db, 0, null).map((r) => r.scope)).toEqual([""]);
|
|
68
68
|
expect(spendUsdSince(db, 0, null, "acme.api")).toBe(0);
|
|
69
69
|
expect(spendUsdSince(db, 0, null)).toBeGreaterThanOrEqual(0);
|
|
70
|
-
expect(ledger.prune?.(0)).toBe(0);
|
|
70
|
+
expect(ledger.prune?.(0)?.deleted).toBe(0);
|
|
71
|
+
// v19: the fixture's row predates `redactions`, so nothing claims a
|
|
72
|
+
// redaction happened on it and the report totals it as zero.
|
|
73
|
+
expect(buildUsageReport(db, { windowDays: 3650 }).totals.redactions).toBe(0);
|
|
71
74
|
} finally {
|
|
72
75
|
db.close();
|
|
73
76
|
try {
|
package/test/reconfigure.test.ts
CHANGED
|
@@ -1,8 +1,12 @@
|
|
|
1
|
+
import { mkdtempSync, rmSync } from "node:fs";
|
|
2
|
+
import { tmpdir } from "node:os";
|
|
3
|
+
import { join } from "node:path";
|
|
1
4
|
import { describe, expect, test } from "bun:test";
|
|
2
5
|
import { applyConfigPatch, assignInPlace, touched } from "../src/config/apply.ts";
|
|
3
6
|
import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
|
|
4
7
|
import type { RouterConfig } from "../src/config/types.ts";
|
|
5
8
|
import { startServer } from "../src/server/http.ts";
|
|
9
|
+
import { openDb } from "../src/util/sqlite.ts";
|
|
6
10
|
|
|
7
11
|
/**
|
|
8
12
|
* Live reconfiguration: a running router follows a config change without a
|
|
@@ -101,6 +105,55 @@ describe("a running server reconfigures", () => {
|
|
|
101
105
|
}
|
|
102
106
|
});
|
|
103
107
|
|
|
108
|
+
test("a benchmarks change ages the feed cache out; an unrelated change leaves it alone", async () => {
|
|
109
|
+
const dir = mkdtempSync(join(tmpdir(), "amr-benchfeed-"));
|
|
110
|
+
const cfg: RouterConfig = {
|
|
111
|
+
...base(),
|
|
112
|
+
ledger: { ...DEFAULT_CONFIG.ledger, path: join(dir, "ledger.db") },
|
|
113
|
+
// Off so the background catalog refresh cannot re-fetch the feeds under the
|
|
114
|
+
// assertion. Invalidation does not depend on it: the point is that the row
|
|
115
|
+
// is aged out before the refresh runs, whatever the refresh then does.
|
|
116
|
+
benchmarks: { ...DEFAULT_CONFIG.benchmarks, enabled: false },
|
|
117
|
+
};
|
|
118
|
+
const started = startServer(cfg);
|
|
119
|
+
const db = openDb(cfg.ledger.path);
|
|
120
|
+
const fetchedAt = (): number | null => {
|
|
121
|
+
const row = db.query("SELECT fetched_at_ms FROM benchmark_cache WHERE id = 1").get() as { fetched_at_ms: number } | null;
|
|
122
|
+
return row === null ? null : row.fetched_at_ms;
|
|
123
|
+
};
|
|
124
|
+
try {
|
|
125
|
+
const seeded = Date.now();
|
|
126
|
+
db.query("INSERT INTO benchmark_cache (id, payload, fetched_at_ms) VALUES (1, ?, ?)").run("[]", seeded);
|
|
127
|
+
|
|
128
|
+
// The ~daily feed cadence is not every settings save's to reset.
|
|
129
|
+
const unrelated = await started.reconfigure({ filters: { latencyWeight: 0.42 } });
|
|
130
|
+
expect(unrelated.catalogRefreshing).toBe(false);
|
|
131
|
+
expect(fetchedAt()).toBe(seeded);
|
|
132
|
+
|
|
133
|
+
// A key an operator just pasted has to reach the catalog now, not tomorrow.
|
|
134
|
+
const r = await started.reconfigure({ benchmarks: { artificialAnalysisApiKey: "aa-key" } });
|
|
135
|
+
expect(r.rejected).toEqual([]);
|
|
136
|
+
expect(r.changed).toEqual(["benchmarks.artificialAnalysisApiKey"]);
|
|
137
|
+
expect(r.catalogRefreshing).toBe(true);
|
|
138
|
+
expect(fetchedAt()).toBe(0);
|
|
139
|
+
// The payload survives the invalidation, so a failed re-fetch has something
|
|
140
|
+
// to fall back on.
|
|
141
|
+
const row = db.query("SELECT payload FROM benchmark_cache WHERE id = 1").get() as { payload: string } | null;
|
|
142
|
+
expect(row?.payload).toBe("[]");
|
|
143
|
+
|
|
144
|
+
// Clearing it again is a benchmarks change too: scores from a key that is
|
|
145
|
+
// gone must stop being used just as promptly.
|
|
146
|
+
db.query("UPDATE benchmark_cache SET fetched_at_ms = ? WHERE id = 1").run(seeded);
|
|
147
|
+
const cleared = await started.reconfigure({ benchmarks: { artificialAnalysisApiKey: "" } });
|
|
148
|
+
expect(cleared.catalogRefreshing).toBe(true);
|
|
149
|
+
expect(fetchedAt()).toBe(0);
|
|
150
|
+
} finally {
|
|
151
|
+
db.close();
|
|
152
|
+
await started.stop();
|
|
153
|
+
rmSync(dir, { recursive: true, force: true });
|
|
154
|
+
}
|
|
155
|
+
});
|
|
156
|
+
|
|
104
157
|
test("the socket and the ledger file are refused rather than half-applied", async () => {
|
|
105
158
|
const cfg = base();
|
|
106
159
|
const started = startServer(cfg);
|