auto-model-router 0.20.0 → 0.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,9 +5,12 @@ import { createProviders } from "./providers.ts";
5
5
  import { createBridgeFromConfig } from "../context/index.ts";
6
6
  import { createFeedbackStore, type Verdict } from "../cost/feedback.ts";
7
7
  import { createLedger } from "../cost/ledger.ts";
8
+ import { createRetentionRunner } from "../cost/retention.ts";
9
+ import { redactionRulesFor } from "../config/redaction.ts";
8
10
  import { createSessionOverrides } from "./overrides.ts";
9
11
  import { catalogView } from "./catalog-view.ts";
10
12
  import { buildUpstreamModels } from "../catalog/static-catalog.ts";
13
+ import { invalidateFeedCache } from "../catalog/benchmark-feeds.ts";
11
14
  import { applyRequestPolicy, resolveProfile } from "../router/index.ts";
12
15
  import { parsePolicyHeader } from "../wire/openai/request.ts";
13
16
  import { createDigester } from "./digest.ts";
@@ -265,6 +268,18 @@ export function startServer(cfg: RouterConfig): StartedServer {
265
268
  },
266
269
  );
267
270
 
271
+ // Compile the redaction rules before the listener exists: a rule that does
272
+ // not load is a hole in the guard, and an operator who configured redaction
273
+ // must not get a router that started and forwarded anyway. Programmatic
274
+ // overrides (an embedder's) never pass through the config schema, so this is
275
+ // the only place their rules are checked.
276
+ const redactionRules = redactionRulesFor(cfg.redaction);
277
+ if (redactionRules.length > 0) {
278
+ // Names only. The patterns describe the secrets and the matches are the
279
+ // secrets; neither belongs in a log line.
280
+ log.info("redaction enabled", { rules: redactionRules.map((r) => r.name).join(","), scanTools: cfg.redaction.scanTools });
281
+ }
282
+
268
283
  if (context.enabled) {
269
284
  log.info("agentdox context bridge enabled", {
270
285
  url: cfg.context.baseUrl,
@@ -296,8 +311,24 @@ export function startServer(cfg: RouterConfig): StartedServer {
296
311
  log.warn("initial catalog fetch failed", { error: err instanceof Error ? err.message : String(err) });
297
312
  });
298
313
 
299
- // One housekeeping timer for both tables. `unref`'d so it never holds the
300
- // process open.
314
+ // Ledger retention. The runner owns the once-an-hour floor, so the minute
315
+ // timer below, the boot run and `POST /v1/router/prune` cannot between them
316
+ // run a whole-ledger delete more often than that. It reads the window live,
317
+ // so a hot reload that lowers it applies on the next tick.
318
+ const retention = createRetentionRunner({ ledger, retentionDays: () => cfg.ledger.retentionDays });
319
+ const retain = (): void => {
320
+ try {
321
+ const result = retention.maybeRun();
322
+ if (result !== null && result.deleted > 0) {
323
+ log.info("pruned ledger rows past retention", { deleted: result.deleted, retentionDays: cfg.ledger.retentionDays });
324
+ }
325
+ } catch (err) {
326
+ log.warn("ledger retention prune failed", { error: err instanceof Error ? err.message : String(err) });
327
+ }
328
+ };
329
+
330
+ // One housekeeping timer for all three tables. `unref`'d so it never holds
331
+ // the process open.
301
332
  const pruneTimer = setInterval(() => {
302
333
  try {
303
334
  const dropped = conversations.prune(cfg.ledger.conversationTtlMs);
@@ -315,21 +346,12 @@ export function startServer(cfg: RouterConfig): StartedServer {
315
346
  } catch (err) {
316
347
  log.warn("context block prune failed", { error: err instanceof Error ? err.message : String(err) });
317
348
  }
349
+ retain();
318
350
  }, 60_000);
319
351
  pruneTimer.unref();
320
352
 
321
- // Ledger retention: hourly, and once at boot so a lowered setting takes
322
- // effect without waiting. Reads the live config, so it hot-reloads.
323
- const retain = (): void => {
324
- try {
325
- const dropped = ledger.prune?.(cfg.ledger.retentionDays) ?? 0;
326
- if (dropped > 0) log.info("pruned ledger rows past retention", { dropped, retentionDays: cfg.ledger.retentionDays });
327
- } catch (err) {
328
- log.warn("ledger retention prune failed", { error: err instanceof Error ? err.message : String(err) });
329
- }
330
- };
331
- const retentionTimer = setInterval(retain, 3_600_000);
332
- retentionTimer.unref();
353
+ // Once shortly after boot, so a lowered window takes effect without waiting
354
+ // out an hour; the runner's own floor governs everything after that.
333
355
  setTimeout(retain, 5_000).unref();
334
356
 
335
357
  // Periodically refetch the (key-scoped) catalog in the background so
@@ -665,6 +687,15 @@ export function startServer(cfg: RouterConfig): StartedServer {
665
687
  }),
666
688
  );
667
689
  }
690
+ if (req.method === "POST" && url.pathname === "/v1/router/prune") {
691
+ // A front door of the team edition holds a READ-ONLY handle on the
692
+ // ledger file by design, so this route is the only way it can act
693
+ // on its own retention policy — and the once-an-hour floor is the
694
+ // runner's, not the caller's, so calling it in a loop is harmless.
695
+ const result = retention.runNow();
696
+ if (result.deleted > 0) log.info("pruned ledger rows past retention", { deleted: result.deleted, retentionDays: cfg.ledger.retentionDays });
697
+ return json({ ...result, retentionDays: cfg.ledger.retentionDays });
698
+ }
668
699
  if (req.method === "POST" && url.pathname === "/v1/router/feedback") {
669
700
  // A user verdict on the newest routed turn of an omp session.
670
701
  const body = (await req.json().catch(() => null)) as Record<string, unknown> | null;
@@ -769,15 +800,25 @@ export function startServer(cfg: RouterConfig): StartedServer {
769
800
  defaultScope: cfg.context.defaultScope === "" ? "(per-request header only)" : cfg.context.defaultScope,
770
801
  });
771
802
  }
772
- if (!touched(changed, "openrouter", "ollama")) return false;
803
+ const benchmarksChanged = touched(changed, "benchmarks");
804
+ if (!touched(changed, "openrouter", "ollama") && !benchmarksChanged) return false;
805
+ if (benchmarksChanged) {
806
+ // The feed cache keeps its own ~daily TTL, so a refresh alone would rebuild
807
+ // the catalog from yesterday's feeds and the new Artificial Analysis key
808
+ // would do nothing until it expired. Age the row out so the refresh below
809
+ // re-fetches; the payload stays put, so a fetch that fails leaves the
810
+ // scores already serving in place. Only a benchmarks change does this —
811
+ // every other reconfigure keeps the cadence the cache is there for.
812
+ invalidateFeedCache(db);
813
+ }
773
814
  // A key change makes the catalog key-scoped (or not), and enabling Ollama adds
774
815
  // its models: the snapshot is rebuilt before the next turn ranks. Started, not
775
816
  // awaited — the caller is a settings save, not a network client, and the
776
817
  // previous snapshot serves turns until the new one lands.
777
818
  void catalog
778
819
  .refresh()
779
- .then((snap) => log.info("catalog refreshed after an upstream change", { models: snap.models.length, ollama: cfg.ollama.enabled }))
780
- .catch((err: unknown) => log.warn("catalog refresh after an upstream change failed; the previous snapshot stands", { error: err instanceof Error ? err.message : String(err) }));
820
+ .then((snap) => log.info("catalog refreshed after a live config change", { models: snap.models.length, ollama: cfg.ollama.enabled, benchmarks: benchmarksChanged }))
821
+ .catch((err: unknown) => log.warn("catalog refresh after a live config change failed; the previous snapshot stands", { error: err instanceof Error ? err.message : String(err) }));
781
822
  return true;
782
823
  }
783
824
 
@@ -0,0 +1,104 @@
1
+ /**
2
+ * Redaction: strings an operator forbids from leaving the process.
3
+ *
4
+ * An operator with a compliance obligation needs two things a router can
5
+ * actually give: certainty that certain shapes of text never reach a provider,
6
+ * and evidence that the guard ran. This module is the first half; the ledger's
7
+ * `redactions` count is the second. Neither ever records WHAT was matched — a
8
+ * redaction log that quotes the secret is just a second copy of the secret.
9
+ *
10
+ * `runTurn` applies this to the rendered upstream body, which is the
11
+ * chat-completions shape every front end normalises to and every provider
12
+ * client renders from, so the guard sits between the router and ALL of them:
13
+ * a new upstream cannot bypass it by construction.
14
+ */
15
+
16
+ import type { CompiledRedactionRule } from "../config/redaction.ts";
17
+
18
+ /** Applies every rule to one string. Returns the text and how many matches were replaced. */
19
+ export function redactText(text: string, rules: readonly CompiledRedactionRule[]): { text: string; count: number } {
20
+ let out = text;
21
+ let count = 0;
22
+ for (const rule of rules) {
23
+ // `replace` with a global regex resets and advances `lastIndex` itself,
24
+ // including over a zero-width position — which compilation refuses anyway.
25
+ out = out.replace(rule.regex, () => {
26
+ count++;
27
+ return rule.replacement;
28
+ });
29
+ }
30
+ return { text: out, count };
31
+ }
32
+
33
+ type Rec = Record<string, unknown>;
34
+ const isRec = (v: unknown): v is Rec => v !== null && typeof v === "object" && !Array.isArray(v);
35
+
36
+ /** Redacts a message's `content`, whether it is a string or an array of parts. Returns matches replaced. */
37
+ function redactContent(message: Rec, rules: readonly CompiledRedactionRule[]): number {
38
+ const content = message.content;
39
+ if (typeof content === "string") {
40
+ const r = redactText(content, rules);
41
+ if (r.count > 0) message.content = r.text;
42
+ return r.count;
43
+ }
44
+ if (!Array.isArray(content)) return 0;
45
+ let count = 0;
46
+ for (const part of content) {
47
+ // Text parts only: an `image_url` part carries a data URI, which no
48
+ // redaction rule can meaningfully read and every rule would be slow over.
49
+ if (!isRec(part) || part.type !== "text" || typeof part.text !== "string") continue;
50
+ const r = redactText(part.text, rules);
51
+ if (r.count > 0) part.text = r.text;
52
+ count += r.count;
53
+ }
54
+ return count;
55
+ }
56
+
57
+ /**
58
+ * Redacts the rendered upstream body in place and returns how many matches
59
+ * were replaced.
60
+ *
61
+ * The body is the chat-completions shape: every front end (chat completions,
62
+ * Responses, Anthropic Messages) translates into it before parsing, and every
63
+ * upstream client renders its own protocol FROM it, so this one pass covers
64
+ * every wire in and every provider out.
65
+ *
66
+ * What is scanned:
67
+ * - the text content of every message that is not a tool result — the system
68
+ * prompt (including the injected agentdox block), the user's words, the
69
+ * assistant's replay;
70
+ * - with `scanTools`, tool-call arguments and tool-result content as well.
71
+ * Off by default because tool results are where the bytes are: a turn's
72
+ * prompt is mostly file content, so scanning them is most of the cost — and
73
+ * also, for an operator who cares about a secret in a file the agent read,
74
+ * most of the point.
75
+ *
76
+ * Tool NAMES, ids and the tool schemas are left alone: they are the harness's
77
+ * own vocabulary, not conversation content, and rewriting one breaks the
78
+ * call/result pairing the model needs.
79
+ */
80
+ export function redactUpstreamBody(
81
+ body: Rec,
82
+ rules: readonly CompiledRedactionRule[],
83
+ opts: { scanTools: boolean },
84
+ ): number {
85
+ if (rules.length === 0) return 0;
86
+ const messages = body.messages;
87
+ if (!Array.isArray(messages)) return 0;
88
+ let count = 0;
89
+ for (const message of messages) {
90
+ if (!isRec(message)) continue;
91
+ const isToolResult = message.role === "tool";
92
+ if (!isToolResult || opts.scanTools) count += redactContent(message, rules);
93
+ if (!opts.scanTools || !Array.isArray(message.tool_calls)) continue;
94
+ for (const call of message.tool_calls) {
95
+ if (!isRec(call) || !isRec(call.function)) continue;
96
+ const args = call.function.arguments;
97
+ if (typeof args !== "string") continue;
98
+ const r = redactText(args, rules);
99
+ if (r.count > 0) call.function.arguments = r.text;
100
+ count += r.count;
101
+ }
102
+ }
103
+ return count;
104
+ }
@@ -29,6 +29,8 @@ import {
29
29
  import { UpstreamError, type Dispatch, type UpstreamClient } from "../upstream/types.ts";
30
30
  import type { SessionOverrides } from "./overrides.ts";
31
31
  import { digestCompactionEdits, type CompactionDigester } from "./compaction-digest.ts";
32
+ import { redactUpstreamBody } from "./redact.ts";
33
+ import { redactionRulesFor } from "../config/redaction.ts";
32
34
  import { createLogger } from "../util/log.ts";
33
35
  import type { NormRequest, ResponseSink, TurnSummary, UpstreamChunk } from "../wire/types.ts";
34
36
 
@@ -165,6 +167,10 @@ export async function runTurn(
165
167
  let pendingDecision: Decision | null = null;
166
168
  // Digests made this turn, by edit; a retry re-plans and must not pay twice.
167
169
  const digestMemo = new Map<string, string>();
170
+ // Compiled once per turn (and memoised across turns on the rules' own text),
171
+ // never per attempt: a retry renders the same request again and would
172
+ // otherwise recompile every pattern.
173
+ const redactionRules = redactionRulesFor(config.redaction);
168
174
 
169
175
  for (let attempt = 0; attempt < maxAttempts; attempt++) {
170
176
  // Client disconnected before anything was dispatched: spend nothing.
@@ -261,6 +267,18 @@ export async function runTurn(
261
267
  ...(decision.compactionPlan.length > 0 ? { compactionPlan: decision.compactionPlan } : {}),
262
268
  });
263
269
 
270
+ // Redaction, immediately after the body is rendered and before ANYTHING
271
+ // can dispatch it. This is the only choke point that covers every path:
272
+ // all three front ends normalise to this shape, and every upstream client
273
+ // renders its own protocol from it, so a provider added later is covered
274
+ // without being told. The count is recorded on the ledger row below; the
275
+ // matched text is never logged, at any level.
276
+ let redactions = 0;
277
+ if (redactionRules.length > 0) {
278
+ redactions = redactUpstreamBody(body, redactionRules, { scanTools: config.redaction.scanTools });
279
+ if (redactions > 0) log.debug("redacted outgoing request", { matches: redactions, rules: redactionRules.length });
280
+ }
281
+
264
282
  // Calibrate the token estimate against the bytes that actually go out —
265
283
  // after compaction shrank the prompt and the context block was appended
266
284
  // — not the raw request the estimate was taken from.
@@ -333,6 +351,9 @@ export async function runTurn(
333
351
  upstreamGenerationId: generationId,
334
352
  error: fields.error,
335
353
  promptTokensSaved: decision.promptTokensSaved,
354
+ // Evidence that the guard ran, as a count and nothing more. Absent
355
+ // while redaction is off, so an untouched row stays NULL.
356
+ ...(redactionRules.length === 0 ? {} : { redactions }),
336
357
  // The ledger prices OpenRouter slugs from its own cached payload; any
337
358
  // other provider's model exists only in the live catalog.
338
359
  ...(priceModel === undefined ? {} : { priceModel }),
@@ -18,7 +18,7 @@ import { mkdirSync } from "node:fs";
18
18
  import { dirname } from "node:path";
19
19
 
20
20
  /** Bump when a migration is added; guarded below so reopening never regresses it. */
21
- const USER_VERSION = 18;
21
+ const USER_VERSION = 19;
22
22
 
23
23
  const MIGRATIONS = `
24
24
  CREATE TABLE IF NOT EXISTS catalog_cache (
@@ -296,6 +296,16 @@ const MIGRATE_V18 = `
296
296
  ALTER TABLE ledger ADD COLUMN scope TEXT;
297
297
  `;
298
298
 
299
+ // v19: ledger records how many strings redaction removed from the turn's
300
+ // outgoing request (see `redaction` in the config). A COUNT and nothing else:
301
+ // the whole point of redaction is that the matched text does not exist outside
302
+ // the client, so the evidence that it ran must not reintroduce it. NULL on
303
+ // every row written before this and on every turn with redaction off; 0 means
304
+ // the rules ran and matched nothing.
305
+ const MIGRATE_V19 = `
306
+ ALTER TABLE ledger ADD COLUMN redactions INTEGER;
307
+ `;
308
+
299
309
  // v9: benchmark_cache holds the external benchmark feeds (Artificial Analysis,
300
310
  // BenchLM) that backfill quality scores OpenRouter leaves unpublished. It is a
301
311
  // whole new table, created idempotently by the MIGRATIONS block above, so there
@@ -320,6 +330,17 @@ export function openDb(path: string): Database {
320
330
  // ":memory:" has no parent directory to create.
321
331
  if (path !== ":memory:") mkdirSync(dirname(path), { recursive: true });
322
332
  const db = new Database(path);
333
+ // Incremental auto-vacuum so retention can hand freed pages back to the
334
+ // filesystem (`PRAGMA incremental_vacuum` after a prune). SQLite only honours
335
+ // a change of vacuum mode on an empty database or across a full VACUUM, so
336
+ // this takes effect for NEW ledgers; an existing one keeps its mode and
337
+ // simply reuses freed pages instead of releasing them. Must precede the
338
+ // journal-mode change, and is best-effort: never fail an open over it.
339
+ try {
340
+ db.exec("PRAGMA auto_vacuum = INCREMENTAL");
341
+ } catch {
342
+ /* an existing database in another mode; freed pages are reused instead */
343
+ }
323
344
  // WAL + NORMAL: single-writer local service; favours read latency on the turn hot path.
324
345
  db.exec("PRAGMA journal_mode = WAL");
325
346
  db.exec("PRAGMA synchronous = NORMAL");
@@ -338,6 +359,7 @@ export function openDb(path: string): Database {
338
359
  if (!ledgerCols.some((c) => c.name === "hold_arm")) db.exec(MIGRATE_V8);
339
360
  if (!ledgerCols.some((c) => c.name === "prompt_tokens_saved")) db.exec(MIGRATE_V12);
340
361
  if (!ledgerCols.some((c) => c.name === "scope")) db.exec(MIGRATE_V18);
362
+ if (!ledgerCols.some((c) => c.name === "redactions")) db.exec(MIGRATE_V19);
341
363
  const convCols = db.query("PRAGMA table_info(conversations)").all() as { name: string }[];
342
364
  if (!convCols.some((c) => c.name === "context_version")) db.exec(MIGRATE_V11);
343
365
  if (!convCols.some((c) => c.name === "compaction_plan")) db.exec(MIGRATE_V13);
@@ -4,6 +4,7 @@ import { normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
4
4
  import {
5
5
  applyFeedScores,
6
6
  fetchBenchlmScores,
7
+ invalidateFeedCache,
7
8
  normalizeModelKey,
8
9
  parseAaModels,
9
10
  parseBenchlmModels,
@@ -198,6 +199,40 @@ describe("refreshFeedScores", () => {
198
199
  db.close();
199
200
  });
200
201
 
202
+ test("invalidateFeedCache re-fetches inside the TTL, and a failed forced fetch keeps the scores", async () => {
203
+ const db = openDb(":memory:");
204
+ let calls = 0;
205
+ let coding = 58;
206
+ const fakeFetch: FetchLike = async () => {
207
+ calls += 1;
208
+ return Response.json({
209
+ models: coding === 0 ? [] : [{ model: "MiniMax M3", creator: "MiniMax", evidenceStatus: "supported", categoryScores: { coding } }],
210
+ });
211
+ };
212
+
213
+ const cfg = cfgWith({ enabled: true, artificialAnalysisApiKey: "", benchlm: true, refreshMs: 1_000_000 });
214
+ await refreshFeedScores(cfg, db, { fetchImpl: fakeFetch, now: 1000 });
215
+ await refreshFeedScores(cfg, db, { fetchImpl: fakeFetch, now: 2000 });
216
+ expect(calls).toBe(1); // deep inside the TTL
217
+
218
+ // An Artificial Analysis key arriving cannot wait out the day still left on
219
+ // the cache; invalidating is what makes the next refresh actually fetch.
220
+ invalidateFeedCache(db);
221
+ coding = 71;
222
+ const forced = await refreshFeedScores(cfg, db, { fetchImpl: fakeFetch, now: 3000 });
223
+ expect(calls).toBe(2);
224
+ expect(forced[0]).toMatchObject({ key: "minimax-m3", coding: 71 });
225
+
226
+ // And the forced fetch is still best-effort: a feed that answers with nothing
227
+ // leaves the scores already serving in place rather than emptying them.
228
+ invalidateFeedCache(db);
229
+ coding = 0;
230
+ const after = await refreshFeedScores(cfg, db, { fetchImpl: fakeFetch, now: 4000 });
231
+ expect(calls).toBe(3);
232
+ expect(after[0]).toMatchObject({ key: "minimax-m3", coding: 71 });
233
+ db.close();
234
+ });
235
+
201
236
  test("falls back to the stale cache when a refresh returns nothing", async () => {
202
237
  const db = openDb(":memory:");
203
238
  const seed = [{ key: "minimax-m3", creator: "minimax", coding: 58, source: "benchlm" }];
@@ -142,7 +142,7 @@ describe("validateField", () => {
142
142
 
143
143
  describe("WIZARD_SECTIONS coverage", () => {
144
144
  /** Leaves that are edited as whole records/arrays rather than fields. */
145
- const RECORD_PATHS = new Set(["ollama.prices", "ollama.twins", "digest.toolAliases", "profiles", "upstreams"]);
145
+ const RECORD_PATHS = new Set(["ollama.prices", "ollama.twins", "digest.toolAliases", "profiles", "upstreams", "redaction.rules"]);
146
146
 
147
147
  function leaves(obj: unknown, prefix = ""): string[] {
148
148
  if (typeof obj !== "object" || obj === null || Array.isArray(obj)) return [prefix];
@@ -83,6 +83,7 @@ function mkConfig(escalation: Partial<EscalationConfig> = {}): RouterConfig {
83
83
  digest: { enabled: false, minBytes: 12_000, maxBytes: 400_000, tools: ["read"], fromTier: "moderate", tier: "simple", model: "", maxOutputTokens: 700, maxCostUsd: 0.02, timeoutMs: 25_000, toolAliases: {} },
84
84
  profiles: [],
85
85
  ledger: { path: ":memory:", blendWindowDays: 7, blendMinSamples: 20, fallbackBlend: { inputPerMtok: 1, outputPerMtok: 4 }, conversationTtlMs: 86_400_000 , retentionDays: 0,},
86
+ redaction: { enabled: false, rules: [], scanTools: false },
86
87
  adaptiveTierFloors: true,
87
88
  adaptivePriceCeilings: false,
88
89
  logLevel: "silent",
@@ -24,7 +24,7 @@ import { openDb } from "../src/util/sqlite.ts";
24
24
 
25
25
  const FIXTURES = join(import.meta.dir, "fixtures", "migrations");
26
26
  const files = readdirSync(FIXTURES).filter((f) => /^router-v\d+\.db$/.test(f)).sort((a, b) => Number(/\d+/.exec(a)![0]) - Number(/\d+/.exec(b)![0]));
27
- const CURRENT_VERSION = 18;
27
+ const CURRENT_VERSION = 19;
28
28
 
29
29
  describe("schema migrations from every shipped version", () => {
30
30
  test("fixtures exist for the versions that shipped", () => {
@@ -44,7 +44,7 @@ describe("schema migrations from every shipped version", () => {
44
44
  expect((db.query("PRAGMA user_version").get() as { user_version: number }).user_version).toBe(CURRENT_VERSION);
45
45
  // Every column the current code writes exists after migration.
46
46
  const ledgerCols = new Set((db.query("PRAGMA table_info(ledger)").all() as { name: string }[]).map((c) => c.name));
47
- for (const c of ["harness_id", "error_kind", "omp_session_id", "features", "explored_from", "hold_arm", "prompt_tokens_saved", "scope"]) expect(ledgerCols.has(c)).toBe(true);
47
+ for (const c of ["harness_id", "error_kind", "omp_session_id", "features", "explored_from", "hold_arm", "prompt_tokens_saved", "scope", "redactions"]) expect(ledgerCols.has(c)).toBe(true);
48
48
  const convCols = new Set((db.query("PRAGMA table_info(conversations)").all() as { name: string }[]).map((c) => c.name));
49
49
  for (const c of ["context_version", "compaction_plan", "compaction_plan_tokens", "upgrade_deferred_tier"]) expect(convCols.has(c)).toBe(true);
50
50
  // The fixture's ledger row survived the ALTERs with its values.
@@ -67,7 +67,10 @@ describe("schema migrations from every shipped version", () => {
67
67
  expect(exportRows(db, 0, null).map((r) => r.scope)).toEqual([""]);
68
68
  expect(spendUsdSince(db, 0, null, "acme.api")).toBe(0);
69
69
  expect(spendUsdSince(db, 0, null)).toBeGreaterThanOrEqual(0);
70
- expect(ledger.prune?.(0)).toBe(0);
70
+ expect(ledger.prune?.(0)?.deleted).toBe(0);
71
+ // v19: the fixture's row predates `redactions`, so nothing claims a
72
+ // redaction happened on it and the report totals it as zero.
73
+ expect(buildUsageReport(db, { windowDays: 3650 }).totals.redactions).toBe(0);
71
74
  } finally {
72
75
  db.close();
73
76
  try {
@@ -1,8 +1,12 @@
1
+ import { mkdtempSync, rmSync } from "node:fs";
2
+ import { tmpdir } from "node:os";
3
+ import { join } from "node:path";
1
4
  import { describe, expect, test } from "bun:test";
2
5
  import { applyConfigPatch, assignInPlace, touched } from "../src/config/apply.ts";
3
6
  import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
4
7
  import type { RouterConfig } from "../src/config/types.ts";
5
8
  import { startServer } from "../src/server/http.ts";
9
+ import { openDb } from "../src/util/sqlite.ts";
6
10
 
7
11
  /**
8
12
  * Live reconfiguration: a running router follows a config change without a
@@ -101,6 +105,55 @@ describe("a running server reconfigures", () => {
101
105
  }
102
106
  });
103
107
 
108
+ test("a benchmarks change ages the feed cache out; an unrelated change leaves it alone", async () => {
109
+ const dir = mkdtempSync(join(tmpdir(), "amr-benchfeed-"));
110
+ const cfg: RouterConfig = {
111
+ ...base(),
112
+ ledger: { ...DEFAULT_CONFIG.ledger, path: join(dir, "ledger.db") },
113
+ // Off so the background catalog refresh cannot re-fetch the feeds under the
114
+ // assertion. Invalidation does not depend on it: the point is that the row
115
+ // is aged out before the refresh runs, whatever the refresh then does.
116
+ benchmarks: { ...DEFAULT_CONFIG.benchmarks, enabled: false },
117
+ };
118
+ const started = startServer(cfg);
119
+ const db = openDb(cfg.ledger.path);
120
+ const fetchedAt = (): number | null => {
121
+ const row = db.query("SELECT fetched_at_ms FROM benchmark_cache WHERE id = 1").get() as { fetched_at_ms: number } | null;
122
+ return row === null ? null : row.fetched_at_ms;
123
+ };
124
+ try {
125
+ const seeded = Date.now();
126
+ db.query("INSERT INTO benchmark_cache (id, payload, fetched_at_ms) VALUES (1, ?, ?)").run("[]", seeded);
127
+
128
+ // The ~daily feed cadence is not every settings save's to reset.
129
+ const unrelated = await started.reconfigure({ filters: { latencyWeight: 0.42 } });
130
+ expect(unrelated.catalogRefreshing).toBe(false);
131
+ expect(fetchedAt()).toBe(seeded);
132
+
133
+ // A key an operator just pasted has to reach the catalog now, not tomorrow.
134
+ const r = await started.reconfigure({ benchmarks: { artificialAnalysisApiKey: "aa-key" } });
135
+ expect(r.rejected).toEqual([]);
136
+ expect(r.changed).toEqual(["benchmarks.artificialAnalysisApiKey"]);
137
+ expect(r.catalogRefreshing).toBe(true);
138
+ expect(fetchedAt()).toBe(0);
139
+ // The payload survives the invalidation, so a failed re-fetch has something
140
+ // to fall back on.
141
+ const row = db.query("SELECT payload FROM benchmark_cache WHERE id = 1").get() as { payload: string } | null;
142
+ expect(row?.payload).toBe("[]");
143
+
144
+ // Clearing it again is a benchmarks change too: scores from a key that is
145
+ // gone must stop being used just as promptly.
146
+ db.query("UPDATE benchmark_cache SET fetched_at_ms = ? WHERE id = 1").run(seeded);
147
+ const cleared = await started.reconfigure({ benchmarks: { artificialAnalysisApiKey: "" } });
148
+ expect(cleared.catalogRefreshing).toBe(true);
149
+ expect(fetchedAt()).toBe(0);
150
+ } finally {
151
+ db.close();
152
+ await started.stop();
153
+ rmSync(dir, { recursive: true, force: true });
154
+ }
155
+ });
156
+
104
157
  test("the socket and the ledger file are refused rather than half-applied", async () => {
105
158
  const cfg = base();
106
159
  const started = startServer(cfg);