auto-model-router 0.31.0 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/.omp-plugin/marketplace.json +2 -2
  2. package/README.md +32 -2
  3. package/omp-extension/router-configure.ts +9 -7
  4. package/package.json +1 -1
  5. package/src/cli/config-cmd.ts +8 -7
  6. package/src/cli/explain.ts +10 -5
  7. package/src/cli/export.ts +6 -5
  8. package/src/cli/models.ts +10 -7
  9. package/src/cli/report.ts +6 -1
  10. package/src/cli/stats.ts +7 -7
  11. package/src/config/load.ts +10 -1
  12. package/src/config/types.ts +10 -1
  13. package/src/context/bridge.ts +7 -7
  14. package/src/context/index.ts +3 -3
  15. package/src/context/store.ts +39 -56
  16. package/src/context/types.ts +7 -6
  17. package/src/cost/blended.ts +28 -7
  18. package/src/cost/feedback.ts +33 -37
  19. package/src/cost/ledger-sql.ts +547 -0
  20. package/src/cost/ledger.ts +30 -459
  21. package/src/cost/report.ts +171 -129
  22. package/src/cost/retention.ts +10 -10
  23. package/src/cost/summary.ts +15 -10
  24. package/src/cost/types.ts +43 -62
  25. package/src/cost/views.ts +79 -49
  26. package/src/lib.ts +6 -2
  27. package/src/router/candidates.ts +7 -15
  28. package/src/router/classify.ts +6 -4
  29. package/src/router/index.ts +95 -9
  30. package/src/router/select.ts +38 -21
  31. package/src/router/state.ts +90 -102
  32. package/src/router/types.ts +11 -5
  33. package/src/server/advise.ts +6 -4
  34. package/src/server/compaction-digest.ts +1 -1
  35. package/src/server/digest.ts +9 -10
  36. package/src/server/http.ts +101 -41
  37. package/src/server/providers.ts +18 -4
  38. package/src/server/turn.ts +32 -9
  39. package/src/tokens/estimate.ts +16 -6
  40. package/src/upstream/ollama-usage.ts +21 -11
  41. package/src/util/schema.ts +201 -0
  42. package/src/util/sql.ts +246 -0
  43. package/src/wire/anthropic/messages.ts +3 -4
  44. package/src/wire/openai/request.ts +1 -0
  45. package/src/wire/types.ts +7 -0
  46. package/test/anthropic-wire.test.ts +9 -9
  47. package/test/benchmark-feeds.test.ts +7 -7
  48. package/test/cache-control.test.ts +7 -7
  49. package/test/cache-estimate.test.ts +5 -5
  50. package/test/catalog-view.test.ts +4 -4
  51. package/test/catalog.test.ts +11 -11
  52. package/test/classify.test.ts +24 -24
  53. package/test/compaction.test.ts +20 -20
  54. package/test/config-wizard.test.ts +32 -32
  55. package/test/config.test.ts +10 -10
  56. package/test/connect-harnesses.test.ts +11 -11
  57. package/test/context-bridge.test.ts +40 -30
  58. package/test/context-prune.test.ts +43 -36
  59. package/test/context-query.test.ts +8 -8
  60. package/test/controls.test.ts +54 -27
  61. package/test/cost.test.ts +12 -12
  62. package/test/digest.test.ts +55 -44
  63. package/test/embed-lifecycle.test.ts +5 -5
  64. package/test/embed-logic.test.ts +26 -26
  65. package/test/escalate.test.ts +17 -17
  66. package/test/eval.test.ts +13 -13
  67. package/test/executable.test.ts +6 -6
  68. package/test/exploration.test.ts +19 -20
  69. package/test/failover.test.ts +22 -21
  70. package/test/fakes.ts +105 -0
  71. package/test/features.test.ts +21 -21
  72. package/test/harness-requests.test.ts +3 -3
  73. package/test/harness-switch.test.ts +5 -5
  74. package/test/hold-exploration.test.ts +13 -13
  75. package/test/hot-reload.test.ts +5 -5
  76. package/test/learned.test.ts +5 -5
  77. package/test/ledger-sql.test.ts +342 -0
  78. package/test/mcp-entry.test.ts +5 -5
  79. package/test/migrations.test.ts +28 -22
  80. package/test/models-yml.test.ts +18 -18
  81. package/test/ollama.test.ts +40 -34
  82. package/test/omp-credentials.test.ts +16 -16
  83. package/test/policy.test.ts +3 -3
  84. package/test/reconfigure.test.ts +4 -4
  85. package/test/redaction.test.ts +41 -35
  86. package/test/remote.test.ts +12 -12
  87. package/test/report-logic.test.ts +8 -8
  88. package/test/report.test.ts +95 -87
  89. package/test/retention.test.ts +79 -66
  90. package/test/schema.test.ts +123 -0
  91. package/test/scope.test.ts +8 -8
  92. package/test/select.test.ts +216 -257
  93. package/test/skills.test.ts +3 -3
  94. package/test/sql-shim.test.ts +154 -0
  95. package/test/state.test.ts +43 -36
  96. package/test/summary.test.ts +38 -27
  97. package/test/tier-plan.test.ts +45 -62
  98. package/test/toast-logic.test.ts +31 -31
  99. package/test/tokens.test.ts +95 -80
  100. package/test/trust-attribution.test.ts +217 -187
  101. package/test/trust-window.test.ts +37 -32
  102. package/test/turn.test.ts +55 -23
  103. package/test/upstreams.test.ts +13 -13
  104. package/test/views.test.ts +81 -59
  105. package/test/wire-request.test.ts +17 -17
  106. package/test/wire-responses.test.ts +4 -4
  107. package/tools/agentdox-e2e.ts +5 -2
  108. package/tools/export-benchmarks.ts +5 -5
  109. package/tools/ledger-parity.ts +266 -0
  110. package/tools/replay.ts +16 -8
package/src/cost/views.ts CHANGED
@@ -5,12 +5,13 @@
5
5
  * cost export by day, harness and model. Served by `/v1/router/spend`,
6
6
  * `GET /v1/router/feedback` and `/v1/router/export`, exported from lib.ts, and
7
7
  * behind `auto-model-router export`. All of them read only long-stable ledger
8
- * columns and accept a read-only database handle.
8
+ * columns, and they run on either engine through `util/sql.ts`: a front door
9
+ * may be reading a local file while the router writes a shared database.
9
10
  */
10
11
 
11
12
  import { providerOfSlug } from "./report.ts";
12
- import type { Database } from "bun:sqlite";
13
- import { toEntry, type LedgerRow } from "./ledger.ts";
13
+ import { num, type SqlDb } from "../util/sql.ts";
14
+ import { entriesOf } from "./ledger-sql.ts";
14
15
  import { harnessFilter } from "./report.ts";
15
16
  import type { LedgerEntry } from "./types.ts";
16
17
 
@@ -89,7 +90,7 @@ export interface DecisionFilter {
89
90
  * reasons, the classifier's view, the cost forecast against the bill, the escalation signal —
90
91
  * and the verdicts `/router good|bad` recorded against each turn ride along.
91
92
  */
92
- export function decisionEntries(db: Database, filter: DecisionFilter): DecisionEntry[] {
93
+ export async function decisionEntries(db: SqlDb, filter: DecisionFilter): Promise<DecisionEntry[]> {
93
94
  const s = scope(filter.harness, "harness_id");
94
95
  if (s === null) return [];
95
96
  const where = ["created_at_ms >= $since", ...s.sql];
@@ -107,20 +108,26 @@ export function decisionEntries(db: Database, filter: DecisionFilter): DecisionE
107
108
  bind.$session = filter.ompSessionId;
108
109
  }
109
110
  const limit = Math.min(Math.max(filter.limit ?? 50, 1), 1_000);
110
- const rows = db.query(`SELECT * FROM ledger WHERE ${where.join(" AND ")} ORDER BY created_at_ms DESC LIMIT ${limit}`).all(bind) as LedgerRow[];
111
- const entries = rows.map(toEntry);
111
+ const rows = await db.query<unknown>(
112
+ `SELECT * FROM ledger WHERE ${where.join(" AND ")} ORDER BY created_at_ms DESC LIMIT ${limit}`,
113
+ bind,
114
+ );
115
+ const entries = entriesOf(rows);
112
116
  if (entries.length === 0) return [];
113
117
  // Verdicts, when the feedback table exists (it does not on a ledger no one has judged).
114
- const hasFeedback = (db.query("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'feedback'").get() as { name: string } | null) !== null;
115
118
  const verdicts = new Map<string, DecisionEntry["feedback"]>();
116
- if (hasFeedback) {
119
+ if (await db.tableExists("feedback")) {
117
120
  const ids = entries.map((e) => e.id);
118
121
  const marks = ids.map((_, i) => `$f${i}`).join(", ");
119
122
  const fb: Record<string, string> = {};
120
123
  ids.forEach((id, i) => (fb[`$f${i}`] = id));
121
- for (const r of db.query(`SELECT ledger_id, verdict, note, created_at_ms FROM feedback WHERE ledger_id IN (${marks}) ORDER BY created_at_ms ASC`).all(fb) as { ledger_id: string; verdict: string; note: string; created_at_ms: number }[]) {
124
+ const rows = await db.query<{ ledger_id: string; verdict: string; note: string; created_at_ms: number }>(
125
+ `SELECT ledger_id, verdict, note, created_at_ms FROM feedback WHERE ledger_id IN (${marks}) ORDER BY created_at_ms ASC`,
126
+ fb,
127
+ );
128
+ for (const r of rows) {
122
129
  const list = verdicts.get(r.ledger_id) ?? [];
123
- list.push({ verdict: r.verdict === "good" ? "good" : "bad", note: r.note, createdAtMs: r.created_at_ms });
130
+ list.push({ verdict: r.verdict === "good" ? "good" : "bad", note: r.note, createdAtMs: num(r.created_at_ms) });
124
131
  verdicts.set(r.ledger_id, list);
125
132
  }
126
133
  }
@@ -132,7 +139,7 @@ export function decisionEntries(db: Database, filter: DecisionFilter): DecisionE
132
139
  * as the ledger counts them. `contextScope`, when given, narrows to the turns that carried
133
140
  * exactly that agentdox scope — what a front door charges back to one project.
134
141
  */
135
- export function spendUsdSince(db: Database, sinceMs: number, harness: HarnessScope, contextScope?: string): number {
142
+ export async function spendUsdSince(db: SqlDb, sinceMs: number, harness: HarnessScope, contextScope?: string): Promise<number> {
136
143
  const s = scope(harness, "harness_id");
137
144
  if (s === null) return 0;
138
145
  const where = ["created_at_ms >= $since", ...s.sql];
@@ -141,36 +148,42 @@ export function spendUsdSince(db: Database, sinceMs: number, harness: HarnessSco
141
148
  where.push("scope = $scope");
142
149
  bind.$scope = contextScope;
143
150
  }
144
- const row = db.query(`SELECT COALESCE(SUM(${USD}), 0) AS usd FROM ledger WHERE ${where.join(" AND ")}`).get(bind) as { usd: number };
145
- return row.usd;
151
+ const row = await db.one<{ usd: unknown }>(
152
+ `SELECT COALESCE(SUM(${USD}), 0) AS usd FROM ledger WHERE ${where.join(" AND ")}`,
153
+ bind,
154
+ );
155
+ return num(row?.usd);
146
156
  }
147
157
 
148
158
  /** Verdicts since `sinceMs`, by model and the most recent 200, joined to the ledger for the judging harness. */
149
- export function feedbackView(db: Database, sinceMs: number, harness: HarnessScope): FeedbackView {
150
- const exists = (db.query("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'feedback'").get() as { name: string } | null) !== null;
151
- if (!exists) return { byModel: [], recent: [] };
159
+ export async function feedbackView(db: SqlDb, sinceMs: number, harness: HarnessScope): Promise<FeedbackView> {
160
+ if (!(await db.tableExists("feedback"))) return { byModel: [], recent: [] };
152
161
  const s = scope(harness, "l.harness_id");
153
162
  if (s === null) return { byModel: [], recent: [] };
154
163
  const where = ["f.created_at_ms >= $since", ...s.sql].join(" AND ");
155
164
  const bind = { $since: sinceMs, ...s.bind };
156
165
  const recent = (
157
- db.query(`SELECT f.created_at_ms AS at_ms, f.slug, f.tier, f.verdict, f.note, COALESCE(l.harness_id, '') AS harness_id FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} ORDER BY f.created_at_ms DESC LIMIT 200`).all(bind) as {
158
- at_ms: number;
159
- slug: string;
160
- tier: string;
161
- verdict: string;
162
- note: string;
163
- harness_id: string;
164
- }[]
165
- ).map((r) => ({ atMs: r.at_ms, slug: r.slug, tier: r.tier, verdict: (r.verdict === "good" ? "good" : "bad") as "good" | "bad", note: r.note, harnessId: r.harness_id }));
166
+ await db.query<{ at_ms: number; slug: string; tier: string; verdict: string; note: string; harness_id: string }>(
167
+ `SELECT f.created_at_ms AS at_ms, f.slug, f.tier, f.verdict, f.note, COALESCE(l.harness_id, '') AS harness_id
168
+ FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} ORDER BY f.created_at_ms DESC LIMIT 200`,
169
+ bind,
170
+ )
171
+ ).map((r) => ({
172
+ atMs: num(r.at_ms),
173
+ slug: r.slug,
174
+ tier: r.tier,
175
+ verdict: (r.verdict === "good" ? "good" : "bad") as "good" | "bad",
176
+ note: r.note,
177
+ harnessId: r.harness_id,
178
+ }));
166
179
  const byModel = (
167
- db
168
- .query(
169
- `SELECT f.slug, SUM(CASE WHEN f.verdict = 'good' THEN 1 ELSE 0 END) AS good, SUM(CASE WHEN f.verdict = 'bad' THEN 1 ELSE 0 END) AS bad, COUNT(DISTINCT COALESCE(l.harness_id, '')) AS judges
170
- FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} GROUP BY f.slug ORDER BY bad DESC, good DESC, f.slug ASC`,
171
- )
172
- .all(bind) as { slug: string; good: number; bad: number; judges: number }[]
173
- ).map((r) => ({ slug: r.slug, good: r.good, bad: r.bad, judges: r.judges }));
180
+ await db.query<{ slug: string; good: unknown; bad: unknown; judges: unknown }>(
181
+ `SELECT f.slug, SUM(CASE WHEN f.verdict = 'good' THEN 1 ELSE 0 END) AS good, SUM(CASE WHEN f.verdict = 'bad' THEN 1 ELSE 0 END) AS bad,
182
+ COUNT(DISTINCT COALESCE(l.harness_id, '')) AS judges
183
+ FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} GROUP BY f.slug ORDER BY bad DESC, good DESC, f.slug ASC`,
184
+ bind,
185
+ )
186
+ ).map((r) => ({ slug: r.slug, good: num(r.good), bad: num(r.bad), judges: num(r.judges) }));
174
187
  return { byModel, recent };
175
188
  }
176
189
 
@@ -180,36 +193,53 @@ export function feedbackView(db: Database, sinceMs: number, harness: HarnessScop
180
193
  * can charge each project its own share; rows from before v18 (and turns that carried no
181
194
  * scope) group under "".
182
195
  */
183
- export function exportRows(db: Database, sinceMs: number, harness: HarnessScope): ExportRow[] {
196
+ export async function exportRows(db: SqlDb, sinceMs: number, harness: HarnessScope): Promise<ExportRow[]> {
184
197
  const s = scope(harness, "harness_id");
185
198
  if (s === null) return [];
186
199
  const where = ["created_at_ms >= $since", "requested_model <> 'digest'", ...s.sql].join(" AND ");
187
- const rows = db
188
- .query(
189
- `SELECT strftime('%Y-%m-%d', created_at_ms / 1000, 'unixepoch') AS day, harness_id, COALESCE(served_slug, slug) AS slug, COALESCE(scope, '') AS scope,
200
+ // GROUP BY names the day expression rather than its alias: Postgres does not
201
+ // allow a select alias in GROUP BY, and repeating it keeps one statement for
202
+ // both engines.
203
+ const day = db.utcDay("created_at_ms");
204
+ const rows = await db.query<{
205
+ day: string;
206
+ harness_id: string;
207
+ slug: string;
208
+ scope: string;
209
+ dispatches: unknown;
210
+ prompt_tokens: unknown;
211
+ cached_tokens: unknown;
212
+ completion_tokens: unknown;
213
+ spend: unknown;
214
+ escalations: unknown;
215
+ errors: unknown;
216
+ }>(
217
+ `SELECT ${day} AS day, harness_id, COALESCE(served_slug, slug) AS slug, COALESCE(scope, '') AS scope,
190
218
  COUNT(*) AS dispatches,
191
- COALESCE(SUM(json_extract(usage, '$.promptTokens')), 0) AS prompt_tokens,
192
- COALESCE(SUM(json_extract(usage, '$.cachedTokens')), 0) AS cached_tokens,
193
- COALESCE(SUM(json_extract(usage, '$.completionTokens')), 0) AS completion_tokens,
219
+ COALESCE(SUM(${db.jsonNum("usage", "promptTokens")}), 0) AS prompt_tokens,
220
+ COALESCE(SUM(${db.jsonNum("usage", "cachedTokens")}), 0) AS cached_tokens,
221
+ COALESCE(SUM(${db.jsonNum("usage", "completionTokens")}), 0) AS completion_tokens,
194
222
  COALESCE(SUM(${USD}), 0) AS spend,
195
223
  SUM(CASE WHEN escalation_signal IS NOT NULL THEN 1 ELSE 0 END) AS escalations,
196
224
  SUM(CASE WHEN error IS NOT NULL THEN 1 ELSE 0 END) AS errors
197
- FROM ledger WHERE ${where} GROUP BY day, harness_id, slug, scope ORDER BY day ASC, harness_id ASC, spend DESC`,
198
- )
199
- .all({ $since: sinceMs, ...s.bind }) as { day: string; harness_id: string; slug: string; scope: string; dispatches: number; prompt_tokens: number; cached_tokens: number; completion_tokens: number; spend: number; escalations: number; errors: number }[];
225
+ FROM ledger WHERE ${where}
226
+ GROUP BY ${day}, harness_id, COALESCE(served_slug, slug), COALESCE(scope, '')
227
+ ORDER BY 1 ASC, harness_id ASC, spend DESC`,
228
+ { $since: sinceMs, ...s.bind },
229
+ );
200
230
  return rows.map((r) => ({
201
231
  day: r.day,
202
232
  harnessId: r.harness_id,
203
233
  slug: r.slug,
204
234
  scope: r.scope,
205
235
  provider: providerOfSlug(r.slug),
206
- dispatches: r.dispatches,
207
- promptTokens: r.prompt_tokens,
208
- cachedTokens: r.cached_tokens,
209
- completionTokens: r.completion_tokens,
210
- spendUsd: r.spend,
211
- escalations: r.escalations,
212
- errors: r.errors,
236
+ dispatches: num(r.dispatches),
237
+ promptTokens: num(r.prompt_tokens),
238
+ cachedTokens: num(r.cached_tokens),
239
+ completionTokens: num(r.completion_tokens),
240
+ spendUsd: num(r.spend),
241
+ escalations: num(r.escalations),
242
+ errors: num(r.errors),
213
243
  }));
214
244
  }
215
245
 
package/src/lib.ts CHANGED
@@ -25,14 +25,18 @@ export type { DeepPartial } from "./config/load.ts";
25
25
  export { buildUsageReport, renderUsageReport, type UsageReport, type ReportTotals } from "./cost/report.ts";
26
26
  export { buildDailySummary, renderDailySummary, type DailySummary } from "./cost/summary.ts";
27
27
  export { openDb } from "./util/sqlite.ts";
28
+ // A front door reads the ledger through the engine-agnostic handle: the store
29
+ // may be a file or a shared database, and the view functions take this.
30
+ export { dialectOf, openSqlDb, num, numOrNull, type Dialect, type SqlDb } from "./util/sql.ts";
28
31
  export { spendUsdSince, feedbackView, exportRows, exportCsv, decisionEntries, harnessScopeParam, type HarnessScope, type ExportRow, type FeedbackRow, type FeedbackByModel, type FeedbackView, type DecisionEntry, type DecisionFilter } from "./cost/views.ts";
29
- export { createLedger } from "./cost/ledger.ts";
32
+ export { createSqlLedger } from "./cost/ledger-sql.ts";
33
+ export { migrateStore, STORE_TABLES } from "./util/schema.ts";
30
34
  export { createFeedbackStore, type FeedbackStore, type FeedbackRecord } from "./cost/feedback.ts";
31
35
  export { buildExecutable, collectPackageFiles, executableFileName, hostTarget, isExecutableTarget, EXECUTABLE_TARGETS, type ExecutableTarget, type BuildExecutableResult } from "./cli/build-executable.ts";
32
36
  export { parseSkillsBundle, type SkillsBundle } from "./cli/skills.ts";
33
37
  export type { RequestPolicy } from "./wire/types.ts";
34
38
  export type { CatalogView, CatalogViewModel } from "./server/catalog-view.ts";
35
- export type { Ledger, LedgerEntry, PruneResult } from "./cost/types.ts";
39
+ export type { AsyncLedger, LedgerEntry, PruneResult } from "./cost/types.ts";
36
40
  // Retention: a front door asks through `POST /v1/router/prune` rather than
37
41
  // deleting from the ledger itself. The interval is exported so it can say when.
38
42
  export { RETENTION_INTERVAL_MS } from "./cost/retention.ts";
@@ -7,7 +7,7 @@
7
7
  import type { CatalogModel, CatalogSnapshot } from "../catalog/types.ts";
8
8
  import type { FilterConfig, QualityAxis, RouterConfig } from "../config/types.ts";
9
9
  import { forecast, priceAt } from "../cost/forecast.ts";
10
- import type { Ledger, LedgerSignals, ModelLatency } from "../cost/types.ts";
10
+ import type { LedgerSignals, ModelLatency } from "../cost/types.ts";
11
11
  import type { NormRequest } from "../wire/types.ts";
12
12
  import { effectivePriceCeiling, effectiveQualityFloor, tierPlanFor } from "./tier-plan.ts";
13
13
  import type { Candidate, Features, Rejection, TaskType, Tier } from "./types.ts";
@@ -27,7 +27,6 @@ export interface BuildCandidatesArgs {
27
27
  /** Task type; its config selects the axis, quality floor, and image filter. */
28
28
  task: TaskType;
29
29
  snapshot: CatalogSnapshot;
30
- ledger: Ledger | null;
31
30
  cfg: RouterConfig;
32
31
  expectedCompletionTokens: number;
33
32
  /** Slug whose prompt cache is warm this turn; wins score ties. */
@@ -154,7 +153,7 @@ function latencyMultiplier(latency: ModelLatency | null, filters: FilterConfig,
154
153
  }
155
154
 
156
155
  export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candidate[]; rejected: Rejection[] } {
157
- const { req, features, tier, task, snapshot, ledger, cfg, expectedCompletionTokens, warmSlug, relaxLevel = 0 } = args;
156
+ const { req, features, tier, task, snapshot, cfg, expectedCompletionTokens, warmSlug, relaxLevel = 0 } = args;
158
157
  // A Set only when non-empty: the common path allocates nothing.
159
158
  const excluded = args.excludeSlugs === undefined || args.excludeSlugs.length === 0 ? null : new Set(args.excludeSlugs);
160
159
  const tierCfg = cfg.tiers[tier];
@@ -318,14 +317,11 @@ export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candid
318
317
  continue;
319
318
  }
320
319
 
321
- // Use pre-fetched signals when available (batch lookup, one query per
322
- // signal kind for the entire candidate set instead of one per model).
323
- // Falls back to per-slug calls when signals is not provided (e.g. tests).
320
+ // Signals are prefetched by `route` — one query per signal kind for the
321
+ // whole candidate set. Absent means "the ledger knows nothing about this
322
+ // model", which is what a cold start looks like and is handled below.
324
323
  const signals = args.signals;
325
- const trust =
326
- signals?.get(slug)?.trust ??
327
- ledger?.trust(slug, filters.trustScopedByHarness ? req.harnessId : undefined, filters.feedbackByTask ? task : undefined) ??
328
- null;
324
+ const trust = signals?.get(slug)?.trust ?? null;
329
325
  if (!relaxTrust && trust !== null && trust.attempts >= filters.minTrustSamples && trust.successRate < filters.minTrust) {
330
326
  rejected.push({
331
327
  slug,
@@ -342,11 +338,7 @@ export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candid
342
338
  // gate can. Only measured models are dropped, so a new model still gets its
343
339
  // cold-start turns to accumulate samples. Relaxed with trust in rescue.
344
340
  const needLatency = filters.latencyWeight > 0 || filters.maxExpectedWaitMs !== undefined;
345
- const latency = needLatency
346
- ? (signals?.get(slug)?.latency ??
347
- ledger?.latency(slug, filters.trustScopedByHarness ? req.harnessId : undefined) ??
348
- null)
349
- : null;
341
+ const latency = needLatency ? (signals?.get(slug)?.latency ?? null) : null;
350
342
  if (
351
343
  !relaxTrust &&
352
344
  filters.maxExpectedWaitMs !== undefined &&
@@ -6,7 +6,7 @@
6
6
  import type { CatalogSource } from "../catalog/types.ts";
7
7
  import type { QualityAxis, RouterConfig } from "../config/types.ts";
8
8
  import { forecast } from "../cost/forecast.ts";
9
- import type { Ledger } from "../cost/types.ts";
9
+ import type { AsyncLedger } from "../cost/types.ts";
10
10
  import { estimateTokens } from "../tokens/estimate.ts";
11
11
  import type { UpstreamClient } from "../upstream/types.ts";
12
12
  import { sha256Hex } from "../util/hash.ts";
@@ -219,7 +219,7 @@ export function classifyTask(f: Features): TaskType {
219
219
 
220
220
  export interface ClassifyDeps {
221
221
  upstream: UpstreamClient;
222
- ledger: Ledger | null;
222
+ ledger: AsyncLedger | null;
223
223
  catalog: CatalogSource | null;
224
224
  }
225
225
 
@@ -334,11 +334,13 @@ export async function classify(
334
334
  }
335
335
 
336
336
  // Cost guard: adjudication must be cheap relative to the turn it classifies.
337
- const blend = deps.ledger?.blendedRate(cfg.ledger.blendWindowDays) ?? null;
337
+ const blend = (await deps.ledger?.blendedRate(cfg.ledger.blendWindowDays)) ?? null;
338
338
  const inputRate = (blend?.inputPerMtok ?? cfg.ledger.fallbackBlend.inputPerMtok) / 1e6;
339
339
  const outputRate = (blend?.outputPerMtok ?? cfg.ledger.fallbackBlend.outputPerMtok) / 1e6;
340
340
  const judge = deps.catalog?.find(cc.model);
341
- const digestTokens = estimateTokens(Buffer.byteLength(ADJUDICATOR_SYSTEM) + Buffer.byteLength(digest), "unknown", deps.ledger);
341
+ // The adjudicator prompt is short and its family unknown, so the default
342
+ // family ratio is as good as a measured one here.
343
+ const digestTokens = estimateTokens(Buffer.byteLength(ADJUDICATOR_SYSTEM) + Buffer.byteLength(digest), "unknown", null);
342
344
  const adjudicatorUsd =
343
345
  judge !== undefined
344
346
  ? forecast(judge, {
@@ -10,23 +10,25 @@
10
10
  * real request without perturbing the conversation it belongs to.
11
11
  */
12
12
 
13
- import type { CatalogSource } from "../catalog/types.ts";
13
+ import type { CatalogSnapshot, CatalogSource } from "../catalog/types.ts";
14
14
  import type { ProfileConfig, RouterConfig } from "../config/types.ts";
15
- import type { Ledger } from "../cost/types.ts";
15
+ import { type AsyncLedger, type LedgerReader } from "../cost/types.ts";
16
16
  import { estimatePromptTokens } from "../tokens/estimate.ts";
17
17
  import type { UpstreamClient } from "../upstream/types.ts";
18
+ import { modelNotFound } from "../wire/openai/errors.ts";
18
19
  import type { NormRequest, RequestPolicy } from "../wire/types.ts";
19
20
  import { classify, classifyTask } from "./classify.ts";
20
21
  import { extractFeatures } from "./features.ts";
21
- import { select } from "./select.ts";
22
+ import { monthStartMs, select, type TurnReads } from "./select.ts";
22
23
  import { TIER_ORDER, type Classification, type ConversationStore, type Decision, type Router, type Tier } from "./types.ts";
23
-
24
24
  export interface RouterDeps {
25
25
  config: RouterConfig;
26
26
  catalog: CatalogSource;
27
- ledger: Ledger;
27
+ ledger: AsyncLedger;
28
28
  conversations: ConversationStore;
29
29
  upstream: UpstreamClient;
30
+ /** Async ledger reads; defaults to reading `ledger` directly. */
31
+ reader?: LedgerReader;
30
32
  }
31
33
 
32
34
  /**
@@ -90,20 +92,89 @@ export function resolveProfile(cfg: RouterConfig, requestedModel: string, isSuba
90
92
  return fallback;
91
93
  }
92
94
 
95
+ /**
96
+ * The `model` a client names is a profile id. When it is neither a profile nor
97
+ * a catalog slug, routing something else and reporting the asked-for name back
98
+ * is a silent substitution: the caller is billed for a model it never chose.
99
+ * A real slug becomes a pin (absolute, the way a session override is); anything
100
+ * else is refused.
101
+ *
102
+ * @param asked the client's `model` before the provider prefix was stripped.
103
+ * @returns the slug to pin, or undefined when a profile matched.
104
+ */
105
+ export function pinForRequestedModel(
106
+ cfg: RouterConfig,
107
+ slugs: readonly string[],
108
+ requestedModel: string,
109
+ asked: string,
110
+ ): string | undefined {
111
+ if (cfg.profiles.some((p) => p.id === requestedModel)) return undefined;
112
+ if (slugs.includes(asked)) return asked;
113
+ throw modelNotFound(asked);
114
+ }
115
+
116
+ /**
117
+ * Reads the ledger for one turn, concurrently, before selection runs.
118
+ *
119
+ * Only what this turn's configuration actually consults is fetched: month and
120
+ * day spend are skipped when no such budget is set, and the escalation-cost
121
+ * term is skipped when its weight is 0. Empty catalog ⇒ no signal queries at
122
+ * all, which is what keeps the tests' fakes cheap.
123
+ */
124
+ export async function prefetchTurnReads(
125
+ reader: LedgerReader | null,
126
+ req: NormRequest,
127
+ profile: ProfileConfig,
128
+ cfg: RouterConfig,
129
+ snapshot: CatalogSnapshot,
130
+ task: string,
131
+ nowMs: number = Date.now(),
132
+ ): Promise<TurnReads> {
133
+ if (reader === null) return {};
134
+ const slugs = snapshot.models.map((m) => m.slug);
135
+ const harness = cfg.filters.trustScopedByHarness ? req.harnessId : undefined;
136
+ const perMonthUsd = profile.budget?.perMonthUsd ?? cfg.budget.perMonthUsd;
137
+ const perDayUsd = profile.budget?.perDayUsd ?? cfg.budget.perDayUsd;
138
+ const [signals, cacheReliability, escalation, monthSpendUsd, daySpendUsd] = await Promise.all([
139
+ slugs.length > 0 ? reader.signals(slugs, harness, cfg.filters.feedbackByTask ? task : undefined) : undefined,
140
+ slugs.length > 0 && cfg.filters.cacheReliabilityMinSamples > 0 ? reader.cacheReliability(slugs) : undefined,
141
+ cfg.filters.escalationCostWeight > 0 ? reader.escalationCost(cfg.ledger.blendWindowDays) : undefined,
142
+ // Month pacing can tighten the daily ceiling, so month spend is needed
143
+ // whenever either budget is set.
144
+ perMonthUsd !== undefined ? reader.spendSince(monthStartMs(nowMs), req.harnessId) : undefined,
145
+ perDayUsd !== undefined || perMonthUsd !== undefined ? reader.spendSince(nowMs - 86_400_000, req.harnessId) : undefined,
146
+ ]);
147
+ return {
148
+ ...(signals === undefined ? {} : { signals }),
149
+ ...(cacheReliability === undefined ? {} : { cacheReliability }),
150
+ ...(escalation === undefined ? {} : { escalationUsdPerPromptToken: escalation?.usdPerPromptToken ?? null }),
151
+ ...(monthSpendUsd === undefined ? {} : { monthSpendUsd }),
152
+ ...(daySpendUsd === undefined ? {} : { daySpendUsd }),
153
+ };
154
+ }
155
+
93
156
  export function createRouter(deps: RouterDeps): Router {
94
157
  const { config, catalog, ledger, conversations, upstream } = deps;
158
+ // A Postgres ledger supplies its own reader; a local one is wrapped, so the
159
+ // prefetch path is identical for both.
160
+ // The unified ledger IS a reader: `LedgerReader` is the narrow half of it.
161
+ const reader = deps.reader ?? ledger;
95
162
 
96
163
  return {
97
164
  async route(
98
165
  req: NormRequest,
99
166
  opts: { attempt: number; escalateFrom?: Tier; excludeSlugs?: readonly string[]; forceTier?: Tier; forceSlug?: string },
100
167
  ): Promise<Decision> {
101
- const state = conversations.get(req.conversationKey) ?? conversations.load(req.conversationKey);
168
+ const state = (await conversations.get(req.conversationKey)) ?? (await conversations.load(req.conversationKey));
102
169
  const snapshot = await catalog.get();
103
170
 
104
171
  const priorTokenizer =
105
172
  state.currentSlug === null ? undefined : catalog.find(state.currentSlug)?.tokenizer;
106
- const promptTokens = estimatePromptTokens(req, priorTokenizer ?? NEUTRAL_TOKENIZER, ledger);
173
+ const tokenizer = priorTokenizer ?? NEUTRAL_TOKENIZER;
174
+ // One ratio, fetched before estimating: the estimate itself runs in
175
+ // synchronous code that a shared store cannot be read from.
176
+ const ratio = await ledger.tokenRatio(tokenizer);
177
+ const promptTokens = estimatePromptTokens(req, tokenizer, ratio);
107
178
  const features = extractFeatures(req, promptTokens);
108
179
 
109
180
  let classification: Classification;
@@ -136,7 +207,22 @@ export function createRouter(deps: RouterDeps): Router {
136
207
  classification = await classify(req, features, config, { upstream, ledger, catalog });
137
208
  }
138
209
 
139
- const policed = applyRequestPolicy(resolveProfile(config, req.requestedModel, req.isSubagent), config, req.policy, opts.forceSlug);
210
+ const pinnedByModel = pinForRequestedModel(
211
+ config,
212
+ snapshot.models.map((m) => m.slug),
213
+ req.requestedModel,
214
+ req.requestedModelFull ?? req.requestedModel,
215
+ );
216
+ const policed = applyRequestPolicy(
217
+ resolveProfile(config, req.requestedModel, req.isSubagent),
218
+ config,
219
+ req.policy,
220
+ opts.forceSlug ?? pinnedByModel,
221
+ );
222
+ // Every ledger read this turn needs, fetched here rather than inside
223
+ // `select`: selection stays a synchronous pure function, and a ledger
224
+ // that can only be read asynchronously (Postgres) works unchanged.
225
+ const reads = await prefetchTurnReads(reader, req, policed.profile, policed.cfg, snapshot, classification.task);
140
226
  const decision = select({
141
227
  req,
142
228
  features,
@@ -144,9 +230,9 @@ export function createRouter(deps: RouterDeps): Router {
144
230
  profile: policed.profile,
145
231
  state,
146
232
  snapshot,
147
- ledger,
148
233
  cfg: policed.cfg,
149
234
  nowMs: Date.now(),
235
+ reads,
150
236
  ...(opts.excludeSlugs === undefined ? {} : { excludeSlugs: opts.excludeSlugs }),
151
237
  ...(policed.forceSlug === undefined ? {} : { forceSlug: policed.forceSlug }),
152
238
  });
@@ -9,7 +9,7 @@
9
9
  import type { CatalogSnapshot } from "../catalog/types.ts";
10
10
  import type { ProfileConfig, RouterConfig } from "../config/types.ts";
11
11
  import { forecast, priceAt } from "../cost/forecast.ts";
12
- import type { Ledger } from "../cost/types.ts";
12
+ import type { LedgerSignals, ModelCacheReliability } from "../cost/types.ts";
13
13
  import { explorationDraw } from "./explore.ts";
14
14
  import type { CompactionEdit, NormRequest, ReasoningLevel } from "../wire/types.ts";
15
15
  import { compactedBytes, planCompaction, validatePlan, type CompactionResult } from "./compaction.ts";
@@ -35,7 +35,6 @@ export interface SelectArgs {
35
35
  profile: ProfileConfig;
36
36
  state: ConversationState;
37
37
  snapshot: CatalogSnapshot;
38
- ledger: Ledger | null;
39
38
  cfg: RouterConfig;
40
39
  nowMs: number;
41
40
  /**
@@ -49,6 +48,30 @@ export interface SelectArgs {
49
48
  * stay/switch comparison. Ignored when the catalog has no such model.
50
49
  */
51
50
  forceSlug?: string;
51
+ /** Ledger reads already performed for this turn; see TurnReads. */
52
+ reads?: TurnReads;
53
+ }
54
+
55
+ /**
56
+ * Every ledger read a turn needs, fetched BEFORE selection.
57
+ *
58
+ * `select` is synchronous on purpose — it is a pure ranking function, and
59
+ * `explain` depends on being able to run it without side effects. A ledger on
60
+ * Postgres cannot be read synchronously, so the reads move up to `route`,
61
+ * which is already async, and arrive here as data. Absent ⇒ read through the
62
+ * absent reads degrade to no signals rather than reaching for the store.
63
+ */
64
+ export interface TurnReads {
65
+ /** Trust and latency per candidate slug; one batch query per signal kind. */
66
+ signals?: Map<string, LedgerSignals>;
67
+ /** Observed warm-cache hit rate per slug. */
68
+ cacheReliability?: Map<string, ModelCacheReliability>;
69
+ /** What an escalated retry bills per prompt token; null ⇒ the term is inert. */
70
+ escalationUsdPerPromptToken?: number | null;
71
+ /** Spend since the start of the UTC month, scoped as the filters say. */
72
+ monthSpendUsd?: number;
73
+ /** Spend over the rolling 24h, scoped as the filters say. */
74
+ daySpendUsd?: number;
52
75
  }
53
76
 
54
77
  /**
@@ -67,7 +90,6 @@ export class BudgetExceededError extends Error {
67
90
  // Mid-range completion assumption for forecasts. Long generations amortize
68
91
  // into prompt-dominated cost anyway; precision here does not move rankings.
69
92
  const EXPECTED_COMPLETION_TOKENS = 1024;
70
- const DAY_MS = 86_400_000;
71
93
 
72
94
  /**
73
95
  * Authors known to accept replayed assistant reasoning over chat completions:
@@ -129,7 +151,7 @@ export function monthPace(nowMs: number, perMonthUsd: number, spentUsd: number):
129
151
  }
130
152
 
131
153
  export function select(args: SelectArgs): Decision {
132
- const { req, features, classification, profile, state, snapshot, ledger, cfg, nowMs } = args;
154
+ const { req, features, classification, profile, state, snapshot, cfg, nowMs } = args;
133
155
  const reasons: string[] = [];
134
156
  const minI = tierIdx(profile.minTier);
135
157
  const maxI = tierIdx(profile.maxTier);
@@ -351,29 +373,23 @@ export function select(args: SelectArgs): Decision {
351
373
  const warmSlug = cacheWarm ? state.cacheWarmSlug : null;
352
374
  // Expected cache hit for a model: its observed rate once enough warm-expected
353
375
  // samples exist (filters.cacheReliabilityMinSamples), else a reliable 1.
376
+ const reads = args.reads;
354
377
  const cacheHitExpectation = (slug: string): { rate: number; measured: boolean; samples: number } => {
355
378
  const min = cfg.filters.cacheReliabilityMinSamples;
356
- const rel = min > 0 ? (ledger?.cacheReliability?.(slug) ?? null) : null;
379
+ const rel = min > 0 ? (reads?.cacheReliability?.get(slug) ?? null) : null;
357
380
  if (rel === null || rel.samples < min) return { rate: 1, measured: false, samples: rel?.samples ?? 0 };
358
381
  return { rate: rel.hitRate, measured: true, samples: rel.samples };
359
382
  };
360
- // Pre-fetch trust/latency signals for all candidate slugs in one batch
361
- // query per signal kind, instead of per-model individual lookups.
362
- const candidateSignals =
363
- ledger !== null && snapshot.models.length > 0
364
- ? ledger.signals?.(
365
- snapshot.models.map((m) => m.slug),
366
- cfg.filters.trustScopedByHarness ? req.harnessId : undefined,
367
- cfg.filters.feedbackByTask ? classification.task : undefined,
368
- )
369
- : undefined;
383
+ // Trust and latency for every candidate, fetched by `route` before this ran.
384
+ // There is no fallback read here on purpose: selection is synchronous and the
385
+ // store may be a shared database, so a missing prefetch must degrade to
386
+ // "no signals" rather than silently reach for a handle it cannot await.
387
+ const candidateSignals = reads?.signals;
370
388
  // What an escalated retry has actually been billing per prompt token, for
371
389
  // the escalation-cost term in candidate scoring. Read once per turn; null
372
390
  // (term inert) when the weight is 0 or the ledger has too few samples.
373
391
  const escalationUsdPerPromptToken =
374
- cfg.filters.escalationCostWeight > 0
375
- ? (ledger?.escalationCost?.(cfg.ledger.blendWindowDays)?.usdPerPromptToken ?? null)
376
- : null;
392
+ cfg.filters.escalationCostWeight > 0 ? (reads?.escalationUsdPerPromptToken ?? null) : null;
377
393
  // A pinned slug is admitted the way a config pin is: into the tier's pin
378
394
  // list for this call only, so the quality floor cannot keep it out.
379
395
  const pinSlug = args.forceSlug !== undefined && snapshot.models.some((m) => m.slug === args.forceSlug) ? args.forceSlug : undefined;
@@ -386,7 +402,6 @@ export function select(args: SelectArgs): Decision {
386
402
  tier: t,
387
403
  task: classification.task,
388
404
  snapshot,
389
- ledger,
390
405
  cfg: buildCfg,
391
406
  expectedCompletionTokens: EXPECTED_COMPLETION_TOKENS,
392
407
  warmSlug,
@@ -515,13 +530,15 @@ export function select(args: SelectArgs): Decision {
515
530
  // left, becomes a daily ceiling that tightens as the month runs ahead.
516
531
  let paceNote = "";
517
532
  if (budget.perMonthUsd !== undefined) {
518
- const pace = monthPace(nowMs, budget.perMonthUsd, ledger?.spendSince(monthStartMs(nowMs), req.harnessId) ?? 0);
533
+ const monthSpend = reads?.monthSpendUsd ?? 0;
534
+ const pace = monthPace(nowMs, budget.perMonthUsd, monthSpend);
519
535
  if (budget.perDayUsd === undefined || pace.dailyCapUsd < budget.perDayUsd) {
520
536
  budget.perDayUsd = pace.dailyCapUsd;
521
537
  paceNote = ` (month pacing: $${pace.spentUsd.toFixed(2)} of $${budget.perMonthUsd} spent, $${pace.dailyCapUsd.toFixed(2)}/day for ${pace.daysLeft} more days)`;
522
538
  }
523
539
  }
524
- const daySpend = budget.perDayUsd !== undefined ? (ledger?.spendSince(nowMs - DAY_MS, req.harnessId) ?? 0) : 0;
540
+ const daySpend =
541
+ budget.perDayUsd !== undefined ? (reads?.daySpendUsd ?? 0) : 0;
525
542
  const breach = (c: Candidate): string | null => {
526
543
  if (budget.perTurnUsd !== undefined && c.forecast.coldUsd > budget.perTurnUsd) {
527
544
  return `cold forecast $${c.forecast.coldUsd.toFixed(4)} > per-turn budget $${budget.perTurnUsd}`;