auto-model-router 0.30.3 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/.omp-plugin/marketplace.json +2 -2
  2. package/README.md +32 -2
  3. package/omp-extension/router-configure.ts +9 -7
  4. package/package.json +1 -1
  5. package/src/cli/config-cmd.ts +8 -7
  6. package/src/cli/explain.ts +10 -5
  7. package/src/cli/export.ts +6 -5
  8. package/src/cli/models.ts +10 -7
  9. package/src/cli/report.ts +6 -1
  10. package/src/cli/stats.ts +7 -7
  11. package/src/config/load.ts +10 -1
  12. package/src/config/types.ts +10 -1
  13. package/src/context/bridge.ts +7 -7
  14. package/src/context/index.ts +3 -3
  15. package/src/context/store.ts +39 -56
  16. package/src/context/types.ts +7 -6
  17. package/src/cost/blended.ts +28 -7
  18. package/src/cost/feedback.ts +33 -37
  19. package/src/cost/ledger-sql.ts +547 -0
  20. package/src/cost/ledger.ts +30 -459
  21. package/src/cost/report.ts +171 -129
  22. package/src/cost/retention.ts +10 -10
  23. package/src/cost/summary.ts +15 -10
  24. package/src/cost/types.ts +43 -62
  25. package/src/cost/views.ts +79 -49
  26. package/src/eval/calibrate.ts +47 -12
  27. package/src/eval/run.ts +18 -2
  28. package/src/lib.ts +6 -2
  29. package/src/router/candidates.ts +7 -15
  30. package/src/router/classify.ts +6 -4
  31. package/src/router/index.ts +95 -9
  32. package/src/router/select.ts +38 -21
  33. package/src/router/state.ts +90 -102
  34. package/src/router/types.ts +11 -5
  35. package/src/server/advise.ts +6 -4
  36. package/src/server/compaction-digest.ts +1 -1
  37. package/src/server/digest.ts +9 -10
  38. package/src/server/http.ts +109 -46
  39. package/src/server/providers.ts +18 -4
  40. package/src/server/turn.ts +32 -9
  41. package/src/tokens/estimate.ts +16 -6
  42. package/src/upstream/ollama-usage.ts +21 -11
  43. package/src/util/schema.ts +201 -0
  44. package/src/util/sql.ts +246 -0
  45. package/src/wire/anthropic/messages.ts +3 -4
  46. package/src/wire/openai/request.ts +1 -0
  47. package/src/wire/types.ts +7 -0
  48. package/test/anthropic-wire.test.ts +9 -9
  49. package/test/benchmark-feeds.test.ts +7 -7
  50. package/test/cache-control.test.ts +7 -7
  51. package/test/cache-estimate.test.ts +5 -5
  52. package/test/catalog-view.test.ts +4 -4
  53. package/test/catalog.test.ts +11 -11
  54. package/test/classify.test.ts +24 -24
  55. package/test/compaction.test.ts +20 -20
  56. package/test/config-wizard.test.ts +32 -32
  57. package/test/config.test.ts +10 -10
  58. package/test/connect-harnesses.test.ts +11 -11
  59. package/test/context-bridge.test.ts +40 -30
  60. package/test/context-prune.test.ts +43 -36
  61. package/test/context-query.test.ts +8 -8
  62. package/test/controls.test.ts +54 -27
  63. package/test/cost.test.ts +12 -12
  64. package/test/digest.test.ts +55 -44
  65. package/test/embed-lifecycle.test.ts +5 -5
  66. package/test/embed-logic.test.ts +26 -26
  67. package/test/escalate.test.ts +17 -17
  68. package/test/eval.test.ts +73 -16
  69. package/test/executable.test.ts +6 -6
  70. package/test/exploration.test.ts +19 -20
  71. package/test/failover.test.ts +22 -21
  72. package/test/fakes.ts +105 -0
  73. package/test/features.test.ts +21 -21
  74. package/test/harness-requests.test.ts +3 -3
  75. package/test/harness-switch.test.ts +5 -5
  76. package/test/hold-exploration.test.ts +13 -13
  77. package/test/hot-reload.test.ts +5 -5
  78. package/test/learned.test.ts +5 -5
  79. package/test/ledger-sql.test.ts +342 -0
  80. package/test/mcp-entry.test.ts +5 -5
  81. package/test/migrations.test.ts +28 -22
  82. package/test/models-yml.test.ts +18 -18
  83. package/test/ollama.test.ts +40 -34
  84. package/test/omp-credentials.test.ts +16 -16
  85. package/test/policy.test.ts +3 -3
  86. package/test/reconfigure.test.ts +4 -4
  87. package/test/redaction.test.ts +41 -35
  88. package/test/remote.test.ts +12 -12
  89. package/test/report-logic.test.ts +8 -8
  90. package/test/report.test.ts +95 -87
  91. package/test/retention.test.ts +79 -66
  92. package/test/schema.test.ts +123 -0
  93. package/test/scope.test.ts +8 -8
  94. package/test/select.test.ts +216 -257
  95. package/test/skills.test.ts +3 -3
  96. package/test/sql-shim.test.ts +154 -0
  97. package/test/state.test.ts +43 -36
  98. package/test/summary.test.ts +38 -27
  99. package/test/tier-plan.test.ts +45 -62
  100. package/test/toast-logic.test.ts +31 -31
  101. package/test/tokens.test.ts +95 -80
  102. package/test/trust-attribution.test.ts +217 -187
  103. package/test/trust-window.test.ts +37 -32
  104. package/test/turn.test.ts +55 -23
  105. package/test/upstreams.test.ts +13 -13
  106. package/test/views.test.ts +81 -59
  107. package/test/wire-request.test.ts +17 -17
  108. package/test/wire-responses.test.ts +4 -4
  109. package/tools/agentdox-e2e.ts +5 -2
  110. package/tools/export-benchmarks.ts +5 -5
  111. package/tools/ledger-parity.ts +266 -0
  112. package/tools/replay.ts +16 -8
package/src/cost/views.ts CHANGED
@@ -5,12 +5,13 @@
5
5
  * cost export by day, harness and model. Served by `/v1/router/spend`,
6
6
  * `GET /v1/router/feedback` and `/v1/router/export`, exported from lib.ts, and
7
7
  * behind `auto-model-router export`. All of them read only long-stable ledger
8
- * columns and accept a read-only database handle.
8
+ * columns, and they run on either engine through `util/sql.ts`: a front door
9
+ * may be reading a local file while the router writes a shared database.
9
10
  */
10
11
 
11
12
  import { providerOfSlug } from "./report.ts";
12
- import type { Database } from "bun:sqlite";
13
- import { toEntry, type LedgerRow } from "./ledger.ts";
13
+ import { num, type SqlDb } from "../util/sql.ts";
14
+ import { entriesOf } from "./ledger-sql.ts";
14
15
  import { harnessFilter } from "./report.ts";
15
16
  import type { LedgerEntry } from "./types.ts";
16
17
 
@@ -89,7 +90,7 @@ export interface DecisionFilter {
89
90
  * reasons, the classifier's view, the cost forecast against the bill, the escalation signal —
90
91
  * and the verdicts `/router good|bad` recorded against each turn ride along.
91
92
  */
92
- export function decisionEntries(db: Database, filter: DecisionFilter): DecisionEntry[] {
93
+ export async function decisionEntries(db: SqlDb, filter: DecisionFilter): Promise<DecisionEntry[]> {
93
94
  const s = scope(filter.harness, "harness_id");
94
95
  if (s === null) return [];
95
96
  const where = ["created_at_ms >= $since", ...s.sql];
@@ -107,20 +108,26 @@ export function decisionEntries(db: Database, filter: DecisionFilter): DecisionE
107
108
  bind.$session = filter.ompSessionId;
108
109
  }
109
110
  const limit = Math.min(Math.max(filter.limit ?? 50, 1), 1_000);
110
- const rows = db.query(`SELECT * FROM ledger WHERE ${where.join(" AND ")} ORDER BY created_at_ms DESC LIMIT ${limit}`).all(bind) as LedgerRow[];
111
- const entries = rows.map(toEntry);
111
+ const rows = await db.query<unknown>(
112
+ `SELECT * FROM ledger WHERE ${where.join(" AND ")} ORDER BY created_at_ms DESC LIMIT ${limit}`,
113
+ bind,
114
+ );
115
+ const entries = entriesOf(rows);
112
116
  if (entries.length === 0) return [];
113
117
  // Verdicts, when the feedback table exists (it does not on a ledger no one has judged).
114
- const hasFeedback = (db.query("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'feedback'").get() as { name: string } | null) !== null;
115
118
  const verdicts = new Map<string, DecisionEntry["feedback"]>();
116
- if (hasFeedback) {
119
+ if (await db.tableExists("feedback")) {
117
120
  const ids = entries.map((e) => e.id);
118
121
  const marks = ids.map((_, i) => `$f${i}`).join(", ");
119
122
  const fb: Record<string, string> = {};
120
123
  ids.forEach((id, i) => (fb[`$f${i}`] = id));
121
- for (const r of db.query(`SELECT ledger_id, verdict, note, created_at_ms FROM feedback WHERE ledger_id IN (${marks}) ORDER BY created_at_ms ASC`).all(fb) as { ledger_id: string; verdict: string; note: string; created_at_ms: number }[]) {
124
+ const rows = await db.query<{ ledger_id: string; verdict: string; note: string; created_at_ms: number }>(
125
+ `SELECT ledger_id, verdict, note, created_at_ms FROM feedback WHERE ledger_id IN (${marks}) ORDER BY created_at_ms ASC`,
126
+ fb,
127
+ );
128
+ for (const r of rows) {
122
129
  const list = verdicts.get(r.ledger_id) ?? [];
123
- list.push({ verdict: r.verdict === "good" ? "good" : "bad", note: r.note, createdAtMs: r.created_at_ms });
130
+ list.push({ verdict: r.verdict === "good" ? "good" : "bad", note: r.note, createdAtMs: num(r.created_at_ms) });
124
131
  verdicts.set(r.ledger_id, list);
125
132
  }
126
133
  }
@@ -132,7 +139,7 @@ export function decisionEntries(db: Database, filter: DecisionFilter): DecisionE
132
139
  * as the ledger counts them. `contextScope`, when given, narrows to the turns that carried
133
140
  * exactly that agentdox scope — what a front door charges back to one project.
134
141
  */
135
- export function spendUsdSince(db: Database, sinceMs: number, harness: HarnessScope, contextScope?: string): number {
142
+ export async function spendUsdSince(db: SqlDb, sinceMs: number, harness: HarnessScope, contextScope?: string): Promise<number> {
136
143
  const s = scope(harness, "harness_id");
137
144
  if (s === null) return 0;
138
145
  const where = ["created_at_ms >= $since", ...s.sql];
@@ -141,36 +148,42 @@ export function spendUsdSince(db: Database, sinceMs: number, harness: HarnessSco
141
148
  where.push("scope = $scope");
142
149
  bind.$scope = contextScope;
143
150
  }
144
- const row = db.query(`SELECT COALESCE(SUM(${USD}), 0) AS usd FROM ledger WHERE ${where.join(" AND ")}`).get(bind) as { usd: number };
145
- return row.usd;
151
+ const row = await db.one<{ usd: unknown }>(
152
+ `SELECT COALESCE(SUM(${USD}), 0) AS usd FROM ledger WHERE ${where.join(" AND ")}`,
153
+ bind,
154
+ );
155
+ return num(row?.usd);
146
156
  }
147
157
 
148
158
  /** Verdicts since `sinceMs`, by model and the most recent 200, joined to the ledger for the judging harness. */
149
- export function feedbackView(db: Database, sinceMs: number, harness: HarnessScope): FeedbackView {
150
- const exists = (db.query("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'feedback'").get() as { name: string } | null) !== null;
151
- if (!exists) return { byModel: [], recent: [] };
159
+ export async function feedbackView(db: SqlDb, sinceMs: number, harness: HarnessScope): Promise<FeedbackView> {
160
+ if (!(await db.tableExists("feedback"))) return { byModel: [], recent: [] };
152
161
  const s = scope(harness, "l.harness_id");
153
162
  if (s === null) return { byModel: [], recent: [] };
154
163
  const where = ["f.created_at_ms >= $since", ...s.sql].join(" AND ");
155
164
  const bind = { $since: sinceMs, ...s.bind };
156
165
  const recent = (
157
- db.query(`SELECT f.created_at_ms AS at_ms, f.slug, f.tier, f.verdict, f.note, COALESCE(l.harness_id, '') AS harness_id FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} ORDER BY f.created_at_ms DESC LIMIT 200`).all(bind) as {
158
- at_ms: number;
159
- slug: string;
160
- tier: string;
161
- verdict: string;
162
- note: string;
163
- harness_id: string;
164
- }[]
165
- ).map((r) => ({ atMs: r.at_ms, slug: r.slug, tier: r.tier, verdict: (r.verdict === "good" ? "good" : "bad") as "good" | "bad", note: r.note, harnessId: r.harness_id }));
166
+ await db.query<{ at_ms: number; slug: string; tier: string; verdict: string; note: string; harness_id: string }>(
167
+ `SELECT f.created_at_ms AS at_ms, f.slug, f.tier, f.verdict, f.note, COALESCE(l.harness_id, '') AS harness_id
168
+ FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} ORDER BY f.created_at_ms DESC LIMIT 200`,
169
+ bind,
170
+ )
171
+ ).map((r) => ({
172
+ atMs: num(r.at_ms),
173
+ slug: r.slug,
174
+ tier: r.tier,
175
+ verdict: (r.verdict === "good" ? "good" : "bad") as "good" | "bad",
176
+ note: r.note,
177
+ harnessId: r.harness_id,
178
+ }));
166
179
  const byModel = (
167
- db
168
- .query(
169
- `SELECT f.slug, SUM(CASE WHEN f.verdict = 'good' THEN 1 ELSE 0 END) AS good, SUM(CASE WHEN f.verdict = 'bad' THEN 1 ELSE 0 END) AS bad, COUNT(DISTINCT COALESCE(l.harness_id, '')) AS judges
170
- FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} GROUP BY f.slug ORDER BY bad DESC, good DESC, f.slug ASC`,
171
- )
172
- .all(bind) as { slug: string; good: number; bad: number; judges: number }[]
173
- ).map((r) => ({ slug: r.slug, good: r.good, bad: r.bad, judges: r.judges }));
180
+ await db.query<{ slug: string; good: unknown; bad: unknown; judges: unknown }>(
181
+ `SELECT f.slug, SUM(CASE WHEN f.verdict = 'good' THEN 1 ELSE 0 END) AS good, SUM(CASE WHEN f.verdict = 'bad' THEN 1 ELSE 0 END) AS bad,
182
+ COUNT(DISTINCT COALESCE(l.harness_id, '')) AS judges
183
+ FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} GROUP BY f.slug ORDER BY bad DESC, good DESC, f.slug ASC`,
184
+ bind,
185
+ )
186
+ ).map((r) => ({ slug: r.slug, good: num(r.good), bad: num(r.bad), judges: num(r.judges) }));
174
187
  return { byModel, recent };
175
188
  }
176
189
 
@@ -180,36 +193,53 @@ export function feedbackView(db: Database, sinceMs: number, harness: HarnessScop
180
193
  * can charge each project its own share; rows from before v18 (and turns that carried no
181
194
  * scope) group under "".
182
195
  */
183
- export function exportRows(db: Database, sinceMs: number, harness: HarnessScope): ExportRow[] {
196
+ export async function exportRows(db: SqlDb, sinceMs: number, harness: HarnessScope): Promise<ExportRow[]> {
184
197
  const s = scope(harness, "harness_id");
185
198
  if (s === null) return [];
186
199
  const where = ["created_at_ms >= $since", "requested_model <> 'digest'", ...s.sql].join(" AND ");
187
- const rows = db
188
- .query(
189
- `SELECT strftime('%Y-%m-%d', created_at_ms / 1000, 'unixepoch') AS day, harness_id, COALESCE(served_slug, slug) AS slug, COALESCE(scope, '') AS scope,
200
+ // GROUP BY names the day expression rather than its alias: Postgres does not
201
+ // allow a select alias in GROUP BY, and repeating it keeps one statement for
202
+ // both engines.
203
+ const day = db.utcDay("created_at_ms");
204
+ const rows = await db.query<{
205
+ day: string;
206
+ harness_id: string;
207
+ slug: string;
208
+ scope: string;
209
+ dispatches: unknown;
210
+ prompt_tokens: unknown;
211
+ cached_tokens: unknown;
212
+ completion_tokens: unknown;
213
+ spend: unknown;
214
+ escalations: unknown;
215
+ errors: unknown;
216
+ }>(
217
+ `SELECT ${day} AS day, harness_id, COALESCE(served_slug, slug) AS slug, COALESCE(scope, '') AS scope,
190
218
  COUNT(*) AS dispatches,
191
- COALESCE(SUM(json_extract(usage, '$.promptTokens')), 0) AS prompt_tokens,
192
- COALESCE(SUM(json_extract(usage, '$.cachedTokens')), 0) AS cached_tokens,
193
- COALESCE(SUM(json_extract(usage, '$.completionTokens')), 0) AS completion_tokens,
219
+ COALESCE(SUM(${db.jsonNum("usage", "promptTokens")}), 0) AS prompt_tokens,
220
+ COALESCE(SUM(${db.jsonNum("usage", "cachedTokens")}), 0) AS cached_tokens,
221
+ COALESCE(SUM(${db.jsonNum("usage", "completionTokens")}), 0) AS completion_tokens,
194
222
  COALESCE(SUM(${USD}), 0) AS spend,
195
223
  SUM(CASE WHEN escalation_signal IS NOT NULL THEN 1 ELSE 0 END) AS escalations,
196
224
  SUM(CASE WHEN error IS NOT NULL THEN 1 ELSE 0 END) AS errors
197
- FROM ledger WHERE ${where} GROUP BY day, harness_id, slug, scope ORDER BY day ASC, harness_id ASC, spend DESC`,
198
- )
199
- .all({ $since: sinceMs, ...s.bind }) as { day: string; harness_id: string; slug: string; scope: string; dispatches: number; prompt_tokens: number; cached_tokens: number; completion_tokens: number; spend: number; escalations: number; errors: number }[];
225
+ FROM ledger WHERE ${where}
226
+ GROUP BY ${day}, harness_id, COALESCE(served_slug, slug), COALESCE(scope, '')
227
+ ORDER BY 1 ASC, harness_id ASC, spend DESC`,
228
+ { $since: sinceMs, ...s.bind },
229
+ );
200
230
  return rows.map((r) => ({
201
231
  day: r.day,
202
232
  harnessId: r.harness_id,
203
233
  slug: r.slug,
204
234
  scope: r.scope,
205
235
  provider: providerOfSlug(r.slug),
206
- dispatches: r.dispatches,
207
- promptTokens: r.prompt_tokens,
208
- cachedTokens: r.cached_tokens,
209
- completionTokens: r.completion_tokens,
210
- spendUsd: r.spend,
211
- escalations: r.escalations,
212
- errors: r.errors,
236
+ dispatches: num(r.dispatches),
237
+ promptTokens: num(r.prompt_tokens),
238
+ cachedTokens: num(r.cached_tokens),
239
+ completionTokens: num(r.completion_tokens),
240
+ spendUsd: num(r.spend),
241
+ escalations: num(r.escalations),
242
+ errors: num(r.errors),
213
243
  }));
214
244
  }
215
245
 
@@ -58,13 +58,26 @@ export interface AnchorPoint {
58
58
  }
59
59
 
60
60
  /**
61
- * OLS fit, or null when the calibration cannot be trusted: too few points, no
62
- * spread, a non-positive slope, or weak correlation. A suite that does not track
63
- * AA positively and with real correlation would turn a target's score into noise
64
- * dressed as signal, so we refuse it and emit nothing for that axis.
61
+ * The raw range the anchors must actually cover before a line through them means anything.
62
+ *
63
+ * Correlation alone does NOT catch a compressed fit: three anchors published 20/50/80 that
64
+ * our suite scores 0.96/0.97/0.98 correlate at r = 1.0, and the line they define has a slope
65
+ * of ~3000 index points per unit of raw score. That fit is arithmetically perfect and
66
+ * completely useless — it is how a model published at 39.5 was calibrated to 22.8. If the
67
+ * anchors barely differ on our suite, our suite cannot place anything between them.
68
+ */
69
+ export const MIN_RAW_SPREAD = 0.15;
70
+
71
+ /**
72
+ * OLS fit, or null when the calibration cannot be trusted: too few points, too narrow a raw
73
+ * range, no spread, a non-positive slope, or weak correlation. A suite that does not track
74
+ * AA positively, with real correlation, over a real range would turn a target's score into
75
+ * noise dressed as signal, so we refuse it and emit nothing for that axis.
65
76
  */
66
77
  export function fitAxis(points: readonly AnchorPoint[]): LineFit | null {
67
78
  if (points.length < MIN_ANCHORS) return null;
79
+ const raws = points.map((p) => p.raw);
80
+ if (Math.max(...raws) - Math.min(...raws) < MIN_RAW_SPREAD) return null;
68
81
  const n = points.length;
69
82
  let sx = 0;
70
83
  let sy = 0;
@@ -101,19 +114,32 @@ export function applyFit(fit: LineFit, raw: number): number {
101
114
  const AXES: readonly QualityAxis[] = ["coding", "intelligence", "agentic"];
102
115
 
103
116
  /**
104
- * Fit every axis from the anchors' raw suite scores paired with their known AA
105
- * scores. `anchorAa` supplies the AA index per slug+axis (absent ⇒ that anchor
106
- * is not used on that axis).
117
+ * The correlation a fit must reach before its numbers are published as scores. `MIN_R` (0.5)
118
+ * is the bar for a fit being computable at all; this is the bar for TRUSTING one. The fit's
119
+ * `r` and `n` were previously computed and then discarded, so an r of 0.51 and one of 0.99
120
+ * produced indistinguishable output — and a shallow fit silently compressed every target.
121
+ */
122
+ export const PUBLISH_MIN_R = 0.8;
123
+
124
+ /**
125
+ * Fit every axis from the anchors' raw suite scores paired with their known AA scores.
126
+ * `anchorAa` supplies the AA index per slug+axis (absent ⇒ that anchor is not used there).
127
+ *
128
+ * `rawOf` selects which observations the fit is built from. It defaults to every band, but
129
+ * callers should pass the HARD band: easy and moderate sit at ~1.0 for every model worth
130
+ * ranking, so including them leaves the regression almost no variation in x against a wide
131
+ * spread in published y, and the slope collapses toward flat.
107
132
  */
108
133
  export function fitCalibration(
109
134
  anchors: readonly EvalResult[],
110
135
  anchorAa: (slug: string, axis: QualityAxis) => number | undefined,
136
+ rawOf: (result: EvalResult, axis: QualityAxis) => number | null = (r, axis) => axisMean(r.axes[axis]),
111
137
  ): Calibration {
112
138
  const cal: Calibration = {};
113
139
  for (const axis of AXES) {
114
140
  const points: AnchorPoint[] = [];
115
141
  for (const r of anchors) {
116
- const raw = axisMean(r.axes[axis]);
142
+ const raw = rawOf(r, axis);
117
143
  const aa = anchorAa(r.slug, axis);
118
144
  if (raw !== null && aa !== undefined) points.push({ raw, aa });
119
145
  }
@@ -123,15 +149,24 @@ export function fitCalibration(
123
149
  return cal;
124
150
  }
125
151
 
126
- function axisMean(a: AxisScore | undefined): number | null {
152
+ export function axisMean(a: AxisScore | undefined): number | null {
127
153
  return a === undefined || a.n === 0 ? null : a.sum / a.n;
128
154
  }
129
155
 
130
- /** Calibrated local FeedScores for the targets, one axis at a time, skipping axes with no fit. */
156
+ /** The hard band alone, for calibration. */
157
+ export const hardRaw = (r: EvalResult, axis: QualityAxis): number | null => axisMean(r.axesHard[axis]);
158
+
159
+ /**
160
+ * Calibrated local FeedScores for the targets, one axis at a time. An axis is skipped when it
161
+ * has no fit, when the fit is weaker than `minR`, or when the target made no observation on
162
+ * it — honest silence rather than a number nobody should act on.
163
+ */
131
164
  export function toLocalFeedScores(
132
165
  targets: readonly EvalResult[],
133
166
  cal: Calibration,
134
167
  authorOf: (slug: string) => string,
168
+ rawOf: (result: EvalResult, axis: QualityAxis) => number | null = (r, axis) => axisMean(r.axes[axis]),
169
+ minR = PUBLISH_MIN_R,
135
170
  ): FeedScore[] {
136
171
  const out: FeedScore[] = [];
137
172
  for (const r of targets) {
@@ -139,8 +174,8 @@ export function toLocalFeedScores(
139
174
  let any = false;
140
175
  for (const axis of AXES) {
141
176
  const fit = cal[axis];
142
- const raw = axisMean(r.axes[axis]);
143
- if (fit === undefined || raw === null) continue;
177
+ const raw = rawOf(r, axis);
178
+ if (fit === undefined || raw === null || fit.r < minR) continue;
144
179
  entry[axis] = applyFit(fit, raw);
145
180
  any = true;
146
181
  }
package/src/eval/run.ts CHANGED
@@ -39,6 +39,14 @@ export interface EvalResult {
39
39
  * questions were easy.
40
40
  */
41
41
  byComplexity: Partial<Record<Complexity, AxisScore>>;
42
+ /**
43
+ * Per-axis scores from the HARD band alone. Calibration fits against these: the easy and
44
+ * moderate bands sit at ~1.0 for every model worth ranking, so including them gives the
45
+ * regression almost no variation in x against a wide spread in published y — the fitted
46
+ * slope goes shallow and every target is dragged toward the middle. Measured: a model
47
+ * published at intelligence 39.5 calibrated to 22.8 across 10 passes.
48
+ */
49
+ axesHard: Record<QualityAxis, AxisScore>;
42
50
  /**
43
51
  * Per-axis spread across passes: max pass mean minus min pass mean, or null under two
44
52
  * passes. A wide spread means the headline is one sample of a noisy quantity, and is the
@@ -144,6 +152,7 @@ async function scorePass(slug: string, args: RunEvalArgs): Promise<EvalResult> {
144
152
  }
145
153
  }
146
154
  const byComplexity: Partial<Record<Complexity, AxisScore>> = {};
155
+ const axesHard: Record<QualityAxis, AxisScore> = { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } };
147
156
  for (const o of [...objective, ...judgedOutcomes, ...scenarioOutcomes]) {
148
157
  if (!o.ok) {
149
158
  // An unobserved task is NOT a zero: a provider's throttle or outage would otherwise
@@ -154,11 +163,15 @@ async function scorePass(slug: string, args: RunEvalArgs): Promise<EvalResult> {
154
163
  }
155
164
  axes[o.axis].sum += o.grade;
156
165
  axes[o.axis].n += 1;
166
+ if (o.complexity === "hard") {
167
+ axesHard[o.axis].sum += o.grade;
168
+ axesHard[o.axis].n += 1;
169
+ }
157
170
  const band = (byComplexity[o.complexity] ??= { sum: 0, n: 0 });
158
171
  band.sum += o.grade;
159
172
  band.n += 1;
160
173
  }
161
- return { slug, axes, errors, repeats: 1, spread: {}, byComplexity };
174
+ return { slug, axes, errors, repeats: 1, spread: {}, byComplexity, axesHard };
162
175
  }
163
176
 
164
177
  const AXES: readonly QualityAxis[] = ["coding", "intelligence", "agentic"];
@@ -174,6 +187,7 @@ async function scoreModel(slug: string, args: RunEvalArgs): Promise<EvalResult>
174
187
  const axes: Record<QualityAxis, AxisScore> = { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } };
175
188
  const means: Record<QualityAxis, number[]> = { coding: [], intelligence: [], agentic: [] };
176
189
  const byComplexity: Partial<Record<Complexity, AxisScore>> = {};
190
+ const axesHard: Record<QualityAxis, AxisScore> = { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } };
177
191
  let errors = 0;
178
192
  for (let i = 0; i < passes; i++) {
179
193
  const pass = await scorePass(slug, args);
@@ -182,6 +196,8 @@ async function scoreModel(slug: string, args: RunEvalArgs): Promise<EvalResult>
182
196
  for (const axis of AXES) {
183
197
  axes[axis].sum += pass.axes[axis].sum;
184
198
  axes[axis].n += pass.axes[axis].n;
199
+ axesHard[axis].sum += pass.axesHard[axis].sum;
200
+ axesHard[axis].n += pass.axesHard[axis].n;
185
201
  if (pass.axes[axis].n > 0) means[axis].push(pass.axes[axis].sum / pass.axes[axis].n);
186
202
  }
187
203
  for (const [band, score] of Object.entries(pass.byComplexity) as [Complexity, AxisScore][]) {
@@ -195,7 +211,7 @@ async function scoreModel(slug: string, args: RunEvalArgs): Promise<EvalResult>
195
211
  const m = means[axis];
196
212
  if (m.length > 1) spread[axis] = Math.max(...m) - Math.min(...m);
197
213
  }
198
- return { slug, axes, errors, repeats: passes, spread, byComplexity };
214
+ return { slug, axes, errors, repeats: passes, spread, byComplexity, axesHard };
199
215
  }
200
216
 
201
217
  export async function runEval(args: RunEvalArgs): Promise<EvalResult[]> {
package/src/lib.ts CHANGED
@@ -25,14 +25,18 @@ export type { DeepPartial } from "./config/load.ts";
25
25
  export { buildUsageReport, renderUsageReport, type UsageReport, type ReportTotals } from "./cost/report.ts";
26
26
  export { buildDailySummary, renderDailySummary, type DailySummary } from "./cost/summary.ts";
27
27
  export { openDb } from "./util/sqlite.ts";
28
+ // A front door reads the ledger through the engine-agnostic handle: the store
29
+ // may be a file or a shared database, and the view functions take this.
30
+ export { dialectOf, openSqlDb, num, numOrNull, type Dialect, type SqlDb } from "./util/sql.ts";
28
31
  export { spendUsdSince, feedbackView, exportRows, exportCsv, decisionEntries, harnessScopeParam, type HarnessScope, type ExportRow, type FeedbackRow, type FeedbackByModel, type FeedbackView, type DecisionEntry, type DecisionFilter } from "./cost/views.ts";
29
- export { createLedger } from "./cost/ledger.ts";
32
+ export { createSqlLedger } from "./cost/ledger-sql.ts";
33
+ export { migrateStore, STORE_TABLES } from "./util/schema.ts";
30
34
  export { createFeedbackStore, type FeedbackStore, type FeedbackRecord } from "./cost/feedback.ts";
31
35
  export { buildExecutable, collectPackageFiles, executableFileName, hostTarget, isExecutableTarget, EXECUTABLE_TARGETS, type ExecutableTarget, type BuildExecutableResult } from "./cli/build-executable.ts";
32
36
  export { parseSkillsBundle, type SkillsBundle } from "./cli/skills.ts";
33
37
  export type { RequestPolicy } from "./wire/types.ts";
34
38
  export type { CatalogView, CatalogViewModel } from "./server/catalog-view.ts";
35
- export type { Ledger, LedgerEntry, PruneResult } from "./cost/types.ts";
39
+ export type { AsyncLedger, LedgerEntry, PruneResult } from "./cost/types.ts";
36
40
  // Retention: a front door asks through `POST /v1/router/prune` rather than
37
41
  // deleting from the ledger itself. The interval is exported so it can say when.
38
42
  export { RETENTION_INTERVAL_MS } from "./cost/retention.ts";
@@ -7,7 +7,7 @@
7
7
  import type { CatalogModel, CatalogSnapshot } from "../catalog/types.ts";
8
8
  import type { FilterConfig, QualityAxis, RouterConfig } from "../config/types.ts";
9
9
  import { forecast, priceAt } from "../cost/forecast.ts";
10
- import type { Ledger, LedgerSignals, ModelLatency } from "../cost/types.ts";
10
+ import type { LedgerSignals, ModelLatency } from "../cost/types.ts";
11
11
  import type { NormRequest } from "../wire/types.ts";
12
12
  import { effectivePriceCeiling, effectiveQualityFloor, tierPlanFor } from "./tier-plan.ts";
13
13
  import type { Candidate, Features, Rejection, TaskType, Tier } from "./types.ts";
@@ -27,7 +27,6 @@ export interface BuildCandidatesArgs {
27
27
  /** Task type; its config selects the axis, quality floor, and image filter. */
28
28
  task: TaskType;
29
29
  snapshot: CatalogSnapshot;
30
- ledger: Ledger | null;
31
30
  cfg: RouterConfig;
32
31
  expectedCompletionTokens: number;
33
32
  /** Slug whose prompt cache is warm this turn; wins score ties. */
@@ -154,7 +153,7 @@ function latencyMultiplier(latency: ModelLatency | null, filters: FilterConfig,
154
153
  }
155
154
 
156
155
  export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candidate[]; rejected: Rejection[] } {
157
- const { req, features, tier, task, snapshot, ledger, cfg, expectedCompletionTokens, warmSlug, relaxLevel = 0 } = args;
156
+ const { req, features, tier, task, snapshot, cfg, expectedCompletionTokens, warmSlug, relaxLevel = 0 } = args;
158
157
  // A Set only when non-empty: the common path allocates nothing.
159
158
  const excluded = args.excludeSlugs === undefined || args.excludeSlugs.length === 0 ? null : new Set(args.excludeSlugs);
160
159
  const tierCfg = cfg.tiers[tier];
@@ -318,14 +317,11 @@ export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candid
318
317
  continue;
319
318
  }
320
319
 
321
- // Use pre-fetched signals when available (batch lookup, one query per
322
- // signal kind for the entire candidate set instead of one per model).
323
- // Falls back to per-slug calls when signals is not provided (e.g. tests).
320
+ // Signals are prefetched by `route` — one query per signal kind for the
321
+ // whole candidate set. Absent means "the ledger knows nothing about this
322
+ // model", which is what a cold start looks like and is handled below.
324
323
  const signals = args.signals;
325
- const trust =
326
- signals?.get(slug)?.trust ??
327
- ledger?.trust(slug, filters.trustScopedByHarness ? req.harnessId : undefined, filters.feedbackByTask ? task : undefined) ??
328
- null;
324
+ const trust = signals?.get(slug)?.trust ?? null;
329
325
  if (!relaxTrust && trust !== null && trust.attempts >= filters.minTrustSamples && trust.successRate < filters.minTrust) {
330
326
  rejected.push({
331
327
  slug,
@@ -342,11 +338,7 @@ export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candid
342
338
  // gate can. Only measured models are dropped, so a new model still gets its
343
339
  // cold-start turns to accumulate samples. Relaxed with trust in rescue.
344
340
  const needLatency = filters.latencyWeight > 0 || filters.maxExpectedWaitMs !== undefined;
345
- const latency = needLatency
346
- ? (signals?.get(slug)?.latency ??
347
- ledger?.latency(slug, filters.trustScopedByHarness ? req.harnessId : undefined) ??
348
- null)
349
- : null;
341
+ const latency = needLatency ? (signals?.get(slug)?.latency ?? null) : null;
350
342
  if (
351
343
  !relaxTrust &&
352
344
  filters.maxExpectedWaitMs !== undefined &&
@@ -6,7 +6,7 @@
6
6
  import type { CatalogSource } from "../catalog/types.ts";
7
7
  import type { QualityAxis, RouterConfig } from "../config/types.ts";
8
8
  import { forecast } from "../cost/forecast.ts";
9
- import type { Ledger } from "../cost/types.ts";
9
+ import type { AsyncLedger } from "../cost/types.ts";
10
10
  import { estimateTokens } from "../tokens/estimate.ts";
11
11
  import type { UpstreamClient } from "../upstream/types.ts";
12
12
  import { sha256Hex } from "../util/hash.ts";
@@ -219,7 +219,7 @@ export function classifyTask(f: Features): TaskType {
219
219
 
220
220
  export interface ClassifyDeps {
221
221
  upstream: UpstreamClient;
222
- ledger: Ledger | null;
222
+ ledger: AsyncLedger | null;
223
223
  catalog: CatalogSource | null;
224
224
  }
225
225
 
@@ -334,11 +334,13 @@ export async function classify(
334
334
  }
335
335
 
336
336
  // Cost guard: adjudication must be cheap relative to the turn it classifies.
337
- const blend = deps.ledger?.blendedRate(cfg.ledger.blendWindowDays) ?? null;
337
+ const blend = (await deps.ledger?.blendedRate(cfg.ledger.blendWindowDays)) ?? null;
338
338
  const inputRate = (blend?.inputPerMtok ?? cfg.ledger.fallbackBlend.inputPerMtok) / 1e6;
339
339
  const outputRate = (blend?.outputPerMtok ?? cfg.ledger.fallbackBlend.outputPerMtok) / 1e6;
340
340
  const judge = deps.catalog?.find(cc.model);
341
- const digestTokens = estimateTokens(Buffer.byteLength(ADJUDICATOR_SYSTEM) + Buffer.byteLength(digest), "unknown", deps.ledger);
341
+ // The adjudicator prompt is short and its family unknown, so the default
342
+ // family ratio is as good as a measured one here.
343
+ const digestTokens = estimateTokens(Buffer.byteLength(ADJUDICATOR_SYSTEM) + Buffer.byteLength(digest), "unknown", null);
342
344
  const adjudicatorUsd =
343
345
  judge !== undefined
344
346
  ? forecast(judge, {
@@ -10,23 +10,25 @@
10
10
  * real request without perturbing the conversation it belongs to.
11
11
  */
12
12
 
13
- import type { CatalogSource } from "../catalog/types.ts";
13
+ import type { CatalogSnapshot, CatalogSource } from "../catalog/types.ts";
14
14
  import type { ProfileConfig, RouterConfig } from "../config/types.ts";
15
- import type { Ledger } from "../cost/types.ts";
15
+ import { type AsyncLedger, type LedgerReader } from "../cost/types.ts";
16
16
  import { estimatePromptTokens } from "../tokens/estimate.ts";
17
17
  import type { UpstreamClient } from "../upstream/types.ts";
18
+ import { modelNotFound } from "../wire/openai/errors.ts";
18
19
  import type { NormRequest, RequestPolicy } from "../wire/types.ts";
19
20
  import { classify, classifyTask } from "./classify.ts";
20
21
  import { extractFeatures } from "./features.ts";
21
- import { select } from "./select.ts";
22
+ import { monthStartMs, select, type TurnReads } from "./select.ts";
22
23
  import { TIER_ORDER, type Classification, type ConversationStore, type Decision, type Router, type Tier } from "./types.ts";
23
-
24
24
  export interface RouterDeps {
25
25
  config: RouterConfig;
26
26
  catalog: CatalogSource;
27
- ledger: Ledger;
27
+ ledger: AsyncLedger;
28
28
  conversations: ConversationStore;
29
29
  upstream: UpstreamClient;
30
+ /** Async ledger reads; defaults to reading `ledger` directly. */
31
+ reader?: LedgerReader;
30
32
  }
31
33
 
32
34
  /**
@@ -90,20 +92,89 @@ export function resolveProfile(cfg: RouterConfig, requestedModel: string, isSuba
90
92
  return fallback;
91
93
  }
92
94
 
95
+ /**
96
+ * The `model` a client names is a profile id. When it is neither a profile nor
97
+ * a catalog slug, routing something else and reporting the asked-for name back
98
+ * is a silent substitution: the caller is billed for a model it never chose.
99
+ * A real slug becomes a pin (absolute, the way a session override is); anything
100
+ * else is refused.
101
+ *
102
+ * @param asked the client's `model` before the provider prefix was stripped.
103
+ * @returns the slug to pin, or undefined when a profile matched.
104
+ */
105
+ export function pinForRequestedModel(
106
+ cfg: RouterConfig,
107
+ slugs: readonly string[],
108
+ requestedModel: string,
109
+ asked: string,
110
+ ): string | undefined {
111
+ if (cfg.profiles.some((p) => p.id === requestedModel)) return undefined;
112
+ if (slugs.includes(asked)) return asked;
113
+ throw modelNotFound(asked);
114
+ }
115
+
116
+ /**
117
+ * Reads the ledger for one turn, concurrently, before selection runs.
118
+ *
119
+ * Only what this turn's configuration actually consults is fetched: month and
120
+ * day spend are skipped when no such budget is set, and the escalation-cost
121
+ * term is skipped when its weight is 0. Empty catalog ⇒ no signal queries at
122
+ * all, which is what keeps the tests' fakes cheap.
123
+ */
124
+ export async function prefetchTurnReads(
125
+ reader: LedgerReader | null,
126
+ req: NormRequest,
127
+ profile: ProfileConfig,
128
+ cfg: RouterConfig,
129
+ snapshot: CatalogSnapshot,
130
+ task: string,
131
+ nowMs: number = Date.now(),
132
+ ): Promise<TurnReads> {
133
+ if (reader === null) return {};
134
+ const slugs = snapshot.models.map((m) => m.slug);
135
+ const harness = cfg.filters.trustScopedByHarness ? req.harnessId : undefined;
136
+ const perMonthUsd = profile.budget?.perMonthUsd ?? cfg.budget.perMonthUsd;
137
+ const perDayUsd = profile.budget?.perDayUsd ?? cfg.budget.perDayUsd;
138
+ const [signals, cacheReliability, escalation, monthSpendUsd, daySpendUsd] = await Promise.all([
139
+ slugs.length > 0 ? reader.signals(slugs, harness, cfg.filters.feedbackByTask ? task : undefined) : undefined,
140
+ slugs.length > 0 && cfg.filters.cacheReliabilityMinSamples > 0 ? reader.cacheReliability(slugs) : undefined,
141
+ cfg.filters.escalationCostWeight > 0 ? reader.escalationCost(cfg.ledger.blendWindowDays) : undefined,
142
+ // Month pacing can tighten the daily ceiling, so month spend is needed
143
+ // whenever either budget is set.
144
+ perMonthUsd !== undefined ? reader.spendSince(monthStartMs(nowMs), req.harnessId) : undefined,
145
+ perDayUsd !== undefined || perMonthUsd !== undefined ? reader.spendSince(nowMs - 86_400_000, req.harnessId) : undefined,
146
+ ]);
147
+ return {
148
+ ...(signals === undefined ? {} : { signals }),
149
+ ...(cacheReliability === undefined ? {} : { cacheReliability }),
150
+ ...(escalation === undefined ? {} : { escalationUsdPerPromptToken: escalation?.usdPerPromptToken ?? null }),
151
+ ...(monthSpendUsd === undefined ? {} : { monthSpendUsd }),
152
+ ...(daySpendUsd === undefined ? {} : { daySpendUsd }),
153
+ };
154
+ }
155
+
93
156
  export function createRouter(deps: RouterDeps): Router {
94
157
  const { config, catalog, ledger, conversations, upstream } = deps;
158
+ // A Postgres ledger supplies its own reader; a local one is wrapped, so the
159
+ // prefetch path is identical for both.
160
+ // The unified ledger IS a reader: `LedgerReader` is the narrow half of it.
161
+ const reader = deps.reader ?? ledger;
95
162
 
96
163
  return {
97
164
  async route(
98
165
  req: NormRequest,
99
166
  opts: { attempt: number; escalateFrom?: Tier; excludeSlugs?: readonly string[]; forceTier?: Tier; forceSlug?: string },
100
167
  ): Promise<Decision> {
101
- const state = conversations.get(req.conversationKey) ?? conversations.load(req.conversationKey);
168
+ const state = (await conversations.get(req.conversationKey)) ?? (await conversations.load(req.conversationKey));
102
169
  const snapshot = await catalog.get();
103
170
 
104
171
  const priorTokenizer =
105
172
  state.currentSlug === null ? undefined : catalog.find(state.currentSlug)?.tokenizer;
106
- const promptTokens = estimatePromptTokens(req, priorTokenizer ?? NEUTRAL_TOKENIZER, ledger);
173
+ const tokenizer = priorTokenizer ?? NEUTRAL_TOKENIZER;
174
+ // One ratio, fetched before estimating: the estimate itself runs in
175
+ // synchronous code that a shared store cannot be read from.
176
+ const ratio = await ledger.tokenRatio(tokenizer);
177
+ const promptTokens = estimatePromptTokens(req, tokenizer, ratio);
107
178
  const features = extractFeatures(req, promptTokens);
108
179
 
109
180
  let classification: Classification;
@@ -136,7 +207,22 @@ export function createRouter(deps: RouterDeps): Router {
136
207
  classification = await classify(req, features, config, { upstream, ledger, catalog });
137
208
  }
138
209
 
139
- const policed = applyRequestPolicy(resolveProfile(config, req.requestedModel, req.isSubagent), config, req.policy, opts.forceSlug);
210
+ const pinnedByModel = pinForRequestedModel(
211
+ config,
212
+ snapshot.models.map((m) => m.slug),
213
+ req.requestedModel,
214
+ req.requestedModelFull ?? req.requestedModel,
215
+ );
216
+ const policed = applyRequestPolicy(
217
+ resolveProfile(config, req.requestedModel, req.isSubagent),
218
+ config,
219
+ req.policy,
220
+ opts.forceSlug ?? pinnedByModel,
221
+ );
222
+ // Every ledger read this turn needs, fetched here rather than inside
223
+ // `select`: selection stays a synchronous pure function, and a ledger
224
+ // that can only be read asynchronously (Postgres) works unchanged.
225
+ const reads = await prefetchTurnReads(reader, req, policed.profile, policed.cfg, snapshot, classification.task);
140
226
  const decision = select({
141
227
  req,
142
228
  features,
@@ -144,9 +230,9 @@ export function createRouter(deps: RouterDeps): Router {
144
230
  profile: policed.profile,
145
231
  state,
146
232
  snapshot,
147
- ledger,
148
233
  cfg: policed.cfg,
149
234
  nowMs: Date.now(),
235
+ reads,
150
236
  ...(opts.excludeSlugs === undefined ? {} : { excludeSlugs: opts.excludeSlugs }),
151
237
  ...(policed.forceSlug === undefined ? {} : { forceSlug: policed.forceSlug }),
152
238
  });