auto-model-router 0.30.3 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/.omp-plugin/marketplace.json +2 -2
  2. package/README.md +32 -2
  3. package/omp-extension/router-configure.ts +9 -7
  4. package/package.json +1 -1
  5. package/src/cli/config-cmd.ts +8 -7
  6. package/src/cli/explain.ts +10 -5
  7. package/src/cli/export.ts +6 -5
  8. package/src/cli/models.ts +10 -7
  9. package/src/cli/report.ts +6 -1
  10. package/src/cli/stats.ts +7 -7
  11. package/src/config/load.ts +10 -1
  12. package/src/config/types.ts +10 -1
  13. package/src/context/bridge.ts +7 -7
  14. package/src/context/index.ts +3 -3
  15. package/src/context/store.ts +39 -56
  16. package/src/context/types.ts +7 -6
  17. package/src/cost/blended.ts +28 -7
  18. package/src/cost/feedback.ts +33 -37
  19. package/src/cost/ledger-sql.ts +547 -0
  20. package/src/cost/ledger.ts +30 -459
  21. package/src/cost/report.ts +171 -129
  22. package/src/cost/retention.ts +10 -10
  23. package/src/cost/summary.ts +15 -10
  24. package/src/cost/types.ts +43 -62
  25. package/src/cost/views.ts +79 -49
  26. package/src/eval/calibrate.ts +47 -12
  27. package/src/eval/run.ts +18 -2
  28. package/src/lib.ts +6 -2
  29. package/src/router/candidates.ts +7 -15
  30. package/src/router/classify.ts +6 -4
  31. package/src/router/index.ts +95 -9
  32. package/src/router/select.ts +38 -21
  33. package/src/router/state.ts +90 -102
  34. package/src/router/types.ts +11 -5
  35. package/src/server/advise.ts +6 -4
  36. package/src/server/compaction-digest.ts +1 -1
  37. package/src/server/digest.ts +9 -10
  38. package/src/server/http.ts +109 -46
  39. package/src/server/providers.ts +18 -4
  40. package/src/server/turn.ts +32 -9
  41. package/src/tokens/estimate.ts +16 -6
  42. package/src/upstream/ollama-usage.ts +21 -11
  43. package/src/util/schema.ts +201 -0
  44. package/src/util/sql.ts +246 -0
  45. package/src/wire/anthropic/messages.ts +3 -4
  46. package/src/wire/openai/request.ts +1 -0
  47. package/src/wire/types.ts +7 -0
  48. package/test/anthropic-wire.test.ts +9 -9
  49. package/test/benchmark-feeds.test.ts +7 -7
  50. package/test/cache-control.test.ts +7 -7
  51. package/test/cache-estimate.test.ts +5 -5
  52. package/test/catalog-view.test.ts +4 -4
  53. package/test/catalog.test.ts +11 -11
  54. package/test/classify.test.ts +24 -24
  55. package/test/compaction.test.ts +20 -20
  56. package/test/config-wizard.test.ts +32 -32
  57. package/test/config.test.ts +10 -10
  58. package/test/connect-harnesses.test.ts +11 -11
  59. package/test/context-bridge.test.ts +40 -30
  60. package/test/context-prune.test.ts +43 -36
  61. package/test/context-query.test.ts +8 -8
  62. package/test/controls.test.ts +54 -27
  63. package/test/cost.test.ts +12 -12
  64. package/test/digest.test.ts +55 -44
  65. package/test/embed-lifecycle.test.ts +5 -5
  66. package/test/embed-logic.test.ts +26 -26
  67. package/test/escalate.test.ts +17 -17
  68. package/test/eval.test.ts +73 -16
  69. package/test/executable.test.ts +6 -6
  70. package/test/exploration.test.ts +19 -20
  71. package/test/failover.test.ts +22 -21
  72. package/test/fakes.ts +105 -0
  73. package/test/features.test.ts +21 -21
  74. package/test/harness-requests.test.ts +3 -3
  75. package/test/harness-switch.test.ts +5 -5
  76. package/test/hold-exploration.test.ts +13 -13
  77. package/test/hot-reload.test.ts +5 -5
  78. package/test/learned.test.ts +5 -5
  79. package/test/ledger-sql.test.ts +342 -0
  80. package/test/mcp-entry.test.ts +5 -5
  81. package/test/migrations.test.ts +28 -22
  82. package/test/models-yml.test.ts +18 -18
  83. package/test/ollama.test.ts +40 -34
  84. package/test/omp-credentials.test.ts +16 -16
  85. package/test/policy.test.ts +3 -3
  86. package/test/reconfigure.test.ts +4 -4
  87. package/test/redaction.test.ts +41 -35
  88. package/test/remote.test.ts +12 -12
  89. package/test/report-logic.test.ts +8 -8
  90. package/test/report.test.ts +95 -87
  91. package/test/retention.test.ts +79 -66
  92. package/test/schema.test.ts +123 -0
  93. package/test/scope.test.ts +8 -8
  94. package/test/select.test.ts +216 -257
  95. package/test/skills.test.ts +3 -3
  96. package/test/sql-shim.test.ts +154 -0
  97. package/test/state.test.ts +43 -36
  98. package/test/summary.test.ts +38 -27
  99. package/test/tier-plan.test.ts +45 -62
  100. package/test/toast-logic.test.ts +31 -31
  101. package/test/tokens.test.ts +95 -80
  102. package/test/trust-attribution.test.ts +217 -187
  103. package/test/trust-window.test.ts +37 -32
  104. package/test/turn.test.ts +55 -23
  105. package/test/upstreams.test.ts +13 -13
  106. package/test/views.test.ts +81 -59
  107. package/test/wire-request.test.ts +17 -17
  108. package/test/wire-responses.test.ts +4 -4
  109. package/tools/agentdox-e2e.ts +5 -2
  110. package/tools/export-benchmarks.ts +5 -5
  111. package/tools/ledger-parity.ts +266 -0
  112. package/tools/replay.ts +16 -8
@@ -0,0 +1,201 @@
1
+ /**
2
+ * The store's schema, for either engine.
3
+ *
4
+ * `util/sqlite.ts` remains the migration path for a SQLite FILE that already
5
+ * exists: nineteen versions have shipped, and an old file still needs its
6
+ * `ALTER TABLE`s applied in order. This module declares the FINAL shape of
7
+ * every table instead, which is what a fresh store needs — a Postgres database
8
+ * has no history to migrate, and on an already-migrated file every statement
9
+ * here is a no-op.
10
+ *
11
+ * The two must agree, so the definitions below are transcribed from
12
+ * `util/sqlite.ts` with its incremental columns folded in, and
13
+ * `test/schema.test.ts` compares the two engine-by-engine rather than trusting
14
+ * that they were copied correctly.
15
+ */
16
+
17
+ import type { SqlDb } from "./sql.ts";
18
+
19
+ /**
20
+ * Creates every table and index the router uses. Idempotent, so boot order
21
+ * never matters — the property the SQLite bootstrap has always had.
22
+ */
23
+ export async function migrateStore(db: SqlDb): Promise<void> {
24
+ const json = db.type("json");
25
+ const float = db.type("float");
26
+ const big = db.type("bigint");
27
+ // A single-row cache keyed on a constant: `CHECK (id = 1)` is portable, and
28
+ // it is what keeps a second payload from ever accumulating.
29
+ const singleton = (name: string, extra = ""): string =>
30
+ `CREATE TABLE IF NOT EXISTS ${name} (
31
+ id INTEGER PRIMARY KEY CHECK (id = 1),
32
+ payload ${json} NOT NULL,
33
+ fetched_at_ms ${big} NOT NULL${extra}
34
+ )`;
35
+
36
+ const statements: string[] = [
37
+ singleton("catalog_cache", `,\n\t\t\tetag TEXT,\n\t\t\tkey_scoped INTEGER NOT NULL DEFAULT 0`),
38
+ singleton("ollama_catalog_cache"),
39
+ singleton("benchmark_cache"),
40
+ singleton("local_scores"),
41
+
42
+ // One row per dispatched upstream generation. The columns nineteen
43
+ // migrations added are declared here as they finally stand.
44
+ `CREATE TABLE IF NOT EXISTS ledger (
45
+ id TEXT PRIMARY KEY,
46
+ created_at_ms ${big} NOT NULL,
47
+ conversation_key TEXT NOT NULL,
48
+ session_id TEXT NOT NULL,
49
+ turn INTEGER NOT NULL,
50
+ requested_model TEXT NOT NULL,
51
+ harness_id TEXT NOT NULL DEFAULT '',
52
+ omp_session_id TEXT NOT NULL DEFAULT '',
53
+ slug TEXT NOT NULL,
54
+ served_slug TEXT,
55
+ tier TEXT NOT NULL,
56
+ classification_source TEXT NOT NULL,
57
+ reasons ${json} NOT NULL,
58
+ predicted_usd ${float} NOT NULL,
59
+ reported_usd ${float},
60
+ usage ${json} NOT NULL,
61
+ cost_breakdown ${json},
62
+ attempt INTEGER NOT NULL,
63
+ escalation_signal TEXT,
64
+ latency_ms INTEGER NOT NULL,
65
+ ttft_ms INTEGER,
66
+ finish_reason TEXT,
67
+ wasted INTEGER NOT NULL DEFAULT 0,
68
+ upstream_generation_id TEXT,
69
+ error TEXT,
70
+ error_kind TEXT,
71
+ features ${json},
72
+ score ${float},
73
+ confidence ${float},
74
+ task TEXT,
75
+ classifier_reasons ${json},
76
+ explored_from TEXT,
77
+ hold_arm INTEGER,
78
+ prompt_tokens_saved INTEGER,
79
+ scope TEXT,
80
+ redactions INTEGER
81
+ )`,
82
+ "CREATE INDEX IF NOT EXISTS idx_ledger_conversation ON ledger (conversation_key)",
83
+ "CREATE INDEX IF NOT EXISTS idx_ledger_created ON ledger (created_at_ms)",
84
+ "CREATE INDEX IF NOT EXISTS idx_ledger_slug ON ledger (slug)",
85
+ // Per-slug newest-first reads (the latency window, any windowed trust).
86
+ // Without this they sorted every row the slug ever had: measured on a
87
+ // real ledger, 5-10ms and RISING with history against a flat 0.04-0.08ms.
88
+ "CREATE INDEX IF NOT EXISTS idx_ledger_slug_created ON ledger (slug, created_at_ms DESC)",
89
+ "CREATE INDEX IF NOT EXISTS idx_ledger_harness_created ON ledger (harness_id, created_at_ms DESC)",
90
+ "CREATE INDEX IF NOT EXISTS idx_ledger_slug_harness_created ON ledger (slug, harness_id, created_at_ms DESC)",
91
+ "CREATE INDEX IF NOT EXISTS idx_ledger_session ON ledger (omp_session_id, created_at_ms DESC)",
92
+
93
+ `CREATE TABLE IF NOT EXISTS token_calibration (
94
+ tokenizer TEXT PRIMARY KEY,
95
+ est_bytes ${big} NOT NULL,
96
+ actual_tokens ${big} NOT NULL,
97
+ samples ${big} NOT NULL
98
+ )`,
99
+
100
+ // Per-conversation routing memory: the sticky window, which model's
101
+ // prompt cache is warm, and the accumulated spend the budget guard reads.
102
+ `CREATE TABLE IF NOT EXISTS conversations (
103
+ key TEXT PRIMARY KEY,
104
+ session_id TEXT NOT NULL,
105
+ turn INTEGER NOT NULL DEFAULT 0,
106
+ current_slug TEXT,
107
+ current_tier TEXT,
108
+ sticky_until_turn INTEGER NOT NULL DEFAULT 0,
109
+ escalations INTEGER NOT NULL DEFAULT 0,
110
+ spent_usd ${float} NOT NULL DEFAULT 0,
111
+ last_prompt_tokens INTEGER NOT NULL DEFAULT 0,
112
+ cache_warm_slug TEXT,
113
+ cache_warm_at_ms ${big} NOT NULL DEFAULT 0,
114
+ context_version TEXT,
115
+ context_fetched_at_ms ${big} NOT NULL DEFAULT 0,
116
+ compaction_plan ${json},
117
+ compaction_plan_tokens INTEGER NOT NULL DEFAULT 0,
118
+ upgrade_deferred_tier TEXT,
119
+ updated_at_ms ${big} NOT NULL DEFAULT 0
120
+ )`,
121
+
122
+ // agentdox bridge. Blocks are content-addressed so many conversations on
123
+ // one project share a copy, and so a restart can re-inject the SAME bytes
124
+ // a conversation was already using (the upstream cache outlives us).
125
+ `CREATE TABLE IF NOT EXISTS context_blocks (
126
+ version TEXT PRIMARY KEY,
127
+ scope TEXT NOT NULL,
128
+ block TEXT NOT NULL,
129
+ fetched_at_ms ${big} NOT NULL
130
+ )`,
131
+
132
+ // Plan-meter readings beside the ledger's own Ollama total at the same
133
+ // instant, so the estimate can be calibrated against the bill.
134
+ `CREATE TABLE IF NOT EXISTS ollama_meter_samples (
135
+ at_ms ${big} PRIMARY KEY,
136
+ meter_usd ${float} NOT NULL,
137
+ ledger_usd ${float} NOT NULL
138
+ )`,
139
+
140
+ // User verdicts on routed turns, tied to the ledger row judged.
141
+ `CREATE TABLE IF NOT EXISTS feedback (
142
+ id TEXT PRIMARY KEY,
143
+ ledger_id TEXT NOT NULL,
144
+ omp_session_id TEXT NOT NULL DEFAULT '',
145
+ slug TEXT NOT NULL,
146
+ tier TEXT NOT NULL,
147
+ verdict TEXT NOT NULL,
148
+ note TEXT NOT NULL DEFAULT '',
149
+ created_at_ms ${big} NOT NULL
150
+ )`,
151
+ "CREATE INDEX IF NOT EXISTS idx_feedback_created ON feedback (created_at_ms)",
152
+ "CREATE INDEX IF NOT EXISTS idx_feedback_ledger ON feedback (ledger_id)",
153
+
154
+ // Small durable markers (when the daily summary was last posted, per harness).
155
+ `CREATE TABLE IF NOT EXISTS router_kv (
156
+ key TEXT PRIMARY KEY,
157
+ value TEXT NOT NULL,
158
+ updated_at_ms ${big} NOT NULL
159
+ )`,
160
+
161
+ `CREATE TABLE IF NOT EXISTS agentdox_sessions (
162
+ conversation_key TEXT PRIMARY KEY,
163
+ scope TEXT NOT NULL,
164
+ session_id TEXT NOT NULL,
165
+ created_at_ms ${big} NOT NULL
166
+ )`,
167
+ ];
168
+
169
+ // `IF NOT EXISTS` is not atomic on Postgres: two replicas booting against a
170
+ // fresh database both pass the existence check and the loser fails on the
171
+ // unique index over pg_type (23505), or on the table name itself (42P07 /
172
+ // 42710). Measured: one of two replicas started together died with
173
+ // "duplicate key value violates unique constraint pg_type_typname_nsp_index".
174
+ // The condition those errors report is the condition the statement asked to
175
+ // tolerate, so they are the success case arriving from the other replica.
176
+ const RACED = new Set(["23505", "42P07", "42710"]);
177
+ for (const statement of statements) {
178
+ try {
179
+ await db.sql.unsafe(statement);
180
+ } catch (err) {
181
+ const code = (err as { errno?: unknown }).errno;
182
+ if (!RACED.has(String(code))) throw err;
183
+ }
184
+ }
185
+ }
186
+
187
+ /** Every table `migrateStore` creates, for tests and for teardown. */
188
+ export const STORE_TABLES = [
189
+ "catalog_cache",
190
+ "ollama_catalog_cache",
191
+ "benchmark_cache",
192
+ "local_scores",
193
+ "ledger",
194
+ "token_calibration",
195
+ "conversations",
196
+ "context_blocks",
197
+ "ollama_meter_samples",
198
+ "feedback",
199
+ "router_kv",
200
+ "agentdox_sessions",
201
+ ] as const;
@@ -0,0 +1,246 @@
1
+ /**
2
+ * One SQL handle for both engines.
3
+ *
4
+ * The ledger has to live in the same database as the rest of a deployment's
5
+ * state when that state is shared (spend read from a file that lags the
6
+ * database it must agree with is a cap that silently over-admits), and in a
7
+ * local file when it is not. `Bun.SQL` speaks both, so the ledger is written
8
+ * once against this shim rather than twice.
9
+ *
10
+ * Only three things actually differ, and they are the three things here:
11
+ *
12
+ * 1. **JSON access.** `json_extract(usage, '$.promptTokens')` against
13
+ * `(usage->>'promptTokens')::numeric`.
14
+ * 2. **Numeric results.** Postgres returns `COUNT(*)` and `SUM(BIGINT)` as
15
+ * STRINGS. Arithmetic on those silently produces wrong answers rather than
16
+ * throwing — a measured trust score came out 0.9756 instead of 0.9726 this
17
+ * way — so every numeric read goes through `num`.
18
+ * 3. **Column types.** `JSONB`/`DOUBLE PRECISION` against `TEXT`/`REAL`.
19
+ *
20
+ * Everything else — parameters, `IN ${sql(array)}`, window functions,
21
+ * `ON CONFLICT`, `DELETE ... RETURNING`, bulk insert, transactions — is
22
+ * identical on both, verified by `tools/dialect-probe.ts`.
23
+ */
24
+
25
+ import { SQL } from "bun";
26
+
27
+ export type Dialect = "sqlite" | "postgres";
28
+
29
+ export interface SqlDb {
30
+ readonly sql: SQL;
31
+ readonly dialect: Dialect;
32
+ /** A JSON member as a NUMBER-typed SQL expression, for aggregates and comparisons. */
33
+ jsonNum(column: string, key: string): string;
34
+ /** A JSON member as TEXT, for equality against string values. */
35
+ jsonText(column: string, key: string): string;
36
+ /**
37
+ * A NESTED JSON member as a number, e.g. `features.anatomy.messages`.
38
+ *
39
+ * `jsonNum` cannot express this: SQLite takes a whole path in one string
40
+ * (`'$.anatomy.messages'`), while Postgres' `->>` reads a SINGLE key, so a
41
+ * dotted key silently reads NULL there — a whole report section came back
42
+ * null rather than failing.
43
+ */
44
+ jsonPathNum(column: string, path: readonly string[]): string;
45
+ /**
46
+ * A JSON member that holds a BOOLEAN, as 1/0.
47
+ *
48
+ * The engines disagree twice over: SQLite's `json_extract` yields the
49
+ * INTEGER 1 for JSON `true`, Postgres' `->>` yields the TEXT 'true', and
50
+ * SQLite's `IN` does not coerce between them. Comparing the wrong way round
51
+ * silently misclassifies every row — it counted router-ESTIMATED cache hits
52
+ * as measured ones and inflated cache reliability samples by 2%.
53
+ */
54
+ jsonBool(column: string, key: string): string;
55
+ /**
56
+ * Substring test as a boolean expression. SQLite has `instr(haystack,
57
+ * needle) > 0`; Postgres spells it `position(needle in haystack) > 0`.
58
+ */
59
+ contains(haystack: string, needle: string): string;
60
+ /**
61
+ * An epoch-millisecond column as a `YYYY-MM-DD` UTC day. SQLite has
62
+ * `strftime(..., 'unixepoch')`; Postgres needs `to_timestamp` plus an
63
+ * explicit UTC conversion, or the server's timezone silently decides which
64
+ * day a turn was billed on.
65
+ */
66
+ utcDay(msColumn: string): string;
67
+ /** Whether a table exists, without reading engine-specific catalog tables. */
68
+ tableExists(name: string): Promise<boolean>;
69
+ /**
70
+ * Runs SQL written with `$name` placeholders, whichever way the engine
71
+ * wants them numbered.
72
+ *
73
+ * The reporting queries are assembled from optional filters (a harness
74
+ * scope, an upper time bound), which a tagged template cannot express — the
75
+ * shape of the statement is decided at runtime. Rewriting them as string
76
+ * concatenation with positional parameters would renumber every bind by
77
+ * hand, which is how a filter ends up reading the wrong column.
78
+ */
79
+ query<T>(text: string, binds?: Record<string, unknown>): Promise<T[]>;
80
+ /** `query`, for a statement that yields at most one row. */
81
+ one<T>(text: string, binds?: Record<string, unknown>): Promise<T | null>;
82
+ /** A column type that differs between engines. */
83
+ type(kind: "json" | "float" | "bigint"): string;
84
+ /**
85
+ * Cast suffix for a parameter that may be NULL. Postgres refuses to infer a
86
+ * type for a bare NULL placeholder ("could not determine data type of
87
+ * parameter $2"), which the `(${x} IS NULL OR col = ${x})` idiom for an
88
+ * optional filter relies on; SQLite has no cast syntax to add. Empty there.
89
+ */
90
+ readonly nullableText: string;
91
+ /**
92
+ * Two-argument scalar minimum. SQLite spells it `MIN(a, b)`; Postgres
93
+ * reserves `MIN` for the aggregate and needs `LEAST(a, b)`.
94
+ */
95
+ least(a: string, b: string): string;
96
+ close(): Promise<void>;
97
+ }
98
+
99
+ /**
100
+ * `sqlite://` (or a bare path) and `postgres://`/`postgresql://` URLs. A bare
101
+ * path is accepted because the router's config has always taken
102
+ * `ledger.path`, and a deployment that never opts into Postgres should not
103
+ * have to learn a URL scheme.
104
+ */
105
+ export function dialectOf(url: string): Dialect {
106
+ return url.startsWith("postgres://") || url.startsWith("postgresql://") ? "postgres" : "sqlite";
107
+ }
108
+
109
+ export function openSqlDb(url: string): SqlDb {
110
+ const dialect = dialectOf(url);
111
+ const target = dialect === "sqlite" && !url.startsWith("sqlite:") ? `sqlite://${url}` : url;
112
+ const sql = new SQL(target);
113
+ const pg = dialect === "postgres";
114
+ return {
115
+ sql,
116
+ dialect,
117
+ jsonNum: (column, key) => (pg ? `(${column}->>'${key}')::numeric` : `json_extract(${column}, '$.${key}')`),
118
+ jsonText: (column, key) => (pg ? `(${column}->>'${key}')` : `json_extract(${column}, '$.${key}')`),
119
+ jsonPathNum: (column, path) =>
120
+ pg
121
+ ? `(${column}#>>'{${path.join(",")}}')::numeric`
122
+ : `json_extract(${column}, '$.${path.join(".")}')`,
123
+ jsonBool: (column, key) =>
124
+ pg
125
+ ? `CASE WHEN (${column}->>'${key}') IN ('true', '1') THEN 1 ELSE 0 END`
126
+ : `CASE WHEN json_extract(${column}, '$.${key}') IN (1, 'true', '1') THEN 1 ELSE 0 END`,
127
+ type: (kind) => {
128
+ if (kind === "json") return pg ? "JSONB" : "TEXT";
129
+ if (kind === "float") return pg ? "DOUBLE PRECISION" : "REAL";
130
+ return pg ? "BIGINT" : "INTEGER";
131
+ },
132
+ nullableText: pg ? "::text" : "",
133
+ least: (a, b) => (pg ? `LEAST(${a}, ${b})` : `MIN(${a}, ${b})`),
134
+ // The haystack is cast to text: a JSON column is `jsonb` on Postgres and
135
+ // `position()` refuses it, while sqlite stores the same column as TEXT.
136
+ contains: (haystack, needle) => (pg ? `position(${needle} in (${haystack})::text) > 0` : `instr(${haystack}, ${needle}) > 0`),
137
+ utcDay: (msColumn) =>
138
+ pg
139
+ ? `to_char(to_timestamp(${msColumn} / 1000) AT TIME ZONE 'UTC', 'YYYY-MM-DD')`
140
+ : `strftime('%Y-%m-%d', ${msColumn} / 1000, 'unixepoch')`,
141
+ tableExists: async (name) => {
142
+ // Cheaper and more portable than either catalog table: ask for nothing
143
+ // from it and see whether the statement plans.
144
+ try {
145
+ await sql.unsafe(`SELECT 1 FROM ${name} WHERE 1 = 0`);
146
+ return true;
147
+ } catch {
148
+ return false;
149
+ }
150
+ },
151
+ query: async <T>(text: string, binds: Record<string, unknown> = {}): Promise<T[]> => {
152
+ const { text: prepared, values } = bindNamed(text, binds, pg);
153
+ return (await sql.unsafe(prepared, values)) as T[];
154
+ },
155
+ one: async <T>(text: string, binds: Record<string, unknown> = {}): Promise<T | null> => {
156
+ const { text: prepared, values } = bindNamed(text, binds, pg);
157
+ const rows = (await sql.unsafe(prepared, values)) as T[];
158
+ return rows[0] ?? null;
159
+ },
160
+ close: async () => {
161
+ await sql.end();
162
+ },
163
+ };
164
+ }
165
+
166
+
167
+ /**
168
+ * Rewrites `$name` placeholders to the engine's positional form, in order of
169
+ * first appearance, and returns the matching value array.
170
+ *
171
+ * A name may repeat — `created_at_ms >= $since` and a `CASE` on `$since` in
172
+ * the same statement is normal — and repeats must reuse one parameter slot on
173
+ * Postgres. JSON paths (`'$.isSubagent'`) are not placeholders: the pattern
174
+ * requires a letter or underscore after the `$`, which `$.` fails.
175
+ */
176
+ export function bindNamed(text: string, binds: Record<string, unknown>, pg: boolean): { text: string; values: unknown[] } {
177
+ // Callers that were written against bun:sqlite pass their binds keyed WITH
178
+ // the sigil (`{ $since: 0 }`), which is how every reporting query in this
179
+ // repo already builds them; both forms resolve.
180
+ const valueOf = (name: string): unknown => {
181
+ if (name in binds) return binds[name];
182
+ const sigil = `$${name}`;
183
+ if (sigil in binds) return binds[sigil];
184
+ throw new Error(`sql: no value bound for $${name}`);
185
+ };
186
+ const order: string[] = [];
187
+ const prepared = text.replace(/\$([a-zA-Z_][a-zA-Z0-9_]*)/g, (_match, name: string) => {
188
+ valueOf(name);
189
+ let index = order.indexOf(name);
190
+ if (index === -1) {
191
+ order.push(name);
192
+ index = order.length - 1;
193
+ }
194
+ return pg ? `$${index + 1}` : "?";
195
+ });
196
+ if (!pg) {
197
+ // SQLite's positional `?` cannot reuse a slot, so a repeated name is
198
+ // bound once per occurrence rather than once per name.
199
+ const values: unknown[] = [];
200
+ text.replace(/\$([a-zA-Z_][a-zA-Z0-9_]*)/g, (_match, name: string) => {
201
+ values.push(valueOf(name));
202
+ return "";
203
+ });
204
+ return { text: prepared, values };
205
+ }
206
+ return { text: prepared, values: order.map(valueOf) };
207
+ }
208
+
209
+ /**
210
+ * Coerces a value Postgres may have returned as a string. Applied at every
211
+ * numeric read: `Number(null)` is 0, which is wrong for a nullable aggregate,
212
+ * so null and undefined are preserved.
213
+ */
214
+ export function num(value: unknown): number {
215
+ return typeof value === "number" ? value : Number(value ?? 0);
216
+ }
217
+
218
+ /** `num`, but a missing value stays missing rather than becoming 0. */
219
+ export function numOrNull(value: unknown): number | null {
220
+ if (value === null || value === undefined) return null;
221
+ return typeof value === "number" ? value : Number(value);
222
+ }
223
+
224
+ /**
225
+ * JSON for a `json` column. Postgres' driver encodes a JS string destined for
226
+ * `JSONB` as a JSON *string* — `jsonb_typeof` reads `'string'` and every
227
+ * `->>` on it returns NULL — so an object must be passed through unstringified
228
+ * there, while SQLite's TEXT column needs the serialised form.
229
+ */
230
+ export function jsonParam(db: SqlDb, value: unknown): unknown {
231
+ if (value === null || value === undefined) return null;
232
+ return db.dialect === "postgres" ? value : JSON.stringify(value);
233
+ }
234
+
235
+ /** Reads a `json` column back, whichever way it was stored. */
236
+ export function jsonValue<T>(value: unknown): T | null {
237
+ if (value === null || value === undefined) return null;
238
+ if (typeof value === "string") {
239
+ try {
240
+ return JSON.parse(value) as T;
241
+ } catch {
242
+ return null;
243
+ }
244
+ }
245
+ return value as T;
246
+ }
@@ -25,8 +25,7 @@
25
25
  */
26
26
 
27
27
  import { encoder, sseDataFrame } from "../../util/sse.ts";
28
- import type { Ledger } from "../../cost/types.ts";
29
- import { estimateTokens } from "../../tokens/estimate.ts";
28
+ import { estimateTokens, type TokenRatio } from "../../tokens/estimate.ts";
30
29
  import type { NormRequest, ResponseSink, TurnSummary, UpstreamChunk, WireError } from "../types.ts";
31
30
  import { invalidRequest, WireErrorException } from "../openai/errors.ts";
32
31
  import { parseChatRequest } from "../openai/request.ts";
@@ -217,9 +216,9 @@ export function parseMessagesRequest(body: unknown, headers: Headers, models: Re
217
216
  }
218
217
 
219
218
  /** `POST /v1/messages/count_tokens`: the router's own estimate over the prompt bytes. */
220
- export function countAnthropicTokens(body: unknown, models: Record<string, string>, ledger: Ledger | null): number {
219
+ export function countAnthropicTokens(body: unknown, models: Record<string, string>, ratio: TokenRatio): number {
221
220
  const norm = parseChatRequest(messagesToChatBody({ ...(isRec(body) ? body : {}), stream: false }, models), new Headers());
222
- return estimateTokens(norm.promptBytes, "anthropic", ledger);
221
+ return estimateTokens(norm.promptBytes, "anthropic", ratio);
223
222
  }
224
223
 
225
224
  // ---------------------------------------------------------------------------
@@ -418,6 +418,7 @@ export function parseChatRequest(body: unknown, headers: Headers): NormRequest {
418
418
  isSubagent,
419
419
  ...(policy === undefined ? {} : { policy }),
420
420
  requestedModel,
421
+ requestedModelFull: b.model,
421
422
  messages,
422
423
  tools,
423
424
  forcedToolChoice,
package/src/wire/types.ts CHANGED
@@ -128,6 +128,13 @@ export interface NormRequest {
128
128
  policy?: RequestPolicy;
129
129
  /** Virtual model the client selected, e.g. `auto`, `auto-cheap`, `auto-max`. */
130
130
  requestedModel: string;
131
+ /**
132
+ * The client's `model` string before the provider prefix was stripped, so a
133
+ * vendor-qualified catalog slug (`deepseek/deepseek-v4.1-flash`) stays
134
+ * distinguishable from a profile id. Absent ⇒ `requestedModel` is the whole
135
+ * of what the client asked for.
136
+ */
137
+ requestedModelFull?: string;
131
138
  messages: NormMessage[];
132
139
  tools: NormTool[];
133
140
  /** True when the client forced a specific tool. */
@@ -42,7 +42,7 @@ const CLAUDE_CODE_BODY = {
42
42
  };
43
43
 
44
44
  describe("messagesToChatBody", () => {
45
- test("translates a Claude Code turn: system blocks, tool_use/tool_result, custom tools only, tool_choice, thinking budget", () => {
45
+ test("translates a Claude Code turn: system blocks, tool_use/tool_result, custom tools only, tool_choice, thinking budget", async () => {
46
46
  const b = messagesToChatBody(CLAUDE_CODE_BODY);
47
47
  expect(b.model).toBe("auto");
48
48
  const messages = b.messages as { role: string; content: unknown; tool_calls?: unknown; tool_call_id?: string }[];
@@ -63,7 +63,7 @@ describe("messagesToChatBody", () => {
63
63
  for (const k of ["system", "metadata", "thinking", "tool_choice_anthropic", "cache_control"]) expect(k in b && k !== "tool_choice").toBe(false);
64
64
  });
65
65
 
66
- test("string bodies, images, documents, error tool results, tool_choice variants, stop sequences, effort", () => {
66
+ test("string bodies, images, documents, error tool results, tool_choice variants, stop sequences, effort", async () => {
67
67
  const b = messagesToChatBody({
68
68
  model: "auto-max",
69
69
  system: "sys",
@@ -97,7 +97,7 @@ describe("messagesToChatBody", () => {
97
97
  expect(messagesToChatBody({ model: "m", messages: [{ role: "user", content: "x" }], thinking: { type: "adaptive" } }).reasoning).toEqual({ effort: "medium" });
98
98
  });
99
99
 
100
- test("rejects what cannot be a turn", () => {
100
+ test("rejects what cannot be a turn", async () => {
101
101
  expect(() => messagesToChatBody("nope")).toThrow("JSON object");
102
102
  expect(() => messagesToChatBody({ model: "", messages: [{ role: "user", content: "x" }] })).toThrow("model");
103
103
  expect(() => messagesToChatBody({ model: "m", messages: [] })).toThrow("messages");
@@ -105,7 +105,7 @@ describe("messagesToChatBody", () => {
105
105
  expect(() => messagesToChatBody({ model: "m", messages: [{ role: "user", content: 5 }] })).toThrow("content");
106
106
  });
107
107
 
108
- test("model names map by glob, first match wins, profile ids pass through", () => {
108
+ test("model names map by glob, first match wins, profile ids pass through", async () => {
109
109
  expect(mapAnthropicModel("claude-haiku-4-5-20251001")).toBe("auto-cheap");
110
110
  expect(mapAnthropicModel("claude-opus-4-8")).toBe("auto");
111
111
  expect(mapAnthropicModel("auto-sub")).toBe("auto-sub");
@@ -115,7 +115,7 @@ describe("messagesToChatBody", () => {
115
115
  });
116
116
 
117
117
  describe("parseMessagesRequest", () => {
118
- test("derives the harness from the user agent and the session from metadata; explicit headers win; the rendered body is chat-shaped", () => {
118
+ test("derives the harness from the user agent and the session from metadata; explicit headers win; the rendered body is chat-shaped", async () => {
119
119
  const headers = new Headers({ "user-agent": "claude-cli/2.1.263 (external, cli)", "x-api-key": "k", "anthropic-version": "2023-06-01" });
120
120
  const norm = parseMessagesRequest(CLAUDE_CODE_BODY, headers);
121
121
  expect(norm.protocol).toBe("anthropic-messages");
@@ -138,7 +138,7 @@ describe("parseMessagesRequest", () => {
138
138
  expect(anthropicIdentityHeaders({}, new Headers({ "user-agent": "python-requests" })).get("x-omp-harness")).toBe("anthropic");
139
139
  });
140
140
 
141
- test("the agentdox scope and layer headers reach the request on the Anthropic path too", () => {
141
+ test("the agentdox scope and layer headers reach the request on the Anthropic path too", async () => {
142
142
  const plain = parseMessagesRequest(CLAUDE_CODE_BODY, new Headers({ "user-agent": "claude-cli/2.1.263" }));
143
143
  expect(plain.agentdoxScope).toBe("");
144
144
  expect(plain.agentdoxGroup).toBe("");
@@ -154,13 +154,13 @@ describe("parseMessagesRequest", () => {
154
154
  expect(parseMessagesRequest(CLAUDE_CODE_BODY, new Headers({ "x-agentdox-group": "Not A Slug" })).agentdoxGroup).toBe("");
155
155
  });
156
156
 
157
- test("the origin fingerprint reaches the request on the Anthropic path too", () => {
157
+ test("the origin fingerprint reaches the request on the Anthropic path too", async () => {
158
158
  expect(parseMessagesRequest(CLAUDE_CODE_BODY, new Headers({ "user-agent": "claude-cli/2.1.263" })).agentdoxOrigin).toBe("");
159
159
  expect(parseMessagesRequest(CLAUDE_CODE_BODY, new Headers({ "x-agentdox-origin": "github.com/drewappling/omp-router" })).agentdoxOrigin).toBe("github.com/drewappling/omp-router");
160
160
  expect(parseMessagesRequest(CLAUDE_CODE_BODY, new Headers({ "x-agentdox-origin": "https://github.com/a/b" })).agentdoxOrigin).toBe("");
161
161
  });
162
162
 
163
- test("a request captured from Claude Code 2.1: system inside messages, JSON user_id, adaptive thinking with effort, 23 custom tools", () => {
163
+ test("a request captured from Claude Code 2.1: system inside messages, JSON user_id, adaptive thinking with effort, 23 custom tools", async () => {
164
164
  const fixture = JSON.parse(readFileSync("test/fixtures/harness/claude-code.json", "utf8")) as { headers: Record<string, string>; body: Record<string, unknown> };
165
165
  const norm = parseMessagesRequest(fixture.body, new Headers(fixture.headers));
166
166
  expect(norm.harnessId).toBe("claude-code");
@@ -179,7 +179,7 @@ describe("parseMessagesRequest", () => {
179
179
  expect(sessionFromUserId("nothing here")).toBeNull();
180
180
  });
181
181
 
182
- test("count_tokens estimates from the prompt bytes", () => {
182
+ test("count_tokens estimates from the prompt bytes", async () => {
183
183
  expect(countAnthropicTokens(CLAUDE_CODE_BODY, DEFAULT_CONFIG.anthropic.models, null)).toBeGreaterThan(50);
184
184
  expect(() => countAnthropicTokens({ model: "m", messages: [] }, {}, null)).toThrow("messages");
185
185
  });
@@ -40,7 +40,7 @@ function blScore(over: Partial<FeedScore> & { key: string }): FeedScore {
40
40
  }
41
41
 
42
42
  describe("normalizeModelKey", () => {
43
- test("strips provider, tilde, and release words but keeps the parameter size", () => {
43
+ test("strips provider, tilde, and release words but keeps the parameter size", async () => {
44
44
  expect(normalizeModelKey("z-ai/glm-5.3-flash")).toBe("glm-5-3-flash");
45
45
  expect(normalizeModelKey("~deepseek/deepseek-v4-flash-latest")).toBe("deepseek-v4-flash");
46
46
  expect(normalizeModelKey("meta/muse-glimmer-30b")).toBe("muse-glimmer-30b");
@@ -51,7 +51,7 @@ describe("normalizeModelKey", () => {
51
51
  });
52
52
 
53
53
  describe("parseAaModels", () => {
54
- test("reads the three indices, keeps in-range values, and skips empty rows", () => {
54
+ test("reads the three indices, keeps in-range values, and skips empty rows", async () => {
55
55
  const body = {
56
56
  data: [
57
57
  {
@@ -74,7 +74,7 @@ describe("parseAaModels", () => {
74
74
  });
75
75
 
76
76
  describe("parseBenchlmModels", () => {
77
- test("maps categories to axes, drops estimated rows, and ignores out-of-range", () => {
77
+ test("maps categories to axes, drops estimated rows, and ignores out-of-range", async () => {
78
78
  const body = {
79
79
  models: [
80
80
  {
@@ -98,7 +98,7 @@ describe("parseBenchlmModels", () => {
98
98
  });
99
99
 
100
100
  describe("applyFeedScores", () => {
101
- test("fills the real gap models and reaches normalizeCatalogModel", () => {
101
+ test("fills the real gap models and reaches normalizeCatalogModel", async () => {
102
102
  const catalog = [
103
103
  raw("meta/muse-glimmer-30b"),
104
104
  raw("z-ai/glm-5.3-flash"),
@@ -135,7 +135,7 @@ describe("applyFeedScores", () => {
135
135
  expect(result.sources.benchlm).toBe(3); // muse coding+agentic, glm agentic
136
136
  });
137
137
 
138
- test("AA wins over BenchLM for the same axis", () => {
138
+ test("AA wins over BenchLM for the same axis", async () => {
139
139
  const catalog = [raw("z-ai/glm-5.3-flash")];
140
140
  const feeds: FeedScore[] = [
141
141
  blScore({ key: "glm-5-3-flash", creator: "z-ai", coding: 10 }),
@@ -145,7 +145,7 @@ describe("applyFeedScores", () => {
145
145
  expect(normalizeCatalogModel(catalog[0])?.quality.coding).toBe(61);
146
146
  });
147
147
 
148
- test("never fuzzy-matches a different model", () => {
148
+ test("never fuzzy-matches a different model", async () => {
149
149
  const catalog = [raw("meta/muse-glimmer-30b")];
150
150
  // Same family, different model — must not lend its score.
151
151
  const feeds: FeedScore[] = [aaScore({ key: "muse-spark-1-2", creator: "meta", coding: 72 })];
@@ -154,7 +154,7 @@ describe("applyFeedScores", () => {
154
154
  expect(normalizeCatalogModel(catalog[0])?.quality).toEqual({});
155
155
  });
156
156
 
157
- test("a shared key with conflicting creators fills only the creator that matches", () => {
157
+ test("a shared key with conflicting creators fills only the creator that matches", async () => {
158
158
  const catalog = [raw("z-ai/glm-5.3-flash")];
159
159
  const feeds: FeedScore[] = [
160
160
  aaScore({ key: "glm-5-3-flash", creator: "someone-else", coding: 5 }),