auto-model-router 0.30.3 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +32 -2
- package/omp-extension/router-configure.ts +9 -7
- package/package.json +1 -1
- package/src/cli/config-cmd.ts +8 -7
- package/src/cli/explain.ts +10 -5
- package/src/cli/export.ts +6 -5
- package/src/cli/models.ts +10 -7
- package/src/cli/report.ts +6 -1
- package/src/cli/stats.ts +7 -7
- package/src/config/load.ts +10 -1
- package/src/config/types.ts +10 -1
- package/src/context/bridge.ts +7 -7
- package/src/context/index.ts +3 -3
- package/src/context/store.ts +39 -56
- package/src/context/types.ts +7 -6
- package/src/cost/blended.ts +28 -7
- package/src/cost/feedback.ts +33 -37
- package/src/cost/ledger-sql.ts +547 -0
- package/src/cost/ledger.ts +30 -459
- package/src/cost/report.ts +171 -129
- package/src/cost/retention.ts +10 -10
- package/src/cost/summary.ts +15 -10
- package/src/cost/types.ts +43 -62
- package/src/cost/views.ts +79 -49
- package/src/eval/calibrate.ts +47 -12
- package/src/eval/run.ts +18 -2
- package/src/lib.ts +6 -2
- package/src/router/candidates.ts +7 -15
- package/src/router/classify.ts +6 -4
- package/src/router/index.ts +95 -9
- package/src/router/select.ts +38 -21
- package/src/router/state.ts +90 -102
- package/src/router/types.ts +11 -5
- package/src/server/advise.ts +6 -4
- package/src/server/compaction-digest.ts +1 -1
- package/src/server/digest.ts +9 -10
- package/src/server/http.ts +109 -46
- package/src/server/providers.ts +18 -4
- package/src/server/turn.ts +32 -9
- package/src/tokens/estimate.ts +16 -6
- package/src/upstream/ollama-usage.ts +21 -11
- package/src/util/schema.ts +201 -0
- package/src/util/sql.ts +246 -0
- package/src/wire/anthropic/messages.ts +3 -4
- package/src/wire/openai/request.ts +1 -0
- package/src/wire/types.ts +7 -0
- package/test/anthropic-wire.test.ts +9 -9
- package/test/benchmark-feeds.test.ts +7 -7
- package/test/cache-control.test.ts +7 -7
- package/test/cache-estimate.test.ts +5 -5
- package/test/catalog-view.test.ts +4 -4
- package/test/catalog.test.ts +11 -11
- package/test/classify.test.ts +24 -24
- package/test/compaction.test.ts +20 -20
- package/test/config-wizard.test.ts +32 -32
- package/test/config.test.ts +10 -10
- package/test/connect-harnesses.test.ts +11 -11
- package/test/context-bridge.test.ts +40 -30
- package/test/context-prune.test.ts +43 -36
- package/test/context-query.test.ts +8 -8
- package/test/controls.test.ts +54 -27
- package/test/cost.test.ts +12 -12
- package/test/digest.test.ts +55 -44
- package/test/embed-lifecycle.test.ts +5 -5
- package/test/embed-logic.test.ts +26 -26
- package/test/escalate.test.ts +17 -17
- package/test/eval.test.ts +73 -16
- package/test/executable.test.ts +6 -6
- package/test/exploration.test.ts +19 -20
- package/test/failover.test.ts +22 -21
- package/test/fakes.ts +105 -0
- package/test/features.test.ts +21 -21
- package/test/harness-requests.test.ts +3 -3
- package/test/harness-switch.test.ts +5 -5
- package/test/hold-exploration.test.ts +13 -13
- package/test/hot-reload.test.ts +5 -5
- package/test/learned.test.ts +5 -5
- package/test/ledger-sql.test.ts +342 -0
- package/test/mcp-entry.test.ts +5 -5
- package/test/migrations.test.ts +28 -22
- package/test/models-yml.test.ts +18 -18
- package/test/ollama.test.ts +40 -34
- package/test/omp-credentials.test.ts +16 -16
- package/test/policy.test.ts +3 -3
- package/test/reconfigure.test.ts +4 -4
- package/test/redaction.test.ts +41 -35
- package/test/remote.test.ts +12 -12
- package/test/report-logic.test.ts +8 -8
- package/test/report.test.ts +95 -87
- package/test/retention.test.ts +79 -66
- package/test/schema.test.ts +123 -0
- package/test/scope.test.ts +8 -8
- package/test/select.test.ts +216 -257
- package/test/skills.test.ts +3 -3
- package/test/sql-shim.test.ts +154 -0
- package/test/state.test.ts +43 -36
- package/test/summary.test.ts +38 -27
- package/test/tier-plan.test.ts +45 -62
- package/test/toast-logic.test.ts +31 -31
- package/test/tokens.test.ts +95 -80
- package/test/trust-attribution.test.ts +217 -187
- package/test/trust-window.test.ts +37 -32
- package/test/turn.test.ts +55 -23
- package/test/upstreams.test.ts +13 -13
- package/test/views.test.ts +81 -59
- package/test/wire-request.test.ts +17 -17
- package/test/wire-responses.test.ts +4 -4
- package/tools/agentdox-e2e.ts +5 -2
- package/tools/export-benchmarks.ts +5 -5
- package/tools/ledger-parity.ts +266 -0
- package/tools/replay.ts +16 -8
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The store's schema, for either engine.
|
|
3
|
+
*
|
|
4
|
+
* `util/sqlite.ts` remains the migration path for a SQLite FILE that already
|
|
5
|
+
* exists: nineteen versions have shipped, and an old file still needs its
|
|
6
|
+
* `ALTER TABLE`s applied in order. This module declares the FINAL shape of
|
|
7
|
+
* every table instead, which is what a fresh store needs — a Postgres database
|
|
8
|
+
* has no history to migrate, and on an already-migrated file every statement
|
|
9
|
+
* here is a no-op.
|
|
10
|
+
*
|
|
11
|
+
* The two must agree, so the definitions below are transcribed from
|
|
12
|
+
* `util/sqlite.ts` with its incremental columns folded in, and
|
|
13
|
+
* `test/schema.test.ts` compares the two engine-by-engine rather than trusting
|
|
14
|
+
* that they were copied correctly.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import type { SqlDb } from "./sql.ts";
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Creates every table and index the router uses. Idempotent, so boot order
|
|
21
|
+
* never matters — the property the SQLite bootstrap has always had.
|
|
22
|
+
*/
|
|
23
|
+
export async function migrateStore(db: SqlDb): Promise<void> {
|
|
24
|
+
const json = db.type("json");
|
|
25
|
+
const float = db.type("float");
|
|
26
|
+
const big = db.type("bigint");
|
|
27
|
+
// A single-row cache keyed on a constant: `CHECK (id = 1)` is portable, and
|
|
28
|
+
// it is what keeps a second payload from ever accumulating.
|
|
29
|
+
const singleton = (name: string, extra = ""): string =>
|
|
30
|
+
`CREATE TABLE IF NOT EXISTS ${name} (
|
|
31
|
+
id INTEGER PRIMARY KEY CHECK (id = 1),
|
|
32
|
+
payload ${json} NOT NULL,
|
|
33
|
+
fetched_at_ms ${big} NOT NULL${extra}
|
|
34
|
+
)`;
|
|
35
|
+
|
|
36
|
+
const statements: string[] = [
|
|
37
|
+
singleton("catalog_cache", `,\n\t\t\tetag TEXT,\n\t\t\tkey_scoped INTEGER NOT NULL DEFAULT 0`),
|
|
38
|
+
singleton("ollama_catalog_cache"),
|
|
39
|
+
singleton("benchmark_cache"),
|
|
40
|
+
singleton("local_scores"),
|
|
41
|
+
|
|
42
|
+
// One row per dispatched upstream generation. The columns nineteen
|
|
43
|
+
// migrations added are declared here as they finally stand.
|
|
44
|
+
`CREATE TABLE IF NOT EXISTS ledger (
|
|
45
|
+
id TEXT PRIMARY KEY,
|
|
46
|
+
created_at_ms ${big} NOT NULL,
|
|
47
|
+
conversation_key TEXT NOT NULL,
|
|
48
|
+
session_id TEXT NOT NULL,
|
|
49
|
+
turn INTEGER NOT NULL,
|
|
50
|
+
requested_model TEXT NOT NULL,
|
|
51
|
+
harness_id TEXT NOT NULL DEFAULT '',
|
|
52
|
+
omp_session_id TEXT NOT NULL DEFAULT '',
|
|
53
|
+
slug TEXT NOT NULL,
|
|
54
|
+
served_slug TEXT,
|
|
55
|
+
tier TEXT NOT NULL,
|
|
56
|
+
classification_source TEXT NOT NULL,
|
|
57
|
+
reasons ${json} NOT NULL,
|
|
58
|
+
predicted_usd ${float} NOT NULL,
|
|
59
|
+
reported_usd ${float},
|
|
60
|
+
usage ${json} NOT NULL,
|
|
61
|
+
cost_breakdown ${json},
|
|
62
|
+
attempt INTEGER NOT NULL,
|
|
63
|
+
escalation_signal TEXT,
|
|
64
|
+
latency_ms INTEGER NOT NULL,
|
|
65
|
+
ttft_ms INTEGER,
|
|
66
|
+
finish_reason TEXT,
|
|
67
|
+
wasted INTEGER NOT NULL DEFAULT 0,
|
|
68
|
+
upstream_generation_id TEXT,
|
|
69
|
+
error TEXT,
|
|
70
|
+
error_kind TEXT,
|
|
71
|
+
features ${json},
|
|
72
|
+
score ${float},
|
|
73
|
+
confidence ${float},
|
|
74
|
+
task TEXT,
|
|
75
|
+
classifier_reasons ${json},
|
|
76
|
+
explored_from TEXT,
|
|
77
|
+
hold_arm INTEGER,
|
|
78
|
+
prompt_tokens_saved INTEGER,
|
|
79
|
+
scope TEXT,
|
|
80
|
+
redactions INTEGER
|
|
81
|
+
)`,
|
|
82
|
+
"CREATE INDEX IF NOT EXISTS idx_ledger_conversation ON ledger (conversation_key)",
|
|
83
|
+
"CREATE INDEX IF NOT EXISTS idx_ledger_created ON ledger (created_at_ms)",
|
|
84
|
+
"CREATE INDEX IF NOT EXISTS idx_ledger_slug ON ledger (slug)",
|
|
85
|
+
// Per-slug newest-first reads (the latency window, any windowed trust).
|
|
86
|
+
// Without this they sorted every row the slug ever had: measured on a
|
|
87
|
+
// real ledger, 5-10ms and RISING with history against a flat 0.04-0.08ms.
|
|
88
|
+
"CREATE INDEX IF NOT EXISTS idx_ledger_slug_created ON ledger (slug, created_at_ms DESC)",
|
|
89
|
+
"CREATE INDEX IF NOT EXISTS idx_ledger_harness_created ON ledger (harness_id, created_at_ms DESC)",
|
|
90
|
+
"CREATE INDEX IF NOT EXISTS idx_ledger_slug_harness_created ON ledger (slug, harness_id, created_at_ms DESC)",
|
|
91
|
+
"CREATE INDEX IF NOT EXISTS idx_ledger_session ON ledger (omp_session_id, created_at_ms DESC)",
|
|
92
|
+
|
|
93
|
+
`CREATE TABLE IF NOT EXISTS token_calibration (
|
|
94
|
+
tokenizer TEXT PRIMARY KEY,
|
|
95
|
+
est_bytes ${big} NOT NULL,
|
|
96
|
+
actual_tokens ${big} NOT NULL,
|
|
97
|
+
samples ${big} NOT NULL
|
|
98
|
+
)`,
|
|
99
|
+
|
|
100
|
+
// Per-conversation routing memory: the sticky window, which model's
|
|
101
|
+
// prompt cache is warm, and the accumulated spend the budget guard reads.
|
|
102
|
+
`CREATE TABLE IF NOT EXISTS conversations (
|
|
103
|
+
key TEXT PRIMARY KEY,
|
|
104
|
+
session_id TEXT NOT NULL,
|
|
105
|
+
turn INTEGER NOT NULL DEFAULT 0,
|
|
106
|
+
current_slug TEXT,
|
|
107
|
+
current_tier TEXT,
|
|
108
|
+
sticky_until_turn INTEGER NOT NULL DEFAULT 0,
|
|
109
|
+
escalations INTEGER NOT NULL DEFAULT 0,
|
|
110
|
+
spent_usd ${float} NOT NULL DEFAULT 0,
|
|
111
|
+
last_prompt_tokens INTEGER NOT NULL DEFAULT 0,
|
|
112
|
+
cache_warm_slug TEXT,
|
|
113
|
+
cache_warm_at_ms ${big} NOT NULL DEFAULT 0,
|
|
114
|
+
context_version TEXT,
|
|
115
|
+
context_fetched_at_ms ${big} NOT NULL DEFAULT 0,
|
|
116
|
+
compaction_plan ${json},
|
|
117
|
+
compaction_plan_tokens INTEGER NOT NULL DEFAULT 0,
|
|
118
|
+
upgrade_deferred_tier TEXT,
|
|
119
|
+
updated_at_ms ${big} NOT NULL DEFAULT 0
|
|
120
|
+
)`,
|
|
121
|
+
|
|
122
|
+
// agentdox bridge. Blocks are content-addressed so many conversations on
|
|
123
|
+
// one project share a copy, and so a restart can re-inject the SAME bytes
|
|
124
|
+
// a conversation was already using (the upstream cache outlives us).
|
|
125
|
+
`CREATE TABLE IF NOT EXISTS context_blocks (
|
|
126
|
+
version TEXT PRIMARY KEY,
|
|
127
|
+
scope TEXT NOT NULL,
|
|
128
|
+
block TEXT NOT NULL,
|
|
129
|
+
fetched_at_ms ${big} NOT NULL
|
|
130
|
+
)`,
|
|
131
|
+
|
|
132
|
+
// Plan-meter readings beside the ledger's own Ollama total at the same
|
|
133
|
+
// instant, so the estimate can be calibrated against the bill.
|
|
134
|
+
`CREATE TABLE IF NOT EXISTS ollama_meter_samples (
|
|
135
|
+
at_ms ${big} PRIMARY KEY,
|
|
136
|
+
meter_usd ${float} NOT NULL,
|
|
137
|
+
ledger_usd ${float} NOT NULL
|
|
138
|
+
)`,
|
|
139
|
+
|
|
140
|
+
// User verdicts on routed turns, tied to the ledger row judged.
|
|
141
|
+
`CREATE TABLE IF NOT EXISTS feedback (
|
|
142
|
+
id TEXT PRIMARY KEY,
|
|
143
|
+
ledger_id TEXT NOT NULL,
|
|
144
|
+
omp_session_id TEXT NOT NULL DEFAULT '',
|
|
145
|
+
slug TEXT NOT NULL,
|
|
146
|
+
tier TEXT NOT NULL,
|
|
147
|
+
verdict TEXT NOT NULL,
|
|
148
|
+
note TEXT NOT NULL DEFAULT '',
|
|
149
|
+
created_at_ms ${big} NOT NULL
|
|
150
|
+
)`,
|
|
151
|
+
"CREATE INDEX IF NOT EXISTS idx_feedback_created ON feedback (created_at_ms)",
|
|
152
|
+
"CREATE INDEX IF NOT EXISTS idx_feedback_ledger ON feedback (ledger_id)",
|
|
153
|
+
|
|
154
|
+
// Small durable markers (when the daily summary was last posted, per harness).
|
|
155
|
+
`CREATE TABLE IF NOT EXISTS router_kv (
|
|
156
|
+
key TEXT PRIMARY KEY,
|
|
157
|
+
value TEXT NOT NULL,
|
|
158
|
+
updated_at_ms ${big} NOT NULL
|
|
159
|
+
)`,
|
|
160
|
+
|
|
161
|
+
`CREATE TABLE IF NOT EXISTS agentdox_sessions (
|
|
162
|
+
conversation_key TEXT PRIMARY KEY,
|
|
163
|
+
scope TEXT NOT NULL,
|
|
164
|
+
session_id TEXT NOT NULL,
|
|
165
|
+
created_at_ms ${big} NOT NULL
|
|
166
|
+
)`,
|
|
167
|
+
];
|
|
168
|
+
|
|
169
|
+
// `IF NOT EXISTS` is not atomic on Postgres: two replicas booting against a
|
|
170
|
+
// fresh database both pass the existence check and the loser fails on the
|
|
171
|
+
// unique index over pg_type (23505), or on the table name itself (42P07 /
|
|
172
|
+
// 42710). Measured: one of two replicas started together died with
|
|
173
|
+
// "duplicate key value violates unique constraint pg_type_typname_nsp_index".
|
|
174
|
+
// The condition those errors report is the condition the statement asked to
|
|
175
|
+
// tolerate, so they are the success case arriving from the other replica.
|
|
176
|
+
const RACED = new Set(["23505", "42P07", "42710"]);
|
|
177
|
+
for (const statement of statements) {
|
|
178
|
+
try {
|
|
179
|
+
await db.sql.unsafe(statement);
|
|
180
|
+
} catch (err) {
|
|
181
|
+
const code = (err as { errno?: unknown }).errno;
|
|
182
|
+
if (!RACED.has(String(code))) throw err;
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/** Every table `migrateStore` creates, for tests and for teardown. */
|
|
188
|
+
export const STORE_TABLES = [
|
|
189
|
+
"catalog_cache",
|
|
190
|
+
"ollama_catalog_cache",
|
|
191
|
+
"benchmark_cache",
|
|
192
|
+
"local_scores",
|
|
193
|
+
"ledger",
|
|
194
|
+
"token_calibration",
|
|
195
|
+
"conversations",
|
|
196
|
+
"context_blocks",
|
|
197
|
+
"ollama_meter_samples",
|
|
198
|
+
"feedback",
|
|
199
|
+
"router_kv",
|
|
200
|
+
"agentdox_sessions",
|
|
201
|
+
] as const;
|
package/src/util/sql.ts
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* One SQL handle for both engines.
|
|
3
|
+
*
|
|
4
|
+
* The ledger has to live in the same database as the rest of a deployment's
|
|
5
|
+
* state when that state is shared (spend read from a file that lags the
|
|
6
|
+
* database it must agree with is a cap that silently over-admits), and in a
|
|
7
|
+
* local file when it is not. `Bun.SQL` speaks both, so the ledger is written
|
|
8
|
+
* once against this shim rather than twice.
|
|
9
|
+
*
|
|
10
|
+
* Only three things actually differ, and they are the three things here:
|
|
11
|
+
*
|
|
12
|
+
* 1. **JSON access.** `json_extract(usage, '$.promptTokens')` against
|
|
13
|
+
* `(usage->>'promptTokens')::numeric`.
|
|
14
|
+
* 2. **Numeric results.** Postgres returns `COUNT(*)` and `SUM(BIGINT)` as
|
|
15
|
+
* STRINGS. Arithmetic on those silently produces wrong answers rather than
|
|
16
|
+
* throwing — a measured trust score came out 0.9756 instead of 0.9726 this
|
|
17
|
+
* way — so every numeric read goes through `num`.
|
|
18
|
+
* 3. **Column types.** `JSONB`/`DOUBLE PRECISION` against `TEXT`/`REAL`.
|
|
19
|
+
*
|
|
20
|
+
* Everything else — parameters, `IN ${sql(array)}`, window functions,
|
|
21
|
+
* `ON CONFLICT`, `DELETE ... RETURNING`, bulk insert, transactions — is
|
|
22
|
+
* identical on both, verified by `tools/dialect-probe.ts`.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
import { SQL } from "bun";
|
|
26
|
+
|
|
27
|
+
export type Dialect = "sqlite" | "postgres";
|
|
28
|
+
|
|
29
|
+
export interface SqlDb {
|
|
30
|
+
readonly sql: SQL;
|
|
31
|
+
readonly dialect: Dialect;
|
|
32
|
+
/** A JSON member as a NUMBER-typed SQL expression, for aggregates and comparisons. */
|
|
33
|
+
jsonNum(column: string, key: string): string;
|
|
34
|
+
/** A JSON member as TEXT, for equality against string values. */
|
|
35
|
+
jsonText(column: string, key: string): string;
|
|
36
|
+
/**
|
|
37
|
+
* A NESTED JSON member as a number, e.g. `features.anatomy.messages`.
|
|
38
|
+
*
|
|
39
|
+
* `jsonNum` cannot express this: SQLite takes a whole path in one string
|
|
40
|
+
* (`'$.anatomy.messages'`), while Postgres' `->>` reads a SINGLE key, so a
|
|
41
|
+
* dotted key silently reads NULL there — a whole report section came back
|
|
42
|
+
* null rather than failing.
|
|
43
|
+
*/
|
|
44
|
+
jsonPathNum(column: string, path: readonly string[]): string;
|
|
45
|
+
/**
|
|
46
|
+
* A JSON member that holds a BOOLEAN, as 1/0.
|
|
47
|
+
*
|
|
48
|
+
* The engines disagree twice over: SQLite's `json_extract` yields the
|
|
49
|
+
* INTEGER 1 for JSON `true`, Postgres' `->>` yields the TEXT 'true', and
|
|
50
|
+
* SQLite's `IN` does not coerce between them. Comparing the wrong way round
|
|
51
|
+
* silently misclassifies every row — it counted router-ESTIMATED cache hits
|
|
52
|
+
* as measured ones and inflated cache reliability samples by 2%.
|
|
53
|
+
*/
|
|
54
|
+
jsonBool(column: string, key: string): string;
|
|
55
|
+
/**
|
|
56
|
+
* Substring test as a boolean expression. SQLite has `instr(haystack,
|
|
57
|
+
* needle) > 0`; Postgres spells it `position(needle in haystack) > 0`.
|
|
58
|
+
*/
|
|
59
|
+
contains(haystack: string, needle: string): string;
|
|
60
|
+
/**
|
|
61
|
+
* An epoch-millisecond column as a `YYYY-MM-DD` UTC day. SQLite has
|
|
62
|
+
* `strftime(..., 'unixepoch')`; Postgres needs `to_timestamp` plus an
|
|
63
|
+
* explicit UTC conversion, or the server's timezone silently decides which
|
|
64
|
+
* day a turn was billed on.
|
|
65
|
+
*/
|
|
66
|
+
utcDay(msColumn: string): string;
|
|
67
|
+
/** Whether a table exists, without reading engine-specific catalog tables. */
|
|
68
|
+
tableExists(name: string): Promise<boolean>;
|
|
69
|
+
/**
|
|
70
|
+
* Runs SQL written with `$name` placeholders, whichever way the engine
|
|
71
|
+
* wants them numbered.
|
|
72
|
+
*
|
|
73
|
+
* The reporting queries are assembled from optional filters (a harness
|
|
74
|
+
* scope, an upper time bound), which a tagged template cannot express — the
|
|
75
|
+
* shape of the statement is decided at runtime. Rewriting them as string
|
|
76
|
+
* concatenation with positional parameters would renumber every bind by
|
|
77
|
+
* hand, which is how a filter ends up reading the wrong column.
|
|
78
|
+
*/
|
|
79
|
+
query<T>(text: string, binds?: Record<string, unknown>): Promise<T[]>;
|
|
80
|
+
/** `query`, for a statement that yields at most one row. */
|
|
81
|
+
one<T>(text: string, binds?: Record<string, unknown>): Promise<T | null>;
|
|
82
|
+
/** A column type that differs between engines. */
|
|
83
|
+
type(kind: "json" | "float" | "bigint"): string;
|
|
84
|
+
/**
|
|
85
|
+
* Cast suffix for a parameter that may be NULL. Postgres refuses to infer a
|
|
86
|
+
* type for a bare NULL placeholder ("could not determine data type of
|
|
87
|
+
* parameter $2"), which the `(${x} IS NULL OR col = ${x})` idiom for an
|
|
88
|
+
* optional filter relies on; SQLite has no cast syntax to add. Empty there.
|
|
89
|
+
*/
|
|
90
|
+
readonly nullableText: string;
|
|
91
|
+
/**
|
|
92
|
+
* Two-argument scalar minimum. SQLite spells it `MIN(a, b)`; Postgres
|
|
93
|
+
* reserves `MIN` for the aggregate and needs `LEAST(a, b)`.
|
|
94
|
+
*/
|
|
95
|
+
least(a: string, b: string): string;
|
|
96
|
+
close(): Promise<void>;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* `sqlite://` (or a bare path) and `postgres://`/`postgresql://` URLs. A bare
|
|
101
|
+
* path is accepted because the router's config has always taken
|
|
102
|
+
* `ledger.path`, and a deployment that never opts into Postgres should not
|
|
103
|
+
* have to learn a URL scheme.
|
|
104
|
+
*/
|
|
105
|
+
export function dialectOf(url: string): Dialect {
|
|
106
|
+
return url.startsWith("postgres://") || url.startsWith("postgresql://") ? "postgres" : "sqlite";
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
export function openSqlDb(url: string): SqlDb {
|
|
110
|
+
const dialect = dialectOf(url);
|
|
111
|
+
const target = dialect === "sqlite" && !url.startsWith("sqlite:") ? `sqlite://${url}` : url;
|
|
112
|
+
const sql = new SQL(target);
|
|
113
|
+
const pg = dialect === "postgres";
|
|
114
|
+
return {
|
|
115
|
+
sql,
|
|
116
|
+
dialect,
|
|
117
|
+
jsonNum: (column, key) => (pg ? `(${column}->>'${key}')::numeric` : `json_extract(${column}, '$.${key}')`),
|
|
118
|
+
jsonText: (column, key) => (pg ? `(${column}->>'${key}')` : `json_extract(${column}, '$.${key}')`),
|
|
119
|
+
jsonPathNum: (column, path) =>
|
|
120
|
+
pg
|
|
121
|
+
? `(${column}#>>'{${path.join(",")}}')::numeric`
|
|
122
|
+
: `json_extract(${column}, '$.${path.join(".")}')`,
|
|
123
|
+
jsonBool: (column, key) =>
|
|
124
|
+
pg
|
|
125
|
+
? `CASE WHEN (${column}->>'${key}') IN ('true', '1') THEN 1 ELSE 0 END`
|
|
126
|
+
: `CASE WHEN json_extract(${column}, '$.${key}') IN (1, 'true', '1') THEN 1 ELSE 0 END`,
|
|
127
|
+
type: (kind) => {
|
|
128
|
+
if (kind === "json") return pg ? "JSONB" : "TEXT";
|
|
129
|
+
if (kind === "float") return pg ? "DOUBLE PRECISION" : "REAL";
|
|
130
|
+
return pg ? "BIGINT" : "INTEGER";
|
|
131
|
+
},
|
|
132
|
+
nullableText: pg ? "::text" : "",
|
|
133
|
+
least: (a, b) => (pg ? `LEAST(${a}, ${b})` : `MIN(${a}, ${b})`),
|
|
134
|
+
// The haystack is cast to text: a JSON column is `jsonb` on Postgres and
|
|
135
|
+
// `position()` refuses it, while sqlite stores the same column as TEXT.
|
|
136
|
+
contains: (haystack, needle) => (pg ? `position(${needle} in (${haystack})::text) > 0` : `instr(${haystack}, ${needle}) > 0`),
|
|
137
|
+
utcDay: (msColumn) =>
|
|
138
|
+
pg
|
|
139
|
+
? `to_char(to_timestamp(${msColumn} / 1000) AT TIME ZONE 'UTC', 'YYYY-MM-DD')`
|
|
140
|
+
: `strftime('%Y-%m-%d', ${msColumn} / 1000, 'unixepoch')`,
|
|
141
|
+
tableExists: async (name) => {
|
|
142
|
+
// Cheaper and more portable than either catalog table: ask for nothing
|
|
143
|
+
// from it and see whether the statement plans.
|
|
144
|
+
try {
|
|
145
|
+
await sql.unsafe(`SELECT 1 FROM ${name} WHERE 1 = 0`);
|
|
146
|
+
return true;
|
|
147
|
+
} catch {
|
|
148
|
+
return false;
|
|
149
|
+
}
|
|
150
|
+
},
|
|
151
|
+
query: async <T>(text: string, binds: Record<string, unknown> = {}): Promise<T[]> => {
|
|
152
|
+
const { text: prepared, values } = bindNamed(text, binds, pg);
|
|
153
|
+
return (await sql.unsafe(prepared, values)) as T[];
|
|
154
|
+
},
|
|
155
|
+
one: async <T>(text: string, binds: Record<string, unknown> = {}): Promise<T | null> => {
|
|
156
|
+
const { text: prepared, values } = bindNamed(text, binds, pg);
|
|
157
|
+
const rows = (await sql.unsafe(prepared, values)) as T[];
|
|
158
|
+
return rows[0] ?? null;
|
|
159
|
+
},
|
|
160
|
+
close: async () => {
|
|
161
|
+
await sql.end();
|
|
162
|
+
},
|
|
163
|
+
};
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Rewrites `$name` placeholders to the engine's positional form, in order of
|
|
169
|
+
* first appearance, and returns the matching value array.
|
|
170
|
+
*
|
|
171
|
+
* A name may repeat — `created_at_ms >= $since` and a `CASE` on `$since` in
|
|
172
|
+
* the same statement is normal — and repeats must reuse one parameter slot on
|
|
173
|
+
* Postgres. JSON paths (`'$.isSubagent'`) are not placeholders: the pattern
|
|
174
|
+
* requires a letter or underscore after the `$`, which `$.` fails.
|
|
175
|
+
*/
|
|
176
|
+
export function bindNamed(text: string, binds: Record<string, unknown>, pg: boolean): { text: string; values: unknown[] } {
|
|
177
|
+
// Callers that were written against bun:sqlite pass their binds keyed WITH
|
|
178
|
+
// the sigil (`{ $since: 0 }`), which is how every reporting query in this
|
|
179
|
+
// repo already builds them; both forms resolve.
|
|
180
|
+
const valueOf = (name: string): unknown => {
|
|
181
|
+
if (name in binds) return binds[name];
|
|
182
|
+
const sigil = `$${name}`;
|
|
183
|
+
if (sigil in binds) return binds[sigil];
|
|
184
|
+
throw new Error(`sql: no value bound for $${name}`);
|
|
185
|
+
};
|
|
186
|
+
const order: string[] = [];
|
|
187
|
+
const prepared = text.replace(/\$([a-zA-Z_][a-zA-Z0-9_]*)/g, (_match, name: string) => {
|
|
188
|
+
valueOf(name);
|
|
189
|
+
let index = order.indexOf(name);
|
|
190
|
+
if (index === -1) {
|
|
191
|
+
order.push(name);
|
|
192
|
+
index = order.length - 1;
|
|
193
|
+
}
|
|
194
|
+
return pg ? `$${index + 1}` : "?";
|
|
195
|
+
});
|
|
196
|
+
if (!pg) {
|
|
197
|
+
// SQLite's positional `?` cannot reuse a slot, so a repeated name is
|
|
198
|
+
// bound once per occurrence rather than once per name.
|
|
199
|
+
const values: unknown[] = [];
|
|
200
|
+
text.replace(/\$([a-zA-Z_][a-zA-Z0-9_]*)/g, (_match, name: string) => {
|
|
201
|
+
values.push(valueOf(name));
|
|
202
|
+
return "";
|
|
203
|
+
});
|
|
204
|
+
return { text: prepared, values };
|
|
205
|
+
}
|
|
206
|
+
return { text: prepared, values: order.map(valueOf) };
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Coerces a value Postgres may have returned as a string. Applied at every
|
|
211
|
+
* numeric read: `Number(null)` is 0, which is wrong for a nullable aggregate,
|
|
212
|
+
* so null and undefined are preserved.
|
|
213
|
+
*/
|
|
214
|
+
export function num(value: unknown): number {
|
|
215
|
+
return typeof value === "number" ? value : Number(value ?? 0);
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
/** `num`, but a missing value stays missing rather than becoming 0. */
|
|
219
|
+
export function numOrNull(value: unknown): number | null {
|
|
220
|
+
if (value === null || value === undefined) return null;
|
|
221
|
+
return typeof value === "number" ? value : Number(value);
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* JSON for a `json` column. Postgres' driver encodes a JS string destined for
|
|
226
|
+
* `JSONB` as a JSON *string* — `jsonb_typeof` reads `'string'` and every
|
|
227
|
+
* `->>` on it returns NULL — so an object must be passed through unstringified
|
|
228
|
+
* there, while SQLite's TEXT column needs the serialised form.
|
|
229
|
+
*/
|
|
230
|
+
export function jsonParam(db: SqlDb, value: unknown): unknown {
|
|
231
|
+
if (value === null || value === undefined) return null;
|
|
232
|
+
return db.dialect === "postgres" ? value : JSON.stringify(value);
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
/** Reads a `json` column back, whichever way it was stored. */
|
|
236
|
+
export function jsonValue<T>(value: unknown): T | null {
|
|
237
|
+
if (value === null || value === undefined) return null;
|
|
238
|
+
if (typeof value === "string") {
|
|
239
|
+
try {
|
|
240
|
+
return JSON.parse(value) as T;
|
|
241
|
+
} catch {
|
|
242
|
+
return null;
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
return value as T;
|
|
246
|
+
}
|
|
@@ -25,8 +25,7 @@
|
|
|
25
25
|
*/
|
|
26
26
|
|
|
27
27
|
import { encoder, sseDataFrame } from "../../util/sse.ts";
|
|
28
|
-
import type
|
|
29
|
-
import { estimateTokens } from "../../tokens/estimate.ts";
|
|
28
|
+
import { estimateTokens, type TokenRatio } from "../../tokens/estimate.ts";
|
|
30
29
|
import type { NormRequest, ResponseSink, TurnSummary, UpstreamChunk, WireError } from "../types.ts";
|
|
31
30
|
import { invalidRequest, WireErrorException } from "../openai/errors.ts";
|
|
32
31
|
import { parseChatRequest } from "../openai/request.ts";
|
|
@@ -217,9 +216,9 @@ export function parseMessagesRequest(body: unknown, headers: Headers, models: Re
|
|
|
217
216
|
}
|
|
218
217
|
|
|
219
218
|
/** `POST /v1/messages/count_tokens`: the router's own estimate over the prompt bytes. */
|
|
220
|
-
export function countAnthropicTokens(body: unknown, models: Record<string, string>,
|
|
219
|
+
export function countAnthropicTokens(body: unknown, models: Record<string, string>, ratio: TokenRatio): number {
|
|
221
220
|
const norm = parseChatRequest(messagesToChatBody({ ...(isRec(body) ? body : {}), stream: false }, models), new Headers());
|
|
222
|
-
return estimateTokens(norm.promptBytes, "anthropic",
|
|
221
|
+
return estimateTokens(norm.promptBytes, "anthropic", ratio);
|
|
223
222
|
}
|
|
224
223
|
|
|
225
224
|
// ---------------------------------------------------------------------------
|
package/src/wire/types.ts
CHANGED
|
@@ -128,6 +128,13 @@ export interface NormRequest {
|
|
|
128
128
|
policy?: RequestPolicy;
|
|
129
129
|
/** Virtual model the client selected, e.g. `auto`, `auto-cheap`, `auto-max`. */
|
|
130
130
|
requestedModel: string;
|
|
131
|
+
/**
|
|
132
|
+
* The client's `model` string before the provider prefix was stripped, so a
|
|
133
|
+
* vendor-qualified catalog slug (`deepseek/deepseek-v4.1-flash`) stays
|
|
134
|
+
* distinguishable from a profile id. Absent ⇒ `requestedModel` is the whole
|
|
135
|
+
* of what the client asked for.
|
|
136
|
+
*/
|
|
137
|
+
requestedModelFull?: string;
|
|
131
138
|
messages: NormMessage[];
|
|
132
139
|
tools: NormTool[];
|
|
133
140
|
/** True when the client forced a specific tool. */
|
|
@@ -42,7 +42,7 @@ const CLAUDE_CODE_BODY = {
|
|
|
42
42
|
};
|
|
43
43
|
|
|
44
44
|
describe("messagesToChatBody", () => {
|
|
45
|
-
test("translates a Claude Code turn: system blocks, tool_use/tool_result, custom tools only, tool_choice, thinking budget", () => {
|
|
45
|
+
test("translates a Claude Code turn: system blocks, tool_use/tool_result, custom tools only, tool_choice, thinking budget", async () => {
|
|
46
46
|
const b = messagesToChatBody(CLAUDE_CODE_BODY);
|
|
47
47
|
expect(b.model).toBe("auto");
|
|
48
48
|
const messages = b.messages as { role: string; content: unknown; tool_calls?: unknown; tool_call_id?: string }[];
|
|
@@ -63,7 +63,7 @@ describe("messagesToChatBody", () => {
|
|
|
63
63
|
for (const k of ["system", "metadata", "thinking", "tool_choice_anthropic", "cache_control"]) expect(k in b && k !== "tool_choice").toBe(false);
|
|
64
64
|
});
|
|
65
65
|
|
|
66
|
-
test("string bodies, images, documents, error tool results, tool_choice variants, stop sequences, effort", () => {
|
|
66
|
+
test("string bodies, images, documents, error tool results, tool_choice variants, stop sequences, effort", async () => {
|
|
67
67
|
const b = messagesToChatBody({
|
|
68
68
|
model: "auto-max",
|
|
69
69
|
system: "sys",
|
|
@@ -97,7 +97,7 @@ describe("messagesToChatBody", () => {
|
|
|
97
97
|
expect(messagesToChatBody({ model: "m", messages: [{ role: "user", content: "x" }], thinking: { type: "adaptive" } }).reasoning).toEqual({ effort: "medium" });
|
|
98
98
|
});
|
|
99
99
|
|
|
100
|
-
test("rejects what cannot be a turn", () => {
|
|
100
|
+
test("rejects what cannot be a turn", async () => {
|
|
101
101
|
expect(() => messagesToChatBody("nope")).toThrow("JSON object");
|
|
102
102
|
expect(() => messagesToChatBody({ model: "", messages: [{ role: "user", content: "x" }] })).toThrow("model");
|
|
103
103
|
expect(() => messagesToChatBody({ model: "m", messages: [] })).toThrow("messages");
|
|
@@ -105,7 +105,7 @@ describe("messagesToChatBody", () => {
|
|
|
105
105
|
expect(() => messagesToChatBody({ model: "m", messages: [{ role: "user", content: 5 }] })).toThrow("content");
|
|
106
106
|
});
|
|
107
107
|
|
|
108
|
-
test("model names map by glob, first match wins, profile ids pass through", () => {
|
|
108
|
+
test("model names map by glob, first match wins, profile ids pass through", async () => {
|
|
109
109
|
expect(mapAnthropicModel("claude-haiku-4-5-20251001")).toBe("auto-cheap");
|
|
110
110
|
expect(mapAnthropicModel("claude-opus-4-8")).toBe("auto");
|
|
111
111
|
expect(mapAnthropicModel("auto-sub")).toBe("auto-sub");
|
|
@@ -115,7 +115,7 @@ describe("messagesToChatBody", () => {
|
|
|
115
115
|
});
|
|
116
116
|
|
|
117
117
|
describe("parseMessagesRequest", () => {
|
|
118
|
-
test("derives the harness from the user agent and the session from metadata; explicit headers win; the rendered body is chat-shaped", () => {
|
|
118
|
+
test("derives the harness from the user agent and the session from metadata; explicit headers win; the rendered body is chat-shaped", async () => {
|
|
119
119
|
const headers = new Headers({ "user-agent": "claude-cli/2.1.263 (external, cli)", "x-api-key": "k", "anthropic-version": "2023-06-01" });
|
|
120
120
|
const norm = parseMessagesRequest(CLAUDE_CODE_BODY, headers);
|
|
121
121
|
expect(norm.protocol).toBe("anthropic-messages");
|
|
@@ -138,7 +138,7 @@ describe("parseMessagesRequest", () => {
|
|
|
138
138
|
expect(anthropicIdentityHeaders({}, new Headers({ "user-agent": "python-requests" })).get("x-omp-harness")).toBe("anthropic");
|
|
139
139
|
});
|
|
140
140
|
|
|
141
|
-
test("the agentdox scope and layer headers reach the request on the Anthropic path too", () => {
|
|
141
|
+
test("the agentdox scope and layer headers reach the request on the Anthropic path too", async () => {
|
|
142
142
|
const plain = parseMessagesRequest(CLAUDE_CODE_BODY, new Headers({ "user-agent": "claude-cli/2.1.263" }));
|
|
143
143
|
expect(plain.agentdoxScope).toBe("");
|
|
144
144
|
expect(plain.agentdoxGroup).toBe("");
|
|
@@ -154,13 +154,13 @@ describe("parseMessagesRequest", () => {
|
|
|
154
154
|
expect(parseMessagesRequest(CLAUDE_CODE_BODY, new Headers({ "x-agentdox-group": "Not A Slug" })).agentdoxGroup).toBe("");
|
|
155
155
|
});
|
|
156
156
|
|
|
157
|
-
test("the origin fingerprint reaches the request on the Anthropic path too", () => {
|
|
157
|
+
test("the origin fingerprint reaches the request on the Anthropic path too", async () => {
|
|
158
158
|
expect(parseMessagesRequest(CLAUDE_CODE_BODY, new Headers({ "user-agent": "claude-cli/2.1.263" })).agentdoxOrigin).toBe("");
|
|
159
159
|
expect(parseMessagesRequest(CLAUDE_CODE_BODY, new Headers({ "x-agentdox-origin": "github.com/drewappling/omp-router" })).agentdoxOrigin).toBe("github.com/drewappling/omp-router");
|
|
160
160
|
expect(parseMessagesRequest(CLAUDE_CODE_BODY, new Headers({ "x-agentdox-origin": "https://github.com/a/b" })).agentdoxOrigin).toBe("");
|
|
161
161
|
});
|
|
162
162
|
|
|
163
|
-
test("a request captured from Claude Code 2.1: system inside messages, JSON user_id, adaptive thinking with effort, 23 custom tools", () => {
|
|
163
|
+
test("a request captured from Claude Code 2.1: system inside messages, JSON user_id, adaptive thinking with effort, 23 custom tools", async () => {
|
|
164
164
|
const fixture = JSON.parse(readFileSync("test/fixtures/harness/claude-code.json", "utf8")) as { headers: Record<string, string>; body: Record<string, unknown> };
|
|
165
165
|
const norm = parseMessagesRequest(fixture.body, new Headers(fixture.headers));
|
|
166
166
|
expect(norm.harnessId).toBe("claude-code");
|
|
@@ -179,7 +179,7 @@ describe("parseMessagesRequest", () => {
|
|
|
179
179
|
expect(sessionFromUserId("nothing here")).toBeNull();
|
|
180
180
|
});
|
|
181
181
|
|
|
182
|
-
test("count_tokens estimates from the prompt bytes", () => {
|
|
182
|
+
test("count_tokens estimates from the prompt bytes", async () => {
|
|
183
183
|
expect(countAnthropicTokens(CLAUDE_CODE_BODY, DEFAULT_CONFIG.anthropic.models, null)).toBeGreaterThan(50);
|
|
184
184
|
expect(() => countAnthropicTokens({ model: "m", messages: [] }, {}, null)).toThrow("messages");
|
|
185
185
|
});
|
|
@@ -40,7 +40,7 @@ function blScore(over: Partial<FeedScore> & { key: string }): FeedScore {
|
|
|
40
40
|
}
|
|
41
41
|
|
|
42
42
|
describe("normalizeModelKey", () => {
|
|
43
|
-
test("strips provider, tilde, and release words but keeps the parameter size", () => {
|
|
43
|
+
test("strips provider, tilde, and release words but keeps the parameter size", async () => {
|
|
44
44
|
expect(normalizeModelKey("z-ai/glm-5.3-flash")).toBe("glm-5-3-flash");
|
|
45
45
|
expect(normalizeModelKey("~deepseek/deepseek-v4-flash-latest")).toBe("deepseek-v4-flash");
|
|
46
46
|
expect(normalizeModelKey("meta/muse-glimmer-30b")).toBe("muse-glimmer-30b");
|
|
@@ -51,7 +51,7 @@ describe("normalizeModelKey", () => {
|
|
|
51
51
|
});
|
|
52
52
|
|
|
53
53
|
describe("parseAaModels", () => {
|
|
54
|
-
test("reads the three indices, keeps in-range values, and skips empty rows", () => {
|
|
54
|
+
test("reads the three indices, keeps in-range values, and skips empty rows", async () => {
|
|
55
55
|
const body = {
|
|
56
56
|
data: [
|
|
57
57
|
{
|
|
@@ -74,7 +74,7 @@ describe("parseAaModels", () => {
|
|
|
74
74
|
});
|
|
75
75
|
|
|
76
76
|
describe("parseBenchlmModels", () => {
|
|
77
|
-
test("maps categories to axes, drops estimated rows, and ignores out-of-range", () => {
|
|
77
|
+
test("maps categories to axes, drops estimated rows, and ignores out-of-range", async () => {
|
|
78
78
|
const body = {
|
|
79
79
|
models: [
|
|
80
80
|
{
|
|
@@ -98,7 +98,7 @@ describe("parseBenchlmModels", () => {
|
|
|
98
98
|
});
|
|
99
99
|
|
|
100
100
|
describe("applyFeedScores", () => {
|
|
101
|
-
test("fills the real gap models and reaches normalizeCatalogModel", () => {
|
|
101
|
+
test("fills the real gap models and reaches normalizeCatalogModel", async () => {
|
|
102
102
|
const catalog = [
|
|
103
103
|
raw("meta/muse-glimmer-30b"),
|
|
104
104
|
raw("z-ai/glm-5.3-flash"),
|
|
@@ -135,7 +135,7 @@ describe("applyFeedScores", () => {
|
|
|
135
135
|
expect(result.sources.benchlm).toBe(3); // muse coding+agentic, glm agentic
|
|
136
136
|
});
|
|
137
137
|
|
|
138
|
-
test("AA wins over BenchLM for the same axis", () => {
|
|
138
|
+
test("AA wins over BenchLM for the same axis", async () => {
|
|
139
139
|
const catalog = [raw("z-ai/glm-5.3-flash")];
|
|
140
140
|
const feeds: FeedScore[] = [
|
|
141
141
|
blScore({ key: "glm-5-3-flash", creator: "z-ai", coding: 10 }),
|
|
@@ -145,7 +145,7 @@ describe("applyFeedScores", () => {
|
|
|
145
145
|
expect(normalizeCatalogModel(catalog[0])?.quality.coding).toBe(61);
|
|
146
146
|
});
|
|
147
147
|
|
|
148
|
-
test("never fuzzy-matches a different model", () => {
|
|
148
|
+
test("never fuzzy-matches a different model", async () => {
|
|
149
149
|
const catalog = [raw("meta/muse-glimmer-30b")];
|
|
150
150
|
// Same family, different model — must not lend its score.
|
|
151
151
|
const feeds: FeedScore[] = [aaScore({ key: "muse-spark-1-2", creator: "meta", coding: 72 })];
|
|
@@ -154,7 +154,7 @@ describe("applyFeedScores", () => {
|
|
|
154
154
|
expect(normalizeCatalogModel(catalog[0])?.quality).toEqual({});
|
|
155
155
|
});
|
|
156
156
|
|
|
157
|
-
test("a shared key with conflicting creators fills only the creator that matches", () => {
|
|
157
|
+
test("a shared key with conflicting creators fills only the creator that matches", async () => {
|
|
158
158
|
const catalog = [raw("z-ai/glm-5.3-flash")];
|
|
159
159
|
const feeds: FeedScore[] = [
|
|
160
160
|
aaScore({ key: "glm-5-3-flash", creator: "someone-else", coding: 5 }),
|