auto-model-router 0.31.0 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +32 -2
- package/omp-extension/router-configure.ts +9 -7
- package/package.json +1 -1
- package/src/cli/config-cmd.ts +8 -7
- package/src/cli/explain.ts +10 -5
- package/src/cli/export.ts +6 -5
- package/src/cli/models.ts +10 -7
- package/src/cli/report.ts +6 -1
- package/src/cli/stats.ts +7 -7
- package/src/config/load.ts +10 -1
- package/src/config/types.ts +10 -1
- package/src/context/bridge.ts +7 -7
- package/src/context/index.ts +3 -3
- package/src/context/store.ts +39 -56
- package/src/context/types.ts +7 -6
- package/src/cost/blended.ts +28 -7
- package/src/cost/feedback.ts +33 -37
- package/src/cost/ledger-sql.ts +547 -0
- package/src/cost/ledger.ts +30 -459
- package/src/cost/report.ts +171 -129
- package/src/cost/retention.ts +10 -10
- package/src/cost/summary.ts +15 -10
- package/src/cost/types.ts +43 -62
- package/src/cost/views.ts +79 -49
- package/src/lib.ts +6 -2
- package/src/router/candidates.ts +7 -15
- package/src/router/classify.ts +6 -4
- package/src/router/index.ts +95 -9
- package/src/router/select.ts +38 -21
- package/src/router/state.ts +90 -102
- package/src/router/types.ts +11 -5
- package/src/server/advise.ts +6 -4
- package/src/server/compaction-digest.ts +1 -1
- package/src/server/digest.ts +9 -10
- package/src/server/http.ts +101 -41
- package/src/server/providers.ts +18 -4
- package/src/server/turn.ts +32 -9
- package/src/tokens/estimate.ts +16 -6
- package/src/upstream/ollama-usage.ts +21 -11
- package/src/util/schema.ts +201 -0
- package/src/util/sql.ts +246 -0
- package/src/wire/anthropic/messages.ts +3 -4
- package/src/wire/openai/request.ts +1 -0
- package/src/wire/types.ts +7 -0
- package/test/anthropic-wire.test.ts +9 -9
- package/test/benchmark-feeds.test.ts +7 -7
- package/test/cache-control.test.ts +7 -7
- package/test/cache-estimate.test.ts +5 -5
- package/test/catalog-view.test.ts +4 -4
- package/test/catalog.test.ts +11 -11
- package/test/classify.test.ts +24 -24
- package/test/compaction.test.ts +20 -20
- package/test/config-wizard.test.ts +32 -32
- package/test/config.test.ts +10 -10
- package/test/connect-harnesses.test.ts +11 -11
- package/test/context-bridge.test.ts +40 -30
- package/test/context-prune.test.ts +43 -36
- package/test/context-query.test.ts +8 -8
- package/test/controls.test.ts +54 -27
- package/test/cost.test.ts +12 -12
- package/test/digest.test.ts +55 -44
- package/test/embed-lifecycle.test.ts +5 -5
- package/test/embed-logic.test.ts +26 -26
- package/test/escalate.test.ts +17 -17
- package/test/eval.test.ts +13 -13
- package/test/executable.test.ts +6 -6
- package/test/exploration.test.ts +19 -20
- package/test/failover.test.ts +22 -21
- package/test/fakes.ts +105 -0
- package/test/features.test.ts +21 -21
- package/test/harness-requests.test.ts +3 -3
- package/test/harness-switch.test.ts +5 -5
- package/test/hold-exploration.test.ts +13 -13
- package/test/hot-reload.test.ts +5 -5
- package/test/learned.test.ts +5 -5
- package/test/ledger-sql.test.ts +342 -0
- package/test/mcp-entry.test.ts +5 -5
- package/test/migrations.test.ts +28 -22
- package/test/models-yml.test.ts +18 -18
- package/test/ollama.test.ts +40 -34
- package/test/omp-credentials.test.ts +16 -16
- package/test/policy.test.ts +3 -3
- package/test/reconfigure.test.ts +4 -4
- package/test/redaction.test.ts +41 -35
- package/test/remote.test.ts +12 -12
- package/test/report-logic.test.ts +8 -8
- package/test/report.test.ts +95 -87
- package/test/retention.test.ts +79 -66
- package/test/schema.test.ts +123 -0
- package/test/scope.test.ts +8 -8
- package/test/select.test.ts +216 -257
- package/test/skills.test.ts +3 -3
- package/test/sql-shim.test.ts +154 -0
- package/test/state.test.ts +43 -36
- package/test/summary.test.ts +38 -27
- package/test/tier-plan.test.ts +45 -62
- package/test/toast-logic.test.ts +31 -31
- package/test/tokens.test.ts +95 -80
- package/test/trust-attribution.test.ts +217 -187
- package/test/trust-window.test.ts +37 -32
- package/test/turn.test.ts +55 -23
- package/test/upstreams.test.ts +13 -13
- package/test/views.test.ts +81 -59
- package/test/wire-request.test.ts +17 -17
- package/test/wire-responses.test.ts +4 -4
- package/tools/agentdox-e2e.ts +5 -2
- package/tools/export-benchmarks.ts +5 -5
- package/tools/ledger-parity.ts +266 -0
- package/tools/replay.ts +16 -8
package/src/cost/views.ts
CHANGED
|
@@ -5,12 +5,13 @@
|
|
|
5
5
|
* cost export by day, harness and model. Served by `/v1/router/spend`,
|
|
6
6
|
* `GET /v1/router/feedback` and `/v1/router/export`, exported from lib.ts, and
|
|
7
7
|
* behind `auto-model-router export`. All of them read only long-stable ledger
|
|
8
|
-
* columns and
|
|
8
|
+
* columns, and they run on either engine through `util/sql.ts`: a front door
|
|
9
|
+
* may be reading a local file while the router writes a shared database.
|
|
9
10
|
*/
|
|
10
11
|
|
|
11
12
|
import { providerOfSlug } from "./report.ts";
|
|
12
|
-
import type
|
|
13
|
-
import {
|
|
13
|
+
import { num, type SqlDb } from "../util/sql.ts";
|
|
14
|
+
import { entriesOf } from "./ledger-sql.ts";
|
|
14
15
|
import { harnessFilter } from "./report.ts";
|
|
15
16
|
import type { LedgerEntry } from "./types.ts";
|
|
16
17
|
|
|
@@ -89,7 +90,7 @@ export interface DecisionFilter {
|
|
|
89
90
|
* reasons, the classifier's view, the cost forecast against the bill, the escalation signal —
|
|
90
91
|
* and the verdicts `/router good|bad` recorded against each turn ride along.
|
|
91
92
|
*/
|
|
92
|
-
export function decisionEntries(db:
|
|
93
|
+
export async function decisionEntries(db: SqlDb, filter: DecisionFilter): Promise<DecisionEntry[]> {
|
|
93
94
|
const s = scope(filter.harness, "harness_id");
|
|
94
95
|
if (s === null) return [];
|
|
95
96
|
const where = ["created_at_ms >= $since", ...s.sql];
|
|
@@ -107,20 +108,26 @@ export function decisionEntries(db: Database, filter: DecisionFilter): DecisionE
|
|
|
107
108
|
bind.$session = filter.ompSessionId;
|
|
108
109
|
}
|
|
109
110
|
const limit = Math.min(Math.max(filter.limit ?? 50, 1), 1_000);
|
|
110
|
-
const rows = db.query(
|
|
111
|
-
|
|
111
|
+
const rows = await db.query<unknown>(
|
|
112
|
+
`SELECT * FROM ledger WHERE ${where.join(" AND ")} ORDER BY created_at_ms DESC LIMIT ${limit}`,
|
|
113
|
+
bind,
|
|
114
|
+
);
|
|
115
|
+
const entries = entriesOf(rows);
|
|
112
116
|
if (entries.length === 0) return [];
|
|
113
117
|
// Verdicts, when the feedback table exists (it does not on a ledger no one has judged).
|
|
114
|
-
const hasFeedback = (db.query("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'feedback'").get() as { name: string } | null) !== null;
|
|
115
118
|
const verdicts = new Map<string, DecisionEntry["feedback"]>();
|
|
116
|
-
if (
|
|
119
|
+
if (await db.tableExists("feedback")) {
|
|
117
120
|
const ids = entries.map((e) => e.id);
|
|
118
121
|
const marks = ids.map((_, i) => `$f${i}`).join(", ");
|
|
119
122
|
const fb: Record<string, string> = {};
|
|
120
123
|
ids.forEach((id, i) => (fb[`$f${i}`] = id));
|
|
121
|
-
|
|
124
|
+
const rows = await db.query<{ ledger_id: string; verdict: string; note: string; created_at_ms: number }>(
|
|
125
|
+
`SELECT ledger_id, verdict, note, created_at_ms FROM feedback WHERE ledger_id IN (${marks}) ORDER BY created_at_ms ASC`,
|
|
126
|
+
fb,
|
|
127
|
+
);
|
|
128
|
+
for (const r of rows) {
|
|
122
129
|
const list = verdicts.get(r.ledger_id) ?? [];
|
|
123
|
-
list.push({ verdict: r.verdict === "good" ? "good" : "bad", note: r.note, createdAtMs: r.created_at_ms });
|
|
130
|
+
list.push({ verdict: r.verdict === "good" ? "good" : "bad", note: r.note, createdAtMs: num(r.created_at_ms) });
|
|
124
131
|
verdicts.set(r.ledger_id, list);
|
|
125
132
|
}
|
|
126
133
|
}
|
|
@@ -132,7 +139,7 @@ export function decisionEntries(db: Database, filter: DecisionFilter): DecisionE
|
|
|
132
139
|
* as the ledger counts them. `contextScope`, when given, narrows to the turns that carried
|
|
133
140
|
* exactly that agentdox scope — what a front door charges back to one project.
|
|
134
141
|
*/
|
|
135
|
-
export function spendUsdSince(db:
|
|
142
|
+
export async function spendUsdSince(db: SqlDb, sinceMs: number, harness: HarnessScope, contextScope?: string): Promise<number> {
|
|
136
143
|
const s = scope(harness, "harness_id");
|
|
137
144
|
if (s === null) return 0;
|
|
138
145
|
const where = ["created_at_ms >= $since", ...s.sql];
|
|
@@ -141,36 +148,42 @@ export function spendUsdSince(db: Database, sinceMs: number, harness: HarnessSco
|
|
|
141
148
|
where.push("scope = $scope");
|
|
142
149
|
bind.$scope = contextScope;
|
|
143
150
|
}
|
|
144
|
-
const row = db.
|
|
145
|
-
|
|
151
|
+
const row = await db.one<{ usd: unknown }>(
|
|
152
|
+
`SELECT COALESCE(SUM(${USD}), 0) AS usd FROM ledger WHERE ${where.join(" AND ")}`,
|
|
153
|
+
bind,
|
|
154
|
+
);
|
|
155
|
+
return num(row?.usd);
|
|
146
156
|
}
|
|
147
157
|
|
|
148
158
|
/** Verdicts since `sinceMs`, by model and the most recent 200, joined to the ledger for the judging harness. */
|
|
149
|
-
export function feedbackView(db:
|
|
150
|
-
|
|
151
|
-
if (!exists) return { byModel: [], recent: [] };
|
|
159
|
+
export async function feedbackView(db: SqlDb, sinceMs: number, harness: HarnessScope): Promise<FeedbackView> {
|
|
160
|
+
if (!(await db.tableExists("feedback"))) return { byModel: [], recent: [] };
|
|
152
161
|
const s = scope(harness, "l.harness_id");
|
|
153
162
|
if (s === null) return { byModel: [], recent: [] };
|
|
154
163
|
const where = ["f.created_at_ms >= $since", ...s.sql].join(" AND ");
|
|
155
164
|
const bind = { $since: sinceMs, ...s.bind };
|
|
156
165
|
const recent = (
|
|
157
|
-
db.query
|
|
158
|
-
at_ms
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
+
await db.query<{ at_ms: number; slug: string; tier: string; verdict: string; note: string; harness_id: string }>(
|
|
167
|
+
`SELECT f.created_at_ms AS at_ms, f.slug, f.tier, f.verdict, f.note, COALESCE(l.harness_id, '') AS harness_id
|
|
168
|
+
FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} ORDER BY f.created_at_ms DESC LIMIT 200`,
|
|
169
|
+
bind,
|
|
170
|
+
)
|
|
171
|
+
).map((r) => ({
|
|
172
|
+
atMs: num(r.at_ms),
|
|
173
|
+
slug: r.slug,
|
|
174
|
+
tier: r.tier,
|
|
175
|
+
verdict: (r.verdict === "good" ? "good" : "bad") as "good" | "bad",
|
|
176
|
+
note: r.note,
|
|
177
|
+
harnessId: r.harness_id,
|
|
178
|
+
}));
|
|
166
179
|
const byModel = (
|
|
167
|
-
db
|
|
168
|
-
.
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
).map((r) => ({ slug: r.slug, good: r.good, bad: r.bad, judges: r.judges }));
|
|
180
|
+
await db.query<{ slug: string; good: unknown; bad: unknown; judges: unknown }>(
|
|
181
|
+
`SELECT f.slug, SUM(CASE WHEN f.verdict = 'good' THEN 1 ELSE 0 END) AS good, SUM(CASE WHEN f.verdict = 'bad' THEN 1 ELSE 0 END) AS bad,
|
|
182
|
+
COUNT(DISTINCT COALESCE(l.harness_id, '')) AS judges
|
|
183
|
+
FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} GROUP BY f.slug ORDER BY bad DESC, good DESC, f.slug ASC`,
|
|
184
|
+
bind,
|
|
185
|
+
)
|
|
186
|
+
).map((r) => ({ slug: r.slug, good: num(r.good), bad: num(r.bad), judges: num(r.judges) }));
|
|
174
187
|
return { byModel, recent };
|
|
175
188
|
}
|
|
176
189
|
|
|
@@ -180,36 +193,53 @@ export function feedbackView(db: Database, sinceMs: number, harness: HarnessScop
|
|
|
180
193
|
* can charge each project its own share; rows from before v18 (and turns that carried no
|
|
181
194
|
* scope) group under "".
|
|
182
195
|
*/
|
|
183
|
-
export function exportRows(db:
|
|
196
|
+
export async function exportRows(db: SqlDb, sinceMs: number, harness: HarnessScope): Promise<ExportRow[]> {
|
|
184
197
|
const s = scope(harness, "harness_id");
|
|
185
198
|
if (s === null) return [];
|
|
186
199
|
const where = ["created_at_ms >= $since", "requested_model <> 'digest'", ...s.sql].join(" AND ");
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
200
|
+
// GROUP BY names the day expression rather than its alias: Postgres does not
|
|
201
|
+
// allow a select alias in GROUP BY, and repeating it keeps one statement for
|
|
202
|
+
// both engines.
|
|
203
|
+
const day = db.utcDay("created_at_ms");
|
|
204
|
+
const rows = await db.query<{
|
|
205
|
+
day: string;
|
|
206
|
+
harness_id: string;
|
|
207
|
+
slug: string;
|
|
208
|
+
scope: string;
|
|
209
|
+
dispatches: unknown;
|
|
210
|
+
prompt_tokens: unknown;
|
|
211
|
+
cached_tokens: unknown;
|
|
212
|
+
completion_tokens: unknown;
|
|
213
|
+
spend: unknown;
|
|
214
|
+
escalations: unknown;
|
|
215
|
+
errors: unknown;
|
|
216
|
+
}>(
|
|
217
|
+
`SELECT ${day} AS day, harness_id, COALESCE(served_slug, slug) AS slug, COALESCE(scope, '') AS scope,
|
|
190
218
|
COUNT(*) AS dispatches,
|
|
191
|
-
COALESCE(SUM(
|
|
192
|
-
COALESCE(SUM(
|
|
193
|
-
COALESCE(SUM(
|
|
219
|
+
COALESCE(SUM(${db.jsonNum("usage", "promptTokens")}), 0) AS prompt_tokens,
|
|
220
|
+
COALESCE(SUM(${db.jsonNum("usage", "cachedTokens")}), 0) AS cached_tokens,
|
|
221
|
+
COALESCE(SUM(${db.jsonNum("usage", "completionTokens")}), 0) AS completion_tokens,
|
|
194
222
|
COALESCE(SUM(${USD}), 0) AS spend,
|
|
195
223
|
SUM(CASE WHEN escalation_signal IS NOT NULL THEN 1 ELSE 0 END) AS escalations,
|
|
196
224
|
SUM(CASE WHEN error IS NOT NULL THEN 1 ELSE 0 END) AS errors
|
|
197
|
-
FROM ledger WHERE ${where}
|
|
198
|
-
|
|
199
|
-
|
|
225
|
+
FROM ledger WHERE ${where}
|
|
226
|
+
GROUP BY ${day}, harness_id, COALESCE(served_slug, slug), COALESCE(scope, '')
|
|
227
|
+
ORDER BY 1 ASC, harness_id ASC, spend DESC`,
|
|
228
|
+
{ $since: sinceMs, ...s.bind },
|
|
229
|
+
);
|
|
200
230
|
return rows.map((r) => ({
|
|
201
231
|
day: r.day,
|
|
202
232
|
harnessId: r.harness_id,
|
|
203
233
|
slug: r.slug,
|
|
204
234
|
scope: r.scope,
|
|
205
235
|
provider: providerOfSlug(r.slug),
|
|
206
|
-
dispatches: r.dispatches,
|
|
207
|
-
promptTokens: r.prompt_tokens,
|
|
208
|
-
cachedTokens: r.cached_tokens,
|
|
209
|
-
completionTokens: r.completion_tokens,
|
|
210
|
-
spendUsd: r.spend,
|
|
211
|
-
escalations: r.escalations,
|
|
212
|
-
errors: r.errors,
|
|
236
|
+
dispatches: num(r.dispatches),
|
|
237
|
+
promptTokens: num(r.prompt_tokens),
|
|
238
|
+
cachedTokens: num(r.cached_tokens),
|
|
239
|
+
completionTokens: num(r.completion_tokens),
|
|
240
|
+
spendUsd: num(r.spend),
|
|
241
|
+
escalations: num(r.escalations),
|
|
242
|
+
errors: num(r.errors),
|
|
213
243
|
}));
|
|
214
244
|
}
|
|
215
245
|
|
package/src/lib.ts
CHANGED
|
@@ -25,14 +25,18 @@ export type { DeepPartial } from "./config/load.ts";
|
|
|
25
25
|
export { buildUsageReport, renderUsageReport, type UsageReport, type ReportTotals } from "./cost/report.ts";
|
|
26
26
|
export { buildDailySummary, renderDailySummary, type DailySummary } from "./cost/summary.ts";
|
|
27
27
|
export { openDb } from "./util/sqlite.ts";
|
|
28
|
+
// A front door reads the ledger through the engine-agnostic handle: the store
|
|
29
|
+
// may be a file or a shared database, and the view functions take this.
|
|
30
|
+
export { dialectOf, openSqlDb, num, numOrNull, type Dialect, type SqlDb } from "./util/sql.ts";
|
|
28
31
|
export { spendUsdSince, feedbackView, exportRows, exportCsv, decisionEntries, harnessScopeParam, type HarnessScope, type ExportRow, type FeedbackRow, type FeedbackByModel, type FeedbackView, type DecisionEntry, type DecisionFilter } from "./cost/views.ts";
|
|
29
|
-
export {
|
|
32
|
+
export { createSqlLedger } from "./cost/ledger-sql.ts";
|
|
33
|
+
export { migrateStore, STORE_TABLES } from "./util/schema.ts";
|
|
30
34
|
export { createFeedbackStore, type FeedbackStore, type FeedbackRecord } from "./cost/feedback.ts";
|
|
31
35
|
export { buildExecutable, collectPackageFiles, executableFileName, hostTarget, isExecutableTarget, EXECUTABLE_TARGETS, type ExecutableTarget, type BuildExecutableResult } from "./cli/build-executable.ts";
|
|
32
36
|
export { parseSkillsBundle, type SkillsBundle } from "./cli/skills.ts";
|
|
33
37
|
export type { RequestPolicy } from "./wire/types.ts";
|
|
34
38
|
export type { CatalogView, CatalogViewModel } from "./server/catalog-view.ts";
|
|
35
|
-
export type {
|
|
39
|
+
export type { AsyncLedger, LedgerEntry, PruneResult } from "./cost/types.ts";
|
|
36
40
|
// Retention: a front door asks through `POST /v1/router/prune` rather than
|
|
37
41
|
// deleting from the ledger itself. The interval is exported so it can say when.
|
|
38
42
|
export { RETENTION_INTERVAL_MS } from "./cost/retention.ts";
|
package/src/router/candidates.ts
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
import type { CatalogModel, CatalogSnapshot } from "../catalog/types.ts";
|
|
8
8
|
import type { FilterConfig, QualityAxis, RouterConfig } from "../config/types.ts";
|
|
9
9
|
import { forecast, priceAt } from "../cost/forecast.ts";
|
|
10
|
-
import type {
|
|
10
|
+
import type { LedgerSignals, ModelLatency } from "../cost/types.ts";
|
|
11
11
|
import type { NormRequest } from "../wire/types.ts";
|
|
12
12
|
import { effectivePriceCeiling, effectiveQualityFloor, tierPlanFor } from "./tier-plan.ts";
|
|
13
13
|
import type { Candidate, Features, Rejection, TaskType, Tier } from "./types.ts";
|
|
@@ -27,7 +27,6 @@ export interface BuildCandidatesArgs {
|
|
|
27
27
|
/** Task type; its config selects the axis, quality floor, and image filter. */
|
|
28
28
|
task: TaskType;
|
|
29
29
|
snapshot: CatalogSnapshot;
|
|
30
|
-
ledger: Ledger | null;
|
|
31
30
|
cfg: RouterConfig;
|
|
32
31
|
expectedCompletionTokens: number;
|
|
33
32
|
/** Slug whose prompt cache is warm this turn; wins score ties. */
|
|
@@ -154,7 +153,7 @@ function latencyMultiplier(latency: ModelLatency | null, filters: FilterConfig,
|
|
|
154
153
|
}
|
|
155
154
|
|
|
156
155
|
export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candidate[]; rejected: Rejection[] } {
|
|
157
|
-
const { req, features, tier, task, snapshot,
|
|
156
|
+
const { req, features, tier, task, snapshot, cfg, expectedCompletionTokens, warmSlug, relaxLevel = 0 } = args;
|
|
158
157
|
// A Set only when non-empty: the common path allocates nothing.
|
|
159
158
|
const excluded = args.excludeSlugs === undefined || args.excludeSlugs.length === 0 ? null : new Set(args.excludeSlugs);
|
|
160
159
|
const tierCfg = cfg.tiers[tier];
|
|
@@ -318,14 +317,11 @@ export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candid
|
|
|
318
317
|
continue;
|
|
319
318
|
}
|
|
320
319
|
|
|
321
|
-
//
|
|
322
|
-
//
|
|
323
|
-
//
|
|
320
|
+
// Signals are prefetched by `route` — one query per signal kind for the
|
|
321
|
+
// whole candidate set. Absent means "the ledger knows nothing about this
|
|
322
|
+
// model", which is what a cold start looks like and is handled below.
|
|
324
323
|
const signals = args.signals;
|
|
325
|
-
const trust =
|
|
326
|
-
signals?.get(slug)?.trust ??
|
|
327
|
-
ledger?.trust(slug, filters.trustScopedByHarness ? req.harnessId : undefined, filters.feedbackByTask ? task : undefined) ??
|
|
328
|
-
null;
|
|
324
|
+
const trust = signals?.get(slug)?.trust ?? null;
|
|
329
325
|
if (!relaxTrust && trust !== null && trust.attempts >= filters.minTrustSamples && trust.successRate < filters.minTrust) {
|
|
330
326
|
rejected.push({
|
|
331
327
|
slug,
|
|
@@ -342,11 +338,7 @@ export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candid
|
|
|
342
338
|
// gate can. Only measured models are dropped, so a new model still gets its
|
|
343
339
|
// cold-start turns to accumulate samples. Relaxed with trust in rescue.
|
|
344
340
|
const needLatency = filters.latencyWeight > 0 || filters.maxExpectedWaitMs !== undefined;
|
|
345
|
-
const latency = needLatency
|
|
346
|
-
? (signals?.get(slug)?.latency ??
|
|
347
|
-
ledger?.latency(slug, filters.trustScopedByHarness ? req.harnessId : undefined) ??
|
|
348
|
-
null)
|
|
349
|
-
: null;
|
|
341
|
+
const latency = needLatency ? (signals?.get(slug)?.latency ?? null) : null;
|
|
350
342
|
if (
|
|
351
343
|
!relaxTrust &&
|
|
352
344
|
filters.maxExpectedWaitMs !== undefined &&
|
package/src/router/classify.ts
CHANGED
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
import type { CatalogSource } from "../catalog/types.ts";
|
|
7
7
|
import type { QualityAxis, RouterConfig } from "../config/types.ts";
|
|
8
8
|
import { forecast } from "../cost/forecast.ts";
|
|
9
|
-
import type {
|
|
9
|
+
import type { AsyncLedger } from "../cost/types.ts";
|
|
10
10
|
import { estimateTokens } from "../tokens/estimate.ts";
|
|
11
11
|
import type { UpstreamClient } from "../upstream/types.ts";
|
|
12
12
|
import { sha256Hex } from "../util/hash.ts";
|
|
@@ -219,7 +219,7 @@ export function classifyTask(f: Features): TaskType {
|
|
|
219
219
|
|
|
220
220
|
export interface ClassifyDeps {
|
|
221
221
|
upstream: UpstreamClient;
|
|
222
|
-
ledger:
|
|
222
|
+
ledger: AsyncLedger | null;
|
|
223
223
|
catalog: CatalogSource | null;
|
|
224
224
|
}
|
|
225
225
|
|
|
@@ -334,11 +334,13 @@ export async function classify(
|
|
|
334
334
|
}
|
|
335
335
|
|
|
336
336
|
// Cost guard: adjudication must be cheap relative to the turn it classifies.
|
|
337
|
-
const blend = deps.ledger?.blendedRate(cfg.ledger.blendWindowDays) ?? null;
|
|
337
|
+
const blend = (await deps.ledger?.blendedRate(cfg.ledger.blendWindowDays)) ?? null;
|
|
338
338
|
const inputRate = (blend?.inputPerMtok ?? cfg.ledger.fallbackBlend.inputPerMtok) / 1e6;
|
|
339
339
|
const outputRate = (blend?.outputPerMtok ?? cfg.ledger.fallbackBlend.outputPerMtok) / 1e6;
|
|
340
340
|
const judge = deps.catalog?.find(cc.model);
|
|
341
|
-
|
|
341
|
+
// The adjudicator prompt is short and its family unknown, so the default
|
|
342
|
+
// family ratio is as good as a measured one here.
|
|
343
|
+
const digestTokens = estimateTokens(Buffer.byteLength(ADJUDICATOR_SYSTEM) + Buffer.byteLength(digest), "unknown", null);
|
|
342
344
|
const adjudicatorUsd =
|
|
343
345
|
judge !== undefined
|
|
344
346
|
? forecast(judge, {
|
package/src/router/index.ts
CHANGED
|
@@ -10,23 +10,25 @@
|
|
|
10
10
|
* real request without perturbing the conversation it belongs to.
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
|
-
import type { CatalogSource } from "../catalog/types.ts";
|
|
13
|
+
import type { CatalogSnapshot, CatalogSource } from "../catalog/types.ts";
|
|
14
14
|
import type { ProfileConfig, RouterConfig } from "../config/types.ts";
|
|
15
|
-
import type
|
|
15
|
+
import { type AsyncLedger, type LedgerReader } from "../cost/types.ts";
|
|
16
16
|
import { estimatePromptTokens } from "../tokens/estimate.ts";
|
|
17
17
|
import type { UpstreamClient } from "../upstream/types.ts";
|
|
18
|
+
import { modelNotFound } from "../wire/openai/errors.ts";
|
|
18
19
|
import type { NormRequest, RequestPolicy } from "../wire/types.ts";
|
|
19
20
|
import { classify, classifyTask } from "./classify.ts";
|
|
20
21
|
import { extractFeatures } from "./features.ts";
|
|
21
|
-
import { select } from "./select.ts";
|
|
22
|
+
import { monthStartMs, select, type TurnReads } from "./select.ts";
|
|
22
23
|
import { TIER_ORDER, type Classification, type ConversationStore, type Decision, type Router, type Tier } from "./types.ts";
|
|
23
|
-
|
|
24
24
|
export interface RouterDeps {
|
|
25
25
|
config: RouterConfig;
|
|
26
26
|
catalog: CatalogSource;
|
|
27
|
-
ledger:
|
|
27
|
+
ledger: AsyncLedger;
|
|
28
28
|
conversations: ConversationStore;
|
|
29
29
|
upstream: UpstreamClient;
|
|
30
|
+
/** Async ledger reads; defaults to reading `ledger` directly. */
|
|
31
|
+
reader?: LedgerReader;
|
|
30
32
|
}
|
|
31
33
|
|
|
32
34
|
/**
|
|
@@ -90,20 +92,89 @@ export function resolveProfile(cfg: RouterConfig, requestedModel: string, isSuba
|
|
|
90
92
|
return fallback;
|
|
91
93
|
}
|
|
92
94
|
|
|
95
|
+
/**
|
|
96
|
+
* The `model` a client names is a profile id. When it is neither a profile nor
|
|
97
|
+
* a catalog slug, routing something else and reporting the asked-for name back
|
|
98
|
+
* is a silent substitution: the caller is billed for a model it never chose.
|
|
99
|
+
* A real slug becomes a pin (absolute, the way a session override is); anything
|
|
100
|
+
* else is refused.
|
|
101
|
+
*
|
|
102
|
+
* @param asked the client's `model` before the provider prefix was stripped.
|
|
103
|
+
* @returns the slug to pin, or undefined when a profile matched.
|
|
104
|
+
*/
|
|
105
|
+
export function pinForRequestedModel(
|
|
106
|
+
cfg: RouterConfig,
|
|
107
|
+
slugs: readonly string[],
|
|
108
|
+
requestedModel: string,
|
|
109
|
+
asked: string,
|
|
110
|
+
): string | undefined {
|
|
111
|
+
if (cfg.profiles.some((p) => p.id === requestedModel)) return undefined;
|
|
112
|
+
if (slugs.includes(asked)) return asked;
|
|
113
|
+
throw modelNotFound(asked);
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* Reads the ledger for one turn, concurrently, before selection runs.
|
|
118
|
+
*
|
|
119
|
+
* Only what this turn's configuration actually consults is fetched: month and
|
|
120
|
+
* day spend are skipped when no such budget is set, and the escalation-cost
|
|
121
|
+
* term is skipped when its weight is 0. Empty catalog ⇒ no signal queries at
|
|
122
|
+
* all, which is what keeps the tests' fakes cheap.
|
|
123
|
+
*/
|
|
124
|
+
export async function prefetchTurnReads(
|
|
125
|
+
reader: LedgerReader | null,
|
|
126
|
+
req: NormRequest,
|
|
127
|
+
profile: ProfileConfig,
|
|
128
|
+
cfg: RouterConfig,
|
|
129
|
+
snapshot: CatalogSnapshot,
|
|
130
|
+
task: string,
|
|
131
|
+
nowMs: number = Date.now(),
|
|
132
|
+
): Promise<TurnReads> {
|
|
133
|
+
if (reader === null) return {};
|
|
134
|
+
const slugs = snapshot.models.map((m) => m.slug);
|
|
135
|
+
const harness = cfg.filters.trustScopedByHarness ? req.harnessId : undefined;
|
|
136
|
+
const perMonthUsd = profile.budget?.perMonthUsd ?? cfg.budget.perMonthUsd;
|
|
137
|
+
const perDayUsd = profile.budget?.perDayUsd ?? cfg.budget.perDayUsd;
|
|
138
|
+
const [signals, cacheReliability, escalation, monthSpendUsd, daySpendUsd] = await Promise.all([
|
|
139
|
+
slugs.length > 0 ? reader.signals(slugs, harness, cfg.filters.feedbackByTask ? task : undefined) : undefined,
|
|
140
|
+
slugs.length > 0 && cfg.filters.cacheReliabilityMinSamples > 0 ? reader.cacheReliability(slugs) : undefined,
|
|
141
|
+
cfg.filters.escalationCostWeight > 0 ? reader.escalationCost(cfg.ledger.blendWindowDays) : undefined,
|
|
142
|
+
// Month pacing can tighten the daily ceiling, so month spend is needed
|
|
143
|
+
// whenever either budget is set.
|
|
144
|
+
perMonthUsd !== undefined ? reader.spendSince(monthStartMs(nowMs), req.harnessId) : undefined,
|
|
145
|
+
perDayUsd !== undefined || perMonthUsd !== undefined ? reader.spendSince(nowMs - 86_400_000, req.harnessId) : undefined,
|
|
146
|
+
]);
|
|
147
|
+
return {
|
|
148
|
+
...(signals === undefined ? {} : { signals }),
|
|
149
|
+
...(cacheReliability === undefined ? {} : { cacheReliability }),
|
|
150
|
+
...(escalation === undefined ? {} : { escalationUsdPerPromptToken: escalation?.usdPerPromptToken ?? null }),
|
|
151
|
+
...(monthSpendUsd === undefined ? {} : { monthSpendUsd }),
|
|
152
|
+
...(daySpendUsd === undefined ? {} : { daySpendUsd }),
|
|
153
|
+
};
|
|
154
|
+
}
|
|
155
|
+
|
|
93
156
|
export function createRouter(deps: RouterDeps): Router {
|
|
94
157
|
const { config, catalog, ledger, conversations, upstream } = deps;
|
|
158
|
+
// A Postgres ledger supplies its own reader; a local one is wrapped, so the
|
|
159
|
+
// prefetch path is identical for both.
|
|
160
|
+
// The unified ledger IS a reader: `LedgerReader` is the narrow half of it.
|
|
161
|
+
const reader = deps.reader ?? ledger;
|
|
95
162
|
|
|
96
163
|
return {
|
|
97
164
|
async route(
|
|
98
165
|
req: NormRequest,
|
|
99
166
|
opts: { attempt: number; escalateFrom?: Tier; excludeSlugs?: readonly string[]; forceTier?: Tier; forceSlug?: string },
|
|
100
167
|
): Promise<Decision> {
|
|
101
|
-
const state = conversations.get(req.conversationKey) ?? conversations.load(req.conversationKey);
|
|
168
|
+
const state = (await conversations.get(req.conversationKey)) ?? (await conversations.load(req.conversationKey));
|
|
102
169
|
const snapshot = await catalog.get();
|
|
103
170
|
|
|
104
171
|
const priorTokenizer =
|
|
105
172
|
state.currentSlug === null ? undefined : catalog.find(state.currentSlug)?.tokenizer;
|
|
106
|
-
const
|
|
173
|
+
const tokenizer = priorTokenizer ?? NEUTRAL_TOKENIZER;
|
|
174
|
+
// One ratio, fetched before estimating: the estimate itself runs in
|
|
175
|
+
// synchronous code that a shared store cannot be read from.
|
|
176
|
+
const ratio = await ledger.tokenRatio(tokenizer);
|
|
177
|
+
const promptTokens = estimatePromptTokens(req, tokenizer, ratio);
|
|
107
178
|
const features = extractFeatures(req, promptTokens);
|
|
108
179
|
|
|
109
180
|
let classification: Classification;
|
|
@@ -136,7 +207,22 @@ export function createRouter(deps: RouterDeps): Router {
|
|
|
136
207
|
classification = await classify(req, features, config, { upstream, ledger, catalog });
|
|
137
208
|
}
|
|
138
209
|
|
|
139
|
-
const
|
|
210
|
+
const pinnedByModel = pinForRequestedModel(
|
|
211
|
+
config,
|
|
212
|
+
snapshot.models.map((m) => m.slug),
|
|
213
|
+
req.requestedModel,
|
|
214
|
+
req.requestedModelFull ?? req.requestedModel,
|
|
215
|
+
);
|
|
216
|
+
const policed = applyRequestPolicy(
|
|
217
|
+
resolveProfile(config, req.requestedModel, req.isSubagent),
|
|
218
|
+
config,
|
|
219
|
+
req.policy,
|
|
220
|
+
opts.forceSlug ?? pinnedByModel,
|
|
221
|
+
);
|
|
222
|
+
// Every ledger read this turn needs, fetched here rather than inside
|
|
223
|
+
// `select`: selection stays a synchronous pure function, and a ledger
|
|
224
|
+
// that can only be read asynchronously (Postgres) works unchanged.
|
|
225
|
+
const reads = await prefetchTurnReads(reader, req, policed.profile, policed.cfg, snapshot, classification.task);
|
|
140
226
|
const decision = select({
|
|
141
227
|
req,
|
|
142
228
|
features,
|
|
@@ -144,9 +230,9 @@ export function createRouter(deps: RouterDeps): Router {
|
|
|
144
230
|
profile: policed.profile,
|
|
145
231
|
state,
|
|
146
232
|
snapshot,
|
|
147
|
-
ledger,
|
|
148
233
|
cfg: policed.cfg,
|
|
149
234
|
nowMs: Date.now(),
|
|
235
|
+
reads,
|
|
150
236
|
...(opts.excludeSlugs === undefined ? {} : { excludeSlugs: opts.excludeSlugs }),
|
|
151
237
|
...(policed.forceSlug === undefined ? {} : { forceSlug: policed.forceSlug }),
|
|
152
238
|
});
|
package/src/router/select.ts
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
import type { CatalogSnapshot } from "../catalog/types.ts";
|
|
10
10
|
import type { ProfileConfig, RouterConfig } from "../config/types.ts";
|
|
11
11
|
import { forecast, priceAt } from "../cost/forecast.ts";
|
|
12
|
-
import type {
|
|
12
|
+
import type { LedgerSignals, ModelCacheReliability } from "../cost/types.ts";
|
|
13
13
|
import { explorationDraw } from "./explore.ts";
|
|
14
14
|
import type { CompactionEdit, NormRequest, ReasoningLevel } from "../wire/types.ts";
|
|
15
15
|
import { compactedBytes, planCompaction, validatePlan, type CompactionResult } from "./compaction.ts";
|
|
@@ -35,7 +35,6 @@ export interface SelectArgs {
|
|
|
35
35
|
profile: ProfileConfig;
|
|
36
36
|
state: ConversationState;
|
|
37
37
|
snapshot: CatalogSnapshot;
|
|
38
|
-
ledger: Ledger | null;
|
|
39
38
|
cfg: RouterConfig;
|
|
40
39
|
nowMs: number;
|
|
41
40
|
/**
|
|
@@ -49,6 +48,30 @@ export interface SelectArgs {
|
|
|
49
48
|
* stay/switch comparison. Ignored when the catalog has no such model.
|
|
50
49
|
*/
|
|
51
50
|
forceSlug?: string;
|
|
51
|
+
/** Ledger reads already performed for this turn; see TurnReads. */
|
|
52
|
+
reads?: TurnReads;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Every ledger read a turn needs, fetched BEFORE selection.
|
|
57
|
+
*
|
|
58
|
+
* `select` is synchronous on purpose — it is a pure ranking function, and
|
|
59
|
+
* `explain` depends on being able to run it without side effects. A ledger on
|
|
60
|
+
* Postgres cannot be read synchronously, so the reads move up to `route`,
|
|
61
|
+
* which is already async, and arrive here as data. Absent ⇒ read through the
|
|
62
|
+
* absent reads degrade to no signals rather than reaching for the store.
|
|
63
|
+
*/
|
|
64
|
+
export interface TurnReads {
|
|
65
|
+
/** Trust and latency per candidate slug; one batch query per signal kind. */
|
|
66
|
+
signals?: Map<string, LedgerSignals>;
|
|
67
|
+
/** Observed warm-cache hit rate per slug. */
|
|
68
|
+
cacheReliability?: Map<string, ModelCacheReliability>;
|
|
69
|
+
/** What an escalated retry bills per prompt token; null ⇒ the term is inert. */
|
|
70
|
+
escalationUsdPerPromptToken?: number | null;
|
|
71
|
+
/** Spend since the start of the UTC month, scoped as the filters say. */
|
|
72
|
+
monthSpendUsd?: number;
|
|
73
|
+
/** Spend over the rolling 24h, scoped as the filters say. */
|
|
74
|
+
daySpendUsd?: number;
|
|
52
75
|
}
|
|
53
76
|
|
|
54
77
|
/**
|
|
@@ -67,7 +90,6 @@ export class BudgetExceededError extends Error {
|
|
|
67
90
|
// Mid-range completion assumption for forecasts. Long generations amortize
|
|
68
91
|
// into prompt-dominated cost anyway; precision here does not move rankings.
|
|
69
92
|
const EXPECTED_COMPLETION_TOKENS = 1024;
|
|
70
|
-
const DAY_MS = 86_400_000;
|
|
71
93
|
|
|
72
94
|
/**
|
|
73
95
|
* Authors known to accept replayed assistant reasoning over chat completions:
|
|
@@ -129,7 +151,7 @@ export function monthPace(nowMs: number, perMonthUsd: number, spentUsd: number):
|
|
|
129
151
|
}
|
|
130
152
|
|
|
131
153
|
export function select(args: SelectArgs): Decision {
|
|
132
|
-
const { req, features, classification, profile, state, snapshot,
|
|
154
|
+
const { req, features, classification, profile, state, snapshot, cfg, nowMs } = args;
|
|
133
155
|
const reasons: string[] = [];
|
|
134
156
|
const minI = tierIdx(profile.minTier);
|
|
135
157
|
const maxI = tierIdx(profile.maxTier);
|
|
@@ -351,29 +373,23 @@ export function select(args: SelectArgs): Decision {
|
|
|
351
373
|
const warmSlug = cacheWarm ? state.cacheWarmSlug : null;
|
|
352
374
|
// Expected cache hit for a model: its observed rate once enough warm-expected
|
|
353
375
|
// samples exist (filters.cacheReliabilityMinSamples), else a reliable 1.
|
|
376
|
+
const reads = args.reads;
|
|
354
377
|
const cacheHitExpectation = (slug: string): { rate: number; measured: boolean; samples: number } => {
|
|
355
378
|
const min = cfg.filters.cacheReliabilityMinSamples;
|
|
356
|
-
const rel = min > 0 ? (
|
|
379
|
+
const rel = min > 0 ? (reads?.cacheReliability?.get(slug) ?? null) : null;
|
|
357
380
|
if (rel === null || rel.samples < min) return { rate: 1, measured: false, samples: rel?.samples ?? 0 };
|
|
358
381
|
return { rate: rel.hitRate, measured: true, samples: rel.samples };
|
|
359
382
|
};
|
|
360
|
-
//
|
|
361
|
-
//
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
snapshot.models.map((m) => m.slug),
|
|
366
|
-
cfg.filters.trustScopedByHarness ? req.harnessId : undefined,
|
|
367
|
-
cfg.filters.feedbackByTask ? classification.task : undefined,
|
|
368
|
-
)
|
|
369
|
-
: undefined;
|
|
383
|
+
// Trust and latency for every candidate, fetched by `route` before this ran.
|
|
384
|
+
// There is no fallback read here on purpose: selection is synchronous and the
|
|
385
|
+
// store may be a shared database, so a missing prefetch must degrade to
|
|
386
|
+
// "no signals" rather than silently reach for a handle it cannot await.
|
|
387
|
+
const candidateSignals = reads?.signals;
|
|
370
388
|
// What an escalated retry has actually been billing per prompt token, for
|
|
371
389
|
// the escalation-cost term in candidate scoring. Read once per turn; null
|
|
372
390
|
// (term inert) when the weight is 0 or the ledger has too few samples.
|
|
373
391
|
const escalationUsdPerPromptToken =
|
|
374
|
-
cfg.filters.escalationCostWeight > 0
|
|
375
|
-
? (ledger?.escalationCost?.(cfg.ledger.blendWindowDays)?.usdPerPromptToken ?? null)
|
|
376
|
-
: null;
|
|
392
|
+
cfg.filters.escalationCostWeight > 0 ? (reads?.escalationUsdPerPromptToken ?? null) : null;
|
|
377
393
|
// A pinned slug is admitted the way a config pin is: into the tier's pin
|
|
378
394
|
// list for this call only, so the quality floor cannot keep it out.
|
|
379
395
|
const pinSlug = args.forceSlug !== undefined && snapshot.models.some((m) => m.slug === args.forceSlug) ? args.forceSlug : undefined;
|
|
@@ -386,7 +402,6 @@ export function select(args: SelectArgs): Decision {
|
|
|
386
402
|
tier: t,
|
|
387
403
|
task: classification.task,
|
|
388
404
|
snapshot,
|
|
389
|
-
ledger,
|
|
390
405
|
cfg: buildCfg,
|
|
391
406
|
expectedCompletionTokens: EXPECTED_COMPLETION_TOKENS,
|
|
392
407
|
warmSlug,
|
|
@@ -515,13 +530,15 @@ export function select(args: SelectArgs): Decision {
|
|
|
515
530
|
// left, becomes a daily ceiling that tightens as the month runs ahead.
|
|
516
531
|
let paceNote = "";
|
|
517
532
|
if (budget.perMonthUsd !== undefined) {
|
|
518
|
-
const
|
|
533
|
+
const monthSpend = reads?.monthSpendUsd ?? 0;
|
|
534
|
+
const pace = monthPace(nowMs, budget.perMonthUsd, monthSpend);
|
|
519
535
|
if (budget.perDayUsd === undefined || pace.dailyCapUsd < budget.perDayUsd) {
|
|
520
536
|
budget.perDayUsd = pace.dailyCapUsd;
|
|
521
537
|
paceNote = ` (month pacing: $${pace.spentUsd.toFixed(2)} of $${budget.perMonthUsd} spent, $${pace.dailyCapUsd.toFixed(2)}/day for ${pace.daysLeft} more days)`;
|
|
522
538
|
}
|
|
523
539
|
}
|
|
524
|
-
const daySpend =
|
|
540
|
+
const daySpend =
|
|
541
|
+
budget.perDayUsd !== undefined ? (reads?.daySpendUsd ?? 0) : 0;
|
|
525
542
|
const breach = (c: Candidate): string | null => {
|
|
526
543
|
if (budget.perTurnUsd !== undefined && c.forecast.coldUsd > budget.perTurnUsd) {
|
|
527
544
|
return `cold forecast $${c.forecast.coldUsd.toFixed(4)} > per-turn budget $${budget.perTurnUsd}`;
|