auto-model-router 0.30.3 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +32 -2
- package/omp-extension/router-configure.ts +9 -7
- package/package.json +1 -1
- package/src/cli/config-cmd.ts +8 -7
- package/src/cli/explain.ts +10 -5
- package/src/cli/export.ts +6 -5
- package/src/cli/models.ts +10 -7
- package/src/cli/report.ts +6 -1
- package/src/cli/stats.ts +7 -7
- package/src/config/load.ts +10 -1
- package/src/config/types.ts +10 -1
- package/src/context/bridge.ts +7 -7
- package/src/context/index.ts +3 -3
- package/src/context/store.ts +39 -56
- package/src/context/types.ts +7 -6
- package/src/cost/blended.ts +28 -7
- package/src/cost/feedback.ts +33 -37
- package/src/cost/ledger-sql.ts +547 -0
- package/src/cost/ledger.ts +30 -459
- package/src/cost/report.ts +171 -129
- package/src/cost/retention.ts +10 -10
- package/src/cost/summary.ts +15 -10
- package/src/cost/types.ts +43 -62
- package/src/cost/views.ts +79 -49
- package/src/eval/calibrate.ts +47 -12
- package/src/eval/run.ts +18 -2
- package/src/lib.ts +6 -2
- package/src/router/candidates.ts +7 -15
- package/src/router/classify.ts +6 -4
- package/src/router/index.ts +95 -9
- package/src/router/select.ts +38 -21
- package/src/router/state.ts +90 -102
- package/src/router/types.ts +11 -5
- package/src/server/advise.ts +6 -4
- package/src/server/compaction-digest.ts +1 -1
- package/src/server/digest.ts +9 -10
- package/src/server/http.ts +109 -46
- package/src/server/providers.ts +18 -4
- package/src/server/turn.ts +32 -9
- package/src/tokens/estimate.ts +16 -6
- package/src/upstream/ollama-usage.ts +21 -11
- package/src/util/schema.ts +201 -0
- package/src/util/sql.ts +246 -0
- package/src/wire/anthropic/messages.ts +3 -4
- package/src/wire/openai/request.ts +1 -0
- package/src/wire/types.ts +7 -0
- package/test/anthropic-wire.test.ts +9 -9
- package/test/benchmark-feeds.test.ts +7 -7
- package/test/cache-control.test.ts +7 -7
- package/test/cache-estimate.test.ts +5 -5
- package/test/catalog-view.test.ts +4 -4
- package/test/catalog.test.ts +11 -11
- package/test/classify.test.ts +24 -24
- package/test/compaction.test.ts +20 -20
- package/test/config-wizard.test.ts +32 -32
- package/test/config.test.ts +10 -10
- package/test/connect-harnesses.test.ts +11 -11
- package/test/context-bridge.test.ts +40 -30
- package/test/context-prune.test.ts +43 -36
- package/test/context-query.test.ts +8 -8
- package/test/controls.test.ts +54 -27
- package/test/cost.test.ts +12 -12
- package/test/digest.test.ts +55 -44
- package/test/embed-lifecycle.test.ts +5 -5
- package/test/embed-logic.test.ts +26 -26
- package/test/escalate.test.ts +17 -17
- package/test/eval.test.ts +73 -16
- package/test/executable.test.ts +6 -6
- package/test/exploration.test.ts +19 -20
- package/test/failover.test.ts +22 -21
- package/test/fakes.ts +105 -0
- package/test/features.test.ts +21 -21
- package/test/harness-requests.test.ts +3 -3
- package/test/harness-switch.test.ts +5 -5
- package/test/hold-exploration.test.ts +13 -13
- package/test/hot-reload.test.ts +5 -5
- package/test/learned.test.ts +5 -5
- package/test/ledger-sql.test.ts +342 -0
- package/test/mcp-entry.test.ts +5 -5
- package/test/migrations.test.ts +28 -22
- package/test/models-yml.test.ts +18 -18
- package/test/ollama.test.ts +40 -34
- package/test/omp-credentials.test.ts +16 -16
- package/test/policy.test.ts +3 -3
- package/test/reconfigure.test.ts +4 -4
- package/test/redaction.test.ts +41 -35
- package/test/remote.test.ts +12 -12
- package/test/report-logic.test.ts +8 -8
- package/test/report.test.ts +95 -87
- package/test/retention.test.ts +79 -66
- package/test/schema.test.ts +123 -0
- package/test/scope.test.ts +8 -8
- package/test/select.test.ts +216 -257
- package/test/skills.test.ts +3 -3
- package/test/sql-shim.test.ts +154 -0
- package/test/state.test.ts +43 -36
- package/test/summary.test.ts +38 -27
- package/test/tier-plan.test.ts +45 -62
- package/test/toast-logic.test.ts +31 -31
- package/test/tokens.test.ts +95 -80
- package/test/trust-attribution.test.ts +217 -187
- package/test/trust-window.test.ts +37 -32
- package/test/turn.test.ts +55 -23
- package/test/upstreams.test.ts +13 -13
- package/test/views.test.ts +81 -59
- package/test/wire-request.test.ts +17 -17
- package/test/wire-responses.test.ts +4 -4
- package/tools/agentdox-e2e.ts +5 -2
- package/tools/export-benchmarks.ts +5 -5
- package/tools/ledger-parity.ts +266 -0
- package/tools/replay.ts +16 -8
package/src/cost/views.ts
CHANGED
|
@@ -5,12 +5,13 @@
|
|
|
5
5
|
* cost export by day, harness and model. Served by `/v1/router/spend`,
|
|
6
6
|
* `GET /v1/router/feedback` and `/v1/router/export`, exported from lib.ts, and
|
|
7
7
|
* behind `auto-model-router export`. All of them read only long-stable ledger
|
|
8
|
-
* columns and
|
|
8
|
+
* columns, and they run on either engine through `util/sql.ts`: a front door
|
|
9
|
+
* may be reading a local file while the router writes a shared database.
|
|
9
10
|
*/
|
|
10
11
|
|
|
11
12
|
import { providerOfSlug } from "./report.ts";
|
|
12
|
-
import type
|
|
13
|
-
import {
|
|
13
|
+
import { num, type SqlDb } from "../util/sql.ts";
|
|
14
|
+
import { entriesOf } from "./ledger-sql.ts";
|
|
14
15
|
import { harnessFilter } from "./report.ts";
|
|
15
16
|
import type { LedgerEntry } from "./types.ts";
|
|
16
17
|
|
|
@@ -89,7 +90,7 @@ export interface DecisionFilter {
|
|
|
89
90
|
* reasons, the classifier's view, the cost forecast against the bill, the escalation signal —
|
|
90
91
|
* and the verdicts `/router good|bad` recorded against each turn ride along.
|
|
91
92
|
*/
|
|
92
|
-
export function decisionEntries(db:
|
|
93
|
+
export async function decisionEntries(db: SqlDb, filter: DecisionFilter): Promise<DecisionEntry[]> {
|
|
93
94
|
const s = scope(filter.harness, "harness_id");
|
|
94
95
|
if (s === null) return [];
|
|
95
96
|
const where = ["created_at_ms >= $since", ...s.sql];
|
|
@@ -107,20 +108,26 @@ export function decisionEntries(db: Database, filter: DecisionFilter): DecisionE
|
|
|
107
108
|
bind.$session = filter.ompSessionId;
|
|
108
109
|
}
|
|
109
110
|
const limit = Math.min(Math.max(filter.limit ?? 50, 1), 1_000);
|
|
110
|
-
const rows = db.query(
|
|
111
|
-
|
|
111
|
+
const rows = await db.query<unknown>(
|
|
112
|
+
`SELECT * FROM ledger WHERE ${where.join(" AND ")} ORDER BY created_at_ms DESC LIMIT ${limit}`,
|
|
113
|
+
bind,
|
|
114
|
+
);
|
|
115
|
+
const entries = entriesOf(rows);
|
|
112
116
|
if (entries.length === 0) return [];
|
|
113
117
|
// Verdicts, when the feedback table exists (it does not on a ledger no one has judged).
|
|
114
|
-
const hasFeedback = (db.query("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'feedback'").get() as { name: string } | null) !== null;
|
|
115
118
|
const verdicts = new Map<string, DecisionEntry["feedback"]>();
|
|
116
|
-
if (
|
|
119
|
+
if (await db.tableExists("feedback")) {
|
|
117
120
|
const ids = entries.map((e) => e.id);
|
|
118
121
|
const marks = ids.map((_, i) => `$f${i}`).join(", ");
|
|
119
122
|
const fb: Record<string, string> = {};
|
|
120
123
|
ids.forEach((id, i) => (fb[`$f${i}`] = id));
|
|
121
|
-
|
|
124
|
+
const rows = await db.query<{ ledger_id: string; verdict: string; note: string; created_at_ms: number }>(
|
|
125
|
+
`SELECT ledger_id, verdict, note, created_at_ms FROM feedback WHERE ledger_id IN (${marks}) ORDER BY created_at_ms ASC`,
|
|
126
|
+
fb,
|
|
127
|
+
);
|
|
128
|
+
for (const r of rows) {
|
|
122
129
|
const list = verdicts.get(r.ledger_id) ?? [];
|
|
123
|
-
list.push({ verdict: r.verdict === "good" ? "good" : "bad", note: r.note, createdAtMs: r.created_at_ms });
|
|
130
|
+
list.push({ verdict: r.verdict === "good" ? "good" : "bad", note: r.note, createdAtMs: num(r.created_at_ms) });
|
|
124
131
|
verdicts.set(r.ledger_id, list);
|
|
125
132
|
}
|
|
126
133
|
}
|
|
@@ -132,7 +139,7 @@ export function decisionEntries(db: Database, filter: DecisionFilter): DecisionE
|
|
|
132
139
|
* as the ledger counts them. `contextScope`, when given, narrows to the turns that carried
|
|
133
140
|
* exactly that agentdox scope — what a front door charges back to one project.
|
|
134
141
|
*/
|
|
135
|
-
export function spendUsdSince(db:
|
|
142
|
+
export async function spendUsdSince(db: SqlDb, sinceMs: number, harness: HarnessScope, contextScope?: string): Promise<number> {
|
|
136
143
|
const s = scope(harness, "harness_id");
|
|
137
144
|
if (s === null) return 0;
|
|
138
145
|
const where = ["created_at_ms >= $since", ...s.sql];
|
|
@@ -141,36 +148,42 @@ export function spendUsdSince(db: Database, sinceMs: number, harness: HarnessSco
|
|
|
141
148
|
where.push("scope = $scope");
|
|
142
149
|
bind.$scope = contextScope;
|
|
143
150
|
}
|
|
144
|
-
const row = db.
|
|
145
|
-
|
|
151
|
+
const row = await db.one<{ usd: unknown }>(
|
|
152
|
+
`SELECT COALESCE(SUM(${USD}), 0) AS usd FROM ledger WHERE ${where.join(" AND ")}`,
|
|
153
|
+
bind,
|
|
154
|
+
);
|
|
155
|
+
return num(row?.usd);
|
|
146
156
|
}
|
|
147
157
|
|
|
148
158
|
/** Verdicts since `sinceMs`, by model and the most recent 200, joined to the ledger for the judging harness. */
|
|
149
|
-
export function feedbackView(db:
|
|
150
|
-
|
|
151
|
-
if (!exists) return { byModel: [], recent: [] };
|
|
159
|
+
export async function feedbackView(db: SqlDb, sinceMs: number, harness: HarnessScope): Promise<FeedbackView> {
|
|
160
|
+
if (!(await db.tableExists("feedback"))) return { byModel: [], recent: [] };
|
|
152
161
|
const s = scope(harness, "l.harness_id");
|
|
153
162
|
if (s === null) return { byModel: [], recent: [] };
|
|
154
163
|
const where = ["f.created_at_ms >= $since", ...s.sql].join(" AND ");
|
|
155
164
|
const bind = { $since: sinceMs, ...s.bind };
|
|
156
165
|
const recent = (
|
|
157
|
-
db.query
|
|
158
|
-
at_ms
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
+
await db.query<{ at_ms: number; slug: string; tier: string; verdict: string; note: string; harness_id: string }>(
|
|
167
|
+
`SELECT f.created_at_ms AS at_ms, f.slug, f.tier, f.verdict, f.note, COALESCE(l.harness_id, '') AS harness_id
|
|
168
|
+
FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} ORDER BY f.created_at_ms DESC LIMIT 200`,
|
|
169
|
+
bind,
|
|
170
|
+
)
|
|
171
|
+
).map((r) => ({
|
|
172
|
+
atMs: num(r.at_ms),
|
|
173
|
+
slug: r.slug,
|
|
174
|
+
tier: r.tier,
|
|
175
|
+
verdict: (r.verdict === "good" ? "good" : "bad") as "good" | "bad",
|
|
176
|
+
note: r.note,
|
|
177
|
+
harnessId: r.harness_id,
|
|
178
|
+
}));
|
|
166
179
|
const byModel = (
|
|
167
|
-
db
|
|
168
|
-
.
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
).map((r) => ({ slug: r.slug, good: r.good, bad: r.bad, judges: r.judges }));
|
|
180
|
+
await db.query<{ slug: string; good: unknown; bad: unknown; judges: unknown }>(
|
|
181
|
+
`SELECT f.slug, SUM(CASE WHEN f.verdict = 'good' THEN 1 ELSE 0 END) AS good, SUM(CASE WHEN f.verdict = 'bad' THEN 1 ELSE 0 END) AS bad,
|
|
182
|
+
COUNT(DISTINCT COALESCE(l.harness_id, '')) AS judges
|
|
183
|
+
FROM feedback f LEFT JOIN ledger l ON l.id = f.ledger_id WHERE ${where} GROUP BY f.slug ORDER BY bad DESC, good DESC, f.slug ASC`,
|
|
184
|
+
bind,
|
|
185
|
+
)
|
|
186
|
+
).map((r) => ({ slug: r.slug, good: num(r.good), bad: num(r.bad), judges: num(r.judges) }));
|
|
174
187
|
return { byModel, recent };
|
|
175
188
|
}
|
|
176
189
|
|
|
@@ -180,36 +193,53 @@ export function feedbackView(db: Database, sinceMs: number, harness: HarnessScop
|
|
|
180
193
|
* can charge each project its own share; rows from before v18 (and turns that carried no
|
|
181
194
|
* scope) group under "".
|
|
182
195
|
*/
|
|
183
|
-
export function exportRows(db:
|
|
196
|
+
export async function exportRows(db: SqlDb, sinceMs: number, harness: HarnessScope): Promise<ExportRow[]> {
|
|
184
197
|
const s = scope(harness, "harness_id");
|
|
185
198
|
if (s === null) return [];
|
|
186
199
|
const where = ["created_at_ms >= $since", "requested_model <> 'digest'", ...s.sql].join(" AND ");
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
200
|
+
// GROUP BY names the day expression rather than its alias: Postgres does not
|
|
201
|
+
// allow a select alias in GROUP BY, and repeating it keeps one statement for
|
|
202
|
+
// both engines.
|
|
203
|
+
const day = db.utcDay("created_at_ms");
|
|
204
|
+
const rows = await db.query<{
|
|
205
|
+
day: string;
|
|
206
|
+
harness_id: string;
|
|
207
|
+
slug: string;
|
|
208
|
+
scope: string;
|
|
209
|
+
dispatches: unknown;
|
|
210
|
+
prompt_tokens: unknown;
|
|
211
|
+
cached_tokens: unknown;
|
|
212
|
+
completion_tokens: unknown;
|
|
213
|
+
spend: unknown;
|
|
214
|
+
escalations: unknown;
|
|
215
|
+
errors: unknown;
|
|
216
|
+
}>(
|
|
217
|
+
`SELECT ${day} AS day, harness_id, COALESCE(served_slug, slug) AS slug, COALESCE(scope, '') AS scope,
|
|
190
218
|
COUNT(*) AS dispatches,
|
|
191
|
-
COALESCE(SUM(
|
|
192
|
-
COALESCE(SUM(
|
|
193
|
-
COALESCE(SUM(
|
|
219
|
+
COALESCE(SUM(${db.jsonNum("usage", "promptTokens")}), 0) AS prompt_tokens,
|
|
220
|
+
COALESCE(SUM(${db.jsonNum("usage", "cachedTokens")}), 0) AS cached_tokens,
|
|
221
|
+
COALESCE(SUM(${db.jsonNum("usage", "completionTokens")}), 0) AS completion_tokens,
|
|
194
222
|
COALESCE(SUM(${USD}), 0) AS spend,
|
|
195
223
|
SUM(CASE WHEN escalation_signal IS NOT NULL THEN 1 ELSE 0 END) AS escalations,
|
|
196
224
|
SUM(CASE WHEN error IS NOT NULL THEN 1 ELSE 0 END) AS errors
|
|
197
|
-
FROM ledger WHERE ${where}
|
|
198
|
-
|
|
199
|
-
|
|
225
|
+
FROM ledger WHERE ${where}
|
|
226
|
+
GROUP BY ${day}, harness_id, COALESCE(served_slug, slug), COALESCE(scope, '')
|
|
227
|
+
ORDER BY 1 ASC, harness_id ASC, spend DESC`,
|
|
228
|
+
{ $since: sinceMs, ...s.bind },
|
|
229
|
+
);
|
|
200
230
|
return rows.map((r) => ({
|
|
201
231
|
day: r.day,
|
|
202
232
|
harnessId: r.harness_id,
|
|
203
233
|
slug: r.slug,
|
|
204
234
|
scope: r.scope,
|
|
205
235
|
provider: providerOfSlug(r.slug),
|
|
206
|
-
dispatches: r.dispatches,
|
|
207
|
-
promptTokens: r.prompt_tokens,
|
|
208
|
-
cachedTokens: r.cached_tokens,
|
|
209
|
-
completionTokens: r.completion_tokens,
|
|
210
|
-
spendUsd: r.spend,
|
|
211
|
-
escalations: r.escalations,
|
|
212
|
-
errors: r.errors,
|
|
236
|
+
dispatches: num(r.dispatches),
|
|
237
|
+
promptTokens: num(r.prompt_tokens),
|
|
238
|
+
cachedTokens: num(r.cached_tokens),
|
|
239
|
+
completionTokens: num(r.completion_tokens),
|
|
240
|
+
spendUsd: num(r.spend),
|
|
241
|
+
escalations: num(r.escalations),
|
|
242
|
+
errors: num(r.errors),
|
|
213
243
|
}));
|
|
214
244
|
}
|
|
215
245
|
|
package/src/eval/calibrate.ts
CHANGED
|
@@ -58,13 +58,26 @@ export interface AnchorPoint {
|
|
|
58
58
|
}
|
|
59
59
|
|
|
60
60
|
/**
|
|
61
|
-
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
61
|
+
* The raw range the anchors must actually cover before a line through them means anything.
|
|
62
|
+
*
|
|
63
|
+
* Correlation alone does NOT catch a compressed fit: three anchors published 20/50/80 that
|
|
64
|
+
* our suite scores 0.96/0.97/0.98 correlate at r = 1.0, and the line they define has a slope
|
|
65
|
+
* of ~3000 index points per unit of raw score. That fit is arithmetically perfect and
|
|
66
|
+
* completely useless — it is how a model published at 39.5 was calibrated to 22.8. If the
|
|
67
|
+
* anchors barely differ on our suite, our suite cannot place anything between them.
|
|
68
|
+
*/
|
|
69
|
+
export const MIN_RAW_SPREAD = 0.15;
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* OLS fit, or null when the calibration cannot be trusted: too few points, too narrow a raw
|
|
73
|
+
* range, no spread, a non-positive slope, or weak correlation. A suite that does not track
|
|
74
|
+
* AA positively, with real correlation, over a real range would turn a target's score into
|
|
75
|
+
* noise dressed as signal, so we refuse it and emit nothing for that axis.
|
|
65
76
|
*/
|
|
66
77
|
export function fitAxis(points: readonly AnchorPoint[]): LineFit | null {
|
|
67
78
|
if (points.length < MIN_ANCHORS) return null;
|
|
79
|
+
const raws = points.map((p) => p.raw);
|
|
80
|
+
if (Math.max(...raws) - Math.min(...raws) < MIN_RAW_SPREAD) return null;
|
|
68
81
|
const n = points.length;
|
|
69
82
|
let sx = 0;
|
|
70
83
|
let sy = 0;
|
|
@@ -101,19 +114,32 @@ export function applyFit(fit: LineFit, raw: number): number {
|
|
|
101
114
|
const AXES: readonly QualityAxis[] = ["coding", "intelligence", "agentic"];
|
|
102
115
|
|
|
103
116
|
/**
|
|
104
|
-
*
|
|
105
|
-
*
|
|
106
|
-
*
|
|
117
|
+
* The correlation a fit must reach before its numbers are published as scores. `MIN_R` (0.5)
|
|
118
|
+
* is the bar for a fit being computable at all; this is the bar for TRUSTING one. The fit's
|
|
119
|
+
* `r` and `n` were previously computed and then discarded, so an r of 0.51 and one of 0.99
|
|
120
|
+
* produced indistinguishable output — and a shallow fit silently compressed every target.
|
|
121
|
+
*/
|
|
122
|
+
export const PUBLISH_MIN_R = 0.8;
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Fit every axis from the anchors' raw suite scores paired with their known AA scores.
|
|
126
|
+
* `anchorAa` supplies the AA index per slug+axis (absent ⇒ that anchor is not used there).
|
|
127
|
+
*
|
|
128
|
+
* `rawOf` selects which observations the fit is built from. It defaults to every band, but
|
|
129
|
+
* callers should pass the HARD band: easy and moderate sit at ~1.0 for every model worth
|
|
130
|
+
* ranking, so including them leaves the regression almost no variation in x against a wide
|
|
131
|
+
* spread in published y, and the slope collapses toward flat.
|
|
107
132
|
*/
|
|
108
133
|
export function fitCalibration(
|
|
109
134
|
anchors: readonly EvalResult[],
|
|
110
135
|
anchorAa: (slug: string, axis: QualityAxis) => number | undefined,
|
|
136
|
+
rawOf: (result: EvalResult, axis: QualityAxis) => number | null = (r, axis) => axisMean(r.axes[axis]),
|
|
111
137
|
): Calibration {
|
|
112
138
|
const cal: Calibration = {};
|
|
113
139
|
for (const axis of AXES) {
|
|
114
140
|
const points: AnchorPoint[] = [];
|
|
115
141
|
for (const r of anchors) {
|
|
116
|
-
const raw =
|
|
142
|
+
const raw = rawOf(r, axis);
|
|
117
143
|
const aa = anchorAa(r.slug, axis);
|
|
118
144
|
if (raw !== null && aa !== undefined) points.push({ raw, aa });
|
|
119
145
|
}
|
|
@@ -123,15 +149,24 @@ export function fitCalibration(
|
|
|
123
149
|
return cal;
|
|
124
150
|
}
|
|
125
151
|
|
|
126
|
-
function axisMean(a: AxisScore | undefined): number | null {
|
|
152
|
+
export function axisMean(a: AxisScore | undefined): number | null {
|
|
127
153
|
return a === undefined || a.n === 0 ? null : a.sum / a.n;
|
|
128
154
|
}
|
|
129
155
|
|
|
130
|
-
/**
|
|
156
|
+
/** The hard band alone, for calibration. */
|
|
157
|
+
export const hardRaw = (r: EvalResult, axis: QualityAxis): number | null => axisMean(r.axesHard[axis]);
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* Calibrated local FeedScores for the targets, one axis at a time. An axis is skipped when it
|
|
161
|
+
* has no fit, when the fit is weaker than `minR`, or when the target made no observation on
|
|
162
|
+
* it — honest silence rather than a number nobody should act on.
|
|
163
|
+
*/
|
|
131
164
|
export function toLocalFeedScores(
|
|
132
165
|
targets: readonly EvalResult[],
|
|
133
166
|
cal: Calibration,
|
|
134
167
|
authorOf: (slug: string) => string,
|
|
168
|
+
rawOf: (result: EvalResult, axis: QualityAxis) => number | null = (r, axis) => axisMean(r.axes[axis]),
|
|
169
|
+
minR = PUBLISH_MIN_R,
|
|
135
170
|
): FeedScore[] {
|
|
136
171
|
const out: FeedScore[] = [];
|
|
137
172
|
for (const r of targets) {
|
|
@@ -139,8 +174,8 @@ export function toLocalFeedScores(
|
|
|
139
174
|
let any = false;
|
|
140
175
|
for (const axis of AXES) {
|
|
141
176
|
const fit = cal[axis];
|
|
142
|
-
const raw =
|
|
143
|
-
if (fit === undefined || raw === null) continue;
|
|
177
|
+
const raw = rawOf(r, axis);
|
|
178
|
+
if (fit === undefined || raw === null || fit.r < minR) continue;
|
|
144
179
|
entry[axis] = applyFit(fit, raw);
|
|
145
180
|
any = true;
|
|
146
181
|
}
|
package/src/eval/run.ts
CHANGED
|
@@ -39,6 +39,14 @@ export interface EvalResult {
|
|
|
39
39
|
* questions were easy.
|
|
40
40
|
*/
|
|
41
41
|
byComplexity: Partial<Record<Complexity, AxisScore>>;
|
|
42
|
+
/**
|
|
43
|
+
* Per-axis scores from the HARD band alone. Calibration fits against these: the easy and
|
|
44
|
+
* moderate bands sit at ~1.0 for every model worth ranking, so including them gives the
|
|
45
|
+
* regression almost no variation in x against a wide spread in published y — the fitted
|
|
46
|
+
* slope goes shallow and every target is dragged toward the middle. Measured: a model
|
|
47
|
+
* published at intelligence 39.5 calibrated to 22.8 across 10 passes.
|
|
48
|
+
*/
|
|
49
|
+
axesHard: Record<QualityAxis, AxisScore>;
|
|
42
50
|
/**
|
|
43
51
|
* Per-axis spread across passes: max pass mean minus min pass mean, or null under two
|
|
44
52
|
* passes. A wide spread means the headline is one sample of a noisy quantity, and is the
|
|
@@ -144,6 +152,7 @@ async function scorePass(slug: string, args: RunEvalArgs): Promise<EvalResult> {
|
|
|
144
152
|
}
|
|
145
153
|
}
|
|
146
154
|
const byComplexity: Partial<Record<Complexity, AxisScore>> = {};
|
|
155
|
+
const axesHard: Record<QualityAxis, AxisScore> = { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } };
|
|
147
156
|
for (const o of [...objective, ...judgedOutcomes, ...scenarioOutcomes]) {
|
|
148
157
|
if (!o.ok) {
|
|
149
158
|
// An unobserved task is NOT a zero: a provider's throttle or outage would otherwise
|
|
@@ -154,11 +163,15 @@ async function scorePass(slug: string, args: RunEvalArgs): Promise<EvalResult> {
|
|
|
154
163
|
}
|
|
155
164
|
axes[o.axis].sum += o.grade;
|
|
156
165
|
axes[o.axis].n += 1;
|
|
166
|
+
if (o.complexity === "hard") {
|
|
167
|
+
axesHard[o.axis].sum += o.grade;
|
|
168
|
+
axesHard[o.axis].n += 1;
|
|
169
|
+
}
|
|
157
170
|
const band = (byComplexity[o.complexity] ??= { sum: 0, n: 0 });
|
|
158
171
|
band.sum += o.grade;
|
|
159
172
|
band.n += 1;
|
|
160
173
|
}
|
|
161
|
-
return { slug, axes, errors, repeats: 1, spread: {}, byComplexity };
|
|
174
|
+
return { slug, axes, errors, repeats: 1, spread: {}, byComplexity, axesHard };
|
|
162
175
|
}
|
|
163
176
|
|
|
164
177
|
const AXES: readonly QualityAxis[] = ["coding", "intelligence", "agentic"];
|
|
@@ -174,6 +187,7 @@ async function scoreModel(slug: string, args: RunEvalArgs): Promise<EvalResult>
|
|
|
174
187
|
const axes: Record<QualityAxis, AxisScore> = { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } };
|
|
175
188
|
const means: Record<QualityAxis, number[]> = { coding: [], intelligence: [], agentic: [] };
|
|
176
189
|
const byComplexity: Partial<Record<Complexity, AxisScore>> = {};
|
|
190
|
+
const axesHard: Record<QualityAxis, AxisScore> = { coding: { sum: 0, n: 0 }, intelligence: { sum: 0, n: 0 }, agentic: { sum: 0, n: 0 } };
|
|
177
191
|
let errors = 0;
|
|
178
192
|
for (let i = 0; i < passes; i++) {
|
|
179
193
|
const pass = await scorePass(slug, args);
|
|
@@ -182,6 +196,8 @@ async function scoreModel(slug: string, args: RunEvalArgs): Promise<EvalResult>
|
|
|
182
196
|
for (const axis of AXES) {
|
|
183
197
|
axes[axis].sum += pass.axes[axis].sum;
|
|
184
198
|
axes[axis].n += pass.axes[axis].n;
|
|
199
|
+
axesHard[axis].sum += pass.axesHard[axis].sum;
|
|
200
|
+
axesHard[axis].n += pass.axesHard[axis].n;
|
|
185
201
|
if (pass.axes[axis].n > 0) means[axis].push(pass.axes[axis].sum / pass.axes[axis].n);
|
|
186
202
|
}
|
|
187
203
|
for (const [band, score] of Object.entries(pass.byComplexity) as [Complexity, AxisScore][]) {
|
|
@@ -195,7 +211,7 @@ async function scoreModel(slug: string, args: RunEvalArgs): Promise<EvalResult>
|
|
|
195
211
|
const m = means[axis];
|
|
196
212
|
if (m.length > 1) spread[axis] = Math.max(...m) - Math.min(...m);
|
|
197
213
|
}
|
|
198
|
-
return { slug, axes, errors, repeats: passes, spread, byComplexity };
|
|
214
|
+
return { slug, axes, errors, repeats: passes, spread, byComplexity, axesHard };
|
|
199
215
|
}
|
|
200
216
|
|
|
201
217
|
export async function runEval(args: RunEvalArgs): Promise<EvalResult[]> {
|
package/src/lib.ts
CHANGED
|
@@ -25,14 +25,18 @@ export type { DeepPartial } from "./config/load.ts";
|
|
|
25
25
|
export { buildUsageReport, renderUsageReport, type UsageReport, type ReportTotals } from "./cost/report.ts";
|
|
26
26
|
export { buildDailySummary, renderDailySummary, type DailySummary } from "./cost/summary.ts";
|
|
27
27
|
export { openDb } from "./util/sqlite.ts";
|
|
28
|
+
// A front door reads the ledger through the engine-agnostic handle: the store
|
|
29
|
+
// may be a file or a shared database, and the view functions take this.
|
|
30
|
+
export { dialectOf, openSqlDb, num, numOrNull, type Dialect, type SqlDb } from "./util/sql.ts";
|
|
28
31
|
export { spendUsdSince, feedbackView, exportRows, exportCsv, decisionEntries, harnessScopeParam, type HarnessScope, type ExportRow, type FeedbackRow, type FeedbackByModel, type FeedbackView, type DecisionEntry, type DecisionFilter } from "./cost/views.ts";
|
|
29
|
-
export {
|
|
32
|
+
export { createSqlLedger } from "./cost/ledger-sql.ts";
|
|
33
|
+
export { migrateStore, STORE_TABLES } from "./util/schema.ts";
|
|
30
34
|
export { createFeedbackStore, type FeedbackStore, type FeedbackRecord } from "./cost/feedback.ts";
|
|
31
35
|
export { buildExecutable, collectPackageFiles, executableFileName, hostTarget, isExecutableTarget, EXECUTABLE_TARGETS, type ExecutableTarget, type BuildExecutableResult } from "./cli/build-executable.ts";
|
|
32
36
|
export { parseSkillsBundle, type SkillsBundle } from "./cli/skills.ts";
|
|
33
37
|
export type { RequestPolicy } from "./wire/types.ts";
|
|
34
38
|
export type { CatalogView, CatalogViewModel } from "./server/catalog-view.ts";
|
|
35
|
-
export type {
|
|
39
|
+
export type { AsyncLedger, LedgerEntry, PruneResult } from "./cost/types.ts";
|
|
36
40
|
// Retention: a front door asks through `POST /v1/router/prune` rather than
|
|
37
41
|
// deleting from the ledger itself. The interval is exported so it can say when.
|
|
38
42
|
export { RETENTION_INTERVAL_MS } from "./cost/retention.ts";
|
package/src/router/candidates.ts
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
import type { CatalogModel, CatalogSnapshot } from "../catalog/types.ts";
|
|
8
8
|
import type { FilterConfig, QualityAxis, RouterConfig } from "../config/types.ts";
|
|
9
9
|
import { forecast, priceAt } from "../cost/forecast.ts";
|
|
10
|
-
import type {
|
|
10
|
+
import type { LedgerSignals, ModelLatency } from "../cost/types.ts";
|
|
11
11
|
import type { NormRequest } from "../wire/types.ts";
|
|
12
12
|
import { effectivePriceCeiling, effectiveQualityFloor, tierPlanFor } from "./tier-plan.ts";
|
|
13
13
|
import type { Candidate, Features, Rejection, TaskType, Tier } from "./types.ts";
|
|
@@ -27,7 +27,6 @@ export interface BuildCandidatesArgs {
|
|
|
27
27
|
/** Task type; its config selects the axis, quality floor, and image filter. */
|
|
28
28
|
task: TaskType;
|
|
29
29
|
snapshot: CatalogSnapshot;
|
|
30
|
-
ledger: Ledger | null;
|
|
31
30
|
cfg: RouterConfig;
|
|
32
31
|
expectedCompletionTokens: number;
|
|
33
32
|
/** Slug whose prompt cache is warm this turn; wins score ties. */
|
|
@@ -154,7 +153,7 @@ function latencyMultiplier(latency: ModelLatency | null, filters: FilterConfig,
|
|
|
154
153
|
}
|
|
155
154
|
|
|
156
155
|
export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candidate[]; rejected: Rejection[] } {
|
|
157
|
-
const { req, features, tier, task, snapshot,
|
|
156
|
+
const { req, features, tier, task, snapshot, cfg, expectedCompletionTokens, warmSlug, relaxLevel = 0 } = args;
|
|
158
157
|
// A Set only when non-empty: the common path allocates nothing.
|
|
159
158
|
const excluded = args.excludeSlugs === undefined || args.excludeSlugs.length === 0 ? null : new Set(args.excludeSlugs);
|
|
160
159
|
const tierCfg = cfg.tiers[tier];
|
|
@@ -318,14 +317,11 @@ export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candid
|
|
|
318
317
|
continue;
|
|
319
318
|
}
|
|
320
319
|
|
|
321
|
-
//
|
|
322
|
-
//
|
|
323
|
-
//
|
|
320
|
+
// Signals are prefetched by `route` — one query per signal kind for the
|
|
321
|
+
// whole candidate set. Absent means "the ledger knows nothing about this
|
|
322
|
+
// model", which is what a cold start looks like and is handled below.
|
|
324
323
|
const signals = args.signals;
|
|
325
|
-
const trust =
|
|
326
|
-
signals?.get(slug)?.trust ??
|
|
327
|
-
ledger?.trust(slug, filters.trustScopedByHarness ? req.harnessId : undefined, filters.feedbackByTask ? task : undefined) ??
|
|
328
|
-
null;
|
|
324
|
+
const trust = signals?.get(slug)?.trust ?? null;
|
|
329
325
|
if (!relaxTrust && trust !== null && trust.attempts >= filters.minTrustSamples && trust.successRate < filters.minTrust) {
|
|
330
326
|
rejected.push({
|
|
331
327
|
slug,
|
|
@@ -342,11 +338,7 @@ export function buildCandidates(args: BuildCandidatesArgs): { candidates: Candid
|
|
|
342
338
|
// gate can. Only measured models are dropped, so a new model still gets its
|
|
343
339
|
// cold-start turns to accumulate samples. Relaxed with trust in rescue.
|
|
344
340
|
const needLatency = filters.latencyWeight > 0 || filters.maxExpectedWaitMs !== undefined;
|
|
345
|
-
const latency = needLatency
|
|
346
|
-
? (signals?.get(slug)?.latency ??
|
|
347
|
-
ledger?.latency(slug, filters.trustScopedByHarness ? req.harnessId : undefined) ??
|
|
348
|
-
null)
|
|
349
|
-
: null;
|
|
341
|
+
const latency = needLatency ? (signals?.get(slug)?.latency ?? null) : null;
|
|
350
342
|
if (
|
|
351
343
|
!relaxTrust &&
|
|
352
344
|
filters.maxExpectedWaitMs !== undefined &&
|
package/src/router/classify.ts
CHANGED
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
import type { CatalogSource } from "../catalog/types.ts";
|
|
7
7
|
import type { QualityAxis, RouterConfig } from "../config/types.ts";
|
|
8
8
|
import { forecast } from "../cost/forecast.ts";
|
|
9
|
-
import type {
|
|
9
|
+
import type { AsyncLedger } from "../cost/types.ts";
|
|
10
10
|
import { estimateTokens } from "../tokens/estimate.ts";
|
|
11
11
|
import type { UpstreamClient } from "../upstream/types.ts";
|
|
12
12
|
import { sha256Hex } from "../util/hash.ts";
|
|
@@ -219,7 +219,7 @@ export function classifyTask(f: Features): TaskType {
|
|
|
219
219
|
|
|
220
220
|
export interface ClassifyDeps {
|
|
221
221
|
upstream: UpstreamClient;
|
|
222
|
-
ledger:
|
|
222
|
+
ledger: AsyncLedger | null;
|
|
223
223
|
catalog: CatalogSource | null;
|
|
224
224
|
}
|
|
225
225
|
|
|
@@ -334,11 +334,13 @@ export async function classify(
|
|
|
334
334
|
}
|
|
335
335
|
|
|
336
336
|
// Cost guard: adjudication must be cheap relative to the turn it classifies.
|
|
337
|
-
const blend = deps.ledger?.blendedRate(cfg.ledger.blendWindowDays) ?? null;
|
|
337
|
+
const blend = (await deps.ledger?.blendedRate(cfg.ledger.blendWindowDays)) ?? null;
|
|
338
338
|
const inputRate = (blend?.inputPerMtok ?? cfg.ledger.fallbackBlend.inputPerMtok) / 1e6;
|
|
339
339
|
const outputRate = (blend?.outputPerMtok ?? cfg.ledger.fallbackBlend.outputPerMtok) / 1e6;
|
|
340
340
|
const judge = deps.catalog?.find(cc.model);
|
|
341
|
-
|
|
341
|
+
// The adjudicator prompt is short and its family unknown, so the default
|
|
342
|
+
// family ratio is as good as a measured one here.
|
|
343
|
+
const digestTokens = estimateTokens(Buffer.byteLength(ADJUDICATOR_SYSTEM) + Buffer.byteLength(digest), "unknown", null);
|
|
342
344
|
const adjudicatorUsd =
|
|
343
345
|
judge !== undefined
|
|
344
346
|
? forecast(judge, {
|
package/src/router/index.ts
CHANGED
|
@@ -10,23 +10,25 @@
|
|
|
10
10
|
* real request without perturbing the conversation it belongs to.
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
|
-
import type { CatalogSource } from "../catalog/types.ts";
|
|
13
|
+
import type { CatalogSnapshot, CatalogSource } from "../catalog/types.ts";
|
|
14
14
|
import type { ProfileConfig, RouterConfig } from "../config/types.ts";
|
|
15
|
-
import type
|
|
15
|
+
import { type AsyncLedger, type LedgerReader } from "../cost/types.ts";
|
|
16
16
|
import { estimatePromptTokens } from "../tokens/estimate.ts";
|
|
17
17
|
import type { UpstreamClient } from "../upstream/types.ts";
|
|
18
|
+
import { modelNotFound } from "../wire/openai/errors.ts";
|
|
18
19
|
import type { NormRequest, RequestPolicy } from "../wire/types.ts";
|
|
19
20
|
import { classify, classifyTask } from "./classify.ts";
|
|
20
21
|
import { extractFeatures } from "./features.ts";
|
|
21
|
-
import { select } from "./select.ts";
|
|
22
|
+
import { monthStartMs, select, type TurnReads } from "./select.ts";
|
|
22
23
|
import { TIER_ORDER, type Classification, type ConversationStore, type Decision, type Router, type Tier } from "./types.ts";
|
|
23
|
-
|
|
24
24
|
export interface RouterDeps {
|
|
25
25
|
config: RouterConfig;
|
|
26
26
|
catalog: CatalogSource;
|
|
27
|
-
ledger:
|
|
27
|
+
ledger: AsyncLedger;
|
|
28
28
|
conversations: ConversationStore;
|
|
29
29
|
upstream: UpstreamClient;
|
|
30
|
+
/** Async ledger reads; defaults to reading `ledger` directly. */
|
|
31
|
+
reader?: LedgerReader;
|
|
30
32
|
}
|
|
31
33
|
|
|
32
34
|
/**
|
|
@@ -90,20 +92,89 @@ export function resolveProfile(cfg: RouterConfig, requestedModel: string, isSuba
|
|
|
90
92
|
return fallback;
|
|
91
93
|
}
|
|
92
94
|
|
|
95
|
+
/**
|
|
96
|
+
* The `model` a client names is a profile id. When it is neither a profile nor
|
|
97
|
+
* a catalog slug, routing something else and reporting the asked-for name back
|
|
98
|
+
* is a silent substitution: the caller is billed for a model it never chose.
|
|
99
|
+
* A real slug becomes a pin (absolute, the way a session override is); anything
|
|
100
|
+
* else is refused.
|
|
101
|
+
*
|
|
102
|
+
* @param asked the client's `model` before the provider prefix was stripped.
|
|
103
|
+
* @returns the slug to pin, or undefined when a profile matched.
|
|
104
|
+
*/
|
|
105
|
+
export function pinForRequestedModel(
|
|
106
|
+
cfg: RouterConfig,
|
|
107
|
+
slugs: readonly string[],
|
|
108
|
+
requestedModel: string,
|
|
109
|
+
asked: string,
|
|
110
|
+
): string | undefined {
|
|
111
|
+
if (cfg.profiles.some((p) => p.id === requestedModel)) return undefined;
|
|
112
|
+
if (slugs.includes(asked)) return asked;
|
|
113
|
+
throw modelNotFound(asked);
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* Reads the ledger for one turn, concurrently, before selection runs.
|
|
118
|
+
*
|
|
119
|
+
* Only what this turn's configuration actually consults is fetched: month and
|
|
120
|
+
* day spend are skipped when no such budget is set, and the escalation-cost
|
|
121
|
+
* term is skipped when its weight is 0. Empty catalog ⇒ no signal queries at
|
|
122
|
+
* all, which is what keeps the tests' fakes cheap.
|
|
123
|
+
*/
|
|
124
|
+
export async function prefetchTurnReads(
|
|
125
|
+
reader: LedgerReader | null,
|
|
126
|
+
req: NormRequest,
|
|
127
|
+
profile: ProfileConfig,
|
|
128
|
+
cfg: RouterConfig,
|
|
129
|
+
snapshot: CatalogSnapshot,
|
|
130
|
+
task: string,
|
|
131
|
+
nowMs: number = Date.now(),
|
|
132
|
+
): Promise<TurnReads> {
|
|
133
|
+
if (reader === null) return {};
|
|
134
|
+
const slugs = snapshot.models.map((m) => m.slug);
|
|
135
|
+
const harness = cfg.filters.trustScopedByHarness ? req.harnessId : undefined;
|
|
136
|
+
const perMonthUsd = profile.budget?.perMonthUsd ?? cfg.budget.perMonthUsd;
|
|
137
|
+
const perDayUsd = profile.budget?.perDayUsd ?? cfg.budget.perDayUsd;
|
|
138
|
+
const [signals, cacheReliability, escalation, monthSpendUsd, daySpendUsd] = await Promise.all([
|
|
139
|
+
slugs.length > 0 ? reader.signals(slugs, harness, cfg.filters.feedbackByTask ? task : undefined) : undefined,
|
|
140
|
+
slugs.length > 0 && cfg.filters.cacheReliabilityMinSamples > 0 ? reader.cacheReliability(slugs) : undefined,
|
|
141
|
+
cfg.filters.escalationCostWeight > 0 ? reader.escalationCost(cfg.ledger.blendWindowDays) : undefined,
|
|
142
|
+
// Month pacing can tighten the daily ceiling, so month spend is needed
|
|
143
|
+
// whenever either budget is set.
|
|
144
|
+
perMonthUsd !== undefined ? reader.spendSince(monthStartMs(nowMs), req.harnessId) : undefined,
|
|
145
|
+
perDayUsd !== undefined || perMonthUsd !== undefined ? reader.spendSince(nowMs - 86_400_000, req.harnessId) : undefined,
|
|
146
|
+
]);
|
|
147
|
+
return {
|
|
148
|
+
...(signals === undefined ? {} : { signals }),
|
|
149
|
+
...(cacheReliability === undefined ? {} : { cacheReliability }),
|
|
150
|
+
...(escalation === undefined ? {} : { escalationUsdPerPromptToken: escalation?.usdPerPromptToken ?? null }),
|
|
151
|
+
...(monthSpendUsd === undefined ? {} : { monthSpendUsd }),
|
|
152
|
+
...(daySpendUsd === undefined ? {} : { daySpendUsd }),
|
|
153
|
+
};
|
|
154
|
+
}
|
|
155
|
+
|
|
93
156
|
export function createRouter(deps: RouterDeps): Router {
|
|
94
157
|
const { config, catalog, ledger, conversations, upstream } = deps;
|
|
158
|
+
// A Postgres ledger supplies its own reader; a local one is wrapped, so the
|
|
159
|
+
// prefetch path is identical for both.
|
|
160
|
+
// The unified ledger IS a reader: `LedgerReader` is the narrow half of it.
|
|
161
|
+
const reader = deps.reader ?? ledger;
|
|
95
162
|
|
|
96
163
|
return {
|
|
97
164
|
async route(
|
|
98
165
|
req: NormRequest,
|
|
99
166
|
opts: { attempt: number; escalateFrom?: Tier; excludeSlugs?: readonly string[]; forceTier?: Tier; forceSlug?: string },
|
|
100
167
|
): Promise<Decision> {
|
|
101
|
-
const state = conversations.get(req.conversationKey) ?? conversations.load(req.conversationKey);
|
|
168
|
+
const state = (await conversations.get(req.conversationKey)) ?? (await conversations.load(req.conversationKey));
|
|
102
169
|
const snapshot = await catalog.get();
|
|
103
170
|
|
|
104
171
|
const priorTokenizer =
|
|
105
172
|
state.currentSlug === null ? undefined : catalog.find(state.currentSlug)?.tokenizer;
|
|
106
|
-
const
|
|
173
|
+
const tokenizer = priorTokenizer ?? NEUTRAL_TOKENIZER;
|
|
174
|
+
// One ratio, fetched before estimating: the estimate itself runs in
|
|
175
|
+
// synchronous code that a shared store cannot be read from.
|
|
176
|
+
const ratio = await ledger.tokenRatio(tokenizer);
|
|
177
|
+
const promptTokens = estimatePromptTokens(req, tokenizer, ratio);
|
|
107
178
|
const features = extractFeatures(req, promptTokens);
|
|
108
179
|
|
|
109
180
|
let classification: Classification;
|
|
@@ -136,7 +207,22 @@ export function createRouter(deps: RouterDeps): Router {
|
|
|
136
207
|
classification = await classify(req, features, config, { upstream, ledger, catalog });
|
|
137
208
|
}
|
|
138
209
|
|
|
139
|
-
const
|
|
210
|
+
const pinnedByModel = pinForRequestedModel(
|
|
211
|
+
config,
|
|
212
|
+
snapshot.models.map((m) => m.slug),
|
|
213
|
+
req.requestedModel,
|
|
214
|
+
req.requestedModelFull ?? req.requestedModel,
|
|
215
|
+
);
|
|
216
|
+
const policed = applyRequestPolicy(
|
|
217
|
+
resolveProfile(config, req.requestedModel, req.isSubagent),
|
|
218
|
+
config,
|
|
219
|
+
req.policy,
|
|
220
|
+
opts.forceSlug ?? pinnedByModel,
|
|
221
|
+
);
|
|
222
|
+
// Every ledger read this turn needs, fetched here rather than inside
|
|
223
|
+
// `select`: selection stays a synchronous pure function, and a ledger
|
|
224
|
+
// that can only be read asynchronously (Postgres) works unchanged.
|
|
225
|
+
const reads = await prefetchTurnReads(reader, req, policed.profile, policed.cfg, snapshot, classification.task);
|
|
140
226
|
const decision = select({
|
|
141
227
|
req,
|
|
142
228
|
features,
|
|
@@ -144,9 +230,9 @@ export function createRouter(deps: RouterDeps): Router {
|
|
|
144
230
|
profile: policed.profile,
|
|
145
231
|
state,
|
|
146
232
|
snapshot,
|
|
147
|
-
ledger,
|
|
148
233
|
cfg: policed.cfg,
|
|
149
234
|
nowMs: Date.now(),
|
|
235
|
+
reads,
|
|
150
236
|
...(opts.excludeSlugs === undefined ? {} : { excludeSlugs: opts.excludeSlugs }),
|
|
151
237
|
...(policed.forceSlug === undefined ? {} : { forceSlug: policed.forceSlug }),
|
|
152
238
|
});
|