auto-model-router 0.1.3 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +127 -46
- package/bun.lock +606 -0
- package/omp-extension/router-embed.ts +14 -6
- package/omp-extension/router-toast.ts +6 -1
- package/omp-extension/toast-logic.ts +7 -0
- package/package.json +2 -1
- package/research/analyze-ledger.ts +173 -0
- package/research/apply-cost-tuning.ts +73 -0
- package/research/cost-analysis.ts +150 -0
- package/research/feed-check.ts +64 -0
- package/research/model-recommendations.ts +86 -0
- package/research/project-yield.ts +96 -0
- package/research/run-eval.ts +133 -0
- package/research/status.ts +55 -0
- package/research/tier-fill.ts +109 -0
- package/research/tier-map.ts +123 -0
- package/src/catalog/benchmark-feeds.ts +397 -0
- package/src/catalog/openrouter-catalog.ts +30 -0
- package/src/config/defaults.ts +30 -0
- package/src/config/load.ts +2 -0
- package/src/config/schema.ts +34 -0
- package/src/config/types.ts +106 -0
- package/src/cost/ledger.ts +27 -3
- package/src/cost/types.ts +30 -0
- package/src/eval/calibrate.ts +131 -0
- package/src/eval/grade.ts +115 -0
- package/src/eval/judge.ts +71 -0
- package/src/eval/run.ts +126 -0
- package/src/eval/tasks.ts +272 -0
- package/src/index.ts +0 -1
- package/src/router/candidates.ts +13 -6
- package/src/router/explore.ts +59 -0
- package/src/router/select.ts +54 -4
- package/src/router/tier-plan.ts +57 -1
- package/src/router/types.ts +13 -0
- package/src/server/turn.ts +10 -2
- package/src/util/sqlite.ts +79 -1
- package/src/wire/openai/request.ts +5 -0
- package/src/wire/types.ts +7 -0
- package/test/benchmark-feeds.test.ts +222 -0
- package/test/escalate.test.ts +1 -0
- package/test/eval.test.ts +184 -0
- package/test/exploration.test.ts +251 -0
- package/test/failover.test.ts +5 -0
- package/test/hold-exploration.test.ts +124 -0
- package/test/tier-plan.test.ts +55 -1
- package/test/toast-logic.test.ts +32 -0
- package/test/tokens.test.ts +8 -0
- package/test/trust-attribution.test.ts +110 -2
- package/test/turn.test.ts +46 -0
- package/test/wire-request.test.ts +11 -0
- package/tools/smoke.ts +2 -0
- package/tools/sync-marketplace-version.ts +60 -0
|
@@ -43,15 +43,20 @@ import {
|
|
|
43
43
|
* Registers the auto-model-router provider (and its virtual models) into omp's model
|
|
44
44
|
* registry at a specific bound port.
|
|
45
45
|
*/
|
|
46
|
-
function registerRouterProvider(pi: ExtensionAPI, port: number, cfg: RouterConfig): void {
|
|
46
|
+
function registerRouterProvider(pi: ExtensionAPI, port: number, cfg: RouterConfig, sessionId: string): void {
|
|
47
47
|
const providerConfig = buildProviderConfig(port, cfg);
|
|
48
|
+
const headers: Record<string, string> = {};
|
|
49
|
+
if (providerConfig.harnessId !== undefined && providerConfig.harnessId !== "") {
|
|
50
|
+
headers["X-Omp-Harness"] = providerConfig.harnessId;
|
|
51
|
+
}
|
|
52
|
+
// Per-session scoping: lets the toast surface only this session's decisions
|
|
53
|
+
// even when several omp sessions share one embedded router's ledger.
|
|
54
|
+
if (sessionId !== "") headers["X-Omp-Session"] = sessionId;
|
|
48
55
|
pi.registerProvider(EMBED_PROVIDER_ID, {
|
|
49
56
|
baseUrl: providerConfig.baseUrl,
|
|
50
57
|
api: "openai-completions",
|
|
51
58
|
apiKey: EMBED_DUMMY_API_KEY,
|
|
52
|
-
...(
|
|
53
|
-
? { headers: { "X-Omp-Harness": providerConfig.harnessId } }
|
|
54
|
-
: {}),
|
|
59
|
+
...(Object.keys(headers).length > 0 ? { headers } : {}),
|
|
55
60
|
models: providerConfig.models.map((m) => ({
|
|
56
61
|
id: m.id,
|
|
57
62
|
name: m.name,
|
|
@@ -90,9 +95,12 @@ export default function (pi: ExtensionAPI): void {
|
|
|
90
95
|
// Subagents and headless sessions do not bind their own router; they
|
|
91
96
|
// route to the main's router via the shared port file. The main writes
|
|
92
97
|
// the file before spawning subagents, so the port is available here.
|
|
98
|
+
// The omp UI session id tags every request so the toast can scope its
|
|
99
|
+
// notifications to that exact session (see router-toast.ts).
|
|
100
|
+
const sessionId = ctx.sessionManager.getSessionId();
|
|
93
101
|
if (!ctx.hasUI) {
|
|
94
102
|
const port = readEmbedPort(portFile);
|
|
95
|
-
if (port !== null) registerRouterProvider(pi, port, cfg);
|
|
103
|
+
if (port !== null) registerRouterProvider(pi, port, cfg, sessionId);
|
|
96
104
|
return;
|
|
97
105
|
}
|
|
98
106
|
|
|
@@ -108,7 +116,7 @@ export default function (pi: ExtensionAPI): void {
|
|
|
108
116
|
|
|
109
117
|
// Publish the shared port; subagents and the toast read it from here.
|
|
110
118
|
writeEmbedPort(portFile, actualPort);
|
|
111
|
-
registerRouterProvider(pi, actualPort, cfg);
|
|
119
|
+
registerRouterProvider(pi, actualPort, cfg, sessionId);
|
|
112
120
|
|
|
113
121
|
pi.on("session_shutdown", () => {
|
|
114
122
|
void app?.stop().catch(() => {});
|
|
@@ -73,6 +73,11 @@ export default function (pi: ExtensionAPI): void {
|
|
|
73
73
|
// poll loop entirely rather than waking every 2s to do nothing.
|
|
74
74
|
if (!ctx.hasUI) return;
|
|
75
75
|
|
|
76
|
+
// This session's omp id. The embed extension tags every request with it
|
|
77
|
+
// (X-Omp-Session), so filtering on it scopes toasts to this session even
|
|
78
|
+
// when several omp sessions share one embedded router's ledger.
|
|
79
|
+
const sessionId = ctx.sessionManager.getSessionId();
|
|
80
|
+
|
|
76
81
|
// The poll request's own deadline (3s) exceeds the poll period (2s), so
|
|
77
82
|
// a slow router could let a second tick start while the first is still in
|
|
78
83
|
// flight — both read the same lastSeenId and raise duplicate toasts. An
|
|
@@ -115,7 +120,7 @@ export default function (pi: ExtensionAPI): void {
|
|
|
115
120
|
const entries = body.entries;
|
|
116
121
|
if (!Array.isArray(entries) || entries.length === 0) return;
|
|
117
122
|
|
|
118
|
-
for (const t of selectToasts(entries, lastSeenId, HARNESS_ID)) {
|
|
123
|
+
for (const t of selectToasts(entries, lastSeenId, HARNESS_ID, sessionId)) {
|
|
119
124
|
ctx.ui.notify(t.text, "info");
|
|
120
125
|
}
|
|
121
126
|
lastSeenId = newestId(entries) ?? lastSeenId;
|
|
@@ -83,6 +83,11 @@ export interface ToastDecision {
|
|
|
83
83
|
wasted: boolean;
|
|
84
84
|
/** Harness id from the request header; empty for the default harness. */
|
|
85
85
|
harnessId: string;
|
|
86
|
+
/**
|
|
87
|
+
* omp UI session id from the `X-Omp-Session` header; empty for the no-header
|
|
88
|
+
* default. Lets the toast scope to a single interactive session.
|
|
89
|
+
*/
|
|
90
|
+
ompSessionId?: string;
|
|
86
91
|
}
|
|
87
92
|
|
|
88
93
|
export interface ToastMessage {
|
|
@@ -111,6 +116,7 @@ export function selectToasts(
|
|
|
111
116
|
entries: ToastDecision[],
|
|
112
117
|
lastSeenId: string | null,
|
|
113
118
|
harnessId = "",
|
|
119
|
+
ompSessionId = "",
|
|
114
120
|
): ToastMessage[] {
|
|
115
121
|
if (lastSeenId === null) return [];
|
|
116
122
|
// `entries` is newest-first. Entries strictly newer than lastSeenId are the
|
|
@@ -124,6 +130,7 @@ export function selectToasts(
|
|
|
124
130
|
if (d === undefined) continue;
|
|
125
131
|
if (d.wasted) continue;
|
|
126
132
|
if (harnessId !== "" && d.harnessId !== harnessId) continue;
|
|
133
|
+
if (ompSessionId !== "" && d.ompSessionId !== ompSessionId) continue;
|
|
127
134
|
out.push({ model: d.servedSlug ?? d.slug, tier: d.tier, costUsd: d.reportedUsd, text: toToastText(d) });
|
|
128
135
|
}
|
|
129
136
|
return out;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "auto-model-router",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.2.0",
|
|
4
4
|
"private": false,
|
|
5
5
|
"description": "Local cost/complexity-aware model router for Oh My Pi, backed by OpenRouter",
|
|
6
6
|
"type": "module",
|
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
"typecheck": "tsc --noEmit",
|
|
12
12
|
"test": "bun test",
|
|
13
13
|
"smoke": "bun run tools/smoke.ts",
|
|
14
|
+
"version": "bun run tools/sync-marketplace-version.ts",
|
|
14
15
|
"release": "npm version $1 && git push --follow-tags"
|
|
15
16
|
},
|
|
16
17
|
"dependencies": {
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Read-only analysis of a ledger snapshot.
|
|
3
|
+
*
|
|
4
|
+
* Answers the questions that decide whether a learned classifier is worth
|
|
5
|
+
* building at all: does the adjudicator ever actually run, how are tiers
|
|
6
|
+
* distributed, and how much label signal (escalations) exists.
|
|
7
|
+
*
|
|
8
|
+
* Usage: bun run research/analyze-ledger.ts [path-to-snapshot.db]
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { Database } from "bun:sqlite";
|
|
12
|
+
|
|
13
|
+
const path = process.argv[2] ?? "research-data/snapshot.db";
|
|
14
|
+
const db = new Database(path, { readonly: true });
|
|
15
|
+
|
|
16
|
+
function section(label: string, sql: string): void {
|
|
17
|
+
console.log(`\n=== ${label} ===`);
|
|
18
|
+
try {
|
|
19
|
+
const rows = db.query(sql).all() as Record<string, unknown>[];
|
|
20
|
+
if (rows.length === 0) {
|
|
21
|
+
console.log(" (no rows)");
|
|
22
|
+
return;
|
|
23
|
+
}
|
|
24
|
+
for (const r of rows) {
|
|
25
|
+
console.log(" " + Object.entries(r).map(([k, v]) => `${k}=${v}`).join(" "));
|
|
26
|
+
}
|
|
27
|
+
} catch (err) {
|
|
28
|
+
console.log(" ERROR: " + (err instanceof Error ? err.message : String(err)));
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
section(
|
|
33
|
+
"tier distribution",
|
|
34
|
+
`SELECT tier, COUNT(*) n, ROUND(100.0 * COUNT(*) / (SELECT COUNT(*) FROM ledger), 1) pct
|
|
35
|
+
FROM ledger GROUP BY tier ORDER BY n DESC`,
|
|
36
|
+
);
|
|
37
|
+
|
|
38
|
+
section(
|
|
39
|
+
"classification source",
|
|
40
|
+
`SELECT classification_source src, COUNT(*) n,
|
|
41
|
+
ROUND(100.0 * COUNT(*) / (SELECT COUNT(*) FROM ledger), 1) pct
|
|
42
|
+
FROM ledger GROUP BY src ORDER BY n DESC`,
|
|
43
|
+
);
|
|
44
|
+
|
|
45
|
+
section(
|
|
46
|
+
"adjudicator activity (from reasons text)",
|
|
47
|
+
`SELECT
|
|
48
|
+
SUM(CASE WHEN reasons LIKE '%adjudicator skipped%' THEN 1 ELSE 0 END) AS skipped_cost_guard,
|
|
49
|
+
SUM(CASE WHEN reasons LIKE '%adjudicator failed%' THEN 1 ELSE 0 END) AS failed,
|
|
50
|
+
SUM(CASE WHEN reasons LIKE '%not a tier word%' THEN 1 ELSE 0 END) AS bad_reply,
|
|
51
|
+
SUM(CASE WHEN reasons LIKE '%adjudicator:%' THEN 1 ELSE 0 END) AS verdict_used,
|
|
52
|
+
COUNT(*) AS total_turns
|
|
53
|
+
FROM ledger`,
|
|
54
|
+
);
|
|
55
|
+
|
|
56
|
+
section(
|
|
57
|
+
"escalation signals",
|
|
58
|
+
`SELECT COALESCE(escalation_signal, '(none)') signal, COUNT(*) n
|
|
59
|
+
FROM ledger GROUP BY signal ORDER BY n DESC`,
|
|
60
|
+
);
|
|
61
|
+
|
|
62
|
+
section(
|
|
63
|
+
"spend and waste",
|
|
64
|
+
`SELECT SUM(wasted) wasted_turns,
|
|
65
|
+
ROUND(SUM(COALESCE(reported_usd, predicted_usd)), 4) total_usd,
|
|
66
|
+
ROUND(SUM(CASE WHEN wasted = 1 THEN COALESCE(reported_usd, predicted_usd) ELSE 0 END), 4) wasted_usd
|
|
67
|
+
FROM ledger`,
|
|
68
|
+
);
|
|
69
|
+
|
|
70
|
+
section(
|
|
71
|
+
"spend by served model",
|
|
72
|
+
`SELECT COALESCE(served_slug, slug) model, COUNT(*) n,
|
|
73
|
+
ROUND(SUM(COALESCE(reported_usd, predicted_usd)), 4) usd
|
|
74
|
+
FROM ledger GROUP BY model ORDER BY usd DESC LIMIT 10`,
|
|
75
|
+
);
|
|
76
|
+
|
|
77
|
+
section(
|
|
78
|
+
"error kinds",
|
|
79
|
+
`SELECT COALESCE(error_kind, '(none)') kind, COUNT(*) n
|
|
80
|
+
FROM ledger GROUP BY kind ORDER BY n DESC`,
|
|
81
|
+
);
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
// ---------------------------------------------------------------------------
|
|
86
|
+
// Exploration (schema v7): the counterfactual natural traffic cannot supply.
|
|
87
|
+
//
|
|
88
|
+
// A turn we deliberately routed one tier cheaper either escalated (the cheap
|
|
89
|
+
// model genuinely could not do it) or committed (the classifier was over-
|
|
90
|
+
// routing, and the cheaper tier would have served the turn fine).
|
|
91
|
+
// ---------------------------------------------------------------------------
|
|
92
|
+
|
|
93
|
+
section(
|
|
94
|
+
"exploration coverage",
|
|
95
|
+
`SELECT
|
|
96
|
+
SUM(CASE WHEN explored_from IS NOT NULL THEN 1 ELSE 0 END) AS explored,
|
|
97
|
+
COUNT(*) AS total,
|
|
98
|
+
ROUND(100.0 * SUM(CASE WHEN explored_from IS NOT NULL THEN 1 ELSE 0 END) / COUNT(*), 2) AS pct
|
|
99
|
+
FROM ledger`,
|
|
100
|
+
);
|
|
101
|
+
|
|
102
|
+
section(
|
|
103
|
+
"exploration verdict by dropped-from tier",
|
|
104
|
+
`SELECT
|
|
105
|
+
explored_from AS dropped_from,
|
|
106
|
+
COUNT(*) AS n,
|
|
107
|
+
SUM(CASE WHEN escalation_signal IS NOT NULL THEN 1 ELSE 0 END) AS escalated,
|
|
108
|
+
ROUND(100.0 * SUM(CASE WHEN escalation_signal IS NULL THEN 1 ELSE 0 END) / COUNT(*), 1) AS pct_cheap_sufficed
|
|
109
|
+
FROM ledger
|
|
110
|
+
WHERE explored_from IS NOT NULL
|
|
111
|
+
GROUP BY explored_from
|
|
112
|
+
ORDER BY n DESC`,
|
|
113
|
+
);
|
|
114
|
+
|
|
115
|
+
section(
|
|
116
|
+
"what exploration cost vs. what it revealed",
|
|
117
|
+
`SELECT
|
|
118
|
+
ROUND(SUM(CASE WHEN explored_from IS NOT NULL AND wasted = 1
|
|
119
|
+
THEN COALESCE(reported_usd, predicted_usd) ELSE 0 END), 4) AS wasted_on_exploration_usd,
|
|
120
|
+
ROUND(SUM(CASE WHEN explored_from IS NOT NULL THEN COALESCE(reported_usd, predicted_usd) ELSE 0 END), 4) AS explored_spend_usd
|
|
121
|
+
FROM ledger`,
|
|
122
|
+
);
|
|
123
|
+
|
|
124
|
+
// Confidence distribution: this is what determines whether the LLM adjudicator
|
|
125
|
+
// is ever reached at all. Turns below the configured ambiguityThreshold should
|
|
126
|
+
// be adjudicated; if that bucket is populated but no row has source 'llm', the
|
|
127
|
+
// adjudicator is failing silently rather than never being needed.
|
|
128
|
+
section(
|
|
129
|
+
"confidence distribution vs. the adjudication band",
|
|
130
|
+
`SELECT
|
|
131
|
+
CASE
|
|
132
|
+
WHEN confidence IS NULL THEN '(uninstrumented)'
|
|
133
|
+
WHEN confidence < 0.6 THEN 'below 0.6 (should adjudicate)'
|
|
134
|
+
WHEN confidence < 0.8 THEN '0.6 - 0.8'
|
|
135
|
+
ELSE '0.8 - 1.0'
|
|
136
|
+
END AS band,
|
|
137
|
+
COUNT(*) AS n,
|
|
138
|
+
SUM(CASE WHEN classification_source = 'llm' THEN 1 ELSE 0 END) AS adjudicated
|
|
139
|
+
FROM ledger
|
|
140
|
+
GROUP BY band
|
|
141
|
+
ORDER BY n DESC`,
|
|
142
|
+
);
|
|
143
|
+
|
|
144
|
+
// ---------------------------------------------------------------------------
|
|
145
|
+
// Hold-length experiment (schema v8).
|
|
146
|
+
//
|
|
147
|
+
// ~95% of hard-tier spend arrives by hysteresis hold rather than by
|
|
148
|
+
// classification, and the hold length is a hand-picked constant. Each
|
|
149
|
+
// conversation is assigned one arm, so arms compare on cost PER
|
|
150
|
+
// CONVERSATION -- comparing per-turn would reward short holds automatically,
|
|
151
|
+
// since a shorter hold simply produces fewer expensive turns.
|
|
152
|
+
//
|
|
153
|
+
// Escalations are the quality guardrail: a shorter hold that costs less but
|
|
154
|
+
// escalates more has moved the cost, not removed it.
|
|
155
|
+
// ---------------------------------------------------------------------------
|
|
156
|
+
|
|
157
|
+
section(
|
|
158
|
+
"hold-length arms",
|
|
159
|
+
`SELECT hold_arm AS arm,
|
|
160
|
+
COUNT(DISTINCT conversation_key) AS conversations,
|
|
161
|
+
COUNT(*) AS turns,
|
|
162
|
+
ROUND(SUM(COALESCE(reported_usd, predicted_usd)), 4) AS usd,
|
|
163
|
+
ROUND(SUM(COALESCE(reported_usd, predicted_usd))
|
|
164
|
+
/ NULLIF(COUNT(DISTINCT conversation_key), 0), 4) AS usd_per_conversation,
|
|
165
|
+
SUM(CASE WHEN escalation_signal IS NOT NULL THEN 1 ELSE 0 END) AS escalations,
|
|
166
|
+
ROUND(100.0 * SUM(CASE WHEN wasted = 1 THEN 1 ELSE 0 END) / COUNT(*), 1) AS pct_wasted
|
|
167
|
+
FROM ledger
|
|
168
|
+
WHERE hold_arm IS NOT NULL
|
|
169
|
+
GROUP BY hold_arm
|
|
170
|
+
ORDER BY hold_arm`,
|
|
171
|
+
);
|
|
172
|
+
|
|
173
|
+
db.close();
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Applies the cost-optimization config changes to the PRODUCTION router config
|
|
3
|
+
* (`~/.auto-model-router/config.yml`), merged + validated + backed up. Never
|
|
4
|
+
* prints the AA key.
|
|
5
|
+
*
|
|
6
|
+
* bun run research/apply-cost-tuning.ts # SAFE NOW: disable the dead
|
|
7
|
+
* # adjudicator only (routing-neutral).
|
|
8
|
+
* bun run research/apply-cost-tuning.ts --full # WINDOW-CLOSE: also raise
|
|
9
|
+
* # switchMargin and lower contextWindow
|
|
10
|
+
* # (these CHANGE live routing).
|
|
11
|
+
*
|
|
12
|
+
* See [[omp-router-cost-optimization]]. Restart all omp windows after applying.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import { chmodSync } from "node:fs";
|
|
16
|
+
import { homedir } from "node:os";
|
|
17
|
+
import { join } from "node:path";
|
|
18
|
+
|
|
19
|
+
// The fork's .env points AUTO_MODEL_ROUTER_HOME at research-data/home; these
|
|
20
|
+
// changes must land in the PRODUCTION config the live routers read.
|
|
21
|
+
process.env.AUTO_MODEL_ROUTER_HOME = join(homedir(), ".auto-model-router");
|
|
22
|
+
|
|
23
|
+
import { routerConfigPath, writeRouterConfig } from "../src/cli/config-cmd.ts";
|
|
24
|
+
import { loadConfig } from "../src/config/load.ts";
|
|
25
|
+
|
|
26
|
+
const full = process.argv.includes("--full");
|
|
27
|
+
const cfg = loadConfig({});
|
|
28
|
+
|
|
29
|
+
// (3) Disable the broken adjudicator: classifier.model is absent from the catalog,
|
|
30
|
+
// so it never returns a verdict — always keeps the heuristic. Setting the
|
|
31
|
+
// threshold to 0 skips the doomed call. Routing-identical, just faster.
|
|
32
|
+
const patch: Record<string, unknown> = { classifier: { ambiguityThreshold: 0 } };
|
|
33
|
+
|
|
34
|
+
if (full) {
|
|
35
|
+
// (2) Stop forfeiting warm cache for tiny completion savings.
|
|
36
|
+
patch.hysteresis = { switchMargin: 4 };
|
|
37
|
+
// (1) Force earlier compaction by advertising a smaller context window.
|
|
38
|
+
// Preserve every other profile field; only contextWindow changes.
|
|
39
|
+
patch.profiles = cfg.profiles.map((p) => ({
|
|
40
|
+
id: p.id,
|
|
41
|
+
name: p.name,
|
|
42
|
+
minTier: p.minTier,
|
|
43
|
+
maxTier: p.maxTier,
|
|
44
|
+
contextWindow: 200_000,
|
|
45
|
+
maxTokens: p.maxTokens,
|
|
46
|
+
}));
|
|
47
|
+
// (4) Cap the input-dominated hard cost. Now SAFE: the key admits ~13 hard
|
|
48
|
+
// candidates, so a $3/Mtok ceiling still leaves gpt-5.6-sol (77.4), grok-4.6,
|
|
49
|
+
// gpt-5.6-terra, glm-5.3, qwen3.8, kimi-k3 with failover; it drops only the
|
|
50
|
+
// $5-10 models (opus-5, gpt-5.5, fable). Raise to ~$5 if you want opus-5
|
|
51
|
+
// available for the very hardest turns. Merges into hard, preserving its floor.
|
|
52
|
+
patch.tiers = { hard: { maxInputPerMtok: 3 } };
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
const target = routerConfigPath();
|
|
56
|
+
const backup = writeRouterConfig(target, patch);
|
|
57
|
+
try {
|
|
58
|
+
chmodSync(target, 0o600); // keep the AA-key-bearing file locked down (advisory on Windows)
|
|
59
|
+
} catch {
|
|
60
|
+
/* best-effort */
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
const after = loadConfig({});
|
|
64
|
+
console.log(`wrote ${target}${backup === null ? "" : `\nbackup ${backup}`}`);
|
|
65
|
+
console.log(`mode ${full ? "FULL (routing-changing)" : "adjudicator-only (safe)"}`);
|
|
66
|
+
console.log(`classifier.ambiguity ${after.classifier.ambiguityThreshold}`);
|
|
67
|
+
console.log(`hysteresis.switchMargin ${after.hysteresis.switchMargin}`);
|
|
68
|
+
console.log(`profiles contextWindow ${after.profiles.map((p) => `${p.id}:${p.contextWindow}`).join(" ")}`);
|
|
69
|
+
// Confirm nothing unrelated was disturbed.
|
|
70
|
+
const k = after.benchmarks.artificialAnalysisApiKey;
|
|
71
|
+
console.log(`AA key intact ${k.trim() === "" ? "MISSING" : `yes (…${k.slice(-4)})`}`);
|
|
72
|
+
console.log(`exploration intact enabled=${after.exploration.enabled} holdArms=[${after.exploration.holdTurns.values.join(",")}] sticky=${after.exploration.stickyPolicy}`);
|
|
73
|
+
console.log("\nrestart all omp windows for this to take effect.");
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Where the money actually goes, from the live ledger. Read-only.
|
|
3
|
+
*
|
|
4
|
+
* Reframes optimization around the finding that ~96% of spend is the PROMPT,
|
|
5
|
+
* not the model's output: decomposes spend by cost component, measures the
|
|
6
|
+
* realized cache hit rate, and LOCALIZES cache misses to their cause (model
|
|
7
|
+
* switching, warm-window expiry, first-turns, exploration). Also quantifies
|
|
8
|
+
* abandoned-attempt waste and whether the LLM adjudicator ever fires.
|
|
9
|
+
*
|
|
10
|
+
* bun run research/cost-analysis.ts
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { homedir } from "node:os";
|
|
14
|
+
import { join } from "node:path";
|
|
15
|
+
|
|
16
|
+
// Force PRODUCTION home; the fork's .env points it at research-data/home.
|
|
17
|
+
process.env.AUTO_MODEL_ROUTER_HOME = join(homedir(), ".auto-model-router");
|
|
18
|
+
|
|
19
|
+
import { Database } from "bun:sqlite";
|
|
20
|
+
import { loadConfig } from "../src/config/load.ts";
|
|
21
|
+
|
|
22
|
+
const cfg = loadConfig({});
|
|
23
|
+
const db = new Database(cfg.ledger.path, { readonly: true });
|
|
24
|
+
|
|
25
|
+
function rows(sql: string): Record<string, unknown>[] {
|
|
26
|
+
return db.query(sql).all() as Record<string, unknown>[];
|
|
27
|
+
}
|
|
28
|
+
function table(title: string, data: Record<string, unknown>[]): void {
|
|
29
|
+
console.log(`\n=== ${title} ===`);
|
|
30
|
+
if (data.length === 0) {
|
|
31
|
+
console.log(" (no rows)");
|
|
32
|
+
return;
|
|
33
|
+
}
|
|
34
|
+
const cols = Object.keys(data[0]!);
|
|
35
|
+
const w = cols.map((c) => Math.max(c.length, ...data.map((r) => String(r[c] ?? "").length)));
|
|
36
|
+
console.log(" " + cols.map((c, i) => c.padEnd(w[i]!)).join(" "));
|
|
37
|
+
for (const r of data) console.log(" " + cols.map((c, i) => String(r[c] ?? "").padEnd(w[i]!)).join(" "));
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
const usd = "json_extract(cost_breakdown,'$.%s')";
|
|
41
|
+
const comp = (k: string) => usd.replace("%s", k);
|
|
42
|
+
|
|
43
|
+
// 1. Spend by cost component.
|
|
44
|
+
table("spend by component (USD)", rows(`
|
|
45
|
+
SELECT
|
|
46
|
+
round(sum(${comp("freshPrompt")}),2) AS fresh_in,
|
|
47
|
+
round(sum(${comp("cacheRead")}),2) AS cache_read,
|
|
48
|
+
round(sum(${comp("cacheWrite")}),2) AS cache_write,
|
|
49
|
+
round(sum(${comp("completion")}),2) AS completion,
|
|
50
|
+
round(sum(${comp("reasoning")}),2) AS reasoning,
|
|
51
|
+
round(sum(reported_usd),2) AS total
|
|
52
|
+
FROM ledger WHERE cost_breakdown IS NOT NULL`));
|
|
53
|
+
|
|
54
|
+
// 2. Token totals + input:output ratio + overall cache hit rate.
|
|
55
|
+
table("tokens", rows(`
|
|
56
|
+
SELECT
|
|
57
|
+
sum(json_extract(usage,'$.promptTokens')) AS prompt_toks,
|
|
58
|
+
sum(json_extract(usage,'$.completionTokens')) AS completion_toks,
|
|
59
|
+
round(sum(json_extract(usage,'$.promptTokens'))*1.0/max(1,sum(json_extract(usage,'$.completionTokens'))),0) AS in_out_ratio,
|
|
60
|
+
round(sum(json_extract(usage,'$.cachedTokens'))*1.0/max(1,sum(json_extract(usage,'$.promptTokens'))),3) AS cache_hit_rate
|
|
61
|
+
FROM ledger WHERE usage IS NOT NULL`));
|
|
62
|
+
|
|
63
|
+
// 3. Cache hit rate + spend by tier.
|
|
64
|
+
table("by tier", rows(`
|
|
65
|
+
SELECT tier,
|
|
66
|
+
count(*) AS n,
|
|
67
|
+
round(sum(reported_usd),2) AS usd,
|
|
68
|
+
round(avg(reported_usd),5) AS per_turn,
|
|
69
|
+
round(sum(json_extract(usage,'$.cachedTokens'))*1.0/max(1,sum(json_extract(usage,'$.promptTokens'))),3) AS cache_hit
|
|
70
|
+
FROM ledger WHERE usage IS NOT NULL GROUP BY tier ORDER BY usd DESC`));
|
|
71
|
+
|
|
72
|
+
// Per-turn view: first attempt only, with previous turn's model + timestamp.
|
|
73
|
+
const SEQ = `
|
|
74
|
+
WITH seq AS (
|
|
75
|
+
SELECT conversation_key, tier, served_slug, created_at_ms, reported_usd, usage, cost_breakdown, explored_from,
|
|
76
|
+
LAG(served_slug) OVER w AS prev_slug,
|
|
77
|
+
LAG(created_at_ms) OVER w AS prev_ms,
|
|
78
|
+
ROW_NUMBER() OVER w AS rn
|
|
79
|
+
FROM ledger
|
|
80
|
+
WHERE served_slug IS NOT NULL AND reported_usd IS NOT NULL AND attempt = 0
|
|
81
|
+
WINDOW w AS (PARTITION BY conversation_key ORDER BY turn, created_at_ms)
|
|
82
|
+
)`;
|
|
83
|
+
|
|
84
|
+
// 4. Cache misses localized to model switching (continuation turns only).
|
|
85
|
+
table("switch effect (continuation turns)", rows(`${SEQ}
|
|
86
|
+
SELECT
|
|
87
|
+
CASE WHEN served_slug = prev_slug THEN 'stayed' ELSE 'switched' END AS kind,
|
|
88
|
+
count(*) AS n,
|
|
89
|
+
round(sum(reported_usd),2) AS usd,
|
|
90
|
+
round(sum(json_extract(usage,'$.cachedTokens'))*1.0/max(1,sum(json_extract(usage,'$.promptTokens'))),3) AS cache_hit,
|
|
91
|
+
round(sum(${comp("cacheWrite")}),2) AS cache_write
|
|
92
|
+
FROM seq WHERE rn > 1 GROUP BY kind`));
|
|
93
|
+
|
|
94
|
+
// 5. Cache misses localized to warm-window expiry (continuation turns only).
|
|
95
|
+
const ttlMin = Math.round(cfg.hysteresis.cacheWarmTtlMs / 60000);
|
|
96
|
+
table(`gap effect (warm window = ${ttlMin} min)`, rows(`${SEQ}
|
|
97
|
+
SELECT
|
|
98
|
+
CASE WHEN (created_at_ms - prev_ms) < ${cfg.hysteresis.cacheWarmTtlMs} THEN 'within_window' ELSE 'expired' END AS gap,
|
|
99
|
+
count(*) AS n,
|
|
100
|
+
round(sum(json_extract(usage,'$.cachedTokens'))*1.0/max(1,sum(json_extract(usage,'$.promptTokens'))),3) AS cache_hit,
|
|
101
|
+
round(sum(${comp("cacheWrite")}),2) AS cache_write
|
|
102
|
+
FROM seq WHERE rn > 1 GROUP BY gap`));
|
|
103
|
+
|
|
104
|
+
// 6. First-turn vs continuation cache-write (first turns are unavoidably cold).
|
|
105
|
+
table("cache-write: first vs continuation", rows(`${SEQ}
|
|
106
|
+
SELECT
|
|
107
|
+
CASE WHEN rn = 1 THEN 'first_turn' ELSE 'continuation' END AS kind,
|
|
108
|
+
count(*) AS n,
|
|
109
|
+
round(sum(${comp("cacheWrite")}),2) AS cache_write,
|
|
110
|
+
round(sum(${comp("freshPrompt")}),2) AS fresh_in
|
|
111
|
+
FROM seq GROUP BY kind`));
|
|
112
|
+
|
|
113
|
+
// 7. Exploration's cache cost (stickyPolicy: always forfeits warm cache).
|
|
114
|
+
table("exploration effect", rows(`${SEQ}
|
|
115
|
+
SELECT
|
|
116
|
+
CASE WHEN explored_from IS NULL THEN 'normal' ELSE 'explored' END AS kind,
|
|
117
|
+
count(*) AS n,
|
|
118
|
+
round(sum(json_extract(usage,'$.cachedTokens'))*1.0/max(1,sum(json_extract(usage,'$.promptTokens'))),3) AS cache_hit,
|
|
119
|
+
round(sum(${comp("cacheWrite")}),2) AS cache_write
|
|
120
|
+
FROM seq WHERE rn > 1 GROUP BY kind`));
|
|
121
|
+
|
|
122
|
+
// 8. Abandoned-attempt waste.
|
|
123
|
+
table("waste (abandoned escalation attempts)", rows(`
|
|
124
|
+
SELECT
|
|
125
|
+
round(sum(CASE WHEN wasted = 1 THEN reported_usd ELSE 0 END),2) AS wasted_usd,
|
|
126
|
+
sum(CASE WHEN wasted = 1 THEN 1 ELSE 0 END) AS wasted_rows,
|
|
127
|
+
round(sum(reported_usd),2) AS total_usd
|
|
128
|
+
FROM ledger WHERE reported_usd IS NOT NULL`));
|
|
129
|
+
|
|
130
|
+
// 9. Does the LLM adjudicator ever fire? confidence vs the ambiguity band.
|
|
131
|
+
table(`adjudicator (ambiguityThreshold = ${cfg.classifier.ambiguityThreshold})`, rows(`
|
|
132
|
+
SELECT classification_source AS source, count(*) AS n,
|
|
133
|
+
round(avg(confidence),3) AS avg_conf,
|
|
134
|
+
sum(CASE WHEN confidence < ${cfg.classifier.ambiguityThreshold} THEN 1 ELSE 0 END) AS below_thresh
|
|
135
|
+
FROM ledger WHERE confidence IS NOT NULL GROUP BY classification_source`));
|
|
136
|
+
|
|
137
|
+
// 10. Spend by prompt-size bucket.
|
|
138
|
+
table("spend by prompt size", rows(`
|
|
139
|
+
SELECT
|
|
140
|
+
CASE
|
|
141
|
+
WHEN json_extract(usage,'$.promptTokens') < 50000 THEN 'a <50k'
|
|
142
|
+
WHEN json_extract(usage,'$.promptTokens') < 150000 THEN 'b 50-150k'
|
|
143
|
+
WHEN json_extract(usage,'$.promptTokens') < 300000 THEN 'c 150-300k'
|
|
144
|
+
ELSE 'd 300k+' END AS bucket,
|
|
145
|
+
count(*) AS n,
|
|
146
|
+
round(sum(reported_usd),2) AS usd,
|
|
147
|
+
round(avg(reported_usd),5) AS per_turn
|
|
148
|
+
FROM ledger WHERE usage IS NOT NULL GROUP BY bucket ORDER BY bucket`));
|
|
149
|
+
|
|
150
|
+
db.close();
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Live proof: fetch the external benchmark feeds for real, apply them to a COPY
|
|
3
|
+
* of the live catalog, and print which previously-unscored models gained scores.
|
|
4
|
+
* Read-only w.r.t. the production DB — never writes benchmark_cache.
|
|
5
|
+
*
|
|
6
|
+
* bun run research/feed-check.ts [db]
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { homedir } from "node:os";
|
|
10
|
+
import { join } from "node:path";
|
|
11
|
+
|
|
12
|
+
import { Database } from "bun:sqlite";
|
|
13
|
+
|
|
14
|
+
import { applyFeedScores, fetchAaScores, fetchBenchlmScores, type FeedScore } from "../src/catalog/benchmark-feeds.ts";
|
|
15
|
+
import { normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
|
|
16
|
+
import { loadConfig } from "../src/config/load.ts";
|
|
17
|
+
|
|
18
|
+
const dbPath = process.argv[2] ?? join(homedir(), ".auto-model-router", "router.db");
|
|
19
|
+
const cfg = loadConfig({});
|
|
20
|
+
|
|
21
|
+
const db = new Database(dbPath, { readonly: true });
|
|
22
|
+
const row = db.query("SELECT payload FROM catalog_cache WHERE id = 1").get() as { payload: string } | null;
|
|
23
|
+
db.close();
|
|
24
|
+
if (row === null) {
|
|
25
|
+
console.log("no cached catalog");
|
|
26
|
+
process.exit(1);
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
const parsed: unknown = JSON.parse(row.payload);
|
|
30
|
+
const rawModels: unknown[] = Array.isArray(parsed) ? parsed : [];
|
|
31
|
+
|
|
32
|
+
// Quality per slug BEFORE the fill.
|
|
33
|
+
function qualityBySlug(models: unknown[]): Map<string, object> {
|
|
34
|
+
const out = new Map<string, object>();
|
|
35
|
+
for (const raw of models) {
|
|
36
|
+
const m = normalizeCatalogModel(raw);
|
|
37
|
+
if (m !== null) out.set(m.slug, m.quality);
|
|
38
|
+
}
|
|
39
|
+
return out;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
const before = qualityBySlug(structuredClone(rawModels));
|
|
43
|
+
|
|
44
|
+
const key = cfg.benchmarks.artificialAnalysisApiKey;
|
|
45
|
+
console.log(`AA key: ${key.trim() === "" ? "(none — AA feed skipped)" : "present"}`);
|
|
46
|
+
const [aa, bl] = await Promise.all([fetchAaScores(key, { timeoutMs: 30_000 }), fetchBenchlmScores({ timeoutMs: 30_000 })]);
|
|
47
|
+
const feeds: FeedScore[] = [...aa, ...bl];
|
|
48
|
+
console.log(`feeds fetched: artificial_analysis=${aa.length} benchlm=${bl.length}\n`);
|
|
49
|
+
|
|
50
|
+
if (feeds.length === 0) {
|
|
51
|
+
console.log("no feed data (endpoints unreachable or empty); nothing to apply.");
|
|
52
|
+
process.exit(0);
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
const result = applyFeedScores(rawModels, feeds);
|
|
56
|
+
const after = qualityBySlug(rawModels);
|
|
57
|
+
|
|
58
|
+
console.log(`filled ${result.modelsFilled} model(s): coding=${result.axes.coding} intelligence=${result.axes.intelligence} agentic=${result.axes.agentic} (aa=${result.sources.artificial_analysis} benchlm=${result.sources.benchlm})\n`);
|
|
59
|
+
|
|
60
|
+
for (const [slug, q] of after) {
|
|
61
|
+
const prev = JSON.stringify(before.get(slug) ?? {});
|
|
62
|
+
const now = JSON.stringify(q);
|
|
63
|
+
if (prev !== now) console.log(`${slug}\n before ${prev}\n after ${now}`);
|
|
64
|
+
}
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Which models to ENABLE on the OpenRouter key. The key-scoped catalog is narrow
|
|
3
|
+
* (guardrails/preferences), so tiers — `hard` especially — resolve to one
|
|
4
|
+
* candidate and can't take a price ceiling or same-tier failover. This fetches
|
|
5
|
+
* the FULL public catalog, runs it through the router's real tier filters, and
|
|
6
|
+
* reports the models that WOULD be eligible per tier but are not currently
|
|
7
|
+
* admitted by the key. Read-only. Needs network.
|
|
8
|
+
*
|
|
9
|
+
* bun run research/model-recommendations.ts
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { homedir } from "node:os";
|
|
13
|
+
import { join } from "node:path";
|
|
14
|
+
|
|
15
|
+
process.env.AUTO_MODEL_ROUTER_HOME = join(homedir(), ".auto-model-router");
|
|
16
|
+
|
|
17
|
+
import { normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
|
|
18
|
+
import type { CatalogModel, CatalogSnapshot } from "../src/catalog/types.ts";
|
|
19
|
+
import { loadConfig } from "../src/config/load.ts";
|
|
20
|
+
import { buildCandidates } from "../src/router/candidates.ts";
|
|
21
|
+
import { extractFeatures } from "../src/router/features.ts";
|
|
22
|
+
import { effectiveQualityFloor, tierPlanFor } from "../src/router/tier-plan.ts";
|
|
23
|
+
import type { Tier } from "../src/router/types.ts";
|
|
24
|
+
import { createOpenRouterClient } from "../src/upstream/openrouter.ts";
|
|
25
|
+
import { parseChatRequest } from "../src/wire/openai/request.ts";
|
|
26
|
+
|
|
27
|
+
const cfg = loadConfig({});
|
|
28
|
+
|
|
29
|
+
const client = createOpenRouterClient(cfg);
|
|
30
|
+
|
|
31
|
+
// Current key-scoped availability, fetched LIVE (/models/user) so it reflects
|
|
32
|
+
// account changes immediately — the cached snapshot lags until a router refresh.
|
|
33
|
+
const userRaw = await client.fetchModelsForUser(AbortSignal.timeout(30_000));
|
|
34
|
+
const keyScoped = new Set<string>(
|
|
35
|
+
userRaw.map(normalizeCatalogModel).filter((m): m is CatalogModel => m !== null).map((m) => m.slug),
|
|
36
|
+
);
|
|
37
|
+
|
|
38
|
+
// Full public catalog (carries AA benchmarks).
|
|
39
|
+
const raw = await client.fetchModels(AbortSignal.timeout(30_000));
|
|
40
|
+
const models = raw.map(normalizeCatalogModel).filter((m): m is CatalogModel => m !== null);
|
|
41
|
+
const snapshot: CatalogSnapshot = { models, fetchedAtMs: Date.now() };
|
|
42
|
+
console.log(`public catalog: ${models.length} models; key-scoped now (live): ${keyScoped.size}\n`);
|
|
43
|
+
|
|
44
|
+
// A representative hard coding turn, so task=coding (axis=coding).
|
|
45
|
+
const req = parseChatRequest(
|
|
46
|
+
{ model: "auto", messages: [{ role: "user", content: "Refactor the scheduler to remove the global lock and prove it stays correct under concurrency." }] },
|
|
47
|
+
new Headers(),
|
|
48
|
+
);
|
|
49
|
+
const features = extractFeatures(req, 8000);
|
|
50
|
+
|
|
51
|
+
// Cover both axes a coding agent actually routes on: coding (task=coding) and
|
|
52
|
+
// intelligence (task=chat). Model choice for `hard` should be strong on both.
|
|
53
|
+
const combos: Array<[Tier, "coding" | "chat"]> = [
|
|
54
|
+
["hard", "coding"],
|
|
55
|
+
["hard", "chat"],
|
|
56
|
+
["moderate", "coding"],
|
|
57
|
+
];
|
|
58
|
+
for (const [tier, task] of combos) {
|
|
59
|
+
const axis = cfg.tasks[task].axis;
|
|
60
|
+
const tc = cfg.tiers[tier];
|
|
61
|
+
// Floor if the FULL catalog were available (not the collapsed key-scoped floor).
|
|
62
|
+
const floor = cfg.adaptiveTierFloors
|
|
63
|
+
? effectiveQualityFloor(tc.minQuality, tier, axis, tierPlanFor(snapshot, cfg))
|
|
64
|
+
: tc.minQuality;
|
|
65
|
+
const { candidates } = buildCandidates({
|
|
66
|
+
req,
|
|
67
|
+
features,
|
|
68
|
+
tier,
|
|
69
|
+
task,
|
|
70
|
+
snapshot,
|
|
71
|
+
ledger: null,
|
|
72
|
+
cfg,
|
|
73
|
+
expectedCompletionTokens: 8000,
|
|
74
|
+
warmSlug: null,
|
|
75
|
+
});
|
|
76
|
+
const ranked = [...candidates].sort((a, b) => b.qualityScore - a.qualityScore);
|
|
77
|
+
const haveNow = ranked.filter((c) => keyScoped.has(c.model.slug)).length;
|
|
78
|
+
console.log(`\n===== ${tier} (axis ${axis}, floor ${floor.toFixed(1)}) — ${ranked.length} eligible in full catalog, ${haveNow} already on your key =====`);
|
|
79
|
+
console.log(" add? slug".padEnd(46) + "q in$/Mtok ctx");
|
|
80
|
+
for (const c of ranked.slice(0, 14)) {
|
|
81
|
+
const tag = keyScoped.has(c.model.slug) ? "have " : "ADD ";
|
|
82
|
+
console.log(
|
|
83
|
+
` ${tag} ${c.model.slug.padEnd(38)} ${String(c.qualityScore).padEnd(5)} ${(c.model.price.prompt * 1e6).toFixed(2).padStart(7)} ${c.model.contextLength}`,
|
|
84
|
+
);
|
|
85
|
+
}
|
|
86
|
+
}
|