ruvnet-brain 4.5.4 → 4.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/bin/install.mjs +162 -27
- package/config/model-router/catalog.template.json +126 -52
- package/config/model-router/claude-terminal-mod/.claude-plugin/plugin.json +1 -0
- package/config/model-router/claude-terminal-mod/README.md +22 -0
- package/config/model-router/claude-terminal-mod/hooks/hooks.json +1 -0
- package/config/model-router/claude-terminal-mod/hooks/policy.default.mjs +2 -0
- package/config/model-router/claude-terminal-mod/hooks/register.js +66 -0
- package/config/model-router/claude-terminal-mod/hooks/routing.js +55 -0
- package/config/model-router/claude-terminal-mod/hooks/runtime.js +2 -0
- package/config/model-router/claude-terminal-mod/tests/native.test.ts +86 -0
- package/config/model-router/policy.default.mjs +97 -75
- package/config/model-router/qualification-contract.json +124 -0
- package/config/model-router/routing-eval-cases.json +275 -0
- package/config/model-router/routing-policy.template.json +76 -0
- package/config/model-router/weekly-analyst-instruction.md +60 -0
- package/data/model-catalog.json +44 -49
- package/package.json +6 -3
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/mcp/managed-cli-interface.mjs +9 -4
- package/plugin/scripts/capability-claim-evidence.mjs +9 -2
- package/plugin/scripts/codex-hook-adapter.mjs +18 -9
- package/plugin/scripts/project-capture-queue.mjs +7 -1
- package/plugin/scripts/project-progression-hook.mjs +16 -7
- package/plugin/scripts/project-progression-outbox.mjs +96 -15
- package/plugin/scripts/project-progression-producer.mjs +12 -1
- package/plugin/scripts/project-progression-store.mjs +53 -1
- package/plugin/scripts/project-transition-hook.mjs +19 -8
- package/plugin/scripts/session-snapshot-hook.mjs +2 -1
- package/scripts/claude-terminal-mod.mjs +89 -0
- package/scripts/codex-hook-trust-reconcile.mjs +247 -0
- package/scripts/codex-routed.sh +3 -36
- package/scripts/goldie-weekly.sh +8 -64
- package/scripts/metaharness-router.mjs +7 -1
- package/scripts/model-analyst-sandbox.mjs +54 -0
- package/scripts/model-currency-evidence.mjs +139 -0
- package/scripts/model-currency.mjs +230 -0
- package/scripts/model-native-catalog.mjs +111 -0
- package/scripts/model-native-qualification.mjs +251 -0
- package/scripts/model-router-agent-hook.mjs +136 -0
- package/scripts/model-router-dispatch.mjs +161 -0
- package/scripts/model-router-engine.mjs +155 -104
- package/scripts/model-routing-eval.mjs +108 -0
- package/scripts/model-routing-gateway.mjs +453 -0
- package/scripts/model-routing-launchers.mjs +209 -0
- package/scripts/model-routing-policy-promotion.mjs +203 -0
- package/scripts/model-terminal-gateway.mjs +310 -0
- package/scripts/model-terminal-launchers.mjs +300 -0
- package/scripts/model-weekly-analyst.mjs +299 -0
- package/scripts/model-weekly-assessment.mjs +91 -0
- package/scripts/model-weekly-cycle.mjs +183 -0
- package/scripts/model-weekly-qualification.mjs +362 -0
- package/scripts/native-subscription-usage.mjs +57 -0
- package/scripts/release-qualification-contract.mjs +50 -1
- package/scripts/release-qualification.mjs +9 -6
- package/scripts/security-guidance-codex-compat.mjs +142 -0
- package/scripts/user-model-prompt-hook.mjs +69 -0
|
@@ -1,49 +1,28 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
7
|
-
//
|
|
8
|
-
//
|
|
9
|
-
// here. (ADR-040 / DRACO, verified via search_ruvnet: a hand-built self-signal threshold routed
|
|
10
|
-
// WORSE than always-cheapest; a learned map from a real feature beat the best fixed model. So
|
|
11
|
-
// the SIGNAL/policy is everything and must never be hard-coded into the engine.)
|
|
12
|
-
// • IS harness-neutral: the SAME CLI is consulted by Claude Code AND Codex. It only DECIDES; it
|
|
13
|
-
// does not launch a model. The caller acts on the JSON. (Codex has no native routing surface —
|
|
14
|
-
// ~/.codex/config.toml launches one model per run — so a consulted CLI is the only way to make
|
|
15
|
-
// selection work for Codex too. That is the fix for "only partially OK for Codex.")
|
|
16
|
-
// • Is NOT an executor. Running a task on a cheap model is route-cheap.mjs's job (OpenRouter).
|
|
17
|
-
// This answers only "which model should handle this prompt?"
|
|
18
|
-
//
|
|
19
|
-
// INTEGRATION:
|
|
20
|
-
// Claude Code : call from a hook/skill, parse the JSON, use .model.
|
|
21
|
-
// node model-router-engine.mjs --harness claude-code --prompt "$PROMPT" --json
|
|
22
|
-
// Codex : wrap the codex launch —
|
|
23
|
-
// M=$(node model-router-engine.mjs --harness codex --prompt "$TASK" --json | jq -r .model)
|
|
24
|
-
// codex --model "$M" ...
|
|
25
|
-
//
|
|
26
|
-
// Config (edit freely): ~/.claude/model-router/catalog.json (candidates + verified pricing)
|
|
27
|
-
// ~/.claude/model-router/policy.mjs (YOUR policy; falls back to policy.default.mjs)
|
|
28
|
-
// Decision log: ~/.claude/metaharness/routing-decisions.jsonl (sibling to route-cheap's execution receipts)
|
|
29
|
-
//
|
|
30
|
-
// Usage:
|
|
31
|
-
// node model-router-engine.mjs --prompt "..." [--harness claude-code|codex] [--policy <path>] [--json|--line]
|
|
32
|
-
// echo "the prompt text" | node model-router-engine.mjs --harness codex
|
|
2
|
+
// Deterministic prompt classification + reviewed per-user native model/effort allocation.
|
|
3
|
+
// This selects only; model-router-dispatch.mjs enforces a managed worker launch. Parent chat
|
|
4
|
+
// model selection is controlled by the host, not a UserPromptSubmit recommendation hook.
|
|
5
|
+
// ~/.claude/model-router: catalog.json, profile.json, routing-policy.json, optional policy.mjs.
|
|
6
|
+
// Learned routes are constrained to the reviewed policy pick, never given first refusal.
|
|
7
|
+
// Usage: node model-router-engine.mjs --harness codex --policy-only --json < prompt.txt
|
|
8
|
+
// Decision receipts retain model/effort/class metadata only; no raw prompt or policy reason.
|
|
33
9
|
|
|
34
10
|
import fs from 'node:fs';
|
|
35
11
|
import path from 'node:path';
|
|
36
12
|
import os from 'node:os';
|
|
13
|
+
import crypto from 'node:crypto';
|
|
37
14
|
import { pathToFileURL, fileURLToPath } from 'node:url';
|
|
38
15
|
import { estTokens } from './route-cheap.mjs'; // reuse the verified char/4 estimator (DRY)
|
|
39
16
|
|
|
40
17
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
41
|
-
export const CONFIG_DIR = path.join(os.homedir(), '.claude', 'model-router');
|
|
18
|
+
export const CONFIG_DIR = process.env.MODEL_ROUTER_CONFIG_DIR || path.join(os.homedir(), '.claude', 'model-router');
|
|
42
19
|
// Overridable for hermetic tests + CI (runners have no ~/.claude): the 2026-07-12 CI redness was
|
|
43
20
|
// exactly this — tests that silently depended on one developer's machine state.
|
|
44
21
|
const CATALOG_PATH = process.env.MODEL_ROUTER_CATALOG || path.join(CONFIG_DIR, 'catalog.json');
|
|
45
22
|
const POLICY_USER = path.join(CONFIG_DIR, 'policy.mjs');
|
|
46
23
|
const POLICY_DEFAULT = path.join(CONFIG_DIR, 'policy.default.mjs');
|
|
24
|
+
const POLICY_SHIPPED = path.join(__dirname, '..', 'config', 'model-router', 'policy.default.mjs');
|
|
25
|
+
export const TASK_CLASSES = ['fast', 'medium', 'substantial', 'hard', 'exceptional'];
|
|
47
26
|
const DECISIONS_LOG =
|
|
48
27
|
process.env.MODEL_ROUTER_DECISIONS ||
|
|
49
28
|
path.join(os.homedir(), '.claude', 'metaharness', 'routing-decisions.jsonl');
|
|
@@ -51,7 +30,7 @@ const DECISIONS_LOG =
|
|
|
51
30
|
// ─── feature extraction: this is "based on what the prompt is" ────────────────────────────────
|
|
52
31
|
// Pure and deterministic. Emits SIGNALS only — it never decides. Policies consume these; extend
|
|
53
32
|
// this object as your research identifies new predictive features (it is the documented surface).
|
|
54
|
-
export function extractFeatures(prompt, harness) {
|
|
33
|
+
export function extractFeatures(prompt, harness, taskFacts) {
|
|
55
34
|
const text = prompt || '';
|
|
56
35
|
const codeFences = Math.floor((text.match(/```/g) || []).length / 2);
|
|
57
36
|
const fileTypes = [...new Set((text.match(/\.[a-z0-9]{1,5}\b/gi) || []).map((s) => s.toLowerCase()))].slice(0, 12);
|
|
@@ -64,8 +43,9 @@ export function extractFeatures(prompt, harness) {
|
|
|
64
43
|
hasCode,
|
|
65
44
|
fileTypes,
|
|
66
45
|
questionCount: (text.match(/\?/g) || []).length,
|
|
67
|
-
taskHints: text
|
|
46
|
+
taskHints: text, // policies may regex over the actual prompt head
|
|
68
47
|
harness,
|
|
48
|
+
taskFacts,
|
|
69
49
|
};
|
|
70
50
|
}
|
|
71
51
|
|
|
@@ -94,44 +74,31 @@ export function applyProfile(candidates, profile) {
|
|
|
94
74
|
}));
|
|
95
75
|
}
|
|
96
76
|
|
|
97
|
-
//
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
export function catalogSource() {
|
|
102
|
-
try {
|
|
103
|
-
const j = JSON.parse(fs.readFileSync(CATALOG_PATH, 'utf8'));
|
|
104
|
-
if (Array.isArray(j.candidates) && j.candidates.length) return 'catalog';
|
|
105
|
-
} catch { /* fall through */ }
|
|
106
|
-
return 'built-in-fallback';
|
|
77
|
+
// Catalog absence is not permission to use stale built-in identities.
|
|
78
|
+
export function catalogSource(file = CATALOG_PATH) {
|
|
79
|
+
try { loadCatalog(file); return 'catalog'; }
|
|
80
|
+
catch { return 'unavailable'; }
|
|
107
81
|
}
|
|
108
82
|
|
|
109
|
-
export function loadCatalog() {
|
|
83
|
+
export function loadCatalog(file = CATALOG_PATH) {
|
|
110
84
|
try {
|
|
111
|
-
const
|
|
112
|
-
if (Array.isArray(
|
|
113
|
-
} catch {
|
|
114
|
-
|
|
115
|
-
}
|
|
116
|
-
// Built-in fallback. Claude launchability was verified against Claude Code 2.1.220 on 2026-08-02;
|
|
117
|
-
// prices remain null where the subscription host, rather than a metered API, is authoritative.
|
|
118
|
-
return [
|
|
119
|
-
{ id: 'deepseek/deepseek-chat', provider: 'openrouter', harness: ['claude-code', 'codex'], tier: 'cheap', costPerMTok: { in: 0.2, out: 0.8 }, verified: '2026-07-07' },
|
|
120
|
-
{ id: 'claude-opus-4-8', provider: 'anthropic', harness: ['claude-code'], tier: 'frontier', costPerMTok: { in: 5.0, out: 25.0 }, verified: '2026-07-07' },
|
|
121
|
-
{ id: 'claude-opus-5', provider: 'anthropic', harness: ['claude-code'], subscription: ['claude-code'], tier: 'frontier', costPerMTok: null, verified: '2026-08-02 Claude Code 2.1.220 launch' },
|
|
122
|
-
{ id: 'claude-fable-5', provider: 'anthropic', harness: ['claude-code'], subscription: ['claude-code'], tier: 'frontier', costPerMTok: null, verified: '2026-08-02 Claude Code 2.1.220 launch' },
|
|
123
|
-
{ id: 'gpt-5.5', provider: 'openai', harness: ['codex'], tier: 'frontier', costPerMTok: null, verified: null },
|
|
124
|
-
];
|
|
85
|
+
const catalog = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
86
|
+
if (Array.isArray(catalog.candidates) && catalog.candidates.length) return catalog.candidates;
|
|
87
|
+
} catch { /* report one bounded configuration error, never substitute old models */ }
|
|
88
|
+
throw new Error('Current per-user model catalog missing or invalid; no built-in model fallback');
|
|
125
89
|
}
|
|
126
90
|
|
|
127
91
|
export async function loadPolicy(explicit) {
|
|
128
|
-
|
|
92
|
+
if (explicit && !fs.existsSync(explicit)) throw new Error(`Explicit routing policy missing: ${explicit}`);
|
|
93
|
+
const candidatePaths = [explicit, POLICY_USER, POLICY_DEFAULT, POLICY_SHIPPED].filter(Boolean);
|
|
129
94
|
for (const p of candidatePaths) {
|
|
130
95
|
if (!fs.existsSync(p)) continue;
|
|
131
96
|
try {
|
|
132
97
|
const mod = await import(pathToFileURL(p).href);
|
|
133
98
|
if (typeof mod.choose === 'function') return { choose: mod.choose, source: p };
|
|
99
|
+
throw new Error('Policy must export choose()');
|
|
134
100
|
} catch (e) {
|
|
101
|
+
if (p === explicit || p === POLICY_USER) throw new Error(`User routing policy failed to load: ${e.message}`);
|
|
135
102
|
process.stderr.write(`[model-router] policy at ${p} failed to load: ${e.message}\n`);
|
|
136
103
|
}
|
|
137
104
|
}
|
|
@@ -139,12 +106,14 @@ export async function loadPolicy(explicit) {
|
|
|
139
106
|
}
|
|
140
107
|
|
|
141
108
|
function parseArgs(argv) {
|
|
142
|
-
const a = { harness: null, prompt: null, policy: null, mode: 'json' };
|
|
109
|
+
const a = { harness: null, prompt: null, policy: null, mode: 'json', policyOnly: false, requestJson: false };
|
|
143
110
|
for (let i = 0; i < argv.length; i++) {
|
|
144
111
|
const k = argv[i];
|
|
145
112
|
if (k === '--prompt') a.prompt = argv[++i];
|
|
146
113
|
else if (k === '--harness') a.harness = argv[++i];
|
|
147
114
|
else if (k === '--policy') a.policy = argv[++i];
|
|
115
|
+
else if (k === '--request-json') a.requestJson = true;
|
|
116
|
+
else if (k === '--policy-only') a.policyOnly = true;
|
|
148
117
|
else if (k === '--line') a.mode = 'line';
|
|
149
118
|
else if (k === '--json') a.mode = 'json';
|
|
150
119
|
else if (k === '--help' || k === '-h') a.help = true;
|
|
@@ -164,6 +133,111 @@ function estInputCost(candidate, inTokens) {
|
|
|
164
133
|
return +((inTokens * p.in) / 1e6).toFixed(6);
|
|
165
134
|
}
|
|
166
135
|
|
|
136
|
+
export function loadSelection(file = process.env.MODEL_ROUTER_SELECTION || path.join(CONFIG_DIR, 'routing-policy.json')) {
|
|
137
|
+
try { return JSON.parse(fs.readFileSync(file, 'utf8')); }
|
|
138
|
+
catch { throw new Error('No reviewed per-user routing-policy.json available'); }
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
export function assertCurrentSelection(selection, now = Date.now()) {
|
|
142
|
+
const age = now - Date.parse(selection?.reviewedAt);
|
|
143
|
+
const configuredMaxAge = selection?.maxAgeMs === undefined ? 604800000 : selection.maxAgeMs;
|
|
144
|
+
if (!Number.isSafeInteger(configuredMaxAge) || configuredMaxAge <= 0) {
|
|
145
|
+
throw new Error('Routing allocation maxAgeMs must be a finite positive integer');
|
|
146
|
+
}
|
|
147
|
+
const reviewedAt = selection?.reviewedAt;
|
|
148
|
+
const isoDate = typeof reviewedAt === 'string' && /^\d{4}-\d{2}-\d{2}(?:T\d{2}:\d{2}:\d{2}(?:\.\d{1,9})?(?:Z|[+-]\d{2}:\d{2}))?$/.test(reviewedAt);
|
|
149
|
+
const calendarDate = isoDate && Date.parse(reviewedAt.slice(0, 10));
|
|
150
|
+
const validCalendar = Number.isFinite(calendarDate) && new Date(calendarDate).toISOString().slice(0, 10) === reviewedAt.slice(0, 10);
|
|
151
|
+
if (selection?.schemaVersion !== 1 || !isoDate || !validCalendar || !Number.isFinite(age) || age < 0) {
|
|
152
|
+
throw new Error('Routing allocation missing, invalid or future-dated; owner-reviewed policy required');
|
|
153
|
+
}
|
|
154
|
+
// Evidence age is not revocation of an approved allocation. Retain the original reviewedAt;
|
|
155
|
+
// every managed launch still rechecks allocation integrity, native support, auth and allowance.
|
|
156
|
+
return selection;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
function normalizedRoutes(value) {
|
|
160
|
+
if (Array.isArray(value)) return value.map(normalizedRoutes);
|
|
161
|
+
if (value && typeof value === 'object') return Object.fromEntries(Object.keys(value).sort()
|
|
162
|
+
.map((key) => [key, normalizedRoutes(value[key])]));
|
|
163
|
+
return value;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
export function selectionEvidenceStatus(selection, now = Date.now()) {
|
|
167
|
+
assertCurrentSelection(selection, now);
|
|
168
|
+
const maxAgeMs = Math.min(selection.maxAgeMs ?? 604800000, 604800000);
|
|
169
|
+
const ageMs = now - Date.parse(selection.reviewedAt);
|
|
170
|
+
const routeDigest = selection.routes && typeof selection.routes === 'object' && !Array.isArray(selection.routes)
|
|
171
|
+
? crypto.createHash('sha256').update(JSON.stringify(normalizedRoutes(selection.routes))).digest('hex') : null;
|
|
172
|
+
return { reviewedAt: selection.reviewedAt, maxAgeMs, ageMs, stale: ageMs > maxAgeMs, routeDigest };
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
// Eligibility is independent of policy and learning: catalog pricing is never spend permission.
|
|
176
|
+
export function eligibleCandidates(candidates, profile, harness) {
|
|
177
|
+
const host = profile?.harnesses?.[harness];
|
|
178
|
+
if (host?.available !== true || host?.subscription !== true) return [];
|
|
179
|
+
const provider = { codex: 'openai', 'claude-code': 'anthropic' }[harness];
|
|
180
|
+
return candidates.filter((m) => m.provider === provider &&
|
|
181
|
+
(m.harness || []).includes(harness) && (m.subscription || []).includes(harness));
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
export async function selectDecision({ prompt, harness, candidates, profile, policy,
|
|
185
|
+
features = extractFeatures(prompt, harness), learnedRoute, selection = loadSelection(), now = Date.now() } = {}) {
|
|
186
|
+
const evidence = selectionEvidenceStatus(selection, now);
|
|
187
|
+
const pool = eligibleCandidates(candidates, profile, harness);
|
|
188
|
+
if (!pool.length) throw new Error(`No available native subscription candidates for ${harness}; no metered fallback`);
|
|
189
|
+
if (!policy?.choose) throw new Error('No routing policy available');
|
|
190
|
+
const classifierFile = fs.existsSync(POLICY_SHIPPED) ? POLICY_SHIPPED : POLICY_DEFAULT;
|
|
191
|
+
const classifier = await import(pathToFileURL(classifierFile).href);
|
|
192
|
+
if (typeof classifier.classify !== 'function' || typeof classifier.validateTaskFacts !== 'function') {
|
|
193
|
+
throw new Error('Managed routing classifier missing required exports; update installed policy.default.mjs before dispatch');
|
|
194
|
+
}
|
|
195
|
+
classifier.validateTaskFacts(features.taskFacts);
|
|
196
|
+
const assessedClass = classifier.classify(features, harness);
|
|
197
|
+
const decision = await policy.choose({ features, candidates: pool, harness, profile, selection });
|
|
198
|
+
if (harness === 'codex' && ['substantial', 'exceptional', 'hard'].includes(assessedClass) && decision?.taskClass !== assessedClass) {
|
|
199
|
+
throw new Error(`Task requires explicit qualified ${assessedClass} route; legacy policy cannot silently use medium`);
|
|
200
|
+
}
|
|
201
|
+
const chosen = pool.find((m) => m.id === decision?.model);
|
|
202
|
+
if (!chosen) throw new Error(`Policy model unavailable or unauthorized: ${decision?.model || 'none'}`);
|
|
203
|
+
const taskClass = decision.taskClass;
|
|
204
|
+
if (!TASK_CLASSES.includes(taskClass)) {
|
|
205
|
+
throw new Error('Routing policy must return an explicit qualified taskClass; update legacy policy');
|
|
206
|
+
}
|
|
207
|
+
const effort = decision.effort || selection.routes?.[harness]?.[taskClass]?.effort;
|
|
208
|
+
if (!TASK_CLASSES.includes(taskClass) || !['low', 'medium', 'high', 'xhigh', 'max'].includes(effort)) {
|
|
209
|
+
throw new Error('Policy must specify a supported task class and effort');
|
|
210
|
+
}
|
|
211
|
+
const approved = selection.routes?.[harness]?.[taskClass];
|
|
212
|
+
const codingEffort = selection.routes?.[harness]?.codingEffort;
|
|
213
|
+
const coding = features.hasCode || /\b(implement|code|coding|debug|refactor|test|endpoint|API|repository|module|function)\b/i.test(features.taskHints || '');
|
|
214
|
+
const approvedEffort = harness === 'claude-code' && taskClass === 'medium' && coding
|
|
215
|
+
? codingEffort || approved?.effort : approved?.effort;
|
|
216
|
+
if (harness === 'codex' && ['xhigh', 'max'].includes(effort) &&
|
|
217
|
+
(taskClass !== 'exceptional' || effort !== 'xhigh' || !approved?.requiresNamedReason ||
|
|
218
|
+
!/^[a-z][a-z0-9-]{2,79}$/.test(decision.exceptionalReason || ''))) {
|
|
219
|
+
throw new Error('Exceptional xhigh requires an explicit qualified route and named reason; no automatic max effort');
|
|
220
|
+
}
|
|
221
|
+
if (approved?.model !== chosen.id || approvedEffort !== effort) {
|
|
222
|
+
throw new Error('Custom policy decision exceeds reviewed model/effort allocation; update per-user routing-policy.json');
|
|
223
|
+
}
|
|
224
|
+
if (chosen.supportedEfforts && !chosen.supportedEfforts.includes(effort)) {
|
|
225
|
+
throw new Error(`Policy effort unavailable for ${chosen.id}: ${effort}`);
|
|
226
|
+
}
|
|
227
|
+
let routedBy = 'user-policy';
|
|
228
|
+
try {
|
|
229
|
+
const route = learnedRoute || (await import('./metaharness-router.mjs')).route;
|
|
230
|
+
// Explicit allocation is authoritative. A learned model outside it never gets first refusal.
|
|
231
|
+
const learned = await route(prompt, [chosen], profile);
|
|
232
|
+
routedBy = learned.routedBy === '@metaharness/router' && learned.model === chosen.id
|
|
233
|
+
? '@metaharness/router (policy constrained)'
|
|
234
|
+
: `user-policy (${learned.routedBy || 'learned decision rejected'})`;
|
|
235
|
+
} catch (e) { routedBy = `user-policy (learned router unavailable: ${e.message})`; }
|
|
236
|
+
return { ...decision, provider: chosen.provider, tier: chosen.tier, taskClass, effort,
|
|
237
|
+
subscriptionCovered: true, selectionReviewedAt: selection.reviewedAt, selectionMaxAgeMs: evidence.maxAgeMs,
|
|
238
|
+
selectionRouteDigest: evidence.routeDigest, selectionEvidence: evidence, routedBy };
|
|
239
|
+
}
|
|
240
|
+
|
|
167
241
|
async function main() {
|
|
168
242
|
const args = parseArgs(process.argv.slice(2));
|
|
169
243
|
if (args.help) {
|
|
@@ -175,7 +249,10 @@ async function main() {
|
|
|
175
249
|
args.harness ||
|
|
176
250
|
(process.env.CODEX_SANDBOX || fs.existsSync(path.join(os.homedir(), '.codex', 'config.toml')) && process.env.CODEX ? 'codex' : null) ||
|
|
177
251
|
'claude-code';
|
|
178
|
-
const
|
|
252
|
+
const raw = args.prompt || readStdin();
|
|
253
|
+
const request = args.requestJson ? JSON.parse(raw) : { prompt: raw };
|
|
254
|
+
const prompt = request.prompt;
|
|
255
|
+
if (typeof prompt !== 'string') throw new Error('Request prompt must be a string');
|
|
179
256
|
if (!prompt || !prompt.trim()) {
|
|
180
257
|
process.stderr.write('model-router-engine: no prompt (use --prompt "..." or pipe text on stdin)\n');
|
|
181
258
|
process.exit(2);
|
|
@@ -184,47 +261,11 @@ async function main() {
|
|
|
184
261
|
const profile = loadProfile();
|
|
185
262
|
const candidates = applyProfile(loadCatalog(), profile);
|
|
186
263
|
const policy = await loadPolicy(args.policy);
|
|
187
|
-
const features = extractFeatures(prompt, harness);
|
|
264
|
+
const features = extractFeatures(prompt, harness, request.taskFacts);
|
|
188
265
|
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
// engine users actually run stayed 100% hand-rolled while the README said otherwise. An honest
|
|
193
|
-
// artifact sitting next to a lying claim is still a lie. The decision now comes from rUv's code
|
|
194
|
-
// whenever it CAN decide, and the local heuristic is a fallback that must ANNOUNCE ITSELF.
|
|
195
|
-
let decision;
|
|
196
|
-
let routedBy;
|
|
197
|
-
const pool = candidates.filter((m) => (m.harness || []).includes(harness));
|
|
198
|
-
try {
|
|
199
|
-
const mh = await import('./metaharness-router.mjs');
|
|
200
|
-
const r = await mh.route(prompt, pool.length ? pool : candidates, profile);
|
|
201
|
-
if (r.routedBy === '@metaharness/router') {
|
|
202
|
-
const pick = candidates.find((m) => m.id === r.model);
|
|
203
|
-
decision = {
|
|
204
|
-
model: r.model,
|
|
205
|
-
provider: pick?.provider ?? null,
|
|
206
|
-
tier: pick?.tier ?? null,
|
|
207
|
-
reason: `@metaharness/router (rUv's learned cost-optimal router): predicted quality ${r.predictedQuality?.toFixed(2)}, ${r.metBar ? 'clears' : 'BELOW'} the bar, ${r.subscriptionCovered ? '$0 (your subscription)' : `$${r.costPerMTok}/Mtok`}, from ${r.labels} labelled example(s)`,
|
|
208
|
-
confidence: r.predictedQuality ?? 0,
|
|
209
|
-
};
|
|
210
|
-
routedBy = '@metaharness/router';
|
|
211
|
-
} else {
|
|
212
|
-
// COLD-START or the package is absent. Say which, out loud — never pass the fallback off as the
|
|
213
|
-
// learned router. That substitution is the entire sin this wiring exists to end.
|
|
214
|
-
routedBy = `local-heuristic (${r.routedBy}: ${r.reason})`;
|
|
215
|
-
}
|
|
216
|
-
} catch (e) {
|
|
217
|
-
routedBy = `local-heuristic (@metaharness/router unavailable: ${e.message})`;
|
|
218
|
-
}
|
|
219
|
-
|
|
220
|
-
if (!decision) {
|
|
221
|
-
if (!policy) {
|
|
222
|
-
const pick = pool.slice().sort((x, y) => (x.costPerMTok?.out ?? Infinity) - (y.costPerMTok?.out ?? Infinity))[0] || candidates[0];
|
|
223
|
-
decision = { model: pick?.id ?? null, provider: pick?.provider ?? null, tier: pick?.tier ?? null, reason: 'NO POLICY FOUND — fell back to cheapest priced candidate for the harness', confidence: 0 };
|
|
224
|
-
} else {
|
|
225
|
-
decision = policy.choose({ features, candidates, harness });
|
|
226
|
-
}
|
|
227
|
-
}
|
|
266
|
+
const decision = await selectDecision({ prompt, harness, candidates, profile, policy, features,
|
|
267
|
+
learnedRoute: args.policyOnly ? async () => ({ routedBy: 'SKIPPED (policy-only)' }) : undefined });
|
|
268
|
+
const routedBy = decision.routedBy;
|
|
228
269
|
|
|
229
270
|
const chosen = candidates.find((m) => m.id === decision.model) || null;
|
|
230
271
|
const out = {
|
|
@@ -233,6 +274,15 @@ async function main() {
|
|
|
233
274
|
model: decision.model,
|
|
234
275
|
provider: decision.provider,
|
|
235
276
|
tier: decision.tier,
|
|
277
|
+
taskClass: decision.taskClass,
|
|
278
|
+
exceptionalReason: decision.exceptionalReason,
|
|
279
|
+
classificationSource: decision.classificationSource,
|
|
280
|
+
effort: decision.effort,
|
|
281
|
+
subscriptionCovered: decision.subscriptionCovered,
|
|
282
|
+
selectionReviewedAt: decision.selectionReviewedAt,
|
|
283
|
+
selectionMaxAgeMs: decision.selectionMaxAgeMs,
|
|
284
|
+
selectionRouteDigest: decision.selectionRouteDigest,
|
|
285
|
+
selectionEvidence: decision.selectionEvidence,
|
|
236
286
|
reason: decision.reason,
|
|
237
287
|
confidence: decision.confidence,
|
|
238
288
|
// WHO decided. Never let a caller assume the learned router made a call the heuristic made.
|
|
@@ -240,14 +290,15 @@ async function main() {
|
|
|
240
290
|
policy_source: policy ? policy.source.replace(os.homedir(), '~') : 'none',
|
|
241
291
|
profile: profile ? PROFILE_PATH.replace(os.homedir(), '~') : 'none (catalog taken as-is — run model-router-setup.mjs)',
|
|
242
292
|
price_verified: chosen ? chosen.verified : null,
|
|
243
|
-
est_input_cost_usd: estInputCost(chosen, features.estTokens),
|
|
293
|
+
est_input_cost_usd: decision.subscriptionCovered ? 0 : estInputCost(chosen, features.estTokens),
|
|
294
|
+
api_list_input_cost_usd: estInputCost(chosen, features.estTokens), // API sticker estimate, not subscription billing
|
|
244
295
|
features: { estTokens: features.estTokens, hasCode: features.hasCode, codeFences: features.codeFences, fileTypes: features.fileTypes, questionCount: features.questionCount },
|
|
245
296
|
};
|
|
246
297
|
|
|
247
298
|
// Durable decision log (append-only; separate from route-cheap's execution/savings ledger).
|
|
248
299
|
try {
|
|
249
300
|
fs.mkdirSync(path.dirname(DECISIONS_LOG), { recursive: true });
|
|
250
|
-
fs.appendFileSync(DECISIONS_LOG, JSON.stringify({
|
|
301
|
+
fs.appendFileSync(DECISIONS_LOG, JSON.stringify({ ts: out.ts, harness, model: out.model, effort: out.effort, taskClass: out.taskClass, exceptionalReason: out.exceptionalReason, subscriptionCovered: out.subscriptionCovered, policy_source: out.policy_source }) + '\n');
|
|
251
302
|
} catch { /* logging must never break selection */ }
|
|
252
303
|
|
|
253
304
|
if (args.mode === 'line') {
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Human-labeled deterministic classifier evaluation. Never dispatches or makes inference calls.
|
|
3
|
+
import fs from 'node:fs';
|
|
4
|
+
import path from 'node:path';
|
|
5
|
+
import crypto from 'node:crypto';
|
|
6
|
+
import { execFileSync } from 'node:child_process';
|
|
7
|
+
import { performance } from 'node:perf_hooks';
|
|
8
|
+
import { fileURLToPath } from 'node:url';
|
|
9
|
+
import { classify } from '../config/model-router/policy.default.mjs';
|
|
10
|
+
import { extractFeatures } from './model-router-engine.mjs';
|
|
11
|
+
|
|
12
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
13
|
+
const DEFAULT_CASES = path.join(ROOT, 'config/model-router/routing-eval-cases.json');
|
|
14
|
+
const CLASSES = ['fast', 'medium', 'substantial', 'hard', 'exceptional'];
|
|
15
|
+
const ROLES = ['mechanical-work', 'ordinary-work', 'substantial-implementation', 'bounded-difficult-reasoning', 'exceptional-reasoning'];
|
|
16
|
+
const RISKS = ['low', 'moderate', 'high', 'critical'];
|
|
17
|
+
const rank = (value) => CLASSES.indexOf(value);
|
|
18
|
+
const digest = (bytes) => crypto.createHash('sha256').update(bytes).digest('hex');
|
|
19
|
+
|
|
20
|
+
export function validateCases(cases) {
|
|
21
|
+
if (!Array.isArray(cases) || !cases.length) throw new Error('Evaluation requires at least one labeled case');
|
|
22
|
+
const ids = new Set();
|
|
23
|
+
for (const row of cases) {
|
|
24
|
+
if (!row.id || ids.has(row.id)) throw new Error('Case IDs must be nonempty and unique');
|
|
25
|
+
ids.add(row.id);
|
|
26
|
+
if (typeof row.request !== 'string' || !row.request.trim() || !row.rationale || !row.group) throw new Error(`Case ${row.id} needs an original request, group and label rationale`);
|
|
27
|
+
const { minimumClass, maximumClass, minimumRole } = row.expected || {};
|
|
28
|
+
if (rank(minimumClass) < 0 || rank(maximumClass) < rank(minimumClass)) throw new Error(`Case ${row.id} has an invalid expected class range`);
|
|
29
|
+
if (minimumRole !== ROLES[rank(minimumClass)] || !RISKS.includes(row.risk)) throw new Error(`Case ${row.id} has an invalid role or risk label`);
|
|
30
|
+
}
|
|
31
|
+
return cases;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const percentile = (values, fraction) => values[Math.max(0, Math.ceil(values.length * fraction) - 1)] ?? null;
|
|
35
|
+
const round = (value) => Number(value.toFixed(6));
|
|
36
|
+
|
|
37
|
+
export function evaluateRouting(cases, { classifier = classify, clock = () => performance.now(), generatedAt = new Date().toISOString() } = {}) {
|
|
38
|
+
validateCases(cases);
|
|
39
|
+
const results = cases.map((row) => {
|
|
40
|
+
const features = extractFeatures(row.request, 'codex', row.taskFacts);
|
|
41
|
+
const began = clock();
|
|
42
|
+
let observedClass = null; let error = null;
|
|
43
|
+
try {
|
|
44
|
+
observedClass = classifier(features, 'codex');
|
|
45
|
+
if (rank(observedClass) < 0) throw new Error(`Classifier returned unknown class: ${String(observedClass)}`);
|
|
46
|
+
} catch (caught) { error = caught?.message || String(caught); }
|
|
47
|
+
const latencyMs = round(Math.max(0, clock() - began));
|
|
48
|
+
const underroute = !error && rank(observedClass) < rank(row.expected.minimumClass);
|
|
49
|
+
const dangerousUnderroute = underroute && ['high', 'critical'].includes(row.risk);
|
|
50
|
+
const needlessEscalation = !error && rank(observedClass) > rank(row.expected.maximumClass);
|
|
51
|
+
return {
|
|
52
|
+
...row, observedClass, observedRole: error ? null : ROLES[rank(observedClass)], latencyMs,
|
|
53
|
+
underroute, dangerousUnderroute, needlessEscalation,
|
|
54
|
+
verdict: error || underroute || needlessEscalation ? 'FAIL' : 'PASS',
|
|
55
|
+
...(error ? { error } : {}),
|
|
56
|
+
};
|
|
57
|
+
});
|
|
58
|
+
const latencies = results.map((row) => row.latencyMs).sort((a, b) => a - b);
|
|
59
|
+
const count = (field) => results.filter((row) => row[field]).length;
|
|
60
|
+
const failures = results.filter((row) => row.verdict === 'FAIL').length;
|
|
61
|
+
return {
|
|
62
|
+
schemaVersion: 1, kind: 'ruvnet-brain.model-routing-classifier-evaluation', generatedAt,
|
|
63
|
+
evidenceScope: 'classifier-only', harness: 'codex', verdict: failures ? 'FAIL' : 'PASS',
|
|
64
|
+
metrics: {
|
|
65
|
+
cases: results.length, passed: results.length - failures, failed: failures,
|
|
66
|
+
underroute: count('underroute'), dangerousUnderroute: count('dangerousUnderroute'),
|
|
67
|
+
needlessEscalation: count('needlessEscalation'), classifierErrors: count('error'),
|
|
68
|
+
classificationLatencyMs: { count: latencies.length, median: percentile(latencies, 0.5), p95: percentile(latencies, 0.95), max: latencies.at(-1),
|
|
69
|
+
scope: 'one classifier invocation per case; excludes feature extraction and module startup' },
|
|
70
|
+
},
|
|
71
|
+
nativeHandoff: { status: 'unsupported', qualified: false, attempted: false,
|
|
72
|
+
reason: 'This evaluation calls classify directly; it does not select a model, dispatch a native host, or observe model/effort/continuation.' },
|
|
73
|
+
limitations: ['Human labels are reviewable policy expectations, not measured model capability.',
|
|
74
|
+
'One bounded synthetic corpus is not population accuracy or project-specific optimality.',
|
|
75
|
+
'Latency describes local deterministic classification, not inference or completion speed.',
|
|
76
|
+
'A passing classifier verdict never qualifies native handoff.'],
|
|
77
|
+
results,
|
|
78
|
+
};
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
export function runEvaluation({ caseFile = DEFAULT_CASES, generatedAt } = {}) {
|
|
82
|
+
const raw = fs.readFileSync(caseFile);
|
|
83
|
+
const dataset = JSON.parse(raw);
|
|
84
|
+
if (dataset.schemaVersion !== 1 || dataset.harness !== 'codex') throw new Error('Unsupported case dataset');
|
|
85
|
+
const report = evaluateRouting(dataset.cases, { generatedAt });
|
|
86
|
+
const policyFile = path.join(ROOT, 'config/model-router/policy.default.mjs');
|
|
87
|
+
const featureFile = path.join(ROOT, 'scripts/model-router-engine.mjs');
|
|
88
|
+
report.source = {
|
|
89
|
+
head: execFileSync('git', ['rev-parse', 'HEAD'], { cwd: ROOT, encoding: 'utf8' }).trim(),
|
|
90
|
+
policyPath: path.relative(ROOT, policyFile), policySha256: digest(fs.readFileSync(policyFile)),
|
|
91
|
+
featureExtractorPath: path.relative(ROOT, featureFile), featureExtractorSha256: digest(fs.readFileSync(featureFile)),
|
|
92
|
+
evaluatorPath: path.relative(ROOT, fileURLToPath(import.meta.url)), evaluatorSha256: digest(fs.readFileSync(fileURLToPath(import.meta.url))),
|
|
93
|
+
casePath: path.relative(ROOT, caseFile), casesSha256: digest(raw),
|
|
94
|
+
labelBasis: dataset.labelBasis,
|
|
95
|
+
};
|
|
96
|
+
return report;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
100
|
+
const args = process.argv.slice(2);
|
|
101
|
+
if (args.length && (args.length !== 2 || args[0] !== '--out')) throw new Error('Usage: node scripts/model-routing-eval.mjs [--out <report.json>]');
|
|
102
|
+
const report = runEvaluation();
|
|
103
|
+
const json = `${JSON.stringify(report, null, 2)}\n`;
|
|
104
|
+
if (args.length) fs.writeFileSync(path.resolve(args[1]), json);
|
|
105
|
+
else process.stdout.write(json);
|
|
106
|
+
// Write the full honest report before returning a failed semantic qualification status.
|
|
107
|
+
process.exitCode = report.verdict === 'PASS' ? 0 : 1;
|
|
108
|
+
}
|