ruvnet-brain 4.5.4 → 4.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/README.md +2 -2
  2. package/bin/install.mjs +162 -27
  3. package/config/model-router/catalog.template.json +126 -52
  4. package/config/model-router/claude-terminal-mod/.claude-plugin/plugin.json +1 -0
  5. package/config/model-router/claude-terminal-mod/README.md +22 -0
  6. package/config/model-router/claude-terminal-mod/hooks/hooks.json +1 -0
  7. package/config/model-router/claude-terminal-mod/hooks/policy.default.mjs +2 -0
  8. package/config/model-router/claude-terminal-mod/hooks/register.js +66 -0
  9. package/config/model-router/claude-terminal-mod/hooks/routing.js +55 -0
  10. package/config/model-router/claude-terminal-mod/hooks/runtime.js +2 -0
  11. package/config/model-router/claude-terminal-mod/tests/native.test.ts +86 -0
  12. package/config/model-router/policy.default.mjs +97 -75
  13. package/config/model-router/qualification-contract.json +124 -0
  14. package/config/model-router/routing-eval-cases.json +275 -0
  15. package/config/model-router/routing-policy.template.json +76 -0
  16. package/config/model-router/weekly-analyst-instruction.md +60 -0
  17. package/data/model-catalog.json +44 -49
  18. package/package.json +6 -3
  19. package/plugin/.claude-plugin/plugin.json +1 -1
  20. package/plugin/.codex-plugin/plugin.json +1 -1
  21. package/plugin/mcp/managed-cli-interface.mjs +9 -4
  22. package/plugin/scripts/capability-claim-evidence.mjs +9 -2
  23. package/plugin/scripts/codex-hook-adapter.mjs +18 -9
  24. package/plugin/scripts/project-capture-queue.mjs +7 -1
  25. package/plugin/scripts/project-progression-hook.mjs +16 -7
  26. package/plugin/scripts/project-progression-outbox.mjs +96 -15
  27. package/plugin/scripts/project-progression-producer.mjs +12 -1
  28. package/plugin/scripts/project-progression-store.mjs +53 -1
  29. package/plugin/scripts/project-transition-hook.mjs +19 -8
  30. package/plugin/scripts/session-snapshot-hook.mjs +2 -1
  31. package/scripts/claude-terminal-mod.mjs +89 -0
  32. package/scripts/codex-hook-trust-reconcile.mjs +247 -0
  33. package/scripts/codex-routed.sh +3 -36
  34. package/scripts/goldie-weekly.sh +8 -64
  35. package/scripts/metaharness-router.mjs +7 -1
  36. package/scripts/model-analyst-sandbox.mjs +54 -0
  37. package/scripts/model-currency-evidence.mjs +139 -0
  38. package/scripts/model-currency.mjs +230 -0
  39. package/scripts/model-native-catalog.mjs +111 -0
  40. package/scripts/model-native-qualification.mjs +251 -0
  41. package/scripts/model-router-agent-hook.mjs +136 -0
  42. package/scripts/model-router-dispatch.mjs +161 -0
  43. package/scripts/model-router-engine.mjs +155 -104
  44. package/scripts/model-routing-eval.mjs +108 -0
  45. package/scripts/model-routing-gateway.mjs +453 -0
  46. package/scripts/model-routing-launchers.mjs +209 -0
  47. package/scripts/model-routing-policy-promotion.mjs +203 -0
  48. package/scripts/model-terminal-gateway.mjs +310 -0
  49. package/scripts/model-terminal-launchers.mjs +300 -0
  50. package/scripts/model-weekly-analyst.mjs +299 -0
  51. package/scripts/model-weekly-assessment.mjs +91 -0
  52. package/scripts/model-weekly-cycle.mjs +183 -0
  53. package/scripts/model-weekly-qualification.mjs +362 -0
  54. package/scripts/native-subscription-usage.mjs +57 -0
  55. package/scripts/release-qualification-contract.mjs +50 -1
  56. package/scripts/release-qualification.mjs +9 -6
  57. package/scripts/security-guidance-codex-compat.mjs +142 -0
  58. package/scripts/user-model-prompt-hook.mjs +69 -0
@@ -1,49 +1,28 @@
1
1
  #!/usr/bin/env node
2
- // scripts/model-router-engine.mjs — the harness-neutral MODEL SELECTION engine.
3
- //
4
- // WHAT THIS IS (and is NOT):
5
- // • IS: a pure `prompt -> {model, provider, reason, cost}` DECISION engine. It extracts features
6
- // from the prompt and hands them to a PLUGGABLE POLICY that decides which model to use. The
7
- // policy is the swappable part — drop your researched heuristics (or a learned router) into
8
- // ~/.claude/model-router/policy.mjs and the engine picks them up. NO heuristics are baked in
9
- // here. (ADR-040 / DRACO, verified via search_ruvnet: a hand-built self-signal threshold routed
10
- // WORSE than always-cheapest; a learned map from a real feature beat the best fixed model. So
11
- // the SIGNAL/policy is everything and must never be hard-coded into the engine.)
12
- // • IS harness-neutral: the SAME CLI is consulted by Claude Code AND Codex. It only DECIDES; it
13
- // does not launch a model. The caller acts on the JSON. (Codex has no native routing surface —
14
- // ~/.codex/config.toml launches one model per run — so a consulted CLI is the only way to make
15
- // selection work for Codex too. That is the fix for "only partially OK for Codex.")
16
- // • Is NOT an executor. Running a task on a cheap model is route-cheap.mjs's job (OpenRouter).
17
- // This answers only "which model should handle this prompt?"
18
- //
19
- // INTEGRATION:
20
- // Claude Code : call from a hook/skill, parse the JSON, use .model.
21
- // node model-router-engine.mjs --harness claude-code --prompt "$PROMPT" --json
22
- // Codex : wrap the codex launch —
23
- // M=$(node model-router-engine.mjs --harness codex --prompt "$TASK" --json | jq -r .model)
24
- // codex --model "$M" ...
25
- //
26
- // Config (edit freely): ~/.claude/model-router/catalog.json (candidates + verified pricing)
27
- // ~/.claude/model-router/policy.mjs (YOUR policy; falls back to policy.default.mjs)
28
- // Decision log: ~/.claude/metaharness/routing-decisions.jsonl (sibling to route-cheap's execution receipts)
29
- //
30
- // Usage:
31
- // node model-router-engine.mjs --prompt "..." [--harness claude-code|codex] [--policy <path>] [--json|--line]
32
- // echo "the prompt text" | node model-router-engine.mjs --harness codex
2
+ // Deterministic prompt classification + reviewed per-user native model/effort allocation.
3
+ // This selects only; model-router-dispatch.mjs enforces a managed worker launch. Parent chat
4
+ // model selection is controlled by the host, not a UserPromptSubmit recommendation hook.
5
+ // ~/.claude/model-router: catalog.json, profile.json, routing-policy.json, optional policy.mjs.
6
+ // Learned routes are constrained to the reviewed policy pick, never given first refusal.
7
+ // Usage: node model-router-engine.mjs --harness codex --policy-only --json < prompt.txt
8
+ // Decision receipts retain model/effort/class metadata only; no raw prompt or policy reason.
33
9
 
34
10
  import fs from 'node:fs';
35
11
  import path from 'node:path';
36
12
  import os from 'node:os';
13
+ import crypto from 'node:crypto';
37
14
  import { pathToFileURL, fileURLToPath } from 'node:url';
38
15
  import { estTokens } from './route-cheap.mjs'; // reuse the verified char/4 estimator (DRY)
39
16
 
40
17
  const __dirname = path.dirname(fileURLToPath(import.meta.url));
41
- export const CONFIG_DIR = path.join(os.homedir(), '.claude', 'model-router');
18
+ export const CONFIG_DIR = process.env.MODEL_ROUTER_CONFIG_DIR || path.join(os.homedir(), '.claude', 'model-router');
42
19
  // Overridable for hermetic tests + CI (runners have no ~/.claude): the 2026-07-12 CI redness was
43
20
  // exactly this — tests that silently depended on one developer's machine state.
44
21
  const CATALOG_PATH = process.env.MODEL_ROUTER_CATALOG || path.join(CONFIG_DIR, 'catalog.json');
45
22
  const POLICY_USER = path.join(CONFIG_DIR, 'policy.mjs');
46
23
  const POLICY_DEFAULT = path.join(CONFIG_DIR, 'policy.default.mjs');
24
+ const POLICY_SHIPPED = path.join(__dirname, '..', 'config', 'model-router', 'policy.default.mjs');
25
+ export const TASK_CLASSES = ['fast', 'medium', 'substantial', 'hard', 'exceptional'];
47
26
  const DECISIONS_LOG =
48
27
  process.env.MODEL_ROUTER_DECISIONS ||
49
28
  path.join(os.homedir(), '.claude', 'metaharness', 'routing-decisions.jsonl');
@@ -51,7 +30,7 @@ const DECISIONS_LOG =
51
30
  // ─── feature extraction: this is "based on what the prompt is" ────────────────────────────────
52
31
  // Pure and deterministic. Emits SIGNALS only — it never decides. Policies consume these; extend
53
32
  // this object as your research identifies new predictive features (it is the documented surface).
54
- export function extractFeatures(prompt, harness) {
33
+ export function extractFeatures(prompt, harness, taskFacts) {
55
34
  const text = prompt || '';
56
35
  const codeFences = Math.floor((text.match(/```/g) || []).length / 2);
57
36
  const fileTypes = [...new Set((text.match(/\.[a-z0-9]{1,5}\b/gi) || []).map((s) => s.toLowerCase()))].slice(0, 12);
@@ -64,8 +43,9 @@ export function extractFeatures(prompt, harness) {
64
43
  hasCode,
65
44
  fileTypes,
66
45
  questionCount: (text.match(/\?/g) || []).length,
67
- taskHints: text.slice(0, 4000), // policies may regex over the actual prompt head
46
+ taskHints: text, // policies may regex over the actual prompt head
68
47
  harness,
48
+ taskFacts,
69
49
  };
70
50
  }
71
51
 
@@ -94,44 +74,31 @@ export function applyProfile(candidates, profile) {
94
74
  }));
95
75
  }
96
76
 
97
- // Honest provenance of the catalog the engine is actually using, so no surface can pass the
98
- // built-in stub off as a real personal catalog (trust rule: never present a fallback as the thing).
99
- // Returns 'catalog' when a real ~/.claude/model-router/catalog.json is present + valid, else
100
- // 'built-in-fallback'. Same check loadCatalog() uses — kept in lockstep.
101
- export function catalogSource() {
102
- try {
103
- const j = JSON.parse(fs.readFileSync(CATALOG_PATH, 'utf8'));
104
- if (Array.isArray(j.candidates) && j.candidates.length) return 'catalog';
105
- } catch { /* fall through */ }
106
- return 'built-in-fallback';
77
+ // Catalog absence is not permission to use stale built-in identities.
78
+ export function catalogSource(file = CATALOG_PATH) {
79
+ try { loadCatalog(file); return 'catalog'; }
80
+ catch { return 'unavailable'; }
107
81
  }
108
82
 
109
- export function loadCatalog() {
83
+ export function loadCatalog(file = CATALOG_PATH) {
110
84
  try {
111
- const j = JSON.parse(fs.readFileSync(CATALOG_PATH, 'utf8'));
112
- if (Array.isArray(j.candidates) && j.candidates.length) return j.candidates;
113
- } catch {
114
- /* fall through to a minimal built-in so the engine still answers */
115
- }
116
- // Built-in fallback. Claude launchability was verified against Claude Code 2.1.220 on 2026-08-02;
117
- // prices remain null where the subscription host, rather than a metered API, is authoritative.
118
- return [
119
- { id: 'deepseek/deepseek-chat', provider: 'openrouter', harness: ['claude-code', 'codex'], tier: 'cheap', costPerMTok: { in: 0.2, out: 0.8 }, verified: '2026-07-07' },
120
- { id: 'claude-opus-4-8', provider: 'anthropic', harness: ['claude-code'], tier: 'frontier', costPerMTok: { in: 5.0, out: 25.0 }, verified: '2026-07-07' },
121
- { id: 'claude-opus-5', provider: 'anthropic', harness: ['claude-code'], subscription: ['claude-code'], tier: 'frontier', costPerMTok: null, verified: '2026-08-02 Claude Code 2.1.220 launch' },
122
- { id: 'claude-fable-5', provider: 'anthropic', harness: ['claude-code'], subscription: ['claude-code'], tier: 'frontier', costPerMTok: null, verified: '2026-08-02 Claude Code 2.1.220 launch' },
123
- { id: 'gpt-5.5', provider: 'openai', harness: ['codex'], tier: 'frontier', costPerMTok: null, verified: null },
124
- ];
85
+ const catalog = JSON.parse(fs.readFileSync(file, 'utf8'));
86
+ if (Array.isArray(catalog.candidates) && catalog.candidates.length) return catalog.candidates;
87
+ } catch { /* report one bounded configuration error, never substitute old models */ }
88
+ throw new Error('Current per-user model catalog missing or invalid; no built-in model fallback');
125
89
  }
126
90
 
127
91
  export async function loadPolicy(explicit) {
128
- const candidatePaths = [explicit, POLICY_USER, POLICY_DEFAULT].filter(Boolean);
92
+ if (explicit && !fs.existsSync(explicit)) throw new Error(`Explicit routing policy missing: ${explicit}`);
93
+ const candidatePaths = [explicit, POLICY_USER, POLICY_DEFAULT, POLICY_SHIPPED].filter(Boolean);
129
94
  for (const p of candidatePaths) {
130
95
  if (!fs.existsSync(p)) continue;
131
96
  try {
132
97
  const mod = await import(pathToFileURL(p).href);
133
98
  if (typeof mod.choose === 'function') return { choose: mod.choose, source: p };
99
+ throw new Error('Policy must export choose()');
134
100
  } catch (e) {
101
+ if (p === explicit || p === POLICY_USER) throw new Error(`User routing policy failed to load: ${e.message}`);
135
102
  process.stderr.write(`[model-router] policy at ${p} failed to load: ${e.message}\n`);
136
103
  }
137
104
  }
@@ -139,12 +106,14 @@ export async function loadPolicy(explicit) {
139
106
  }
140
107
 
141
108
  function parseArgs(argv) {
142
- const a = { harness: null, prompt: null, policy: null, mode: 'json' };
109
+ const a = { harness: null, prompt: null, policy: null, mode: 'json', policyOnly: false, requestJson: false };
143
110
  for (let i = 0; i < argv.length; i++) {
144
111
  const k = argv[i];
145
112
  if (k === '--prompt') a.prompt = argv[++i];
146
113
  else if (k === '--harness') a.harness = argv[++i];
147
114
  else if (k === '--policy') a.policy = argv[++i];
115
+ else if (k === '--request-json') a.requestJson = true;
116
+ else if (k === '--policy-only') a.policyOnly = true;
148
117
  else if (k === '--line') a.mode = 'line';
149
118
  else if (k === '--json') a.mode = 'json';
150
119
  else if (k === '--help' || k === '-h') a.help = true;
@@ -164,6 +133,111 @@ function estInputCost(candidate, inTokens) {
164
133
  return +((inTokens * p.in) / 1e6).toFixed(6);
165
134
  }
166
135
 
136
+ export function loadSelection(file = process.env.MODEL_ROUTER_SELECTION || path.join(CONFIG_DIR, 'routing-policy.json')) {
137
+ try { return JSON.parse(fs.readFileSync(file, 'utf8')); }
138
+ catch { throw new Error('No reviewed per-user routing-policy.json available'); }
139
+ }
140
+
141
+ export function assertCurrentSelection(selection, now = Date.now()) {
142
+ const age = now - Date.parse(selection?.reviewedAt);
143
+ const configuredMaxAge = selection?.maxAgeMs === undefined ? 604800000 : selection.maxAgeMs;
144
+ if (!Number.isSafeInteger(configuredMaxAge) || configuredMaxAge <= 0) {
145
+ throw new Error('Routing allocation maxAgeMs must be a finite positive integer');
146
+ }
147
+ const reviewedAt = selection?.reviewedAt;
148
+ const isoDate = typeof reviewedAt === 'string' && /^\d{4}-\d{2}-\d{2}(?:T\d{2}:\d{2}:\d{2}(?:\.\d{1,9})?(?:Z|[+-]\d{2}:\d{2}))?$/.test(reviewedAt);
149
+ const calendarDate = isoDate && Date.parse(reviewedAt.slice(0, 10));
150
+ const validCalendar = Number.isFinite(calendarDate) && new Date(calendarDate).toISOString().slice(0, 10) === reviewedAt.slice(0, 10);
151
+ if (selection?.schemaVersion !== 1 || !isoDate || !validCalendar || !Number.isFinite(age) || age < 0) {
152
+ throw new Error('Routing allocation missing, invalid or future-dated; owner-reviewed policy required');
153
+ }
154
+ // Evidence age is not revocation of an approved allocation. Retain the original reviewedAt;
155
+ // every managed launch still rechecks allocation integrity, native support, auth and allowance.
156
+ return selection;
157
+ }
158
+
159
+ function normalizedRoutes(value) {
160
+ if (Array.isArray(value)) return value.map(normalizedRoutes);
161
+ if (value && typeof value === 'object') return Object.fromEntries(Object.keys(value).sort()
162
+ .map((key) => [key, normalizedRoutes(value[key])]));
163
+ return value;
164
+ }
165
+
166
+ export function selectionEvidenceStatus(selection, now = Date.now()) {
167
+ assertCurrentSelection(selection, now);
168
+ const maxAgeMs = Math.min(selection.maxAgeMs ?? 604800000, 604800000);
169
+ const ageMs = now - Date.parse(selection.reviewedAt);
170
+ const routeDigest = selection.routes && typeof selection.routes === 'object' && !Array.isArray(selection.routes)
171
+ ? crypto.createHash('sha256').update(JSON.stringify(normalizedRoutes(selection.routes))).digest('hex') : null;
172
+ return { reviewedAt: selection.reviewedAt, maxAgeMs, ageMs, stale: ageMs > maxAgeMs, routeDigest };
173
+ }
174
+
175
+ // Eligibility is independent of policy and learning: catalog pricing is never spend permission.
176
+ export function eligibleCandidates(candidates, profile, harness) {
177
+ const host = profile?.harnesses?.[harness];
178
+ if (host?.available !== true || host?.subscription !== true) return [];
179
+ const provider = { codex: 'openai', 'claude-code': 'anthropic' }[harness];
180
+ return candidates.filter((m) => m.provider === provider &&
181
+ (m.harness || []).includes(harness) && (m.subscription || []).includes(harness));
182
+ }
183
+
184
+ export async function selectDecision({ prompt, harness, candidates, profile, policy,
185
+ features = extractFeatures(prompt, harness), learnedRoute, selection = loadSelection(), now = Date.now() } = {}) {
186
+ const evidence = selectionEvidenceStatus(selection, now);
187
+ const pool = eligibleCandidates(candidates, profile, harness);
188
+ if (!pool.length) throw new Error(`No available native subscription candidates for ${harness}; no metered fallback`);
189
+ if (!policy?.choose) throw new Error('No routing policy available');
190
+ const classifierFile = fs.existsSync(POLICY_SHIPPED) ? POLICY_SHIPPED : POLICY_DEFAULT;
191
+ const classifier = await import(pathToFileURL(classifierFile).href);
192
+ if (typeof classifier.classify !== 'function' || typeof classifier.validateTaskFacts !== 'function') {
193
+ throw new Error('Managed routing classifier missing required exports; update installed policy.default.mjs before dispatch');
194
+ }
195
+ classifier.validateTaskFacts(features.taskFacts);
196
+ const assessedClass = classifier.classify(features, harness);
197
+ const decision = await policy.choose({ features, candidates: pool, harness, profile, selection });
198
+ if (harness === 'codex' && ['substantial', 'exceptional', 'hard'].includes(assessedClass) && decision?.taskClass !== assessedClass) {
199
+ throw new Error(`Task requires explicit qualified ${assessedClass} route; legacy policy cannot silently use medium`);
200
+ }
201
+ const chosen = pool.find((m) => m.id === decision?.model);
202
+ if (!chosen) throw new Error(`Policy model unavailable or unauthorized: ${decision?.model || 'none'}`);
203
+ const taskClass = decision.taskClass;
204
+ if (!TASK_CLASSES.includes(taskClass)) {
205
+ throw new Error('Routing policy must return an explicit qualified taskClass; update legacy policy');
206
+ }
207
+ const effort = decision.effort || selection.routes?.[harness]?.[taskClass]?.effort;
208
+ if (!TASK_CLASSES.includes(taskClass) || !['low', 'medium', 'high', 'xhigh', 'max'].includes(effort)) {
209
+ throw new Error('Policy must specify a supported task class and effort');
210
+ }
211
+ const approved = selection.routes?.[harness]?.[taskClass];
212
+ const codingEffort = selection.routes?.[harness]?.codingEffort;
213
+ const coding = features.hasCode || /\b(implement|code|coding|debug|refactor|test|endpoint|API|repository|module|function)\b/i.test(features.taskHints || '');
214
+ const approvedEffort = harness === 'claude-code' && taskClass === 'medium' && coding
215
+ ? codingEffort || approved?.effort : approved?.effort;
216
+ if (harness === 'codex' && ['xhigh', 'max'].includes(effort) &&
217
+ (taskClass !== 'exceptional' || effort !== 'xhigh' || !approved?.requiresNamedReason ||
218
+ !/^[a-z][a-z0-9-]{2,79}$/.test(decision.exceptionalReason || ''))) {
219
+ throw new Error('Exceptional xhigh requires an explicit qualified route and named reason; no automatic max effort');
220
+ }
221
+ if (approved?.model !== chosen.id || approvedEffort !== effort) {
222
+ throw new Error('Custom policy decision exceeds reviewed model/effort allocation; update per-user routing-policy.json');
223
+ }
224
+ if (chosen.supportedEfforts && !chosen.supportedEfforts.includes(effort)) {
225
+ throw new Error(`Policy effort unavailable for ${chosen.id}: ${effort}`);
226
+ }
227
+ let routedBy = 'user-policy';
228
+ try {
229
+ const route = learnedRoute || (await import('./metaharness-router.mjs')).route;
230
+ // Explicit allocation is authoritative. A learned model outside it never gets first refusal.
231
+ const learned = await route(prompt, [chosen], profile);
232
+ routedBy = learned.routedBy === '@metaharness/router' && learned.model === chosen.id
233
+ ? '@metaharness/router (policy constrained)'
234
+ : `user-policy (${learned.routedBy || 'learned decision rejected'})`;
235
+ } catch (e) { routedBy = `user-policy (learned router unavailable: ${e.message})`; }
236
+ return { ...decision, provider: chosen.provider, tier: chosen.tier, taskClass, effort,
237
+ subscriptionCovered: true, selectionReviewedAt: selection.reviewedAt, selectionMaxAgeMs: evidence.maxAgeMs,
238
+ selectionRouteDigest: evidence.routeDigest, selectionEvidence: evidence, routedBy };
239
+ }
240
+
167
241
  async function main() {
168
242
  const args = parseArgs(process.argv.slice(2));
169
243
  if (args.help) {
@@ -175,7 +249,10 @@ async function main() {
175
249
  args.harness ||
176
250
  (process.env.CODEX_SANDBOX || fs.existsSync(path.join(os.homedir(), '.codex', 'config.toml')) && process.env.CODEX ? 'codex' : null) ||
177
251
  'claude-code';
178
- const prompt = args.prompt || readStdin();
252
+ const raw = args.prompt || readStdin();
253
+ const request = args.requestJson ? JSON.parse(raw) : { prompt: raw };
254
+ const prompt = request.prompt;
255
+ if (typeof prompt !== 'string') throw new Error('Request prompt must be a string');
179
256
  if (!prompt || !prompt.trim()) {
180
257
  process.stderr.write('model-router-engine: no prompt (use --prompt "..." or pipe text on stdin)\n');
181
258
  process.exit(2);
@@ -184,47 +261,11 @@ async function main() {
184
261
  const profile = loadProfile();
185
262
  const candidates = applyProfile(loadCatalog(), profile);
186
263
  const policy = await loadPolicy(args.policy);
187
- const features = extractFeatures(prompt, harness);
264
+ const features = extractFeatures(prompt, harness, request.taskFacts);
188
265
 
189
- // ── rUv's REAL router gets FIRST REFUSAL. ────────────────────────────────────────────────────────
190
- // 2026-07-13: this is the fix for a lie I shipped. v2.5's headline was "it uses @metaharness/router",
191
- // I wrote the wrapper, tested it, gated CI against faking — and NEVER WIRED IT INTO THIS FILE. The
192
- // engine users actually run stayed 100% hand-rolled while the README said otherwise. An honest
193
- // artifact sitting next to a lying claim is still a lie. The decision now comes from rUv's code
194
- // whenever it CAN decide, and the local heuristic is a fallback that must ANNOUNCE ITSELF.
195
- let decision;
196
- let routedBy;
197
- const pool = candidates.filter((m) => (m.harness || []).includes(harness));
198
- try {
199
- const mh = await import('./metaharness-router.mjs');
200
- const r = await mh.route(prompt, pool.length ? pool : candidates, profile);
201
- if (r.routedBy === '@metaharness/router') {
202
- const pick = candidates.find((m) => m.id === r.model);
203
- decision = {
204
- model: r.model,
205
- provider: pick?.provider ?? null,
206
- tier: pick?.tier ?? null,
207
- reason: `@metaharness/router (rUv's learned cost-optimal router): predicted quality ${r.predictedQuality?.toFixed(2)}, ${r.metBar ? 'clears' : 'BELOW'} the bar, ${r.subscriptionCovered ? '$0 (your subscription)' : `$${r.costPerMTok}/Mtok`}, from ${r.labels} labelled example(s)`,
208
- confidence: r.predictedQuality ?? 0,
209
- };
210
- routedBy = '@metaharness/router';
211
- } else {
212
- // COLD-START or the package is absent. Say which, out loud — never pass the fallback off as the
213
- // learned router. That substitution is the entire sin this wiring exists to end.
214
- routedBy = `local-heuristic (${r.routedBy}: ${r.reason})`;
215
- }
216
- } catch (e) {
217
- routedBy = `local-heuristic (@metaharness/router unavailable: ${e.message})`;
218
- }
219
-
220
- if (!decision) {
221
- if (!policy) {
222
- const pick = pool.slice().sort((x, y) => (x.costPerMTok?.out ?? Infinity) - (y.costPerMTok?.out ?? Infinity))[0] || candidates[0];
223
- decision = { model: pick?.id ?? null, provider: pick?.provider ?? null, tier: pick?.tier ?? null, reason: 'NO POLICY FOUND — fell back to cheapest priced candidate for the harness', confidence: 0 };
224
- } else {
225
- decision = policy.choose({ features, candidates, harness });
226
- }
227
- }
266
+ const decision = await selectDecision({ prompt, harness, candidates, profile, policy, features,
267
+ learnedRoute: args.policyOnly ? async () => ({ routedBy: 'SKIPPED (policy-only)' }) : undefined });
268
+ const routedBy = decision.routedBy;
228
269
 
229
270
  const chosen = candidates.find((m) => m.id === decision.model) || null;
230
271
  const out = {
@@ -233,6 +274,15 @@ async function main() {
233
274
  model: decision.model,
234
275
  provider: decision.provider,
235
276
  tier: decision.tier,
277
+ taskClass: decision.taskClass,
278
+ exceptionalReason: decision.exceptionalReason,
279
+ classificationSource: decision.classificationSource,
280
+ effort: decision.effort,
281
+ subscriptionCovered: decision.subscriptionCovered,
282
+ selectionReviewedAt: decision.selectionReviewedAt,
283
+ selectionMaxAgeMs: decision.selectionMaxAgeMs,
284
+ selectionRouteDigest: decision.selectionRouteDigest,
285
+ selectionEvidence: decision.selectionEvidence,
236
286
  reason: decision.reason,
237
287
  confidence: decision.confidence,
238
288
  // WHO decided. Never let a caller assume the learned router made a call the heuristic made.
@@ -240,14 +290,15 @@ async function main() {
240
290
  policy_source: policy ? policy.source.replace(os.homedir(), '~') : 'none',
241
291
  profile: profile ? PROFILE_PATH.replace(os.homedir(), '~') : 'none (catalog taken as-is — run model-router-setup.mjs)',
242
292
  price_verified: chosen ? chosen.verified : null,
243
- est_input_cost_usd: estInputCost(chosen, features.estTokens), // null if price unknown — never invented
293
+ est_input_cost_usd: decision.subscriptionCovered ? 0 : estInputCost(chosen, features.estTokens),
294
+ api_list_input_cost_usd: estInputCost(chosen, features.estTokens), // API sticker estimate, not subscription billing
244
295
  features: { estTokens: features.estTokens, hasCode: features.hasCode, codeFences: features.codeFences, fileTypes: features.fileTypes, questionCount: features.questionCount },
245
296
  };
246
297
 
247
298
  // Durable decision log (append-only; separate from route-cheap's execution/savings ledger).
248
299
  try {
249
300
  fs.mkdirSync(path.dirname(DECISIONS_LOG), { recursive: true });
250
- fs.appendFileSync(DECISIONS_LOG, JSON.stringify({ ...out, features: undefined, prompt_head: prompt.slice(0, 120) }) + '\n');
301
+ fs.appendFileSync(DECISIONS_LOG, JSON.stringify({ ts: out.ts, harness, model: out.model, effort: out.effort, taskClass: out.taskClass, exceptionalReason: out.exceptionalReason, subscriptionCovered: out.subscriptionCovered, policy_source: out.policy_source }) + '\n');
251
302
  } catch { /* logging must never break selection */ }
252
303
 
253
304
  if (args.mode === 'line') {
@@ -0,0 +1,108 @@
1
+ #!/usr/bin/env node
2
+ // Human-labeled deterministic classifier evaluation. Never dispatches or makes inference calls.
3
+ import fs from 'node:fs';
4
+ import path from 'node:path';
5
+ import crypto from 'node:crypto';
6
+ import { execFileSync } from 'node:child_process';
7
+ import { performance } from 'node:perf_hooks';
8
+ import { fileURLToPath } from 'node:url';
9
+ import { classify } from '../config/model-router/policy.default.mjs';
10
+ import { extractFeatures } from './model-router-engine.mjs';
11
+
12
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
13
+ const DEFAULT_CASES = path.join(ROOT, 'config/model-router/routing-eval-cases.json');
14
+ const CLASSES = ['fast', 'medium', 'substantial', 'hard', 'exceptional'];
15
+ const ROLES = ['mechanical-work', 'ordinary-work', 'substantial-implementation', 'bounded-difficult-reasoning', 'exceptional-reasoning'];
16
+ const RISKS = ['low', 'moderate', 'high', 'critical'];
17
+ const rank = (value) => CLASSES.indexOf(value);
18
+ const digest = (bytes) => crypto.createHash('sha256').update(bytes).digest('hex');
19
+
20
+ export function validateCases(cases) {
21
+ if (!Array.isArray(cases) || !cases.length) throw new Error('Evaluation requires at least one labeled case');
22
+ const ids = new Set();
23
+ for (const row of cases) {
24
+ if (!row.id || ids.has(row.id)) throw new Error('Case IDs must be nonempty and unique');
25
+ ids.add(row.id);
26
+ if (typeof row.request !== 'string' || !row.request.trim() || !row.rationale || !row.group) throw new Error(`Case ${row.id} needs an original request, group and label rationale`);
27
+ const { minimumClass, maximumClass, minimumRole } = row.expected || {};
28
+ if (rank(minimumClass) < 0 || rank(maximumClass) < rank(minimumClass)) throw new Error(`Case ${row.id} has an invalid expected class range`);
29
+ if (minimumRole !== ROLES[rank(minimumClass)] || !RISKS.includes(row.risk)) throw new Error(`Case ${row.id} has an invalid role or risk label`);
30
+ }
31
+ return cases;
32
+ }
33
+
34
+ const percentile = (values, fraction) => values[Math.max(0, Math.ceil(values.length * fraction) - 1)] ?? null;
35
+ const round = (value) => Number(value.toFixed(6));
36
+
37
+ export function evaluateRouting(cases, { classifier = classify, clock = () => performance.now(), generatedAt = new Date().toISOString() } = {}) {
38
+ validateCases(cases);
39
+ const results = cases.map((row) => {
40
+ const features = extractFeatures(row.request, 'codex', row.taskFacts);
41
+ const began = clock();
42
+ let observedClass = null; let error = null;
43
+ try {
44
+ observedClass = classifier(features, 'codex');
45
+ if (rank(observedClass) < 0) throw new Error(`Classifier returned unknown class: ${String(observedClass)}`);
46
+ } catch (caught) { error = caught?.message || String(caught); }
47
+ const latencyMs = round(Math.max(0, clock() - began));
48
+ const underroute = !error && rank(observedClass) < rank(row.expected.minimumClass);
49
+ const dangerousUnderroute = underroute && ['high', 'critical'].includes(row.risk);
50
+ const needlessEscalation = !error && rank(observedClass) > rank(row.expected.maximumClass);
51
+ return {
52
+ ...row, observedClass, observedRole: error ? null : ROLES[rank(observedClass)], latencyMs,
53
+ underroute, dangerousUnderroute, needlessEscalation,
54
+ verdict: error || underroute || needlessEscalation ? 'FAIL' : 'PASS',
55
+ ...(error ? { error } : {}),
56
+ };
57
+ });
58
+ const latencies = results.map((row) => row.latencyMs).sort((a, b) => a - b);
59
+ const count = (field) => results.filter((row) => row[field]).length;
60
+ const failures = results.filter((row) => row.verdict === 'FAIL').length;
61
+ return {
62
+ schemaVersion: 1, kind: 'ruvnet-brain.model-routing-classifier-evaluation', generatedAt,
63
+ evidenceScope: 'classifier-only', harness: 'codex', verdict: failures ? 'FAIL' : 'PASS',
64
+ metrics: {
65
+ cases: results.length, passed: results.length - failures, failed: failures,
66
+ underroute: count('underroute'), dangerousUnderroute: count('dangerousUnderroute'),
67
+ needlessEscalation: count('needlessEscalation'), classifierErrors: count('error'),
68
+ classificationLatencyMs: { count: latencies.length, median: percentile(latencies, 0.5), p95: percentile(latencies, 0.95), max: latencies.at(-1),
69
+ scope: 'one classifier invocation per case; excludes feature extraction and module startup' },
70
+ },
71
+ nativeHandoff: { status: 'unsupported', qualified: false, attempted: false,
72
+ reason: 'This evaluation calls classify directly; it does not select a model, dispatch a native host, or observe model/effort/continuation.' },
73
+ limitations: ['Human labels are reviewable policy expectations, not measured model capability.',
74
+ 'One bounded synthetic corpus is not population accuracy or project-specific optimality.',
75
+ 'Latency describes local deterministic classification, not inference or completion speed.',
76
+ 'A passing classifier verdict never qualifies native handoff.'],
77
+ results,
78
+ };
79
+ }
80
+
81
+ export function runEvaluation({ caseFile = DEFAULT_CASES, generatedAt } = {}) {
82
+ const raw = fs.readFileSync(caseFile);
83
+ const dataset = JSON.parse(raw);
84
+ if (dataset.schemaVersion !== 1 || dataset.harness !== 'codex') throw new Error('Unsupported case dataset');
85
+ const report = evaluateRouting(dataset.cases, { generatedAt });
86
+ const policyFile = path.join(ROOT, 'config/model-router/policy.default.mjs');
87
+ const featureFile = path.join(ROOT, 'scripts/model-router-engine.mjs');
88
+ report.source = {
89
+ head: execFileSync('git', ['rev-parse', 'HEAD'], { cwd: ROOT, encoding: 'utf8' }).trim(),
90
+ policyPath: path.relative(ROOT, policyFile), policySha256: digest(fs.readFileSync(policyFile)),
91
+ featureExtractorPath: path.relative(ROOT, featureFile), featureExtractorSha256: digest(fs.readFileSync(featureFile)),
92
+ evaluatorPath: path.relative(ROOT, fileURLToPath(import.meta.url)), evaluatorSha256: digest(fs.readFileSync(fileURLToPath(import.meta.url))),
93
+ casePath: path.relative(ROOT, caseFile), casesSha256: digest(raw),
94
+ labelBasis: dataset.labelBasis,
95
+ };
96
+ return report;
97
+ }
98
+
99
+ if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
100
+ const args = process.argv.slice(2);
101
+ if (args.length && (args.length !== 2 || args[0] !== '--out')) throw new Error('Usage: node scripts/model-routing-eval.mjs [--out <report.json>]');
102
+ const report = runEvaluation();
103
+ const json = `${JSON.stringify(report, null, 2)}\n`;
104
+ if (args.length) fs.writeFileSync(path.resolve(args[1]), json);
105
+ else process.stdout.write(json);
106
+ // Write the full honest report before returning a failed semantic qualification status.
107
+ process.exitCode = report.verdict === 'PASS' ? 0 : 1;
108
+ }