ruvnet-brain 4.0.1 → 4.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -0
- package/README.md +4 -4
- package/bin/install.mjs +100 -5
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +8 -22
- package/plugin/.claude-plugin/marketplace.json +1 -0
- package/plugin/.claude-plugin/plugin.json +2 -3
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +2 -2
- package/plugin/commands/configure.md +3 -2
- package/plugin/commands/rvbc.md +4 -3
- package/plugin/commands/rvcb.md +2 -2
- package/plugin/hooks/hooks.json +1 -2
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +21 -0
- package/plugin/scripts/detach.mjs +14 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/ground-ruvnet.sh +16 -6
- package/plugin/scripts/hook-shim.mjs +7 -7
- package/plugin/scripts/learn-capture.sh +22 -3
- package/plugin/scripts/learn-flush.mjs +21 -4
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +477 -0
- package/plugin/scripts/session-start.sh +3 -858
- package/plugin/skills/brain-console/SKILL.md +4 -2
- package/plugin/skills/release-proof/SKILL.md +81 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
- package/plugin/skills/rvbc/SKILL.md +9 -6
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +522 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +250 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +639 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +271 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +180 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2749 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +395 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +508 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +864 -0
- package/plugin/scripts/finalize-token-meter.mjs +0 -25
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// rerank-cap-eval.mjs — does bounding the cross-encoder pool change the ANSWERS?
|
|
3
|
+
//
|
|
4
|
+
// The cross-encoder is 84.7% of a query's wall and reads 605 (query, passage) pairs for an
|
|
5
|
+
// all-repos question. Capping that pool is the only lever that moves the number. But a cap
|
|
6
|
+
// re-orders results, and a naive one was MEASURED to lose the right answer — so no cap ships
|
|
7
|
+
// without a before/after on a real question set.
|
|
8
|
+
//
|
|
9
|
+
// TWO PHASES, because the measurement is expensive and the policy search is not:
|
|
10
|
+
//
|
|
11
|
+
// --collect Runs the FROZEN held-out set (evals/held-out.json — the same corpus
|
|
12
|
+
// scripts/eval-brain.mjs gates on) UNCAPPED, recording every scored candidate:
|
|
13
|
+
// repo, path, lane, within-lane depth, pool position, and its cross-encoder score.
|
|
14
|
+
// One ~8-minute query per question. This is the whole cost of the experiment.
|
|
15
|
+
//
|
|
16
|
+
// --report Replays capRerankPool + selectResults — the SHIPPING functions, imported, not
|
|
17
|
+
// re-implemented — against those recorded scores, for every candidate budget. The
|
|
18
|
+
// cross-encoder is deterministic per (query, passage) pair and pairs are scored
|
|
19
|
+
// independently, so a replay of a subset is EXACT: it is the same arithmetic on the
|
|
20
|
+
// same numbers, not a model of it.
|
|
21
|
+
//
|
|
22
|
+
// Reported metrics are the ones a wrong answer would move, graded by ground truth (never a model
|
|
23
|
+
// judge — an LLM panel once scored a zero-citation answer 98/100 on this repo):
|
|
24
|
+
// pairs — cross-encoder pairs actually scored. The load-independent primary evidence.
|
|
25
|
+
// top1-same — the winning document is byte-identical to the uncapped winner.
|
|
26
|
+
// kept — the uncapped top-k documents still present in the capped top-k.
|
|
27
|
+
// routed — top-1 lands in an expected repo (eval-brain's own metric, same Wilson bound).
|
|
28
|
+
// abstain — adversarial questions still decline (top ce < 0).
|
|
29
|
+
// banner — a winning gist chunk still carries its provenance banner.
|
|
30
|
+
//
|
|
31
|
+
// node scripts/rerank-cap-eval.mjs --collect [--conc 3] [--only ho-01,ho-02]
|
|
32
|
+
// node scripts/rerank-cap-eval.mjs --report [--budgets 605,272,136,69]
|
|
33
|
+
|
|
34
|
+
import fs from 'node:fs';
|
|
35
|
+
import os from 'node:os';
|
|
36
|
+
import path from 'node:path';
|
|
37
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
38
|
+
import { execFile } from 'node:child_process';
|
|
39
|
+
import { gradeQuestion, aggregate } from './eval-brain.mjs';
|
|
40
|
+
|
|
41
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
42
|
+
const KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
|
|
43
|
+
// The code under test is THIS checkout's, run against whatever corpus KB points at.
|
|
44
|
+
const CODE = path.join(ROOT, 'kb');
|
|
45
|
+
const argv = process.argv.slice(2);
|
|
46
|
+
const arg = (f, d) => { const i = argv.indexOf(f); return i >= 0 && argv[i + 1] ? argv[i + 1] : d; };
|
|
47
|
+
const TRACES = arg('--traces', path.join(os.tmpdir(), 'ruvnet-brain-ce-cap-traces'));
|
|
48
|
+
|
|
49
|
+
// The frozen 120, plus probes for the paths a cap could silently break. These are NOT scored
|
|
50
|
+
// against expectRepo (they are not part of the frozen set and must never contaminate it) — they
|
|
51
|
+
// are watched for one thing only: does the cap change the winner?
|
|
52
|
+
const PROBES = [
|
|
53
|
+
{ id: 'px-mcp-policy', query: 'which file enforces the MCP tool policy in ruvector', why: 'the case a naive cap was measured to lose (mcp-policy.js -> an ADR)' },
|
|
54
|
+
{ id: 'px-adr-085', query: 'ADR-085', why: 'bare ADR number — per-repo collision disclosure reads the whole pool' },
|
|
55
|
+
{ id: 'px-pkg-rvf', query: 'what is @ruvector/rvf', why: 'exact @scope/name — exercises the RESCUE lane the cap must never drop' },
|
|
56
|
+
{ id: 'px-meetings', query: 'how many commits and contributors does the project have', why: 'answer is BM25-only in a transcript store; dense buries it past rank 40' },
|
|
57
|
+
];
|
|
58
|
+
|
|
59
|
+
// Probes first, then the frozen set DEALT ROUND-ROBIN ACROSS STRATA. The held-out file is grouped
|
|
60
|
+
// by stratum, and at ~8 minutes a question a run can be interrupted — grouped order would make any
|
|
61
|
+
// prefix a biased sample (all 'described', no 'adversarial'), which is the kind of partial evidence
|
|
62
|
+
// that reads as a result and is not one. Round-robin makes every prefix stratified.
|
|
63
|
+
function loadQuestions() {
|
|
64
|
+
const { questions } = JSON.parse(fs.readFileSync(path.join(ROOT, 'evals', 'held-out.json'), 'utf8'));
|
|
65
|
+
const byStratum = new Map();
|
|
66
|
+
for (const q of questions) (byStratum.get(q.stratum) ?? byStratum.set(q.stratum, []).get(q.stratum)).push(q);
|
|
67
|
+
const lanes = [...byStratum.values()];
|
|
68
|
+
const dealt = [];
|
|
69
|
+
for (let i = 0; dealt.length < questions.length; i++) for (const lane of lanes) if (lane[i]) dealt.push(lane[i]);
|
|
70
|
+
return [...PROBES.map((p) => ({ ...p, stratum: 'probe' })), ...dealt];
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// ── collect ─────────────────────────────────────────────────────────────────────────────────────
|
|
74
|
+
async function collect() {
|
|
75
|
+
const only = arg('--only', '') ? new Set(arg('--only', '').split(',')) : null;
|
|
76
|
+
const questions = loadQuestions().filter((q) => !only || only.has(q.id));
|
|
77
|
+
const CONC = Math.max(1, parseInt(arg('--conc', '3'), 10) || 3);
|
|
78
|
+
fs.mkdirSync(TRACES, { recursive: true });
|
|
79
|
+
|
|
80
|
+
const run = (q) => new Promise((resolve) => {
|
|
81
|
+
const trace = path.join(TRACES, `${q.id}.jsonl`);
|
|
82
|
+
// A partial trace from a killed run must never be read as a complete one: write to .part and
|
|
83
|
+
// rename only on a clean exit, so --report can never score a truncated pool as a real pool.
|
|
84
|
+
const part = `${trace}.part`;
|
|
85
|
+
try { fs.rmSync(part, { force: true }); } catch { /* fresh anyway */ }
|
|
86
|
+
const t0 = Date.now();
|
|
87
|
+
execFile('node', ['forge-ask-all.mjs', '--dir', KB, '--q', q.query, '--k', '3'], {
|
|
88
|
+
cwd: CODE,
|
|
89
|
+
timeout: 1800000,
|
|
90
|
+
maxBuffer: 256 * 1024 * 1024,
|
|
91
|
+
// --cap 0 (the default) is the uncapped baseline. A NON-zero --cap re-collects the same
|
|
92
|
+
// questions with the cap actually engaged, which is the only way to check the replay against
|
|
93
|
+
// reality: capping changes which pairs share a batch, and batch composition was MEASURED to
|
|
94
|
+
// move scores by up to 0.26 logits, so a replayed cap is an approximation of a real one.
|
|
95
|
+
// --cascade K does the same for ADR-058's two-stage cascade. Both are collected FOR REAL for
|
|
96
|
+
// exactly that reason: the cascade re-batches its survivors too, so its scores are its own.
|
|
97
|
+
env: {
|
|
98
|
+
...process.env,
|
|
99
|
+
KB_CE_TRACE: part,
|
|
100
|
+
KB_CE_MAX_PAIRS: arg('--cap', '0'),
|
|
101
|
+
KB_CE_CASCADE_K: arg('--cascade', '0'),
|
|
102
|
+
KB_CE_CASCADE_TOKENS: arg('--tokens', '192'),
|
|
103
|
+
},
|
|
104
|
+
}, (err) => {
|
|
105
|
+
const ms = Date.now() - t0;
|
|
106
|
+
if (!err && fs.existsSync(part) && fs.statSync(part).size > 0) {
|
|
107
|
+
const rec = JSON.parse(fs.readFileSync(part, 'utf8').trim().split('\n')[0]);
|
|
108
|
+
fs.writeFileSync(trace, JSON.stringify({ id: q.id, stratum: q.stratum, expectRepo: q.expectRepo ?? null, ms, ...rec }) + '\n');
|
|
109
|
+
fs.rmSync(part, { force: true });
|
|
110
|
+
resolve({ id: q.id, ok: true, ms, pooled: rec.pooledAll });
|
|
111
|
+
} else {
|
|
112
|
+
try { fs.rmSync(part, { force: true }); } catch { /* nothing to clean */ }
|
|
113
|
+
resolve({ id: q.id, ok: false, ms, err: String(err?.message || 'no trace written').slice(0, 120) });
|
|
114
|
+
}
|
|
115
|
+
});
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
let cursor = 0, done = 0;
|
|
119
|
+
const t0 = Date.now();
|
|
120
|
+
const worker = async () => {
|
|
121
|
+
while (cursor < questions.length) {
|
|
122
|
+
const q = questions[cursor++];
|
|
123
|
+
if (fs.existsSync(path.join(TRACES, `${q.id}.jsonl`))) { done++; continue; } // resumable: an 8h run must survive a restart
|
|
124
|
+
const r = await run(q);
|
|
125
|
+
done++;
|
|
126
|
+
const la = os.loadavg()[0].toFixed(0);
|
|
127
|
+
process.stderr.write(`[cap-eval] ${done}/${questions.length} ${r.ok ? 'ok' : 'FAIL'} ${r.id} ${(r.ms / 1000).toFixed(0)}s pooled=${r.pooled ?? '-'} load=${la} elapsed=${((Date.now() - t0) / 60000).toFixed(0)}m${r.ok ? '' : ` :: ${r.err}`}\n`);
|
|
128
|
+
}
|
|
129
|
+
};
|
|
130
|
+
await Promise.all(Array.from({ length: Math.min(CONC, questions.length) }, worker));
|
|
131
|
+
console.error(`[cap-eval] traces in ${TRACES}`);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// ── report ──────────────────────────────────────────────────────────────────────────────────────
|
|
135
|
+
async function report() {
|
|
136
|
+
const { capRerankPool, selectResults } = await import(pathToFileURL(path.join(CODE, 'forge-ask-all.mjs')).href);
|
|
137
|
+
if (!fs.existsSync(TRACES)) { console.error(`no traces at ${TRACES} — run --collect first`); process.exit(2); }
|
|
138
|
+
const files = fs.readdirSync(TRACES).filter((f) => f.endsWith('.jsonl'));
|
|
139
|
+
if (!files.length) { console.error(`no traces at ${TRACES} — run --collect first`); process.exit(2); }
|
|
140
|
+
const traces = files.map((f) => JSON.parse(fs.readFileSync(path.join(TRACES, f), 'utf8').trim().split('\n')[0]));
|
|
141
|
+
|
|
142
|
+
const budgets = (arg('--budgets', '') ? arg('--budgets', '').split(',').map(Number)
|
|
143
|
+
: [0, 408, 272, 204, 170, 136, 102, 69, 48, 24]).sort((a, b) => b - a);
|
|
144
|
+
|
|
145
|
+
// Rebuild each trace's pool in its ORIGINAL fan-out order — the cap's tie-break depends on it.
|
|
146
|
+
const pools = traces.map((t) => ({
|
|
147
|
+
...t,
|
|
148
|
+
pool: [...t.cands].sort((a, b) => a.poolIdx - b.poolIdx)
|
|
149
|
+
.map((c) => ({ ...c, ceScore: c.ce, fullText: '', text: '', _lane: c.lane, _srcRank: c.rank, _poolIdx: c.poolIdx })),
|
|
150
|
+
}));
|
|
151
|
+
|
|
152
|
+
// The exact-package boost reads a candidate's BODY, which a trace deliberately does not carry
|
|
153
|
+
// (605 whole documents x 120 questions). It fires only on an `@scope/name` query, and it fires
|
|
154
|
+
// identically on any candidate that survives the cap, so it cannot flip a comparison between two
|
|
155
|
+
// survivors — but say so out loud rather than let a reader assume the replay is total.
|
|
156
|
+
const bodyDependent = pools.filter((p) => /@[a-z0-9][a-z0-9._-]*\/[a-z0-9._-]+/i.test(p.query)).map((p) => p.id);
|
|
157
|
+
|
|
158
|
+
const idOf = (r) => `${r.repo}/${r.path}`;
|
|
159
|
+
|
|
160
|
+
// CASCADE — the alternative policy, kept because it was measured and lost. Score every store's
|
|
161
|
+
// best passage first (~1 pair per store), then spend the rest of the budget only on the R stores
|
|
162
|
+
// whose best passage scored highest. It is the obvious "let the cross-encoder decide where to
|
|
163
|
+
// dig" design; the table below is the reason it is not what ships.
|
|
164
|
+
const cascadeKeep = (pool, R) => {
|
|
165
|
+
const best = new Map();
|
|
166
|
+
for (const c of pool) if (c._srcRank === 0 && (!best.has(c.repo) || c.ceScore > best.get(c.repo))) best.set(c.repo, c.ceScore);
|
|
167
|
+
const top = new Set([...best.entries()].sort((a, b) => b[1] - a[1]).slice(0, R).map(([r]) => r));
|
|
168
|
+
return pool.filter((c) => c._srcRank === 0 || c._lane === 'rescue' || top.has(c.repo));
|
|
169
|
+
};
|
|
170
|
+
|
|
171
|
+
const runKept = (p, kept) => {
|
|
172
|
+
const ranked = [...kept].sort((a, b) => (b.ceScore ?? -Infinity) - (a.ceScore ?? -Infinity));
|
|
173
|
+
const { results } = selectResults({ query: p.query, ranked, k: p.k });
|
|
174
|
+
return { pairs: kept.length, results };
|
|
175
|
+
};
|
|
176
|
+
const runPolicy = (p, limit) => runKept(p, capRerankPool(p.pool, { limit }).kept);
|
|
177
|
+
|
|
178
|
+
const base = new Map(pools.map((p) => [p.id, runPolicy(p, 0)]));
|
|
179
|
+
|
|
180
|
+
// ── ROOT CAUSE ────────────────────────────────────────────────────────────────────────────────
|
|
181
|
+
// How deep in its OWN store's vector ranking does the winning document sit? This one histogram
|
|
182
|
+
// decides whether any pre-score cap can be safe. If winners clustered at depth 0, a tiny budget
|
|
183
|
+
// would be free. They do not: the cross-encoder routinely promotes a passage its own store
|
|
184
|
+
// ranked 4th or 7th, which means depth carries little information about who wins, which means
|
|
185
|
+
// every pair a depth cut removes is a real chance of removing the answer.
|
|
186
|
+
const depthTop1 = {}, depthTopK = {};
|
|
187
|
+
for (const p of pools) {
|
|
188
|
+
const r = base.get(p.id).results;
|
|
189
|
+
if (!r.length) continue;
|
|
190
|
+
depthTop1[r[0]._srcRank] = (depthTop1[r[0]._srcRank] || 0) + 1;
|
|
191
|
+
for (const x of r) depthTopK[x._srcRank] = (depthTopK[x._srcRank] || 0) + 1;
|
|
192
|
+
}
|
|
193
|
+
const hist = (h) => Object.keys(h).map(Number).sort((a, b) => a - b).map((d) => `depth ${d}: ${h[d]}`).join(' | ');
|
|
194
|
+
|
|
195
|
+
console.log(`\n# cross-encoder pool cap — quality vs pair count`);
|
|
196
|
+
console.log(`corpus: ${pools.length} questions (${pools.filter((p) => p.stratum !== 'probe').length} frozen held-out + ${pools.filter((p) => p.stratum === 'probe').length} probes)`);
|
|
197
|
+
console.log(`uncapped pool: min ${Math.min(...pools.map((p) => p.pooledAll))}, median ${median(pools.map((p) => p.pooledAll))}, max ${Math.max(...pools.map((p) => p.pooledAll))} pairs`);
|
|
198
|
+
console.log(`body-dependent boost not replayed for: ${bodyDependent.length ? bodyDependent.join(', ') : '(none — no @scope/name query in the set)'}`);
|
|
199
|
+
console.log(`\n## where the answer actually lives, in its own store's vector ranking`);
|
|
200
|
+
console.log(` winning document : ${hist(depthTop1)}`);
|
|
201
|
+
console.log(` every top-k hit : ${hist(depthTopK)}`);
|
|
202
|
+
console.log(` (a budget of B pairs across S stores reaches depth B/S. Winners spread across depth`);
|
|
203
|
+
console.log(` means a depth cut drops answers roughly in proportion to what it saves.)\n`);
|
|
204
|
+
console.log('| policy | pairs (median) | top1-same | top-k kept | routed k/n (Wilson lo) | abstain | banner |');
|
|
205
|
+
console.log('|---|---|---|---|---|---|---|');
|
|
206
|
+
|
|
207
|
+
// Every policy the run compares: the shipping depth cap at a range of budgets, plus cascade.
|
|
208
|
+
const policies = [
|
|
209
|
+
...budgets.map((limit) => ({ label: limit === 0 ? 'uncapped' : `depth B=${limit}`, key: `depth-${limit}`, keep: (p) => capRerankPool(p.pool, { limit }).kept })),
|
|
210
|
+
...[4, 8, 12, 16, 24, 32].map((R) => ({ label: `cascade R=${R}`, key: `cascade-${R}`, keep: (p) => cascadeKeep(p.pool, R) })),
|
|
211
|
+
];
|
|
212
|
+
|
|
213
|
+
const detail = {};
|
|
214
|
+
for (const policy of policies) {
|
|
215
|
+
const rows = [];
|
|
216
|
+
let top1Same = 0, keptNum = 0, keptDen = 0;
|
|
217
|
+
const changed = [];
|
|
218
|
+
for (const p of pools) {
|
|
219
|
+
const cur = runKept(p, policy.keep(p));
|
|
220
|
+
const b = base.get(p.id);
|
|
221
|
+
const same = idOf2(cur.results[0]) === idOf2(b.results[0]);
|
|
222
|
+
if (same) top1Same++; else changed.push(`${p.id}: ${idOf2(b.results[0])} -> ${idOf2(cur.results[0])}`);
|
|
223
|
+
const curSet = new Set(cur.results.map(idOf));
|
|
224
|
+
for (const r of b.results) { keptDen++; if (curSet.has(idOf(r))) keptNum++; }
|
|
225
|
+
if (p.stratum !== 'probe') {
|
|
226
|
+
const top = cur.results[0] ?? null;
|
|
227
|
+
// Same grading rule as scripts/eval-brain.mjs, fed from the replayed result set. `grounded`
|
|
228
|
+
// is structurally true here: every candidate is a passage the retriever actually returned.
|
|
229
|
+
rows.push({
|
|
230
|
+
id: p.id, stratum: p.stratum,
|
|
231
|
+
...gradeQuestion({ stratum: p.stratum, expectRepo: p.expectRepo },
|
|
232
|
+
{ grounded: cur.results.length > 0,
|
|
233
|
+
citations: top ? [{ repo: top.repo, fullPath: idOf(top), ce: top.ceScore }] : [],
|
|
234
|
+
bannerPresent: cur.results.some((r) => r.gist) }),
|
|
235
|
+
});
|
|
236
|
+
}
|
|
237
|
+
detail[policy.key] ??= {};
|
|
238
|
+
detail[policy.key][p.id] = { pairs: cur.pairs, top1: idOf2(cur.results[0]) };
|
|
239
|
+
}
|
|
240
|
+
const agg = aggregate(rows);
|
|
241
|
+
const pairs = median(pools.map((p) => policy.keep(p).length));
|
|
242
|
+
console.log(`| ${policy.label} | ${pairs} | ${top1Same}/${pools.length} (${pct(top1Same / pools.length)}) | ${keptNum}/${keptDen} (${pct(keptNum / keptDen)}) | ${agg.routed.k}/${agg.routed.n} (${pct(agg.routed.lo)}) | ${agg.abstain.k}/${agg.abstain.n} | ${agg.banner.k}/${agg.banner.n} |`);
|
|
243
|
+
if (changed.length && changed.length <= 12) detail[`changed@${policy.key}`] = changed;
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
console.log('\n## winners that changed, per policy (empty = the policy changed no answer)');
|
|
247
|
+
for (const policy of policies) {
|
|
248
|
+
if (policy.key === 'depth-0') continue;
|
|
249
|
+
const c = detail[`changed@${policy.key}`];
|
|
250
|
+
console.log(`\n### ${policy.label}`);
|
|
251
|
+
if (!c) console.log(' (too many to list — see the top1-same column)');
|
|
252
|
+
else if (!c.length) console.log(' none');
|
|
253
|
+
else for (const line of c) console.log(` ${line}`);
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
if (argv.includes('--json')) fs.writeFileSync(path.join(TRACES, 'report.json'), JSON.stringify(detail, null, 2));
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
const idOf2 = (r) => (r ? `${r.repo}/${r.path}` : '(no result)');
|
|
260
|
+
const pct = (x) => `${(x * 100).toFixed(1)}%`;
|
|
261
|
+
const median = (a) => { const s = [...a].sort((x, y) => x - y); return s.length % 2 ? s[(s.length - 1) / 2] : Math.round((s[s.length / 2 - 1] + s[s.length / 2]) / 2); };
|
|
262
|
+
|
|
263
|
+
if (argv.includes('--collect')) await collect();
|
|
264
|
+
else if (argv.includes('--report')) await report();
|
|
265
|
+
else { console.error('usage: rerank-cap-eval.mjs --collect | --report'); process.exit(2); }
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// rerank-cap-warm-ab.mjs — the paired, WARM before/after for the cross-encoder pool cap.
|
|
3
|
+
//
|
|
4
|
+
// Why this exists rather than timing the CLI: a cold `forge-ask-all.mjs` spends ~53s loading two
|
|
5
|
+
// ONNX models before it scores anything, and the cap cannot touch that. Timing cold runs would
|
|
6
|
+
// dilute the effect being measured by roughly 3x and would also compare runs taken hours apart on a
|
|
7
|
+
// machine whose load moved underneath them. This harness loads the models ONCE and then runs each
|
|
8
|
+
// question twice in the same process — uncapped and capped — so the only difference between the two
|
|
9
|
+
// numbers is the thing under test.
|
|
10
|
+
//
|
|
11
|
+
// PAIRED AND ORDER-ALTERNATED: question i runs uncapped-then-capped on even i and capped-then-
|
|
12
|
+
// uncapped on odd i, so any residual warm-up or thermal drift cannot systematically favour one arm.
|
|
13
|
+
// Both arms' ANSWERS are recorded, not just their times: a cap that is fast and wrong is a failure,
|
|
14
|
+
// and this is the file that would catch it.
|
|
15
|
+
//
|
|
16
|
+
// TWO POLICIES SHARE THIS HARNESS, because a number is only comparable to another number taken
|
|
17
|
+
// the same way. --cap is ADR-057's flat pool cap (select by vector distance, one full read each).
|
|
18
|
+
// --cascade is ADR-058's two-stage cascade (read every pooled pair at a truncated length, then
|
|
19
|
+
// re-read the top K in full). Same questions, same pairing, same warm process, same table — so
|
|
20
|
+
// "-30.4%" and whatever the cascade measures can be put side by side honestly.
|
|
21
|
+
//
|
|
22
|
+
// node scripts/rerank-cap-warm-ab.mjs --cap 408 [--n 24] [--out result.json]
|
|
23
|
+
// node scripts/rerank-cap-warm-ab.mjs --cascade 64 [--n 24] [--tokens 192]
|
|
24
|
+
|
|
25
|
+
import fs from 'node:fs';
|
|
26
|
+
import os from 'node:os';
|
|
27
|
+
import path from 'node:path';
|
|
28
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
29
|
+
|
|
30
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
31
|
+
const KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
|
|
32
|
+
const argv = process.argv.slice(2);
|
|
33
|
+
const arg = (f, d) => { const i = argv.indexOf(f); return i >= 0 && argv[i + 1] ? argv[i + 1] : d; };
|
|
34
|
+
// Exactly one arm is under test. --cascade wins if both are given, and says so rather than
|
|
35
|
+
// silently measuring a policy the operator did not ask for.
|
|
36
|
+
const CASCADE = arg('--cascade', '');
|
|
37
|
+
const CAP = arg('--cap', CASCADE ? '0' : '408');
|
|
38
|
+
const TOKENS = arg('--tokens', '192');
|
|
39
|
+
const MODE = CASCADE ? 'cascade' : 'cap';
|
|
40
|
+
const LABEL = CASCADE ? `cascade K=${CASCADE} @${TOKENS}tok` : `capped B=${CAP}`;
|
|
41
|
+
const N = parseInt(arg('--n', '24'), 10);
|
|
42
|
+
const OUT = arg('--out', path.join(os.tmpdir(), `ce-${MODE}-warm-ab-${CASCADE || CAP}.json`));
|
|
43
|
+
|
|
44
|
+
const { searchAll } = await import(pathToFileURL(path.join(ROOT, 'kb', 'forge-ask-all.mjs')).href);
|
|
45
|
+
const { gradeQuestion, aggregate } = await import(pathToFileURL(path.join(ROOT, 'scripts', 'eval-brain.mjs')).href);
|
|
46
|
+
|
|
47
|
+
// Stratified subset of the frozen held-out set, dealt round-robin so every stratum is represented
|
|
48
|
+
// even at small n — a prefix of a grouped file would be all 'described' and no 'adversarial'.
|
|
49
|
+
const { questions } = JSON.parse(fs.readFileSync(path.join(ROOT, 'evals', 'held-out.json'), 'utf8'));
|
|
50
|
+
const byStratum = new Map();
|
|
51
|
+
for (const q of questions) (byStratum.get(q.stratum) ?? byStratum.set(q.stratum, []).get(q.stratum)).push(q);
|
|
52
|
+
const lanes = [...byStratum.values()];
|
|
53
|
+
const dealt = [];
|
|
54
|
+
for (let i = 0; dealt.length < questions.length; i++) for (const lane of lanes) if (lane[i]) dealt.push(lane[i]);
|
|
55
|
+
const set = dealt.slice(0, N);
|
|
56
|
+
|
|
57
|
+
const idOf = (r) => (r ? `${r.repo}/${r.path}` : '(none)');
|
|
58
|
+
// `on` selects the arm under test; `off` is always the untouched uncapped, uncascaded path. Both
|
|
59
|
+
// env knobs are set on EVERY call rather than only when engaged — a leftover value from the
|
|
60
|
+
// previous call is exactly how an A/B measures the same arm twice and reports a 0% delta.
|
|
61
|
+
async function once(query, on) {
|
|
62
|
+
process.env.KB_CE_MAX_PAIRS = String(on && MODE === 'cap' ? CAP : 0);
|
|
63
|
+
process.env.KB_CE_CASCADE_K = String(on && MODE === 'cascade' ? CASCADE : 0);
|
|
64
|
+
process.env.KB_CE_CASCADE_TOKENS = String(TOKENS);
|
|
65
|
+
const t0 = Date.now();
|
|
66
|
+
const out = await searchAll({ dir: KB, query, k: 3 });
|
|
67
|
+
return {
|
|
68
|
+
ms: Date.now() - t0, pairs: out.pooled, pooledAll: out.pooledAll,
|
|
69
|
+
prefiltered: out.prefiltered ?? 0, prefilterMs: out.prefilterMs ?? 0,
|
|
70
|
+
results: out.results.map((r) => ({ id: idOf(r), repo: r.repo, ce: r.ceScore, gist: !!r.gist })),
|
|
71
|
+
};
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// Warm-up: the first query of a process pays both model loads. It is thrown away deliberately —
|
|
75
|
+
// including it would credit the cap with a saving it did not produce.
|
|
76
|
+
process.stderr.write(`[warm-ab] mode=${MODE} (${LABEL}) — loading models (first query is discarded)...\n`);
|
|
77
|
+
const w0 = Date.now();
|
|
78
|
+
await once('what is ruvector', false);
|
|
79
|
+
process.stderr.write(`[warm-ab] warm after ${((Date.now() - w0) / 1000).toFixed(1)}s\n`);
|
|
80
|
+
|
|
81
|
+
const rows = [];
|
|
82
|
+
for (let i = 0; i < set.length; i++) {
|
|
83
|
+
const q = set[i];
|
|
84
|
+
const onFirst = i % 2 === 1;
|
|
85
|
+
const a = await once(q.query, onFirst);
|
|
86
|
+
const b = await once(q.query, !onFirst);
|
|
87
|
+
const [off, on] = onFirst ? [b, a] : [a, b];
|
|
88
|
+
rows.push({ id: q.id, stratum: q.stratum, expectRepo: q.expectRepo ?? null, onFirst, off, on });
|
|
89
|
+
process.stderr.write(`[warm-ab] ${i + 1}/${set.length} ${q.id} off=${(off.ms / 1000).toFixed(1)}s/${off.pairs}p on=${(on.ms / 1000).toFixed(1)}s/${on.pairs}p top1${idOf2(off) === idOf2(on) ? '=same' : ' CHANGED'}\n`);
|
|
90
|
+
}
|
|
91
|
+
function idOf2(x) { return x.results[0]?.id ?? '(none)'; }
|
|
92
|
+
|
|
93
|
+
fs.writeFileSync(OUT, JSON.stringify({ mode: MODE, cap: CAP, cascade: CASCADE, tokens: TOKENS, kb: KB, n: set.length, load: os.loadavg(), rows }, null, 2));
|
|
94
|
+
|
|
95
|
+
// ── the table ───────────────────────────────────────────────────────────────────────────────────
|
|
96
|
+
const med = (a) => { const s = [...a].sort((x, y) => x - y); return s.length % 2 ? s[(s.length - 1) / 2] : Math.round((s[s.length / 2 - 1] + s[s.length / 2]) / 2); };
|
|
97
|
+
const pct = (x) => `${(x * 100).toFixed(1)}%`;
|
|
98
|
+
const grade = (arm) => aggregate(rows.map((r) => {
|
|
99
|
+
const top = r[arm].results[0] ?? null;
|
|
100
|
+
return { stratum: r.stratum, ...gradeQuestion({ stratum: r.stratum, expectRepo: r.expectRepo },
|
|
101
|
+
{ grounded: r[arm].results.length > 0,
|
|
102
|
+
citations: top ? [{ repo: top.repo, fullPath: top.id, ce: top.ce }] : [],
|
|
103
|
+
bannerPresent: r[arm].results.some((x) => x.gist) }) };
|
|
104
|
+
}));
|
|
105
|
+
const top1Same = rows.filter((r) => idOf2(r.off) === idOf2(r.on)).length;
|
|
106
|
+
let kn = 0, kd = 0;
|
|
107
|
+
for (const r of rows) { const s = new Set(r.on.results.map((x) => x.id)); for (const x of r.off.results) { kd++; if (s.has(x.id)) kn++; } }
|
|
108
|
+
const gOff = grade('off'), gOn = grade('on');
|
|
109
|
+
|
|
110
|
+
console.log(`\n# warm A/B — ${MODE === 'cascade' ? `cross-encoder CASCADE KB_CE_CASCADE_K=${CASCADE} KB_CE_CASCADE_TOKENS=${TOKENS}` : `cross-encoder pool cap KB_CE_MAX_PAIRS=${CAP}`}`);
|
|
111
|
+
console.log(`${rows.length} questions from the frozen held-out set, paired, order-alternated, one warm process. load1=${os.loadavg()[0].toFixed(1)} on ${os.cpus().length} cores.\n`);
|
|
112
|
+
console.log('| | full reads (median) | warm wall median | warm wall mean | routed | abstain | banner |');
|
|
113
|
+
console.log('|---|---|---|---|---|---|---|');
|
|
114
|
+
const fmt = (arm, g) => `| ${med(rows.map((r) => r[arm].pairs))} | ${(med(rows.map((r) => r[arm].ms)) / 1000).toFixed(2)}s | ${(rows.reduce((a, r) => a + r[arm].ms, 0) / rows.length / 1000).toFixed(2)}s | ${g.routed.k}/${g.routed.n} | ${g.abstain.k}/${g.abstain.n} | ${g.banner.k}/${g.banner.n} |`;
|
|
115
|
+
console.log(`| baseline, no policy (before) ${fmt('off', gOff)}`);
|
|
116
|
+
console.log(`| ${LABEL} (after) ${fmt('on', gOn)}`);
|
|
117
|
+
if (MODE === 'cascade') {
|
|
118
|
+
// Stage 1 is real work and must be visible, or the table reads as if 64 pairs were the whole cost.
|
|
119
|
+
console.log(`\nstage-1 prefilter: ${med(rows.map((r) => r.on.prefiltered))} pairs read at ${TOKENS} tokens, median ${(med(rows.map((r) => r.on.prefilterMs)) / 1000).toFixed(2)}s of the after-time above`);
|
|
120
|
+
}
|
|
121
|
+
const dMed = 1 - med(rows.map((r) => r.on.ms)) / med(rows.map((r) => r.off.ms));
|
|
122
|
+
console.log(`\nwall-time change (median, paired): ${dMed >= 0 ? '-' : '+'}${pct(Math.abs(dMed))}`);
|
|
123
|
+
console.log(`top-1 cited path identical : ${top1Same}/${rows.length} (${pct(top1Same / rows.length)})`);
|
|
124
|
+
console.log(`top-3 cited paths retained : ${kn}/${kd} (${pct(kn / kd)})`);
|
|
125
|
+
console.log('\n## every question whose top-1 changed');
|
|
126
|
+
const changed = rows.filter((r) => idOf2(r.off) !== idOf2(r.on));
|
|
127
|
+
if (!changed.length) console.log(' none');
|
|
128
|
+
for (const r of changed) console.log(` ${r.id} [${r.stratum}] expect=${(r.expectRepo || ['-']).join('|')}\n before: ${idOf2(r.off)} (ce ${r.off.results[0]?.ce?.toFixed(3)})\n after : ${idOf2(r.on)} (ce ${r.on.results[0]?.ce?.toFixed(3)})`);
|
|
129
|
+
console.log(`\nraw: ${OUT}`);
|
package/scripts/route-cheap.mjs
CHANGED
|
@@ -27,6 +27,7 @@ import path from 'node:path';
|
|
|
27
27
|
import os from 'node:os';
|
|
28
28
|
import { spawnSync } from 'node:child_process';
|
|
29
29
|
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
30
|
+
import { loadRuntimePreferences, runtimeChildEnv } from '../plugin/scripts/runtime-preferences.mjs';
|
|
30
31
|
|
|
31
32
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
32
33
|
|
|
@@ -90,20 +91,15 @@ export function receiptLine(model, costs) {
|
|
|
90
91
|
return `\x1b[2m⚡ MetaHarness: routed to ${model} — ~${pct}% cheaper (est. ${fmt$(costs.cost)} vs ${fmt$(costs.frontier)} ${ref}, saved ~${fmt$(costs.saved)})\x1b[0m`;
|
|
91
92
|
}
|
|
92
93
|
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
94
|
+
function agenticFlowExecutable(env) {
|
|
95
|
+
const home = env.HOME || os.homedir();
|
|
96
|
+
const global = path.join(home, '.npm-global', 'bin', process.platform === 'win32' ? 'agentic-flow.cmd' : 'agentic-flow');
|
|
96
97
|
try {
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
if (m && m[1].trim()) {
|
|
100
|
-
process.env.OPENROUTER_API_KEY = m[1].trim().replace(/^["']|["']$/g, '');
|
|
101
|
-
return true;
|
|
102
|
-
}
|
|
98
|
+
fs.accessSync(global, fs.constants.X_OK);
|
|
99
|
+
return global;
|
|
103
100
|
} catch {
|
|
104
|
-
|
|
101
|
+
return 'agentic-flow';
|
|
105
102
|
}
|
|
106
|
-
return false;
|
|
107
103
|
}
|
|
108
104
|
|
|
109
105
|
function parseArgs(argv) {
|
|
@@ -125,16 +121,25 @@ function main() {
|
|
|
125
121
|
console.error(`Unknown model "${args.model}" — no verified pricing, refusing to invent savings. Known: ${Object.keys(PRICING).join(', ')}`);
|
|
126
122
|
process.exit(2);
|
|
127
123
|
}
|
|
128
|
-
|
|
129
|
-
|
|
124
|
+
const policy = loadRuntimePreferences();
|
|
125
|
+
if (policy.values.routing !== 'auto') {
|
|
126
|
+
console.error(policy.values.routing === 'off'
|
|
127
|
+
? 'Token-smart routing is off in RuvNet Brain Console — nothing was dispatched.'
|
|
128
|
+
: 'Token-smart routing has not been enabled in RuvNet Brain Console — nothing was dispatched.');
|
|
129
|
+
process.exit(1);
|
|
130
|
+
}
|
|
131
|
+
const childEnv = runtimeChildEnv();
|
|
132
|
+
if (!childEnv.OPENROUTER_API_KEY) {
|
|
133
|
+
console.error('OPENROUTER_API_KEY is not configured in the environment or encrypted Brain credential store — cannot route. (Key value is never printed.)');
|
|
130
134
|
process.exit(1);
|
|
131
135
|
}
|
|
132
136
|
|
|
133
137
|
const started = Date.now();
|
|
134
|
-
const run = spawnSync(
|
|
138
|
+
const run = spawnSync(agenticFlowExecutable(childEnv), ['--agent', args.agent, '--model', args.model, '--task', args.task], {
|
|
135
139
|
encoding: 'utf8',
|
|
136
140
|
timeout: 180_000,
|
|
137
|
-
env:
|
|
141
|
+
env: childEnv,
|
|
142
|
+
shell: false,
|
|
138
143
|
});
|
|
139
144
|
|
|
140
145
|
if (run.error || run.status !== 0) {
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// scripts/router-utilization.mjs — the ONGOING view. router-optimizer.mjs answers "what SHOULD each
|
|
3
|
+
// bucket route to"; this answers "once you've actually been using it, how many tasks landed in each
|
|
4
|
+
// bucket, and what did you save versus sending every one to the frontier model?"
|
|
5
|
+
//
|
|
6
|
+
// It implements the shape rUv specifies in ruflo ADR-149 §6 (Observability, Proposed):
|
|
7
|
+
// • modelDistribution — per-model/per-band task counts ("how many are going to each bucket")
|
|
8
|
+
// • costOptimalitySaved — USD vs always paying the frontier ("what the savings are off the frontier")
|
|
9
|
+
//
|
|
10
|
+
// Pure/deterministic: it reads the real receipts ledger (route-cheap.mjs + subagent dispatch write it)
|
|
11
|
+
// and makes NO model or network call. Every number is grounded:
|
|
12
|
+
// • realized cost = the receipt's own recorded est_cost (what the routed model actually cost).
|
|
13
|
+
// • frontier baseline = RECOMPUTED for every receipt against the CURRENT frontier model
|
|
14
|
+
// (route-cheap FRONTIER — now Fable 5) from that receipt's own est token counts. Recomputing (rather
|
|
15
|
+
// than trusting each receipt's stored est_frontier_cost) means switching the frontier re-prices the
|
|
16
|
+
// whole history consistently, so the "vs Fable 5" figure is never a stale Opus-era number.
|
|
17
|
+
// • a receipt whose model has no verified price is counted under `unpriced` and left OUT of the $ math
|
|
18
|
+
// — never assigned an invented cost.
|
|
19
|
+
//
|
|
20
|
+
// Usage: node scripts/router-utilization.mjs [--json]
|
|
21
|
+
|
|
22
|
+
import fs from 'node:fs';
|
|
23
|
+
import { FRONTIER, priceOf, receiptsPath } from './route-cheap.mjs';
|
|
24
|
+
|
|
25
|
+
// Band = a human-legible grouping over the continuous complexity/cost axis (ADR-149: "tier_label is
|
|
26
|
+
// metadata, not control flow"). The four bands mirror router-optimizer.mjs. Known models are mapped
|
|
27
|
+
// explicitly; anything unknown is bucketed by its blended $/Mtok via bandOf()'s fallback.
|
|
28
|
+
const BAND_BY_MODEL = {
|
|
29
|
+
'agent-booster': 'mechanical',
|
|
30
|
+
'inclusionai/ling-2.6-flash': 'cheap',
|
|
31
|
+
'claude-haiku-4.5': 'cheap',
|
|
32
|
+
'deepseek/deepseek-chat': 'cheap',
|
|
33
|
+
'deepseek/deepseek-v4-flash': 'cheap',
|
|
34
|
+
'meta-llama/llama-3.3-70b-instruct': 'mid',
|
|
35
|
+
'openai/gpt-4.1': 'mid',
|
|
36
|
+
'x-ai/grok-4.5': 'mid',
|
|
37
|
+
'claude-sonnet-5': 'mid',
|
|
38
|
+
'claude-opus-4.8': 'frontier',
|
|
39
|
+
'claude-fable-5': 'frontier',
|
|
40
|
+
};
|
|
41
|
+
export const BAND_ORDER = ['mechanical', 'cheap', 'mid', 'frontier'];
|
|
42
|
+
const BAND_LABEL = { mechanical: 'Mechanical', cheap: 'Cheap', mid: 'Mid', frontier: 'Frontier' };
|
|
43
|
+
|
|
44
|
+
/** Which band did this model's task land in? Known → explicit map; unknown → blended-price fallback. */
|
|
45
|
+
export function bandOf(model) {
|
|
46
|
+
if (BAND_BY_MODEL[model]) return BAND_BY_MODEL[model];
|
|
47
|
+
const p = priceOf(model);
|
|
48
|
+
if (!p) return null; // unpriced → caller records it under `unpriced`, never invents a cost
|
|
49
|
+
const blended = (p.in + p.out) / 2;
|
|
50
|
+
if (blended === 0) return 'mechanical';
|
|
51
|
+
if (blended < 2) return 'cheap';
|
|
52
|
+
if (blended < 15) return 'mid';
|
|
53
|
+
return 'frontier';
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
function tokensOf(r) {
|
|
57
|
+
const i = Number(r.est_in_tokens);
|
|
58
|
+
const o = Number(r.est_out_tokens);
|
|
59
|
+
if (Number.isFinite(i) && Number.isFinite(o)) return { in: i, out: o };
|
|
60
|
+
return null;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Summarize the receipts ledger into a per-band distribution + realized-vs-frontier savings.
|
|
65
|
+
* @param {{ receiptsFile?: string }} [opts]
|
|
66
|
+
*/
|
|
67
|
+
export function utilization({ receiptsFile, frontier } = {}) {
|
|
68
|
+
const file = receiptsFile || receiptsPath();
|
|
69
|
+
// Frontier = the user's OWN house flagship (passed in by the console from the live-verified catalog);
|
|
70
|
+
// falls back to route-cheap's default. The counterfactual is recomputed vs THIS, so an OpenAI shop sees
|
|
71
|
+
// "vs GPT-5.6 Sol" and a Claude shop sees "vs Fable 5" — personalized, never one house for everyone.
|
|
72
|
+
const frModel = frontier?.model || FRONTIER.name;
|
|
73
|
+
const fr = (frontier && Number.isFinite(frontier.in) && Number.isFinite(frontier.out))
|
|
74
|
+
? { in: frontier.in, out: frontier.out }
|
|
75
|
+
: priceOf(FRONTIER.name);
|
|
76
|
+
|
|
77
|
+
const bands = Object.fromEntries(
|
|
78
|
+
BAND_ORDER.map((b) => [b, { band: b, label: BAND_LABEL[b], tasks: 0, realizedUsd: 0, frontierUsd: 0, models: {} }])
|
|
79
|
+
);
|
|
80
|
+
let tasks = 0;
|
|
81
|
+
let unpriced = 0;
|
|
82
|
+
let realizedUsd = 0;
|
|
83
|
+
let frontierUsd = 0;
|
|
84
|
+
let since = null;
|
|
85
|
+
let until = null;
|
|
86
|
+
|
|
87
|
+
let lines = [];
|
|
88
|
+
try {
|
|
89
|
+
lines = fs.readFileSync(file, 'utf8').split('\n');
|
|
90
|
+
} catch {
|
|
91
|
+
/* no receipts yet — return the empty shape below */
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
for (const line of lines) {
|
|
95
|
+
if (!line.trim()) continue;
|
|
96
|
+
let r;
|
|
97
|
+
try { r = JSON.parse(line); } catch { continue; }
|
|
98
|
+
if (!r.model) continue;
|
|
99
|
+
tasks++;
|
|
100
|
+
if (r.ts) {
|
|
101
|
+
since = since && since < r.ts ? since : r.ts;
|
|
102
|
+
until = until && until > r.ts ? until : r.ts;
|
|
103
|
+
}
|
|
104
|
+
const band = bandOf(r.model);
|
|
105
|
+
const p = priceOf(r.model);
|
|
106
|
+
const isMech = band === 'mechanical';
|
|
107
|
+
// Mechanical ($0, no LLM — e.g. Agent Booster) has no per-token price but a real realized cost of $0.
|
|
108
|
+
// Everything else needs a verified price AND a priced frontier, or it's excluded (never invented).
|
|
109
|
+
if (!band || (!isMech && (!p || !fr))) { unpriced++; continue; }
|
|
110
|
+
|
|
111
|
+
const tok = tokensOf(r);
|
|
112
|
+
const realized = isMech ? 0
|
|
113
|
+
: Number.isFinite(Number(r.est_cost)) ? Number(r.est_cost)
|
|
114
|
+
: tok ? (tok.in * p.in + tok.out * p.out) / 1e6 : 0;
|
|
115
|
+
// Frontier counterfactual — ALWAYS recomputed vs the current frontier from the receipt's tokens.
|
|
116
|
+
const front = (tok && fr) ? (tok.in * fr.in + tok.out * fr.out) / 1e6 : realized;
|
|
117
|
+
|
|
118
|
+
const b = bands[band];
|
|
119
|
+
b.tasks++;
|
|
120
|
+
b.realizedUsd += realized;
|
|
121
|
+
b.frontierUsd += front;
|
|
122
|
+
b.models[r.model] = (b.models[r.model] || 0) + 1;
|
|
123
|
+
realizedUsd += realized;
|
|
124
|
+
frontierUsd += front;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
const round = (n) => +Number(n).toFixed(4);
|
|
128
|
+
const distribution = BAND_ORDER.map((b) => {
|
|
129
|
+
const x = bands[b];
|
|
130
|
+
return {
|
|
131
|
+
band: b,
|
|
132
|
+
label: x.label,
|
|
133
|
+
tasks: x.tasks,
|
|
134
|
+
pctOfTasks: tasks ? Math.round((x.tasks / tasks) * 100) : 0,
|
|
135
|
+
realizedUsd: round(x.realizedUsd),
|
|
136
|
+
frontierUsd: round(x.frontierUsd),
|
|
137
|
+
savedUsd: round(x.frontierUsd - x.realizedUsd),
|
|
138
|
+
models: Object.entries(x.models)
|
|
139
|
+
.sort((a, c) => c[1] - a[1])
|
|
140
|
+
.map(([model, n]) => ({ model, tasks: n })),
|
|
141
|
+
};
|
|
142
|
+
});
|
|
143
|
+
|
|
144
|
+
const savedUsd = round(frontierUsd - realizedUsd);
|
|
145
|
+
return {
|
|
146
|
+
generatedAt: new Date().toISOString(),
|
|
147
|
+
frontierModel: frModel,
|
|
148
|
+
tasks,
|
|
149
|
+
unpriced,
|
|
150
|
+
since,
|
|
151
|
+
until,
|
|
152
|
+
realizedUsd: round(realizedUsd),
|
|
153
|
+
frontierUsd: round(frontierUsd),
|
|
154
|
+
costOptimalitySaved: savedUsd, // ADR-149 name
|
|
155
|
+
pctSaved: frontierUsd > 0 ? Math.round((savedUsd / frontierUsd) * 100) : null,
|
|
156
|
+
distribution, // ADR-149 modelDistribution, grouped into the four legible bands
|
|
157
|
+
note: `Recomputed live from ${tasks} receipt(s) against your frontier (${frModel}); token counts are the receipts’ own est. values, never projected.`,
|
|
158
|
+
};
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
export function printUtil(u) {
|
|
162
|
+
const money = (v) => (v == null ? '—' : v === 0 ? '$0' : '$' + v);
|
|
163
|
+
console.log(`\nRouting utilization — ${u.tasks} task(s), frontier = ${u.frontierModel}`);
|
|
164
|
+
if (u.since) console.log(` window: ${u.since} → ${u.until}`);
|
|
165
|
+
console.log(` ${'band'.padEnd(11)} ${'tasks'.padStart(6)} ${'%'.padStart(4)} ${'realized'.padStart(11)} ${'vs frontier'.padStart(12)} ${'saved'.padStart(11)}`);
|
|
166
|
+
for (const d of u.distribution) {
|
|
167
|
+
console.log(` ${d.label.padEnd(11)} ${String(d.tasks).padStart(6)} ${String(d.pctOfTasks).padStart(3)}% ${money(d.realizedUsd).padStart(11)} ${money(d.frontierUsd).padStart(12)} ${money(d.savedUsd).padStart(11)}`);
|
|
168
|
+
}
|
|
169
|
+
console.log(` ${'—'.repeat(11)}`);
|
|
170
|
+
console.log(` TOTAL saved vs all-frontier: ${money(u.costOptimalitySaved)} (${u.pctSaved == null ? '—' : u.pctSaved + '%'}); realized ${money(u.realizedUsd)} vs ${money(u.frontierUsd)} on ${u.frontierModel}.`);
|
|
171
|
+
if (u.unpriced) console.log(` (${u.unpriced} receipt(s) had an unpriced model and were excluded from the $ math.)`);
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
export function mainUtil() {
|
|
175
|
+
const u = utilization();
|
|
176
|
+
if (process.argv.includes('--json')) console.log(JSON.stringify(u, null, 2));
|
|
177
|
+
else printUtil(u);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
import { fileURLToPath } from 'node:url';
|
|
181
|
+
import path from 'node:path';
|
|
182
|
+
if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) mainUtil();
|