ruvnet-brain 4.0.1 → 4.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -0
- package/README.md +4 -4
- package/bin/install.mjs +303 -24
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +8 -22
- package/plugin/.claude-plugin/marketplace.json +1 -0
- package/plugin/.claude-plugin/plugin.json +2 -3
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +2 -2
- package/plugin/commands/configure.md +3 -2
- package/plugin/commands/rvbc.md +4 -3
- package/plugin/commands/rvcb.md +2 -2
- package/plugin/commands/whats-new.md +6 -6
- package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
- package/plugin/hooks/hooks.json +1 -2
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +90 -32
- package/plugin/scripts/detach.mjs +14 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/ground-ruvnet.sh +16 -6
- package/plugin/scripts/hook-shim.mjs +34 -29
- package/plugin/scripts/learn-capture.sh +22 -3
- package/plugin/scripts/learn-flush.mjs +21 -4
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +503 -0
- package/plugin/scripts/session-start.sh +3 -858
- package/plugin/scripts/whats-new.mjs +42 -0
- package/plugin/skills/brain-console/SKILL.md +4 -2
- package/plugin/skills/release-proof/SKILL.md +98 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
- package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
- package/plugin/skills/rvbc/SKILL.md +9 -6
- package/plugin/skills/whats-new/SKILL.md +4 -4
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +531 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/fix-workstream.mjs +291 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +237 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +558 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +326 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +172 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2922 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/protected-release-invocation.mjs +76 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/publication-receipt.mjs +307 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-authority.mjs +93 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +439 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +296 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stabilization-receipt.mjs +108 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +867 -0
- package/plugin/scripts/finalize-token-meter.mjs +0 -25
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// scripts/behavioral-l1-l4.mjs — the 4-level BEHAVIORAL test harness for the RuvNet brain.
|
|
3
|
+
//
|
|
4
|
+
// The existing eval scripts (prove.mjs, brain-capability-check.mjs, brain-grade-groundtruth.mjs)
|
|
5
|
+
// test RETRIEVAL: does the right source come back, is it confident, is the answer graded high. This
|
|
6
|
+
// harness tests the four BEHAVIORS a user actually depends on — including the one the brain is meant
|
|
7
|
+
// to DRIVE (orchestration), which no prior script covered:
|
|
8
|
+
//
|
|
9
|
+
// L1 ROUTE — "which repo solves X?" → the brain routes the top hit to the correct repo.
|
|
10
|
+
// L2 DEEP-RECALL — "how is X implemented?" → the brain returns CODE-level source (full bodies),
|
|
11
|
+
// not just doc-comments — proof it walked the repo to the implementation.
|
|
12
|
+
// L3 IMPLEMENT — "implement X using <repo>" → the #1 cited source actually contains the API you
|
|
13
|
+
// would build against, so an implementation grounded in it is correct (mechanical
|
|
14
|
+
// correctness proxy; full multi-vendor grading stays brain-grade-groundtruth.mjs).
|
|
15
|
+
// L4 ORCHESTRATE — the newbie path. Run the real hooks — session-start.sh ONCE per harness run
|
|
16
|
+
// (since ADR-0011 Phase 2 it carries THE PLAYBOOK: ground → SPARC → DDD/ADR →
|
|
17
|
+
// parallel swarm → QA → score ≥98 → frontend-design → AI image-gen →
|
|
18
|
+
// ask-for-API-key → prove) AND the per-turn UserPromptSubmit hook
|
|
19
|
+
// (ground-ruvnet.sh) on "go make magic / here's a spec, build it" prompts — and
|
|
20
|
+
// assert every directive appears at the layer that now owns it (see the ADR-0011
|
|
21
|
+
// Phase 2 relocation note above the L4 table). Hook injection is the only
|
|
22
|
+
// enforcement primitive that exists (ADR-0005), so testing the injected text IS
|
|
23
|
+
// testing whether the brain takes the wheel.
|
|
24
|
+
//
|
|
25
|
+
// L1–L3 call searchAll() — the EXACT engine search_ruvnet wraps (forge-mcp-all.mjs:77). L4 shells the
|
|
26
|
+
// real hooks. No mutation, read-only. Needs KB_MODEL_CACHE pointing at the ONNX model cache for L1–L3.
|
|
27
|
+
//
|
|
28
|
+
// Usage:
|
|
29
|
+
// KB_MODEL_CACHE=<cache> node scripts/behavioral-l1-l4.mjs [--dir kb] [--repos a,b,c] [--levels L1,L4]
|
|
30
|
+
// --repos restrict the search pool (collision-safe while a build sweep is running)
|
|
31
|
+
// --levels comma list of L1,L2,L3,L4 to run (default all)
|
|
32
|
+
// Exit 0 iff every selected check passes.
|
|
33
|
+
|
|
34
|
+
import { execFileSync } from 'node:child_process';
|
|
35
|
+
import fs from 'node:fs';
|
|
36
|
+
import os from 'node:os';
|
|
37
|
+
import path from 'node:path';
|
|
38
|
+
import { fileURLToPath } from 'node:url';
|
|
39
|
+
import { searchAll } from '../kb/forge-ask-all.mjs';
|
|
40
|
+
|
|
41
|
+
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
42
|
+
const ROOT = path.resolve(__dirname, '..');
|
|
43
|
+
const arg = (f, d) => { const i = process.argv.indexOf(f); return i >= 0 ? process.argv[i + 1] : d; };
|
|
44
|
+
const KB_DIR = path.resolve(arg('--dir', path.join(ROOT, 'kb')));
|
|
45
|
+
const HOOK = path.resolve(arg('--hook', path.join(ROOT, 'plugin/scripts/ground-ruvnet.sh')));
|
|
46
|
+
const SESSION_HOOK = path.resolve(arg('--session-hook', path.join(ROOT, 'plugin/scripts/session-start.sh')));
|
|
47
|
+
const reposArg = arg('--repos', '');
|
|
48
|
+
const REPOS = reposArg ? reposArg.split(',').map((s) => s.trim()).filter(Boolean) : undefined;
|
|
49
|
+
const K = parseInt(arg('--k', '6'), 10);
|
|
50
|
+
const LEVELS = new Set((arg('--levels', 'L1,L2,L3,L4')).split(',').map((s) => s.trim().toUpperCase()));
|
|
51
|
+
|
|
52
|
+
const CODE_RX = /(\bfn\s|\bfunction\s|\bimpl\s|\bstruct\s|\bclass\s|\bexport\s|\basync\s|=>|\(full body\))/;
|
|
53
|
+
const G = (s) => `\x1b[32m${s}\x1b[0m`, R = (s) => `\x1b[31m${s}\x1b[0m`, DIM = (s) => `\x1b[2m${s}\x1b[0m`;
|
|
54
|
+
|
|
55
|
+
// ── L1 ROUTE: question → repo it should route to ──────────────────────────────────────────────
|
|
56
|
+
const L1 = [
|
|
57
|
+
{ q: 'single-file HNSW vector store with no server', expect: 'ruvector' },
|
|
58
|
+
{ q: 'orchestrate a swarm of agents across multiple steps', expect: 'ruflo' },
|
|
59
|
+
{ q: 'causal explainable agent memory — why did I recall that', expect: 'agentdb' },
|
|
60
|
+
{ q: 'fork a million vectors cheaply, branch agent memory copy-on-write', expect: 'agenticow' },
|
|
61
|
+
{ q: 'WiFi CSI sensing of vital signs on an edge device', expect: 'ruview' },
|
|
62
|
+
];
|
|
63
|
+
// ── L2 DEEP-RECALL: must return CODE (full bodies) from the right repo, not doc-comments ──────────
|
|
64
|
+
const L2 = [
|
|
65
|
+
{ q: 'how is the HNSW graph insert / neighbour selection implemented', expect: 'ruvector' },
|
|
66
|
+
{ q: 'how does the swarm decide how many agents to spawn in code', expect: 'ruflo' },
|
|
67
|
+
{ q: 'how is a copy-on-write branch created in code', expect: 'agenticow' },
|
|
68
|
+
];
|
|
69
|
+
// ── L3 IMPLEMENT: #1 cited source must contain the API you'd build against ────────────────────────
|
|
70
|
+
const L3 = [
|
|
71
|
+
{ q: 'implement a nearest-neighbour query against an .rvf store', expect: 'ruvector', apiRx: /query|search|knn|nearest/i },
|
|
72
|
+
{ q: 'store and recall an agent memory entry with AgentDB', expect: 'agentdb', apiRx: /store|recall|memory|insert|search/i },
|
|
73
|
+
];
|
|
74
|
+
// ── L4 ORCHESTRATE: run the real hook; assert the injected directive is complete ──────────────────
|
|
75
|
+
const L4 = [
|
|
76
|
+
{
|
|
77
|
+
name: 'newbie "go make magic"',
|
|
78
|
+
prompt: 'go make magic and build me an app that helps people learn',
|
|
79
|
+
// pure build prompt, no RuvNet keyword → Gate 3 (take-the-wheel) must fire with the full pipeline
|
|
80
|
+
must: ['take the wheel', 'SPARC', 'DDD', 'ADR', 'swarm', 'QA gate', '98', 'frontend-design', 'image generation', 'API key', 'PROVEN', 'PARALLEL'],
|
|
81
|
+
},
|
|
82
|
+
{
|
|
83
|
+
name: 'spec → build (RuvNet-named)',
|
|
84
|
+
prompt: 'here is a product spec, build a knowledge base feature using ruvector',
|
|
85
|
+
// names ruvector → Gate 1 (ground) AND Gate 3 (build) must both fire
|
|
86
|
+
must: ['search_ruvnet', 'ground', 'SPARC', 'DDD', 'frontend-design', 'image generation', 'API key', '98'],
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
name: 'classical-default drift',
|
|
90
|
+
prompt: 'set up pinecone and langchain to do rag over my docs',
|
|
91
|
+
// drift keywords → Gate 2 (hijack) must fire AND build keyword "set up" → Gate 3
|
|
92
|
+
must: ['classical default', 'RuVector', 'Ruflo', 'AgentDB', 'take the wheel'],
|
|
93
|
+
},
|
|
94
|
+
{
|
|
95
|
+
name: 'pure recall (no build)',
|
|
96
|
+
prompt: 'what can ruflo actually do',
|
|
97
|
+
must: ['search_ruvnet', 'ground'],
|
|
98
|
+
mustNot: ['take the wheel'], // no build verb → Gate 3 must stay silent
|
|
99
|
+
},
|
|
100
|
+
];
|
|
101
|
+
|
|
102
|
+
const results = { L1: [], L2: [], L3: [], L4: [] };
|
|
103
|
+
|
|
104
|
+
async function runL1() {
|
|
105
|
+
for (const t of L1) {
|
|
106
|
+
const { results: r } = await searchAll({ dir: KB_DIR, query: t.q, k: K, repos: REPOS });
|
|
107
|
+
const top = r[0];
|
|
108
|
+
// Routing (breadth) is correct if the #1 hit (a) IS the expected repo, (b) is a `concepts`
|
|
109
|
+
// capability-card naming it (the by-description router that took routing 33%→96%), or (c) is a
|
|
110
|
+
// doc explicitly ABOUT it in a sibling repo (e.g. a cognitum-cogs page packaging ruview) — all
|
|
111
|
+
// three correctly point the user at the right building block. Annotated transparently below.
|
|
112
|
+
const named = !!top && new RegExp(`\\b${t.expect}\\b`, 'i').test(`${top.path} ${top.title} ${(top.fullText || '').slice(0, 400)}`);
|
|
113
|
+
const viaCard = !!top && top.repo === 'concepts' && named;
|
|
114
|
+
const viaTopic = !!top && top.repo !== t.expect && top.repo !== 'concepts' && named;
|
|
115
|
+
const pass = !!top && (top.repo === t.expect || viaCard || viaTopic);
|
|
116
|
+
const tag = !top ? '' : top.repo === t.expect ? '' : viaCard ? ' via card' : viaTopic ? ` via ${top.repo} doc` : '';
|
|
117
|
+
results.L1.push({ pass, info: `${t.q.slice(0, 38)}… → ${top ? top.repo + '/' + (top.path || '').split('/').pop() : 'none'} (want ${t.expect}${tag}; ce ${top?.ceScore?.toFixed(2) ?? 'n/a'})` });
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
async function runL2() {
|
|
121
|
+
// DEPTH is scoped to the target repo: "does repo X actually contain the implementation of Y?"
|
|
122
|
+
// (full-ecosystem routing is L1's job; here we ask the repo directly so cards/sibling docs can't
|
|
123
|
+
// mask whether the code is indexed). k bumped a bit — deep-recall may sit below overview chunks.
|
|
124
|
+
for (const t of L2) {
|
|
125
|
+
const { results: r } = await searchAll({ dir: KB_DIR, query: t.q, k: 10, repos: [t.expect] });
|
|
126
|
+
const hit = r.find((x) => CODE_RX.test(x.fullText || x.text || ''));
|
|
127
|
+
const pass = !!hit;
|
|
128
|
+
results.L2.push({ pass, info: `${t.q.slice(0, 42)}… → ${pass ? `CODE @ ${t.expect}/${hit.path}` : r[0] ? `doc-only @ ${t.expect}/${r[0].path}` : `no ${t.expect} hit`}` });
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
async function runL3() {
|
|
132
|
+
// Implementability scoped to the target repo: its #1 source for the task must contain the API.
|
|
133
|
+
for (const t of L3) {
|
|
134
|
+
const { results: r } = await searchAll({ dir: KB_DIR, query: t.q, k: K, repos: [t.expect] });
|
|
135
|
+
const top = r[0];
|
|
136
|
+
const apiOk = top && t.apiRx.test(top.fullText || top.text || '');
|
|
137
|
+
const pass = !!top && !!apiOk;
|
|
138
|
+
results.L3.push({ pass, info: `${t.q.slice(0, 40)}… → #1 ${top ? `${t.expect}/${top.path}` : 'none'} apiMatch=${!!apiOk}` });
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
function runL4() {
|
|
142
|
+
// Phase 2 (2026-07-09) relocated the full playbook into session-start.sh (once per session); the
|
|
143
|
+
// per-turn hook carries only a slim reminder. So `must` markers are asserted against the UNION of
|
|
144
|
+
// session-start output (captured once, exactly as a real session receives it) + the scenario's
|
|
145
|
+
// per-turn hook output. `mustNot` leak checks stay on the PER-TURN output alone — session-start
|
|
146
|
+
// legitimately always contains the playbook; the leak under test is a non-build turn wrongly
|
|
147
|
+
// triggering build directives.
|
|
148
|
+
const hookEnv = { ...process.env, CLAUDE_PLUGIN_ROOT: path.join(ROOT, 'plugin') };
|
|
149
|
+
let session = '';
|
|
150
|
+
try { session = execFileSync('bash', [SESSION_HOOK], { input: '', encoding: 'utf8', env: hookEnv }); } catch (e) { session = (e.stdout || ''); }
|
|
151
|
+
const sessionLc = session.toLowerCase();
|
|
152
|
+
for (const t of L4) {
|
|
153
|
+
let out = '';
|
|
154
|
+
try { out = execFileSync('sh', [HOOK], { input: JSON.stringify({ prompt: t.prompt }), encoding: 'utf8', env: hookEnv }); } catch (e) { out = (e.stdout || '') + (e.stderr || ''); }
|
|
155
|
+
const lc = out.toLowerCase();
|
|
156
|
+
const union = `${lc}\n${sessionLc}`;
|
|
157
|
+
const missing = (t.must || []).filter((m) => !union.includes(m.toLowerCase()));
|
|
158
|
+
const leaked = (t.mustNot || []).filter((m) => lc.includes(m.toLowerCase()));
|
|
159
|
+
const pass = missing.length === 0 && leaked.length === 0;
|
|
160
|
+
results.L4.push({ pass, info: `${t.name}${missing.length ? ` — MISSING: ${missing.join(', ')}` : ''}${leaked.length ? ` — LEAKED: ${leaked.join(', ')}` : ''}` });
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
(async () => {
|
|
165
|
+
console.log(`\n=== RuvNet Brain — L1–L4 behavioral harness ===`);
|
|
166
|
+
console.log(DIM(`kb=${KB_DIR} pool=${REPOS ? REPOS.join(',') : 'ALL'} hook=${path.relative(ROOT, HOOK)}\n`));
|
|
167
|
+
if (LEVELS.has('L1')) await runL1();
|
|
168
|
+
if (LEVELS.has('L2')) await runL2();
|
|
169
|
+
if (LEVELS.has('L3')) await runL3();
|
|
170
|
+
if (LEVELS.has('L4')) runL4();
|
|
171
|
+
|
|
172
|
+
// VACUOUS TRUTH IS NOT A PASS (2026-07-28). `allPass` started true and the loop below `continue`s
|
|
173
|
+
// over any level with no results, so `--levels L5` — a level that does not exist — ran ZERO checks
|
|
174
|
+
// and printed OVERALL: PASS, exit 0. GPT-5.6-Sol reproduced it during the Gen-2 QE grading and I
|
|
175
|
+
// reproduced it again here before fixing. A harness that certifies an empty run is worse than no
|
|
176
|
+
// harness: it is the mechanism by which README:484/526 could claim "L1-L4 behavioral harness -
|
|
177
|
+
// all pass" as evidence the hook "drives the full pipeline", while two independent graders scored
|
|
178
|
+
// the QE apparatus 38/100 and 53/100. Nothing could contradict the optimistic claim, because the
|
|
179
|
+
// thing meant to contradict it passed by running nothing.
|
|
180
|
+
//
|
|
181
|
+
// Two guards: an unknown level is an ERROR (not a silent no-op), and zero executed checks is a
|
|
182
|
+
// FAIL. "I checked nothing and found no problems" is the sentence this exists to make impossible.
|
|
183
|
+
const KNOWN = ['L1', 'L2', 'L3', 'L4'];
|
|
184
|
+
const unknown = [...LEVELS].filter((l) => !KNOWN.includes(l));
|
|
185
|
+
if (unknown.length) {
|
|
186
|
+
console.log(`${R('ERROR')} unknown level(s): ${unknown.join(', ')} — known: ${KNOWN.join(', ')}`);
|
|
187
|
+
console.log(`=== OVERALL: ${R('FAIL')} (nothing was verified) ===\n`);
|
|
188
|
+
process.exit(2);
|
|
189
|
+
}
|
|
190
|
+
const executed = KNOWN.reduce((n, l) => n + (results[l]?.length || 0), 0);
|
|
191
|
+
if (executed === 0) {
|
|
192
|
+
console.log(`${R('ERROR')} ZERO checks executed — an empty run is not a pass.`);
|
|
193
|
+
console.log(`=== OVERALL: ${R('FAIL')} (nothing was verified) ===\n`);
|
|
194
|
+
process.exit(2);
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
let allPass = true;
|
|
198
|
+
const titles = { L1: 'ROUTE', L2: 'DEEP-RECALL', L3: 'IMPLEMENT', L4: 'ORCHESTRATE' };
|
|
199
|
+
for (const lvl of ['L1', 'L2', 'L3', 'L4']) {
|
|
200
|
+
if (!LEVELS.has(lvl) || !results[lvl].length) continue;
|
|
201
|
+
const passN = results[lvl].filter((x) => x.pass).length;
|
|
202
|
+
const ok = passN === results[lvl].length;
|
|
203
|
+
allPass = allPass && ok;
|
|
204
|
+
console.log(`${ok ? G('PASS') : R('FAIL')} ${lvl} ${titles[lvl]} (${passN}/${results[lvl].length})`);
|
|
205
|
+
for (const x of results[lvl]) console.log(` ${x.pass ? G('✓') : R('✗')} ${x.info}`);
|
|
206
|
+
console.log();
|
|
207
|
+
}
|
|
208
|
+
console.log(`=== OVERALL: ${allPass ? G('PASS') : R('FAIL')} ===\n`);
|
|
209
|
+
process.exit(allPass ? 0 : 1);
|
|
210
|
+
})();
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// brain-capability-check.mjs — the "never doubt a real capability" gate.
|
|
3
|
+
//
|
|
4
|
+
// Stuart's #1 ask: Claude Code must STOP guessing "I don't think RuvNet can do X" when X exists in
|
|
5
|
+
// Ruv's real source. This measures exactly that. For each "Can <repo> do <Y>?" question — where Y is
|
|
6
|
+
// a REAL capability — the bundle must return a CONFIDENTLY-relevant source document (positive cross-
|
|
7
|
+
// encoder relevance) from the right repo whose text actually evidences the capability. A weak/negative
|
|
8
|
+
// top hit is precisely where Claude would have doubted → a measured FAIL to drive to zero.
|
|
9
|
+
//
|
|
10
|
+
// It also supports CONTROL questions (truth:"no") — capabilities a repo genuinely does NOT have — to
|
|
11
|
+
// confirm the bundle does not hallucinate capabilities either (it should NOT return a confident,
|
|
12
|
+
// evidence-bearing hit for those).
|
|
13
|
+
//
|
|
14
|
+
// node scripts/brain-capability-check.mjs --dir kb --questions kb/capability.agentdb.json [--tau 0] [--k 5] [--pool 12]
|
|
15
|
+
import fs from 'node:fs';
|
|
16
|
+
import path from 'node:path';
|
|
17
|
+
import { fileURLToPath } from 'node:url';
|
|
18
|
+
import { searchAll } from '../kb/forge-ask-all.mjs';
|
|
19
|
+
|
|
20
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
21
|
+
const arg = (f, d) => { const i = process.argv.indexOf(f); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
|
|
22
|
+
const DIR = path.resolve(ROOT, arg('--dir', 'kb'));
|
|
23
|
+
const QFILE = path.resolve(ROOT, arg('--questions', 'kb/capability.agentdb.json'));
|
|
24
|
+
const TAU = parseFloat(arg('--tau', '0')); // min cross-encoder relevance to count as "confident"
|
|
25
|
+
const K = parseInt(arg('--k', '5'), 10);
|
|
26
|
+
const POOL = parseInt(arg('--pool', '12'), 10);
|
|
27
|
+
|
|
28
|
+
const spec = JSON.parse(fs.readFileSync(QFILE, 'utf8'));
|
|
29
|
+
const questions = spec.questions || spec;
|
|
30
|
+
|
|
31
|
+
// A capability is "evidenced + confident" if, within the top-k cross-repo hits, some hit is from the
|
|
32
|
+
// expected repo (when specified), scores >= TAU, and its text contains >= half of the evidence terms.
|
|
33
|
+
function evaluate(results, qq) {
|
|
34
|
+
const ev = (qq.evidence || []).map((s) => s.toLowerCase());
|
|
35
|
+
const need = Math.max(1, Math.ceil(ev.length / 2));
|
|
36
|
+
let best = null;
|
|
37
|
+
for (const r of results) {
|
|
38
|
+
// accept hits from the expected repo, OR a concepts-store hit attributed to it via its path prefix
|
|
39
|
+
if (qq.expectRepo && r.repo !== qq.expectRepo && !(r.repo === 'concepts' && (r.path || '').startsWith(qq.expectRepo + '/'))) continue;
|
|
40
|
+
const txt = (r.fullText || r.text || '').toLowerCase();
|
|
41
|
+
const hits = ev.filter((t) => txt.includes(t)).length;
|
|
42
|
+
const confident = (r.ceScore ?? -Infinity) >= TAU;
|
|
43
|
+
const cand = { repo: r.repo, path: r.path, ce: r.ceScore, evHits: hits, evNeed: need, confident, evidenced: hits >= need };
|
|
44
|
+
if (!best || (cand.confident && cand.evidenced && !(best.confident && best.evidenced)) || (r.ceScore ?? -Infinity) > (best.ce ?? -Infinity)) {
|
|
45
|
+
if (!best || (r.ceScore ?? -Infinity) > (best.ce ?? -Infinity) || (cand.confident && cand.evidenced)) best = cand;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
return best;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
const rows = [];
|
|
52
|
+
for (let i = 0; i < questions.length; i++) {
|
|
53
|
+
const qq = questions[i];
|
|
54
|
+
const { results } = await searchAll({ dir: DIR, query: qq.q, k: K, pool: POOL });
|
|
55
|
+
const best = evaluate(results, qq);
|
|
56
|
+
const truthYes = (qq.truth || 'yes').toLowerCase() !== 'no';
|
|
57
|
+
// YES-capability: pass = confident + evidenced from expected repo. NO-capability (control):
|
|
58
|
+
// pass = the bundle does NOT return a confident, evidenced hit (it correctly has nothing to assert).
|
|
59
|
+
const evidencedConfident = !!(best && best.confident && best.evidenced);
|
|
60
|
+
const pass = truthYes ? evidencedConfident : !evidencedConfident;
|
|
61
|
+
rows.push({ i: i + 1, q: qq.q, truth: truthYes ? 'YES' : 'NO(control)', pass, best });
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
const passN = rows.filter((r) => r.pass).length;
|
|
65
|
+
console.log(`\n=== capability-confidence gate — ${path.basename(QFILE)} (τ=${TAU}, k=${K}, pool=${POOL}) ===\n`);
|
|
66
|
+
for (const r of rows) {
|
|
67
|
+
const b = r.best;
|
|
68
|
+
console.log(`${r.pass ? 'PASS' : 'FAIL'} [${r.truth}] ${r.q}`);
|
|
69
|
+
console.log(` → top@repo=${b ? b.repo : '-'} ce=${b && b.ce != null ? b.ce.toFixed(3) : 'n/a'} evidence=${b ? `${b.evHits}/${b.evNeed}` : '-'} ${b ? (b.path) : ''}`);
|
|
70
|
+
}
|
|
71
|
+
console.log(`\nCAPABILITY-CONFIDENCE: ${passN}/${rows.length} pass (${(100 * passN / rows.length).toFixed(0)}%) | gate target = 100% on YES + 100% on NO controls`);
|
|
72
|
+
process.exit(passN === rows.length ? 0 : 1);
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// brain-grade-groundtruth.mjs — the ADR-0002 gate of record.
|
|
3
|
+
// For each question: (1) run the REAL consumer path (searchKb) → top-k cited source;
|
|
4
|
+
// (2) MECHANICAL ground-truth: does the #1 cited path actually exist in the repo clone?
|
|
5
|
+
// (3) MULTI-VENDOR panel: independent-vendor LLMs (GPT, Gemini — different families from the
|
|
6
|
+
// builder) grade STRICT (#1 doc alone) + REAL-USE (top-5) 1-100 against the RETURNED SOURCE,
|
|
7
|
+
// with the poison rule (incomplete-but-not-wrong < 50). No same-family LLM is the final word.
|
|
8
|
+
// Emits a JSON report + console summary with REAL NUMBERS. Never claims PASS itself — prints the data.
|
|
9
|
+
//
|
|
10
|
+
// node scripts/brain-grade-groundtruth.mjs --name ruflo --variant big \
|
|
11
|
+
// --questions kb/questions.ruflo.json --repo ../ruvnet-repos/ruflo (or set RUVNET_REPO)
|
|
12
|
+
import fs from 'node:fs';
|
|
13
|
+
import path from 'node:path';
|
|
14
|
+
import { fileURLToPath } from 'node:url';
|
|
15
|
+
import { searchKb } from '../kb/forge-ask.mjs';
|
|
16
|
+
import { rerankKb } from '../kb/forge-rerank.mjs';
|
|
17
|
+
const RERANK = process.argv.includes('--rerank');
|
|
18
|
+
|
|
19
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
20
|
+
const arg = (f, d) => { const i = process.argv.indexOf(f); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
|
|
21
|
+
const NAME = arg('--name', 'ruflo');
|
|
22
|
+
const VARIANT = arg('--variant', 'big');
|
|
23
|
+
const QFILE = arg('--questions', path.join(ROOT, 'kb/questions.ruflo.json'));
|
|
24
|
+
const REPO = arg('--repo', process.env.RUVNET_REPO || path.join(ROOT, '..', 'ruvnet-repos', NAME));
|
|
25
|
+
const MODELS = (arg('--models', 'openai/gpt-4o-mini,deepseek/deepseek-chat,meta-llama/llama-3.3-70b-instruct')).split(',').map(s => s.trim()).filter(Boolean);
|
|
26
|
+
|
|
27
|
+
// OpenRouter key (presence already confirmed) — read from env or Ask-Ruvnet .env; never logged.
|
|
28
|
+
function readKey() {
|
|
29
|
+
if (process.env.OPENROUTER_API_KEY) return process.env.OPENROUTER_API_KEY;
|
|
30
|
+
try {
|
|
31
|
+
const env = fs.readFileSync((process.env.RUVNET_ENV_FILE || '.env'), 'utf8');
|
|
32
|
+
const m = env.match(/^OPENROUTER_API_KEY=(.+)$/m); return m ? m[1].trim() : null;
|
|
33
|
+
} catch { return null; }
|
|
34
|
+
}
|
|
35
|
+
const KEY = readKey();
|
|
36
|
+
const clip = (s, n = 2600) => (s && s.length > n ? s.slice(0, n) + `\n…[+${s.length - n} chars]` : (s || ''));
|
|
37
|
+
|
|
38
|
+
async function gradeWith(model, q, strictEvidence, realUseEvidence) {
|
|
39
|
+
const sys = 'You are a STRICT senior code reviewer grading whether a knowledge-base answer is COMPLETE and CORRECT, judged ONLY against the retrieved source shown. Scale 1-100: 98=perfect/complete/actionable; an INCOMPLETE-but-not-wrong answer is POISON (<50). If the source does not actually answer the question, score low. Return ONLY compact JSON: {"strict":N,"realUse":N,"reason":"<=20 words"}.';
|
|
40
|
+
const usr = `QUESTION: ${q}\n\n--- STRICT: the #1 retrieved document ALONE ---\n${strictEvidence}\n\n--- REAL-USE: the top-5 documents a consumer would synthesize ---\n${realUseEvidence}\n\nGrade STRICT (can the #1 doc alone fully answer?) and REAL-USE (can the top-5 together?).`;
|
|
41
|
+
const res = await fetch('https://openrouter.ai/api/v1/chat/completions', {
|
|
42
|
+
method: 'POST',
|
|
43
|
+
headers: { 'Authorization': `Bearer ${KEY}`, 'Content-Type': 'application/json' },
|
|
44
|
+
body: JSON.stringify({ model, messages: [{ role: 'system', content: sys }, { role: 'user', content: usr }], temperature: 0, max_tokens: 200 }),
|
|
45
|
+
});
|
|
46
|
+
if (!res.ok) throw new Error(`${model} HTTP ${res.status}`);
|
|
47
|
+
const j = await res.json();
|
|
48
|
+
const txt = j.choices?.[0]?.message?.content || '';
|
|
49
|
+
const m = txt.match(/\{[\s\S]*\}/);
|
|
50
|
+
if (!m) throw new Error(`${model} no-json`);
|
|
51
|
+
const o = JSON.parse(m[0]);
|
|
52
|
+
return { strict: Number(o.strict), realUse: Number(o.realUse), reason: String(o.reason || '').slice(0, 120) };
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
const questions = JSON.parse(fs.readFileSync(QFILE, 'utf8'));
|
|
56
|
+
if (!KEY) { console.error('No OPENROUTER_API_KEY — cannot run multi-vendor panel.'); process.exit(2); }
|
|
57
|
+
console.log(`# Ground-truth + multi-vendor grade — ${NAME} [${VARIANT}] — ${questions.length} Q × ${MODELS.length} vendors\n`);
|
|
58
|
+
|
|
59
|
+
const report = [];
|
|
60
|
+
for (let i = 0; i < questions.length; i++) {
|
|
61
|
+
const q = questions[i].q;
|
|
62
|
+
const results = RERANK
|
|
63
|
+
? await rerankKb({ dir: path.join(ROOT, 'kb'), name: NAME, query: q, k: 6, variant: VARIANT })
|
|
64
|
+
: await searchKb({ dir: path.join(ROOT, 'kb'), name: NAME, query: q, k: 6, variant: VARIANT });
|
|
65
|
+
const top = results[0];
|
|
66
|
+
const gt = top ? fs.existsSync(path.join(REPO, top.path)) : false; // mechanical ground-truth: cited path real?
|
|
67
|
+
const strictEv = top ? `path: ${top.path}\n${clip(top.fullText)}` : '(no results)';
|
|
68
|
+
const realUseEv = results.slice(0, 5).map((r, j) => `${j + 1}. ${r.path}\n${clip(r.fullText, 1100)}`).join('\n\n');
|
|
69
|
+
const grades = [];
|
|
70
|
+
for (const model of MODELS) {
|
|
71
|
+
try { grades.push({ model, ...await gradeWith(model, q, strictEv, realUseEv) }); }
|
|
72
|
+
catch (e) { grades.push({ model, error: e.message }); }
|
|
73
|
+
}
|
|
74
|
+
const ok = grades.filter(g => Number.isFinite(g.strict));
|
|
75
|
+
const avgS = ok.length ? ok.reduce((s, g) => s + g.strict, 0) / ok.length : null;
|
|
76
|
+
const avgR = ok.length ? ok.reduce((s, g) => s + g.realUse, 0) / ok.length : null;
|
|
77
|
+
report.push({ q, topPath: top?.path || null, groundTruthPathExists: gt, grades, avgStrict: avgS, avgRealUse: avgR });
|
|
78
|
+
console.log(`Q${i + 1}. ${q}`);
|
|
79
|
+
console.log(` #1: ${top?.path || '(none)'} | path-exists: ${gt ? 'YES' : 'NO'}`);
|
|
80
|
+
for (const g of grades) console.log(` ${g.model}: ${g.error ? 'ERR ' + g.error : `strict=${g.strict} realUse=${g.realUse} — ${g.reason}`}`);
|
|
81
|
+
console.log(` → avg strict=${avgS?.toFixed(1) ?? '?'} realUse=${avgR?.toFixed(1) ?? '?'}\n`);
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// Aggregate
|
|
85
|
+
const valid = report.filter(r => Number.isFinite(r.avgStrict));
|
|
86
|
+
const mean = (k) => valid.reduce((s, r) => s + r[k], 0) / valid.length;
|
|
87
|
+
const minK = (k) => Math.min(...valid.map(r => r[k]));
|
|
88
|
+
const gtFail = report.filter(r => !r.groundTruthPathExists).length;
|
|
89
|
+
const summary = {
|
|
90
|
+
name: NAME, variant: VARIANT, questions: questions.length, models: MODELS,
|
|
91
|
+
avgStrict: +mean('avgStrict').toFixed(2), avgRealUse: +mean('avgRealUse').toFixed(2),
|
|
92
|
+
minStrict: minK('avgStrict'), minRealUse: minK('avgRealUse'),
|
|
93
|
+
poisonStrict: valid.filter(r => r.avgStrict < 50).length, poisonRealUse: valid.filter(r => r.avgRealUse < 50).length,
|
|
94
|
+
groundTruthCitationFailures: gtFail,
|
|
95
|
+
};
|
|
96
|
+
fs.writeFileSync(path.join(ROOT, `data/grade-${NAME}-${VARIANT}.json`), JSON.stringify({ summary, report }, null, 2));
|
|
97
|
+
console.log('=== SUMMARY (real numbers) ===');
|
|
98
|
+
console.log(JSON.stringify(summary, null, 2));
|
|
99
|
+
console.log(`\nGate (ADR-0002): PASS needs avg>=98 AND min>=95 on BOTH metrics, 0 poison, 0 citation failures.`);
|
|
100
|
+
console.log(`Report: data/grade-${NAME}-${VARIANT}.json`);
|
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { spawn } from 'node:child_process';
|
|
3
|
+
import { createInterface } from 'node:readline';
|
|
4
|
+
import fs from 'node:fs';
|
|
5
|
+
import os from 'node:os';
|
|
6
|
+
import path from 'node:path';
|
|
7
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
8
|
+
|
|
9
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
10
|
+
|
|
11
|
+
export const QUESTIONS = [
|
|
12
|
+
['How does RuvNet Brain expose the rvbc console in Claude Code and Codex?', /repo=ruvnet-brain/i],
|
|
13
|
+
['What command shows the RuvNet Brain 4.0 upgrade highlights?', /repo=ruvnet-brain/i],
|
|
14
|
+
['How does RuvNet Brain build and include its own source RVF?', /repo=ruvnet-brain/i],
|
|
15
|
+
['How does RuvNet Brain prevent a broad product query from searching every repository?', /repo=ruvnet-brain/i],
|
|
16
|
+
['What release gates prevent RuvNet Brain from publishing an unverified artifact?', /repo=ruvnet-brain/i],
|
|
17
|
+
['How does RuvNet Brain verify Claude Code plugin installation?', /repo=ruvnet-brain/i],
|
|
18
|
+
['How does RuvNet Brain verify Codex skill discovery?', /repo=ruvnet-brain/i],
|
|
19
|
+
['How does RuvNet Brain display what changed after a 4.0 upgrade?', /repo=ruvnet-brain/i],
|
|
20
|
+
['How does RuvNet Brain detect stale project memory without interrupting the user?', /repo=ruvnet-brain/i],
|
|
21
|
+
['How does RuvNet Brain bind release evidence to an exact Git SHA and artifact digest?', /repo=ruvnet-brain/i],
|
|
22
|
+
['How does RuvNet Brain test narrow, broad, and concurrent source searches?', /repo=ruvnet-brain/i],
|
|
23
|
+
['How does RuvNet Brain distinguish proposed ADR capabilities from shipped code?', /repo=ruvnet-brain/i],
|
|
24
|
+
['How does RuvNet Brain prevent a zero-test QE run from passing?', /repo=ruvnet-brain/i],
|
|
25
|
+
['How does RuvNet Brain verify the public npm and GitHub release surfaces agree?', /repo=ruvnet-brain/i],
|
|
26
|
+
['How does RuvNet Brain update both Claude Code and Codex installations?', /repo=ruvnet-brain/i],
|
|
27
|
+
['How does RuvNet Brain handle model-cache initialization across concurrent processes?', /repo=ruvnet-brain/i],
|
|
28
|
+
['How does RuvNet Brain measure whether grounding evidence is thin or strong?', /repo=ruvnet-brain/i],
|
|
29
|
+
['How does RuvNet Brain stop private stores from entering a public bundle?', /repo=ruvnet-brain/i],
|
|
30
|
+
['How does RuvNet Brain sign and verify its downloadable RVF bundle?', /repo=ruvnet-brain/i],
|
|
31
|
+
['What does the RuvNet Brain doctor command verify after installation?', /repo=ruvnet-brain/i],
|
|
32
|
+
['How does RVF provide persistent HNSW vector search?', /repo=ruvector/i],
|
|
33
|
+
['Which RuVector source implements RVF database creation and querying?', /repo=ruvector/i],
|
|
34
|
+
['How does RuVector avoid hand-written cosine search for persisted knowledge?', /repo=ruvector/i],
|
|
35
|
+
['How are witness chains represented in the RVF architecture?', /repo=ruvector/i],
|
|
36
|
+
['How does RuVector support local embedding generation for vector search?', /repo=ruvector/i],
|
|
37
|
+
['How does Ruflo initialize and coordinate a hierarchical agent swarm?', /repo=ruflo/i],
|
|
38
|
+
['How does Ruflo route work to an architect agent?', /repo=ruflo/i],
|
|
39
|
+
['How does Ruflo store and search project memory?', /repo=ruflo/i],
|
|
40
|
+
['How does Ruflo track agent execution separately from agent registration?', /repo=ruflo/i],
|
|
41
|
+
['How does Ruflo enforce policy before consequential actions?', /repo=ruflo/i],
|
|
42
|
+
['How does AgentDB store structured agent memories?', /repo=agentdb/i],
|
|
43
|
+
['How does AgentDB perform semantic memory search?', /repo=agentdb/i],
|
|
44
|
+
['How does AgentDB handle SQLite WAL-backed persistence?', /repo=agentdb/i],
|
|
45
|
+
['How does AgentDB represent graph relationships between memories?', /repo=agentdb/i],
|
|
46
|
+
['How does AgentDB prevent unsafe memory input at its boundaries?', /repo=agentdb/i],
|
|
47
|
+
['How does Agentic QE generate and execute a test fleet?', /repo=agentic-qe/i],
|
|
48
|
+
['How does Agentic QE report test coverage gaps?', /repo=agentic-qe/i],
|
|
49
|
+
['How does Agentic QE run security-focused quality checks?', /repo=agentic-qe/i],
|
|
50
|
+
['How does Agentic QE prevent vacuous success when no tests execute?', /repo=(?:agentic-qe|ruvnet-brain)/i],
|
|
51
|
+
['How does Agentic QE coordinate specialized testing agents?', /repo=agentic-qe/i],
|
|
52
|
+
['What phases are defined by the SPARC methodology?', /repo=sparc/i],
|
|
53
|
+
['How does RuvNet Brain package the managed CLI boundary users invoke?', /repo=ruvnet-brain/i],
|
|
54
|
+
['How does RuvNet Brain validate that every advertised file exists in the npm tarball?', /repo=ruvnet-brain/i],
|
|
55
|
+
['How does RuLake act as a read cache over vector collections?', /repo=rulake/i],
|
|
56
|
+
['How does RuView process radar sensor data?', /repo=ruview/i],
|
|
57
|
+
['Compare RuvNet Brain project memory with AgentDB structured storage.', /repo=(?:ruvnet-brain|agentdb)/i],
|
|
58
|
+
['Compare RuvNet Brain source grounding with Ruflo orchestration.', /repo=(?:ruvnet-brain|ruflo)/i],
|
|
59
|
+
['How should RuvNet Brain enforce an exact-artifact deployment process with clean lineage, GitHub checks, Agentic QE, independent graders, and post-publication verification?', /repo=ruvnet-brain/i],
|
|
60
|
+
['Which shipped files implement the RuvNet Brain release-proof protocol?', /repo=ruvnet-brain/i],
|
|
61
|
+
['What evidence proves RuvNet Brain answers its own product questions from RVF?', /repo=ruvnet-brain/i],
|
|
62
|
+
];
|
|
63
|
+
|
|
64
|
+
function percentile(values, fraction) {
|
|
65
|
+
if (!values.length) return null;
|
|
66
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
67
|
+
return sorted[Math.ceil(sorted.length * fraction) - 1];
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function parseArgs(argv) {
|
|
71
|
+
const value = (flag, fallback) => {
|
|
72
|
+
const index = argv.indexOf(flag);
|
|
73
|
+
return index >= 0 ? argv[index + 1] : fallback;
|
|
74
|
+
};
|
|
75
|
+
return {
|
|
76
|
+
server: path.resolve(value('--server', path.join(ROOT, 'plugin/mcp/server.mjs'))),
|
|
77
|
+
kb: path.resolve(value('--kb', path.join(ROOT, 'kb'))),
|
|
78
|
+
timeoutMs: Number(value('--timeout-ms', '30000')),
|
|
79
|
+
modelCache: path.resolve(value('--model-cache', path.join(ROOT, 'kb', 'models-cache'))),
|
|
80
|
+
json: argv.includes('--json'),
|
|
81
|
+
from: Number(value('--from', '1')),
|
|
82
|
+
to: Number(value('--to', String(QUESTIONS.length))),
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export async function benchmark(options) {
|
|
87
|
+
if (QUESTIONS.length !== 50) throw new Error(`benchmark must contain exactly 50 questions, found ${QUESTIONS.length}`);
|
|
88
|
+
const brainHome = fs.mkdtempSync(path.join(os.tmpdir(), 'ruvnet-brain-latency-'));
|
|
89
|
+
const child = spawn(process.execPath, [options.server], {
|
|
90
|
+
env: {
|
|
91
|
+
...process.env,
|
|
92
|
+
KB_DIR: options.kb,
|
|
93
|
+
RUVNET_BRAIN_KB: options.kb,
|
|
94
|
+
RUVNET_BRAIN_CHILD_MCP: path.join(options.kb, 'forge-mcp-all.mjs'),
|
|
95
|
+
RUVNET_BRAIN_HOME: brainHome,
|
|
96
|
+
KB_MODEL_CACHE: options.modelCache,
|
|
97
|
+
RUVNET_BRAIN_CALL_TIMEOUT_MS: String(options.timeoutMs),
|
|
98
|
+
RUVNET_BRAIN_METER: '0',
|
|
99
|
+
},
|
|
100
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
101
|
+
});
|
|
102
|
+
const lines = createInterface({ input: child.stdout });
|
|
103
|
+
const pending = new Map();
|
|
104
|
+
let stderr = '';
|
|
105
|
+
let nextId = 1;
|
|
106
|
+
child.stderr.on('data', (chunk) => { stderr += chunk.toString(); });
|
|
107
|
+
lines.on('line', (line) => {
|
|
108
|
+
let message;
|
|
109
|
+
try { message = JSON.parse(line); } catch { return; }
|
|
110
|
+
const waiter = pending.get(message.id);
|
|
111
|
+
if (waiter) {
|
|
112
|
+
pending.delete(message.id);
|
|
113
|
+
waiter(message);
|
|
114
|
+
}
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
const request = (method, params, timeoutMs) => {
|
|
118
|
+
const id = nextId++;
|
|
119
|
+
return new Promise((resolve, reject) => {
|
|
120
|
+
const timer = setTimeout(() => {
|
|
121
|
+
pending.delete(id);
|
|
122
|
+
reject(new Error(`MCP ${method} timed out after ${timeoutMs}ms`));
|
|
123
|
+
}, timeoutMs);
|
|
124
|
+
pending.set(id, (message) => {
|
|
125
|
+
clearTimeout(timer);
|
|
126
|
+
resolve(message);
|
|
127
|
+
});
|
|
128
|
+
child.stdin.write(`${JSON.stringify({ jsonrpc: '2.0', id, method, params })}\n`);
|
|
129
|
+
});
|
|
130
|
+
};
|
|
131
|
+
|
|
132
|
+
const ask = (query, expected) => {
|
|
133
|
+
const id = nextId++;
|
|
134
|
+
const started = performance.now();
|
|
135
|
+
return new Promise((resolve) => {
|
|
136
|
+
const timer = setTimeout(() => {
|
|
137
|
+
pending.delete(id);
|
|
138
|
+
resolve({ query, elapsedMs: Math.round(performance.now() - started), status: 'TIMEOUT', cited: false });
|
|
139
|
+
}, options.timeoutMs + 5_000);
|
|
140
|
+
pending.set(id, (message) => {
|
|
141
|
+
clearTimeout(timer);
|
|
142
|
+
const text = message.result?.content?.[0]?.text || '';
|
|
143
|
+
const cited = expected.test(text);
|
|
144
|
+
const observedRepos = [...text.matchAll(/repo=([a-z0-9._-]+)/gi)].map((match) => match[1]);
|
|
145
|
+
resolve({
|
|
146
|
+
query,
|
|
147
|
+
elapsedMs: Math.round(performance.now() - started),
|
|
148
|
+
status: !message.result?.isError && cited ? 'PASS' : 'FAIL',
|
|
149
|
+
cited,
|
|
150
|
+
observedRepos: [...new Set(observedRepos)],
|
|
151
|
+
error: message.error?.message || (message.result?.isError ? text.slice(0, 500) : undefined),
|
|
152
|
+
preview: text.slice(0, 500),
|
|
153
|
+
});
|
|
154
|
+
});
|
|
155
|
+
child.stdin.write(`${JSON.stringify({
|
|
156
|
+
jsonrpc: '2.0',
|
|
157
|
+
id,
|
|
158
|
+
method: 'tools/call',
|
|
159
|
+
params: { name: 'search_ruvnet', arguments: { query, k: 5 } },
|
|
160
|
+
})}\n`);
|
|
161
|
+
});
|
|
162
|
+
};
|
|
163
|
+
|
|
164
|
+
const results = [];
|
|
165
|
+
const readinessStarted = performance.now();
|
|
166
|
+
const initialize = await request('initialize', {
|
|
167
|
+
protocolVersion: '2024-11-05',
|
|
168
|
+
capabilities: {},
|
|
169
|
+
clientInfo: { name: 'brain-latency-50', version: '1.0.0' },
|
|
170
|
+
}, 120_000);
|
|
171
|
+
if (initialize.error) throw new Error(`MCP initialize failed: ${initialize.error.message}`);
|
|
172
|
+
const listed = await request('tools/list', {}, 120_000);
|
|
173
|
+
if (!listed.result?.tools?.some((tool) => tool.name === 'search_ruvnet')) {
|
|
174
|
+
throw new Error('MCP readiness failed: search_ruvnet was not advertised');
|
|
175
|
+
}
|
|
176
|
+
const readinessMs = Math.round(performance.now() - readinessStarted);
|
|
177
|
+
try {
|
|
178
|
+
const selected = QUESTIONS.slice(Math.max(0, options.from - 1), Math.min(QUESTIONS.length, options.to));
|
|
179
|
+
for (const [query, expected] of selected) {
|
|
180
|
+
const result = await ask(query, expected);
|
|
181
|
+
results.push(result);
|
|
182
|
+
if (!options.json) {
|
|
183
|
+
process.stdout.write(`${String(results.length).padStart(2, '0')} ${result.status.padEnd(7)} ${String(result.elapsedMs).padStart(6)}ms ${query}\n`);
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
} finally {
|
|
187
|
+
child.stdin.end();
|
|
188
|
+
await Promise.race([
|
|
189
|
+
new Promise((resolve) => child.once('exit', resolve)),
|
|
190
|
+
new Promise((resolve) => setTimeout(resolve, 5_000)),
|
|
191
|
+
]);
|
|
192
|
+
if (child.exitCode === null) child.kill('SIGKILL');
|
|
193
|
+
lines.close();
|
|
194
|
+
fs.rmSync(brainHome, { recursive: true, force: true });
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
const times = results.map((result) => result.elapsedMs);
|
|
198
|
+
const summary = {
|
|
199
|
+
server: options.server,
|
|
200
|
+
kb: options.kb,
|
|
201
|
+
questions: results.length,
|
|
202
|
+
passed: results.filter((result) => result.status === 'PASS').length,
|
|
203
|
+
failed: results.filter((result) => result.status === 'FAIL').length,
|
|
204
|
+
timedOut: results.filter((result) => result.status === 'TIMEOUT').length,
|
|
205
|
+
cited: results.filter((result) => result.cited).length,
|
|
206
|
+
averageMs: Math.round(times.reduce((sum, value) => sum + value, 0) / times.length),
|
|
207
|
+
medianMs: percentile(times, 0.5),
|
|
208
|
+
p95Ms: percentile(times, 0.95),
|
|
209
|
+
longestMs: Math.max(...times),
|
|
210
|
+
deadlineMs: options.timeoutMs,
|
|
211
|
+
readinessMs,
|
|
212
|
+
stderr: stderr.trim(),
|
|
213
|
+
results,
|
|
214
|
+
};
|
|
215
|
+
if (options.json) console.log(JSON.stringify(summary, null, 2));
|
|
216
|
+
else console.log(`SUMMARY ${summary.passed}/${summary.questions} pass · avg ${summary.averageMs}ms · median ${summary.medianMs}ms · p95 ${summary.p95Ms}ms · max ${summary.longestMs}ms · timeouts ${summary.timedOut}`);
|
|
217
|
+
return summary;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
export async function main(argv = process.argv.slice(2)) {
|
|
221
|
+
const summary = await benchmark(parseArgs(argv));
|
|
222
|
+
return summary.passed === summary.questions && summary.timedOut === 0 ? 0 : 1;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
if (process.argv[1] && pathToFileURL(path.resolve(process.argv[1])).href === import.meta.url) {
|
|
226
|
+
process.exitCode = await main();
|
|
227
|
+
}
|