ruvnet-brain 4.0.1 → 4.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/.claude-plugin/marketplace.json +1 -0
  2. package/README.md +4 -4
  3. package/bin/install.mjs +303 -24
  4. package/console/CONTRACT.md +172 -0
  5. package/console/activity.js +753 -0
  6. package/console/app.js +4189 -0
  7. package/console/architecture.html +1221 -0
  8. package/console/assets/depth-1.webp +0 -0
  9. package/console/assets/depth-2.webp +0 -0
  10. package/console/assets/depth-3.webp +0 -0
  11. package/console/assets/harness-vs-plain.svg +259 -0
  12. package/console/assets/hero.webp +0 -0
  13. package/console/assets/memory.webp +0 -0
  14. package/console/assets/metaharness.svg +247 -0
  15. package/console/index.html +777 -0
  16. package/console/install-architecture.html +162 -0
  17. package/console/install-mockup.html +543 -0
  18. package/console/style.css +2144 -0
  19. package/console/tips.css +926 -0
  20. package/console/tips.html +858 -0
  21. package/console/tips.js +128 -0
  22. package/docs/RELEASE-NOTES-4.0.md +88 -0
  23. package/kb/model-requirements.mjs +37 -6
  24. package/keys/ruvnet-brain-signing.pub.pem +3 -0
  25. package/package.json +8 -22
  26. package/plugin/.claude-plugin/marketplace.json +1 -0
  27. package/plugin/.claude-plugin/plugin.json +2 -3
  28. package/plugin/.codex-plugin/plugin.json +1 -1
  29. package/plugin/commands/brain-console.md +2 -2
  30. package/plugin/commands/configure.md +3 -2
  31. package/plugin/commands/rvbc.md +4 -3
  32. package/plugin/commands/rvcb.md +2 -2
  33. package/plugin/commands/whats-new.md +6 -6
  34. package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
  35. package/plugin/hooks/hooks.json +1 -2
  36. package/plugin/mcp/managed-cli-interface.mjs +47 -4
  37. package/plugin/mcp/server.mjs +90 -32
  38. package/plugin/scripts/detach.mjs +14 -0
  39. package/plugin/scripts/first-session-worker.mjs +38 -0
  40. package/plugin/scripts/ground-ruvnet.sh +16 -6
  41. package/plugin/scripts/hook-shim.mjs +34 -29
  42. package/plugin/scripts/learn-capture.sh +22 -3
  43. package/plugin/scripts/learn-flush.mjs +21 -4
  44. package/plugin/scripts/runtime-preferences.mjs +269 -0
  45. package/plugin/scripts/session-start-core.mjs +503 -0
  46. package/plugin/scripts/session-start.sh +3 -858
  47. package/plugin/scripts/whats-new.mjs +42 -0
  48. package/plugin/skills/brain-console/SKILL.md +4 -2
  49. package/plugin/skills/release-proof/SKILL.md +98 -0
  50. package/plugin/skills/release-proof/agents/openai.yaml +4 -0
  51. package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
  52. package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
  53. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
  54. package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
  55. package/plugin/skills/rvbc/SKILL.md +9 -6
  56. package/plugin/skills/whats-new/SKILL.md +4 -4
  57. package/scripts/adr-backfill.mjs +107 -0
  58. package/scripts/advocacy-outcomes.mjs +808 -0
  59. package/scripts/agentdb-context.mjs +216 -0
  60. package/scripts/agentdb-fleet-doctor.mjs +101 -0
  61. package/scripts/ascii-drift.mjs +236 -0
  62. package/scripts/behavioral-l1-l4.mjs +210 -0
  63. package/scripts/brain-capability-check.mjs +72 -0
  64. package/scripts/brain-grade-groundtruth.mjs +100 -0
  65. package/scripts/brain-latency-50.mjs +227 -0
  66. package/scripts/brain-novice-50.mjs +189 -0
  67. package/scripts/brain-stamp.mjs +94 -0
  68. package/scripts/brain-state.mjs +212 -0
  69. package/scripts/build-bundle.mjs +531 -0
  70. package/scripts/build-concepts.mjs +132 -0
  71. package/scripts/build-l2.mjs +71 -0
  72. package/scripts/build-primer.mjs +73 -0
  73. package/scripts/build-symbols.mjs +68 -0
  74. package/scripts/calibrate-router.mjs +97 -0
  75. package/scripts/capability-audit.mjs +321 -0
  76. package/scripts/capability-registry.mjs +876 -0
  77. package/scripts/check-indexation.mjs +108 -0
  78. package/scripts/check-legibility.mjs +189 -0
  79. package/scripts/ci/build-fixture-kb.mjs +67 -0
  80. package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
  81. package/scripts/ci/learning-replay-recorder.mjs +59 -0
  82. package/scripts/ci/mutate-hook-timeout.mjs +70 -0
  83. package/scripts/ci/stranger-fixture-stage.mjs +17 -0
  84. package/scripts/ci/stranger-scenario.mjs +228 -0
  85. package/scripts/ci/stranger-timeout.mjs +25 -0
  86. package/scripts/ci-verdict.mjs +29 -0
  87. package/scripts/claims-verify.mjs +710 -0
  88. package/scripts/clear-claude-tmp.sh +31 -0
  89. package/scripts/console-engine.mjs +434 -0
  90. package/scripts/console-engine.test.mjs +125 -0
  91. package/scripts/corpus-qa.mjs +250 -0
  92. package/scripts/correction-detect-embed.mjs +346 -0
  93. package/scripts/correction-detect-measure.mjs +270 -0
  94. package/scripts/correction-detect.mjs +686 -0
  95. package/scripts/count-chunks.mjs +54 -0
  96. package/scripts/described-questions.json +30 -0
  97. package/scripts/design-grade.mjs +58 -0
  98. package/scripts/dev-plugin-link.sh +105 -0
  99. package/scripts/distill-project.mjs +200 -0
  100. package/scripts/doc-currency.mjs +801 -0
  101. package/scripts/eval-brain.mjs +244 -0
  102. package/scripts/fix-metaharness-memretrieve.mjs +121 -0
  103. package/scripts/fix-workstream.mjs +291 -0
  104. package/scripts/full-hints.mjs +87 -0
  105. package/scripts/gate.sh +39 -0
  106. package/scripts/gates.mjs +146 -0
  107. package/scripts/gen-console-images.mjs +54 -0
  108. package/scripts/gen-images.mjs +47 -0
  109. package/scripts/git-clone-refresh.mjs +52 -0
  110. package/scripts/git-hooks/pre-push +126 -0
  111. package/scripts/goal-match.mjs +398 -0
  112. package/scripts/goldie-research.mjs +223 -0
  113. package/scripts/goldie-weekly.sh +67 -0
  114. package/scripts/health-repair.mjs +237 -0
  115. package/scripts/helix-scenario-questions.json +10 -0
  116. package/scripts/ingest-gists.mjs +230 -0
  117. package/scripts/ingest-meeting.mjs +115 -0
  118. package/scripts/ingest-repo.mjs +79 -0
  119. package/scripts/install-npx-witness.sh +49 -0
  120. package/scripts/issue-fix.mjs +558 -0
  121. package/scripts/issue-watch.mjs +276 -0
  122. package/scripts/issue4-close-note.md +31 -0
  123. package/scripts/key-canary.mjs +91 -0
  124. package/scripts/latency-to-surface.mjs +233 -0
  125. package/scripts/learning-enable.mjs +380 -0
  126. package/scripts/learning-replay.mjs +1570 -0
  127. package/scripts/learnings.mjs +62 -0
  128. package/scripts/lesson-gate.mjs +680 -0
  129. package/scripts/lesson-lifecycle.mjs +449 -0
  130. package/scripts/lesson-promote.mjs +262 -0
  131. package/scripts/lesson-ratify.mjs +98 -0
  132. package/scripts/lesson-seed.mjs +252 -0
  133. package/scripts/lesson-store.mjs +447 -0
  134. package/scripts/loop-checkpoint.mjs +86 -0
  135. package/scripts/memdb-health.sh +14 -0
  136. package/scripts/memory-doctor.mjs +326 -0
  137. package/scripts/model-catalog.mjs +79 -0
  138. package/scripts/nightly-controller.mjs +66 -0
  139. package/scripts/nightly-gists.sh +72 -0
  140. package/scripts/nightly-wrapper.sh +172 -0
  141. package/scripts/notify.sh +12 -0
  142. package/scripts/npx-witness.sh +56 -0
  143. package/scripts/onboarding-console.mjs +2922 -0
  144. package/scripts/private-fence.mjs +69 -0
  145. package/scripts/proactivity-metrics.mjs +118 -0
  146. package/scripts/proof-questions.json +56 -0
  147. package/scripts/protected-release-invocation.mjs +76 -0
  148. package/scripts/prove.mjs +95 -0
  149. package/scripts/proxy/claude-proxied.sh +57 -0
  150. package/scripts/proxy/proxy-revert.sh +59 -0
  151. package/scripts/proxy/proxy-up.sh +60 -0
  152. package/scripts/proxy/proxy-verify.mjs +142 -0
  153. package/scripts/publication-receipt.mjs +307 -0
  154. package/scripts/published-surface-probe.mjs +241 -0
  155. package/scripts/qe/card-lane-gate.mjs +162 -0
  156. package/scripts/qe/session-start-gate.mjs +229 -0
  157. package/scripts/qe/ux-suite.mjs +323 -0
  158. package/scripts/reconcile-project.mjs +0 -0
  159. package/scripts/record-lesson.mjs +113 -0
  160. package/scripts/refresh-model-catalog.mjs +99 -0
  161. package/scripts/release-authority.mjs +93 -0
  162. package/scripts/release-proof.mjs +9 -0
  163. package/scripts/release-vector.mjs +281 -0
  164. package/scripts/release.mjs +439 -0
  165. package/scripts/remedy-registry.mjs +247 -0
  166. package/scripts/rerank-cap-eval.mjs +265 -0
  167. package/scripts/rerank-cap-warm-ab.mjs +129 -0
  168. package/scripts/route-cheap.mjs +20 -15
  169. package/scripts/router-utilization.mjs +182 -0
  170. package/scripts/routing-flywheel.mjs +596 -0
  171. package/scripts/rvf-generation.mjs +104 -0
  172. package/scripts/rvf-index-audit.mjs +138 -0
  173. package/scripts/self-update.mjs +296 -0
  174. package/scripts/selfcheck.mjs +7 -1
  175. package/scripts/sign-bundle.mjs +69 -0
  176. package/scripts/signal-watch.mjs +171 -0
  177. package/scripts/stabilization-receipt.mjs +108 -0
  178. package/scripts/stack-sync.mjs +469 -0
  179. package/scripts/stamp-existing-rvf-generations.mjs +53 -0
  180. package/scripts/stamp-sweep.mjs +144 -0
  181. package/scripts/status-honesty.mjs +102 -0
  182. package/scripts/sync-version.mjs +217 -0
  183. package/scripts/token-report.mjs +102 -0
  184. package/scripts/top100-benchmark.mjs +479 -0
  185. package/scripts/top100-corpus.mjs +112 -0
  186. package/scripts/top100-semantic-assertions.mjs +449 -0
  187. package/scripts/update-apply.mjs +9 -0
  188. package/scripts/upgrade-notice.mjs +14 -0
  189. package/scripts/verify-bundle.mjs +51 -0
  190. package/scripts/verify-channels.mjs +184 -0
  191. package/scripts/verify-model-catalog.mjs +104 -0
  192. package/scripts/verify-nightly-close-issue4.sh +31 -0
  193. package/scripts/version.mjs +40 -0
  194. package/scripts/wired-check.mjs +867 -0
  195. package/plugin/scripts/finalize-token-meter.mjs +0 -25
@@ -0,0 +1,244 @@
1
+ #!/usr/bin/env node
2
+ // eval-brain.mjs — the eval flywheel. Ask the frozen held-out questions, and judge the answers by
3
+ // GROUND TRUTH rather than by a model's opinion of them. (ADR-0011 Phase 0.)
4
+ //
5
+ // FIVE STRATA, because a gate that only asks easy questions cannot fail:
6
+ // named — the repo is named in the question pass = grounded AND routed
7
+ // described — capability described, no names pass = grounded AND routed
8
+ // scenario — a real-world situation, no names pass = grounded AND routed
9
+ // adversarial — the correct answer is NOT in this corpus pass = ABSTAINED (top ce < 0, or no hits)
10
+ // provenance — gist-shaped content pass = grounded AND (if the top hit IS a
11
+ // gist chunk, it must carry its GIST STATUS banner — better repo grounding also passes)
12
+ //
13
+ // GATING IS ON THE WILSON LOWER BOUND, never the point estimate. With n=12, routed 10/12 had a 95%
14
+ // CI of [55.2%, 95.3%] and 9/12's upper bound (91.1%) overlapped it completely — the gate could not
15
+ // detect the regression it existed to catch. n=120 gives ≥80% power for a 0.90 -> 0.80 drop (n≈69
16
+ // suffices; computed 2026-07-09).
17
+ //
18
+ // FAIL-CLOSED PROMOTION: `--gate` compares each metric's lower bound against evals/baseline.json and
19
+ // exits 1 on any drop. A missing baseline is a failure too — you cannot promote against nothing.
20
+ // Baselines are only ever written deliberately with `--record`.
21
+ //
22
+ // Why no model judge: an LLM panel scored a ZERO-CITATION answer 98/100 on this repo.
23
+ //
24
+ // node scripts/eval-brain.mjs # run + table
25
+ // node scripts/eval-brain.mjs --gate # run + exit 1 on regression (or missing baseline)
26
+ // node scripts/eval-brain.mjs --record # run + write evals/baseline.json (deliberate)
27
+ // node scripts/eval-brain.mjs --json # machine-readable
28
+ // node scripts/eval-brain.mjs --strata named,adversarial # subset (never gate on a subset)
29
+
30
+ import fs from 'node:fs';
31
+ import os from 'node:os';
32
+ import path from 'node:path';
33
+ import { spawnSync } from 'node:child_process';
34
+ import { fileURLToPath, pathToFileURL } from 'node:url';
35
+
36
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
37
+ const KB = process.env.RUVNET_BRAIN_KB || path.join(os.homedir(), '.cache', 'ruvnet-brain', 'kb');
38
+ const HELD_OUT = path.join(ROOT, 'evals', 'held-out.json');
39
+ const BASELINE = path.join(ROOT, 'evals', 'baseline.json');
40
+
41
+ // The cross-encoder emits a relevance logit per (query, passage): strongly negative when unrelated.
42
+ // An adversarial question "passes" when the brain effectively found nothing relevant. 0 is the
43
+ // neutral cut; per-question ce values are recorded so this stays inspectable, and the Wilson-gated
44
+ // baseline makes the stratum a regression detector even if the absolute rate is imperfect.
45
+ export const ABSTAIN_CE = 0;
46
+
47
+ /**
48
+ * Order-independent, tamper-evident hash of the held-out set — rUv's own frozen-eval pattern
49
+ * (ruflo harness-frozen-eval: humanEvalHash / FROZEN_HUMAN_EVAL_HASH). The pinned constant lives in
50
+ * tests/unit/eval-brain-gate.test.mjs; editing ANY question turns that test red, which is what
51
+ * "frozen" means mechanically. Reordering does not (sorting makes the hash order-independent).
52
+ */
53
+ export async function heldOutHash(questions) {
54
+ const { createHash } = await import('node:crypto');
55
+ const per = questions.map((q) =>
56
+ createHash('sha256').update(JSON.stringify({ id: q.id, stratum: q.stratum, query: q.query, expectRepo: q.expectRepo ?? null })).digest('hex'));
57
+ return createHash('sha256').update(per.sort().join('')).digest('hex');
58
+ }
59
+
60
+ /** Wilson score interval — the same instrument rUv uses on every SWE-bench number. */
61
+ export function wilson(k, n, z = 1.96) {
62
+ if (!n) return { p: 0, lo: 0, hi: 1 };
63
+ const p = k / n;
64
+ const d = 1 + (z * z) / n;
65
+ const c = (p + (z * z) / (2 * n)) / d;
66
+ const h = (z * Math.sqrt((p * (1 - p)) / n + (z * z) / (4 * n * n))) / d;
67
+ return { p, lo: Math.max(0, c - h), hi: Math.min(1, c + h) };
68
+ }
69
+
70
+ /**
71
+ * Grade one answered question by its stratum's rule. Pure — takes the verifier verdict, the top
72
+ * citation (with its ce score), and whether the raw output carried the gist provenance banner.
73
+ */
74
+ export function gradeQuestion(q, { grounded, citations, bannerPresent }) {
75
+ const top = citations?.[0] ?? null;
76
+ const routed = !!(grounded && q.expectRepo?.length && top && q.expectRepo.includes(top.repo));
77
+ const abstained = !top || (typeof top.ce === 'number' && top.ce < ABSTAIN_CE);
78
+ switch (q.stratum) {
79
+ case 'adversarial':
80
+ return { grounded, routed: null, abstained, pass: abstained };
81
+ case 'provenance':
82
+ // The mechanism under test: IF a gist chunk wins, it must carry its own status banner.
83
+ // A better hit from the real repo is not a failure — it is better grounding.
84
+ return { grounded, routed: null, abstained, pass: !!grounded && (top?.repo !== 'ruv-gists' || bannerPresent) };
85
+ default:
86
+ return { grounded, routed, abstained, pass: !!grounded && routed };
87
+ }
88
+ }
89
+
90
+ /** Aggregate graded rows into the four gated metrics, each with its Wilson interval. */
91
+ export function aggregate(rows) {
92
+ const by = (pred) => rows.filter(pred);
93
+ const routedStrata = (r) => ['named', 'described', 'scenario'].includes(r.stratum);
94
+ const groundable = by((r) => r.stratum !== 'adversarial');
95
+ const routed = by(routedStrata);
96
+ const adversarial = by((r) => r.stratum === 'adversarial');
97
+ const provenance = by((r) => r.stratum === 'provenance');
98
+ const metric = (arr, key) => {
99
+ const k = arr.filter((r) => r[key]).length;
100
+ return { k, n: arr.length, ...wilson(k, arr.length) };
101
+ };
102
+ return {
103
+ grounded: metric(groundable, 'grounded'),
104
+ routed: metric(routed, 'pass'),
105
+ abstain: metric(adversarial, 'pass'),
106
+ banner: metric(provenance, 'pass'),
107
+ };
108
+ }
109
+
110
+ /** The fail-closed comparison: every metric's lower bound must hold the baseline's lower bound. */
111
+ export function gateAgainst(current, baseline) {
112
+ if (!baseline) return { pass: false, regressions: ['no baseline to promote against — record one deliberately (--record)'] };
113
+ // An old-schema baseline (pre-strata counts, no {k,n,lo}) must FAIL, not pass vacuously — a gate
114
+ // that silently compares against nothing is the decorated-green failure this whole phase kills.
115
+ if (!['grounded', 'routed', 'abstain', 'banner'].some((m) => baseline[m]?.n)) {
116
+ return { pass: false, regressions: ['baseline is from an older schema — re-record deliberately (--record)'] };
117
+ }
118
+ const regressions = [];
119
+ for (const m of ['grounded', 'routed', 'abstain', 'banner']) {
120
+ const cur = current[m];
121
+ const base = baseline[m];
122
+ if (!base || !base.n) continue; // metric absent from an older baseline — cannot regress against nothing
123
+ if (!cur.n) { regressions.push(`${m}: stratum is empty but the baseline has n=${base.n}`); continue; }
124
+ if (cur.lo < base.lo - 1e-9) regressions.push(`${m}: lower bound ${(cur.lo * 100).toFixed(1)}% < baseline ${(base.lo * 100).toFixed(1)}%`);
125
+ }
126
+ return { pass: regressions.length === 0, regressions };
127
+ }
128
+
129
+ async function main() {
130
+ const argv = process.argv.slice(2);
131
+ const GATE = argv.includes('--gate');
132
+ const RECORD = argv.includes('--record');
133
+ const JSON_OUT = argv.includes('--json');
134
+ const strataArg = argv.includes('--strata') ? argv[argv.indexOf('--strata') + 1]?.split(',') : null;
135
+ const die = (msg) => { console.error(`eval-brain: ${msg}`); process.exit(2); };
136
+
137
+ if (!fs.existsSync(path.join(KB, 'forge-ask-all.mjs'))) die(`no brain at ${KB} — run: npx ruvnet-brain`);
138
+ const verifierPath = path.join(KB, 'verify-citation.mjs');
139
+ if (!fs.existsSync(verifierPath)) die('this bundle predates verify-citation.mjs — refusing to score grounding without a way to check it');
140
+ const { verifyGrounding } = await import(pathToFileURL(verifierPath).href);
141
+
142
+ let { questions } = JSON.parse(fs.readFileSync(HELD_OUT, 'utf8'));
143
+ for (const q of questions) if (!q.stratum) die(`question ${q.id} has no stratum`);
144
+ if (strataArg) {
145
+ questions = questions.filter((q) => strataArg.includes(q.stratum));
146
+ if (GATE || RECORD) die('--strata is for local iteration only; never gate or record on a subset');
147
+ }
148
+
149
+ // Each question is an independent subprocess with its own ONNX thread, so the pool parallelizes
150
+ // cleanly across cores — spawnSync would serialize the whole run on the event loop. Measured
151
+ // serial cost was ~13.5s/question ≈ 27 min for 120; at concurrency 6 the wall drops ~6×.
152
+ const { execFile } = await import('node:child_process');
153
+ const ask = (query) => new Promise((resolve) => {
154
+ execFile('node', ['forge-ask-all.mjs', '--dir', KB, '--q', query, '--k', '3'],
155
+ { cwd: KB, timeout: 240000, env: process.env, maxBuffer: 64 * 1024 * 1024 },
156
+ (err, stdout) => resolve(err ? '' : String(stdout || '')));
157
+ });
158
+
159
+ const CONC = Math.max(1, Number(process.env.EVAL_CONCURRENCY ?? (argv.includes('--concurrency') ? argv[argv.indexOf('--concurrency') + 1] : 6)) || 6);
160
+ const rows = new Array(questions.length);
161
+ let cursor = 0;
162
+ let done = 0;
163
+ // An empty answer means the SUBPROCESS failed (contention, timeout, OOM) — an infrastructure
164
+ // event, not a retrieval verdict. Scoring it as "not grounded" pollutes the quality metric with
165
+ // ops noise: one dead process under 6-wide load dragged grounded's lower bound below baseline and
166
+ // failed a gate that retrieval never failed (caught live, 2026-07-10, question ho-04 → "ce=—").
167
+ // Retry once; if it still returns nothing, the row is an INFRA ERROR and the run is inconclusive.
168
+ const infraErrors = [];
169
+ const runOne = async () => {
170
+ while (true) {
171
+ const i = cursor++;
172
+ if (i >= questions.length) return;
173
+ const q = questions[i];
174
+ let out = await ask(q.query);
175
+ if (!out) { await new Promise((r) => setTimeout(r, 2000)); out = await ask(q.query); }
176
+ if (!out) {
177
+ infraErrors.push(q.id);
178
+ done++;
179
+ process.stderr.write(`\r[eval] ${done}/${questions.length} ! ${q.id} (infra) `);
180
+ continue;
181
+ }
182
+ const v = await verifyGrounding(out, KB);
183
+ const graded = gradeQuestion(q, { grounded: v.grounded, citations: v.citations, bannerPresent: /GIST STATUS/.test(out) });
184
+ const top = v.citations?.[0] ?? null;
185
+ rows[i] = {
186
+ id: q.id, stratum: q.stratum, query: q.query,
187
+ citedRepo: top?.repo ?? null, citedPath: top?.fullPath ?? null, ce: top?.ce ?? null,
188
+ ...graded,
189
+ };
190
+ done++;
191
+ process.stderr.write(`\r[eval] ${done}/${questions.length} ${graded.pass ? '✓' : '✗'} ${q.id} `);
192
+ }
193
+ };
194
+ await Promise.all(Array.from({ length: Math.min(CONC, questions.length) }, runOne));
195
+ process.stderr.write('\n');
196
+
197
+ if (infraErrors.length) {
198
+ console.error(`[eval-brain] INCONCLUSIVE — ${infraErrors.length} infrastructure failure(s) after retry: ${infraErrors.join(', ')}`);
199
+ console.error(' A dead subprocess is not a retrieval verdict. Re-run (consider EVAL_CONCURRENCY=3 on a loaded machine).');
200
+ process.exit(2); // never a quality verdict, never a recorded baseline
201
+ }
202
+
203
+ const score = aggregate(rows.filter(Boolean));
204
+
205
+ if (JSON_OUT) {
206
+ console.log(JSON.stringify({ score, rows }, null, 2));
207
+ } else {
208
+ console.log(`\n# eval-brain — frozen held-out set (${rows.length} questions)\n`);
209
+ console.log('| metric | k/n | rate | 95% Wilson |');
210
+ console.log('|---|---|---|---|');
211
+ for (const [m, s] of Object.entries(score)) {
212
+ if (!s.n) continue;
213
+ console.log(`| ${m} | ${s.k}/${s.n} | ${(s.p * 100).toFixed(1)}% | [${(s.lo * 100).toFixed(1)}%, ${(s.hi * 100).toFixed(1)}%] |`);
214
+ }
215
+ const fails = rows.filter((r) => !r.pass);
216
+ if (fails.length) {
217
+ console.log(`\nFailures (${fails.length}):`);
218
+ for (const f of fails) console.log(` ✗ ${f.id} [${f.stratum}] → ${f.citedRepo ?? '—'} ce=${f.ce ?? '—'} ${f.query.slice(0, 64)}`);
219
+ }
220
+ console.log('\nGrounded = the cited passage exists on disk. Routed = the owning repo answered.');
221
+ console.log('Abstain = the brain declined an out-of-corpus question. Banner = a winning gist chunk carried its provenance.');
222
+ console.log('No model graded anything here.\n');
223
+ }
224
+
225
+ if (RECORD) {
226
+ fs.mkdirSync(path.dirname(BASELINE), { recursive: true });
227
+ fs.writeFileSync(BASELINE, JSON.stringify({ recorded: new Date().toISOString(), n: rows.length, score }, null, 2) + '\n');
228
+ console.error(`[eval-brain] baseline recorded (${rows.length} questions)`);
229
+ }
230
+
231
+ if (GATE) {
232
+ const baseline = fs.existsSync(BASELINE) ? JSON.parse(fs.readFileSync(BASELINE, 'utf8')).score : null;
233
+ const g = gateAgainst(score, baseline);
234
+ if (!g.pass) {
235
+ console.error(`[eval-brain] FAIL (fail-closed): ${g.regressions.join('; ')}`);
236
+ process.exit(1);
237
+ }
238
+ console.error('[eval-brain] PASS: every metric holds its baseline Wilson lower bound');
239
+ }
240
+ }
241
+
242
+ if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
243
+ await main();
244
+ }
@@ -0,0 +1,121 @@
1
+ #!/usr/bin/env node
2
+ // fix-metaharness-memretrieve.mjs — repair (and guard) a real bug in Ruflo's metaharness plugin.
3
+ //
4
+ // THE BUG
5
+ // -------
6
+ // `audit-list.mjs` and `audit-trend.mjs` read a stored audit record with:
7
+ // npx @claude-flow/cli memory retrieve --namespace <ns> --key <key>
8
+ // ...WITHOUT `--format json`, then greedily `JSON.parse` whatever `{...}` they can scrape from
9
+ // human-formatted stdout. But `memory retrieve --format json` returns an ENVELOPE:
10
+ // { id, key, namespace, content: "<the audit record, JSON-stringified>", ... }
11
+ // so the record lives in `.content` (a string) and must be unwrapped. Without both changes,
12
+ // memRetrieve() returns null for EVERY key, and `metaharness_audit_list` /
13
+ // `metaharness_drift_from_history` silently report `records: []` while `totalInNamespace > 0`.
14
+ // (Symptom: "0 audit records" even right after a successful oia-audit persisted one.)
15
+ //
16
+ // WHY THIS SCRIPT EXISTS
17
+ // ----------------------
18
+ // Both files live inside the GLOBAL npm package `@claude-flow/cli`, so `npm update -g` (or any
19
+ // reinstall) silently reverts the patch. This has already happened twice. Rather than hand-edit
20
+ // a third time, this script re-applies it idempotently AND can verify it in CI / a doctor check.
21
+ //
22
+ // node scripts/fix-metaharness-memretrieve.mjs --check # exit 1 if reverted (guard)
23
+ // node scripts/fix-metaharness-memretrieve.mjs --apply # patch in place (idempotent)
24
+ //
25
+ // The durable fix is upstream; this keeps the local install honest until that lands.
26
+
27
+ import fs from 'node:fs';
28
+ import path from 'node:path';
29
+ import os from 'node:os';
30
+ import { pathToFileURL } from 'node:url';
31
+
32
+ const TARGET_DIR =
33
+ process.env.METAHARNESS_SCRIPTS_DIR ||
34
+ path.join(os.homedir(), '.npm-global/lib/node_modules/@claude-flow/cli/plugins/ruflo-metaharness/scripts');
35
+
36
+ export const FILES = ['audit-list.mjs', 'audit-trend.mjs'];
37
+ const SENTINEL = 'outer.content'; // present only when the fix is applied
38
+
39
+ const FIXED = `function memRetrieve(key) {
40
+ const r = spawnSync('npx', [
41
+ CLI_PKG, 'memory', 'retrieve',
42
+ '--namespace', NS, '--key', key, '--format', 'json',
43
+ ], { stdio: ['ignore', 'pipe', 'pipe'], encoding: 'utf-8', shell: process.platform === 'win32' });
44
+ if (r.status !== 0) return null;
45
+ const m = /\\{[\\s\\S]*\\}/.exec(r.stdout || '');
46
+ if (!m) return null;
47
+ try {
48
+ const outer = JSON.parse(m[0]);
49
+ // \`memory retrieve --format json\` wraps the record: { id, key, content: "<record JSON>" }.
50
+ // Unwrap \`content\` — parsing the envelope AS the record is the empty-results bug.
51
+ const inner = typeof outer.content === 'string' ? outer.content
52
+ : typeof outer.value === 'string' ? outer.value
53
+ : null;
54
+ if (inner) { try { return JSON.parse(inner); } catch { return null; } }
55
+ return (outer.startedAt || outer.composite) ? outer : null;
56
+ } catch { return null; }
57
+ }`;
58
+
59
+ // Match the whole memRetrieve function, up to the first closing brace at column 0.
60
+ const FN_RE = /function memRetrieve\(key\) \{[\s\S]*?\n\}/;
61
+
62
+ export function statusOf(file, dir = TARGET_DIR) {
63
+ const p = path.join(dir, file);
64
+ if (!fs.existsSync(p)) return { file, p, state: 'missing' };
65
+ const src = fs.readFileSync(p, 'utf-8');
66
+ if (src.includes(SENTINEL)) return { file, p, state: 'fixed', src };
67
+ if (FN_RE.test(src)) return { file, p, state: 'reverted', src };
68
+ return { file, p, state: 'unrecognized', src };
69
+ }
70
+
71
+ export function apply(dir = TARGET_DIR, log = console.log) {
72
+ let changed = 0;
73
+ for (const file of FILES) {
74
+ const s = statusOf(file, dir);
75
+ if (s.state === 'missing') { log(` – ${file}: not installed here — nothing to patch`); continue; }
76
+ if (s.state === 'fixed') { log(` ✓ ${file}: already fixed`); continue; }
77
+ if (s.state === 'unrecognized') { log(` ⚠ ${file}: memRetrieve() not recognized — upstream changed shape; skipping`); continue; }
78
+ fs.writeFileSync(s.p, s.src.replace(FN_RE, FIXED), 'utf-8');
79
+ log(` ✓ ${file}: PATCHED (envelope unwrap + --format json)`);
80
+ changed++;
81
+ }
82
+ log(changed ? `\napplied to ${changed} file(s).` : '\nnothing to do — already healthy.');
83
+ return 0;
84
+ }
85
+
86
+ // A guard that cries wolf gets ignored. `missing` means the metaharness plugin simply isn't
87
+ // installed on this machine (the common case in CI) — that is NOT a failure. Only a file that
88
+ // EXISTS and has lost the fix is a failure, because that is the exact state an `npm update -g`
89
+ // leaves behind, and the symptom is silent (`records: []`, never an error).
90
+ export function check(dir = TARGET_DIR, log = console.log) {
91
+ let reverted = 0;
92
+ let present = 0;
93
+ for (const file of FILES) {
94
+ const s = statusOf(file, dir);
95
+ if (s.state === 'missing') { log(` – ${file}: n/a (metaharness plugin not installed)`); continue; }
96
+ present++;
97
+ if (s.state === 'reverted') reverted++;
98
+ log(` ${s.state === 'fixed' ? '✓' : '✗'} ${file}: ${s.state}`);
99
+ }
100
+ if (!present) {
101
+ log('\n– metaharness plugin not installed — guard not applicable (pass).');
102
+ return 0;
103
+ }
104
+ if (reverted) {
105
+ log(`\n✗ ${reverted} file(s) reverted — an npm update wiped the fix.`);
106
+ log(' Repair: node scripts/fix-metaharness-memretrieve.mjs --apply');
107
+ return 1;
108
+ }
109
+ log('\n✓ metaharness memRetrieve fix intact.');
110
+ return 0;
111
+ }
112
+
113
+ // Only act when run directly, so tests can import the pure functions above.
114
+ // Compare in URL space (pathToFileURL), not by decoding a URL into a path: `new URL(...).pathname`
115
+ // yields "/D:/..." on Windows and never matches, so `--apply` would silently no-op there.
116
+ const invokedDirectly = process.argv[1] && pathToFileURL(path.resolve(process.argv[1])).href === import.meta.url;
117
+ if (invokedDirectly) {
118
+ const mode = process.argv.includes('--check') ? 'check' : 'apply';
119
+ console.log(`metaharness memRetrieve ${mode} — ${TARGET_DIR}\n`);
120
+ process.exit(mode === 'check' ? check() : apply());
121
+ }
@@ -0,0 +1,291 @@
1
+ #!/usr/bin/env node
2
+ // Supervised fix lifecycle boundary: create one isolated writer lane, then seal a clean,
3
+ // content-bound handoff. This command intentionally cannot merge, push, publish, promote,
4
+ // delete, reset, or force-remove anything. Immutable release authority remains release-proof.
5
+ import { createHash } from 'node:crypto';
6
+ import { spawnSync } from 'node:child_process';
7
+ import fs from 'node:fs';
8
+ import path from 'node:path';
9
+ import { pathToFileURL } from 'node:url';
10
+
11
+ const FULL_SHA = /^[a-f0-9]{40}$/i;
12
+ const SCOPED_BRANCH = /^(fix|bugfix|hotfix|issue|codex)\/[A-Za-z0-9][A-Za-z0-9._/-]*$/;
13
+ const SAFE_SCOPE = /^[A-Za-z0-9][A-Za-z0-9._-]*$/;
14
+ const RELEASE_COMMAND = 'node scripts/release-proof.mjs --candidate <candidate-receipt.json>';
15
+
16
+ function fail(message) {
17
+ throw new Error(message);
18
+ }
19
+
20
+ function command(commandName, args, options = {}) {
21
+ return spawnSync(commandName, args, {
22
+ encoding: options.encoding ?? 'utf8',
23
+ cwd: options.cwd,
24
+ maxBuffer: 16 * 1024 * 1024,
25
+ });
26
+ }
27
+
28
+ function git(cwd, args, { allowFailure = false, binary = false } = {}) {
29
+ const result = command('git', ['-C', cwd, ...args], {
30
+ cwd,
31
+ encoding: binary ? null : 'utf8',
32
+ });
33
+ if (!allowFailure && result.status !== 0) {
34
+ const detail = String(result.stderr || result.stdout || result.error?.message || 'unknown git failure').trim();
35
+ fail(`git ${args.join(' ')} failed: ${detail}`);
36
+ }
37
+ return result;
38
+ }
39
+
40
+ function output(cwd, args) {
41
+ return String(git(cwd, args).stdout || '').trim();
42
+ }
43
+
44
+ function parseArgs(args) {
45
+ const parsed = { evidence: [] };
46
+ for (let index = 0; index < args.length; index += 1) {
47
+ const token = args[index];
48
+ if (!token.startsWith('--')) fail(`unexpected argument: ${token}`);
49
+ const key = token.slice(2);
50
+ const value = args[index + 1];
51
+ if (!value || value.startsWith('--')) fail(`missing value for --${key}`);
52
+ index += 1;
53
+ if (key === 'evidence') parsed.evidence.push(value);
54
+ else if (['integration', 'base', 'branch', 'worktree', 'scope', 'out'].includes(key)) parsed[key] = value;
55
+ else fail(`unknown option: --${key}`);
56
+ }
57
+ return parsed;
58
+ }
59
+
60
+ function requireAbsolute(value, label) {
61
+ if (!value) fail(`${label} is required`);
62
+ if (!path.isAbsolute(value)) fail(`${label} must be an absolute path`);
63
+ return path.resolve(value);
64
+ }
65
+
66
+ function requireScope(value) {
67
+ if (!SAFE_SCOPE.test(String(value || ''))) fail('scope must be a bounded identifier using letters, digits, dot, underscore, or dash');
68
+ return value;
69
+ }
70
+
71
+ function requireClean(cwd, role) {
72
+ const dirty = output(cwd, ['status', '--porcelain=v1', '--untracked-files=all']);
73
+ if (dirty) fail(`${role} worktree is dirty; preserve it and reconcile before continuing`);
74
+ }
75
+
76
+ function commonDirectory(cwd) {
77
+ const raw = output(cwd, ['rev-parse', '--git-common-dir']);
78
+ return fs.realpathSync(path.resolve(cwd, raw));
79
+ }
80
+
81
+ function currentBranch(cwd) {
82
+ const branch = output(cwd, ['branch', '--show-current']);
83
+ if (!branch) fail(`${cwd} is detached; a named branch is required`);
84
+ return branch;
85
+ }
86
+
87
+ function validateBase(cwd, base) {
88
+ if (!FULL_SHA.test(String(base || ''))) fail('base lineage must be an explicit full 40-character commit SHA');
89
+ const resolved = output(cwd, ['rev-parse', '--verify', `${base}^{commit}`]);
90
+ if (resolved !== base) fail('base lineage does not resolve to the explicit commit SHA');
91
+ }
92
+
93
+ function requireScopedBranch(branch, integrationBranch) {
94
+ if (!SCOPED_BRANCH.test(String(branch || '')) || branch === integrationBranch) {
95
+ fail('writer must use a scoped non-main branch (fix/, bugfix/, hotfix/, issue/, or codex/) distinct from integration');
96
+ }
97
+ }
98
+
99
+ function worktrees(cwd) {
100
+ const text = output(cwd, ['worktree', 'list', '--porcelain']);
101
+ return text.split(/\n\n+/).filter(Boolean).map((block) => {
102
+ const fields = {};
103
+ for (const line of block.split('\n')) {
104
+ const space = line.indexOf(' ');
105
+ fields[space === -1 ? line : line.slice(0, space)] = space === -1 ? true : line.slice(space + 1);
106
+ }
107
+ return fields;
108
+ });
109
+ }
110
+
111
+ function sameFilesystemEntry(left, right) {
112
+ try {
113
+ const leftStat = fs.statSync(left, { bigint: true });
114
+ const rightStat = fs.statSync(right, { bigint: true });
115
+ return leftStat.dev === rightStat.dev && leftStat.ino === rightStat.ino;
116
+ } catch {
117
+ return false;
118
+ }
119
+ }
120
+
121
+ function assertSingleOwner(cwd, branch, expectedPath, role) {
122
+ const ref = `refs/heads/${branch}`;
123
+ const owners = worktrees(cwd).filter((item) => item.branch === ref);
124
+ if (owners.length !== 1 || !sameFilesystemEntry(owners[0].worktree, expectedPath)) {
125
+ fail(`${role} branch has a shared writer or is not owned by the declared worktree`);
126
+ }
127
+ }
128
+
129
+ function receipt(payload) {
130
+ const encoded = JSON.stringify(payload);
131
+ return { ...payload, receiptId: `sha256:${createHash('sha256').update(encoded).digest('hex')}` };
132
+ }
133
+
134
+ function emit(value, outPath = null) {
135
+ const rendered = `${JSON.stringify(value, null, 2)}\n`;
136
+ if (outPath) {
137
+ const target = requireAbsolute(outPath, 'out');
138
+ if (fs.existsSync(target)) fail(`refusing to overwrite existing receipt: ${target}`);
139
+ fs.writeFileSync(target, rendered, { flag: 'wx', mode: 0o600 });
140
+ }
141
+ process.stdout.write(rendered);
142
+ }
143
+
144
+ function pendingRealPath(value, label) {
145
+ const absolute = requireAbsolute(value, label);
146
+ const parent = path.dirname(absolute);
147
+ if (!fs.existsSync(parent) || !fs.statSync(parent).isDirectory()) fail(`${label} parent directory does not exist: ${parent}`);
148
+ return path.join(fs.realpathSync(parent), path.basename(absolute));
149
+ }
150
+
151
+ function isWithin(root, candidate) {
152
+ const relative = path.relative(fs.realpathSync(root), candidate);
153
+ return relative === '' || (!relative.startsWith(`..${path.sep}`) && relative !== '..' && !path.isAbsolute(relative));
154
+ }
155
+
156
+ export function startFixWorkstream(options) {
157
+ const integration = requireAbsolute(options.integration, 'integration');
158
+ const writer = requireAbsolute(options.worktree, 'worktree');
159
+ const scope = requireScope(options.scope);
160
+ validateBase(integration, options.base);
161
+ requireClean(integration, 'integration');
162
+ const integrationSha = output(integration, ['rev-parse', 'HEAD']);
163
+ if (integrationSha !== options.base) fail('base lineage must equal the current integration HEAD');
164
+ const integrationBranch = currentBranch(integration);
165
+ requireScopedBranch(options.branch, integrationBranch);
166
+ assertSingleOwner(integration, integrationBranch, integration, 'integration');
167
+
168
+ if (writer === integration || writer.startsWith(`${integration}${path.sep}`)) fail('writer worktree must be isolated from the integration worktree');
169
+ if (fs.existsSync(writer)) fail('writer worktree path already exists; it will not be removed or overwritten');
170
+ const branchRef = git(integration, ['show-ref', '--verify', `refs/heads/${options.branch}`], { allowFailure: true });
171
+ if (branchRef.status === 0) {
172
+ const owner = worktrees(integration).find((item) => item.branch === `refs/heads/${options.branch}`);
173
+ fail(owner ? `branch is already checked out at ${owner.worktree}; shared writers are forbidden` : 'writer branch already exists; branch reuse is forbidden');
174
+ }
175
+
176
+ const created = git(integration, ['worktree', 'add', '-b', options.branch, writer, options.base], { allowFailure: true });
177
+ if (created.status !== 0) {
178
+ const detail = String(created.stderr || created.stdout || created.error?.message || 'unknown git failure').trim();
179
+ fail(`git worktree add failed without cleanup or force removal: ${detail}`);
180
+ }
181
+
182
+ if (commonDirectory(writer) !== commonDirectory(integration)) fail('created writer is not in the integration repository lineage');
183
+ if (output(writer, ['rev-parse', 'HEAD']) !== options.base) fail('created writer HEAD does not match the explicit integration base');
184
+ requireClean(writer, 'writer');
185
+ assertSingleOwner(writer, options.branch, writer, 'writer');
186
+
187
+ return receipt({
188
+ schemaVersion: 1,
189
+ phase: 'workstream-start',
190
+ scope,
191
+ branch: options.branch,
192
+ worktree: writer,
193
+ integrationWorktree: integration,
194
+ integrationBranch,
195
+ baseSha: options.base,
196
+ baseTree: output(integration, ['rev-parse', `${options.base}^{tree}`]),
197
+ repositoryCommonDir: commonDirectory(integration),
198
+ publishAuthorized: false,
199
+ promotionAuthorized: false,
200
+ releaseGate: { authority: 'release-proof', command: RELEASE_COMMAND },
201
+ });
202
+ }
203
+
204
+ function requireAncestor(cwd, ancestor, descendant, message) {
205
+ const result = git(cwd, ['merge-base', '--is-ancestor', ancestor, descendant], { allowFailure: true });
206
+ if (result.status !== 0) fail(message);
207
+ }
208
+
209
+ function evidenceFiles(values) {
210
+ if (!Array.isArray(values) || values.length === 0) fail('at least one focused test or review evidence file is required');
211
+ return values.map((value) => {
212
+ const absolute = requireAbsolute(value, 'evidence');
213
+ if (!fs.existsSync(absolute) || !fs.statSync(absolute).isFile()) fail(`evidence file does not exist or is not a file: ${absolute}`);
214
+ const bytes = fs.readFileSync(absolute);
215
+ return { path: absolute, sha256: createHash('sha256').update(bytes).digest('hex'), bytes: bytes.length };
216
+ });
217
+ }
218
+
219
+ export function sealFixHandoff(options) {
220
+ const integration = requireAbsolute(options.integration, 'integration');
221
+ const writer = requireAbsolute(options.worktree, 'worktree');
222
+ const scope = requireScope(options.scope);
223
+ if (options.out) {
224
+ const out = pendingRealPath(options.out, 'out');
225
+ if (isWithin(integration, out) || isWithin(writer, out)) fail('handoff receipt must be outside the integration and writer worktrees');
226
+ if (fs.existsSync(out)) fail(`refusing to overwrite existing receipt: ${out}`);
227
+ }
228
+ validateBase(integration, options.base);
229
+ validateBase(writer, options.base);
230
+ requireClean(integration, 'integration');
231
+ requireClean(writer, 'writer');
232
+ if (commonDirectory(writer) !== commonDirectory(integration)) fail('writer and integration do not share repository lineage');
233
+
234
+ const integrationBranch = currentBranch(integration);
235
+ const writerBranch = currentBranch(writer);
236
+ requireScopedBranch(writerBranch, integrationBranch);
237
+ assertSingleOwner(integration, integrationBranch, integration, 'integration');
238
+ assertSingleOwner(writer, writerBranch, writer, 'writer');
239
+ const integrationSha = output(integration, ['rev-parse', 'HEAD']);
240
+ const writerSha = output(writer, ['rev-parse', 'HEAD']);
241
+ requireAncestor(integration, options.base, integrationSha, 'integration lineage does not descend from the declared base');
242
+ requireAncestor(writer, options.base, writerSha, 'writer lineage does not descend from the declared base');
243
+ if (writerSha === options.base) fail('writer contains no committed change to hand off');
244
+
245
+ const patchBytes = git(writer, ['diff', '--binary', `${options.base}..${writerSha}`], { binary: true }).stdout;
246
+ const changedPaths = output(writer, ['diff', '--name-status', `${options.base}..${writerSha}`])
247
+ .split('\n').filter(Boolean).map((line) => {
248
+ const [status, ...names] = line.split('\t');
249
+ return { status, path: names.join(' -> ') };
250
+ });
251
+ if (changedPaths.length === 0) fail('writer commit has no content change to hand off');
252
+
253
+ return receipt({
254
+ schemaVersion: 1,
255
+ phase: 'fix-handoff',
256
+ scope,
257
+ branch: writerBranch,
258
+ worktree: writer,
259
+ integrationWorktree: integration,
260
+ baseSha: options.base,
261
+ integrationSha,
262
+ writerSha,
263
+ writerTree: output(writer, ['rev-parse', `${writerSha}^{tree}`]),
264
+ patchSha256: createHash('sha256').update(patchBytes).digest('hex'),
265
+ changedPaths,
266
+ evidence: evidenceFiles(options.evidence),
267
+ publishAuthorized: false,
268
+ promotionAuthorized: false,
269
+ nextAuthority: 'integration-owner',
270
+ releaseGate: { authority: 'release-proof', command: RELEASE_COMMAND },
271
+ });
272
+ }
273
+
274
+ export function main(args = process.argv.slice(2)) {
275
+ const [operation, ...rest] = args;
276
+ if (!['start', 'handoff'].includes(operation)) {
277
+ console.error('fix-workstream: supported commands are start and handoff; this tool cannot merge, publish, promote, or clean up');
278
+ return 2;
279
+ }
280
+ try {
281
+ const options = parseArgs(rest);
282
+ const result = operation === 'start' ? startFixWorkstream(options) : sealFixHandoff(options);
283
+ emit(result, operation === 'handoff' ? options.out : null);
284
+ return 0;
285
+ } catch (error) {
286
+ console.error(`fix-workstream: ${error.message}`);
287
+ return 1;
288
+ }
289
+ }
290
+
291
+ if (process.argv[1] && pathToFileURL(path.resolve(process.argv[1])).href === import.meta.url) process.exitCode = main();