ruvnet-brain 4.0.1 → 4.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (185) hide show
  1. package/.claude-plugin/marketplace.json +1 -0
  2. package/README.md +4 -4
  3. package/bin/install.mjs +100 -5
  4. package/console/CONTRACT.md +172 -0
  5. package/console/activity.js +753 -0
  6. package/console/app.js +4189 -0
  7. package/console/architecture.html +1221 -0
  8. package/console/assets/depth-1.webp +0 -0
  9. package/console/assets/depth-2.webp +0 -0
  10. package/console/assets/depth-3.webp +0 -0
  11. package/console/assets/harness-vs-plain.svg +259 -0
  12. package/console/assets/hero.webp +0 -0
  13. package/console/assets/memory.webp +0 -0
  14. package/console/assets/metaharness.svg +247 -0
  15. package/console/index.html +777 -0
  16. package/console/install-architecture.html +162 -0
  17. package/console/install-mockup.html +543 -0
  18. package/console/style.css +2144 -0
  19. package/console/tips.css +926 -0
  20. package/console/tips.html +858 -0
  21. package/console/tips.js +128 -0
  22. package/docs/RELEASE-NOTES-4.0.md +88 -0
  23. package/kb/model-requirements.mjs +37 -6
  24. package/keys/ruvnet-brain-signing.pub.pem +3 -0
  25. package/package.json +8 -22
  26. package/plugin/.claude-plugin/marketplace.json +1 -0
  27. package/plugin/.claude-plugin/plugin.json +2 -3
  28. package/plugin/.codex-plugin/plugin.json +1 -1
  29. package/plugin/commands/brain-console.md +2 -2
  30. package/plugin/commands/configure.md +3 -2
  31. package/plugin/commands/rvbc.md +4 -3
  32. package/plugin/commands/rvcb.md +2 -2
  33. package/plugin/hooks/hooks.json +1 -2
  34. package/plugin/mcp/managed-cli-interface.mjs +47 -4
  35. package/plugin/mcp/server.mjs +21 -0
  36. package/plugin/scripts/detach.mjs +14 -0
  37. package/plugin/scripts/first-session-worker.mjs +38 -0
  38. package/plugin/scripts/ground-ruvnet.sh +16 -6
  39. package/plugin/scripts/hook-shim.mjs +7 -7
  40. package/plugin/scripts/learn-capture.sh +22 -3
  41. package/plugin/scripts/learn-flush.mjs +21 -4
  42. package/plugin/scripts/runtime-preferences.mjs +269 -0
  43. package/plugin/scripts/session-start-core.mjs +477 -0
  44. package/plugin/scripts/session-start.sh +3 -858
  45. package/plugin/skills/brain-console/SKILL.md +4 -2
  46. package/plugin/skills/release-proof/SKILL.md +81 -0
  47. package/plugin/skills/release-proof/agents/openai.yaml +4 -0
  48. package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
  49. package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
  50. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
  51. package/plugin/skills/rvbc/SKILL.md +9 -6
  52. package/scripts/adr-backfill.mjs +107 -0
  53. package/scripts/advocacy-outcomes.mjs +808 -0
  54. package/scripts/agentdb-context.mjs +216 -0
  55. package/scripts/agentdb-fleet-doctor.mjs +101 -0
  56. package/scripts/ascii-drift.mjs +236 -0
  57. package/scripts/behavioral-l1-l4.mjs +210 -0
  58. package/scripts/brain-capability-check.mjs +72 -0
  59. package/scripts/brain-grade-groundtruth.mjs +100 -0
  60. package/scripts/brain-latency-50.mjs +227 -0
  61. package/scripts/brain-novice-50.mjs +189 -0
  62. package/scripts/brain-stamp.mjs +94 -0
  63. package/scripts/brain-state.mjs +212 -0
  64. package/scripts/build-bundle.mjs +522 -0
  65. package/scripts/build-concepts.mjs +132 -0
  66. package/scripts/build-l2.mjs +71 -0
  67. package/scripts/build-primer.mjs +73 -0
  68. package/scripts/build-symbols.mjs +68 -0
  69. package/scripts/calibrate-router.mjs +97 -0
  70. package/scripts/capability-audit.mjs +321 -0
  71. package/scripts/capability-registry.mjs +876 -0
  72. package/scripts/check-indexation.mjs +108 -0
  73. package/scripts/check-legibility.mjs +189 -0
  74. package/scripts/ci/build-fixture-kb.mjs +67 -0
  75. package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
  76. package/scripts/ci/learning-replay-recorder.mjs +59 -0
  77. package/scripts/ci/mutate-hook-timeout.mjs +70 -0
  78. package/scripts/ci/stranger-fixture-stage.mjs +17 -0
  79. package/scripts/ci/stranger-scenario.mjs +228 -0
  80. package/scripts/ci/stranger-timeout.mjs +25 -0
  81. package/scripts/ci-verdict.mjs +29 -0
  82. package/scripts/claims-verify.mjs +710 -0
  83. package/scripts/clear-claude-tmp.sh +31 -0
  84. package/scripts/console-engine.mjs +434 -0
  85. package/scripts/console-engine.test.mjs +125 -0
  86. package/scripts/corpus-qa.mjs +250 -0
  87. package/scripts/correction-detect-embed.mjs +346 -0
  88. package/scripts/correction-detect-measure.mjs +270 -0
  89. package/scripts/correction-detect.mjs +686 -0
  90. package/scripts/count-chunks.mjs +54 -0
  91. package/scripts/described-questions.json +30 -0
  92. package/scripts/design-grade.mjs +58 -0
  93. package/scripts/dev-plugin-link.sh +105 -0
  94. package/scripts/distill-project.mjs +200 -0
  95. package/scripts/doc-currency.mjs +801 -0
  96. package/scripts/eval-brain.mjs +244 -0
  97. package/scripts/fix-metaharness-memretrieve.mjs +121 -0
  98. package/scripts/full-hints.mjs +87 -0
  99. package/scripts/gate.sh +39 -0
  100. package/scripts/gates.mjs +146 -0
  101. package/scripts/gen-console-images.mjs +54 -0
  102. package/scripts/gen-images.mjs +47 -0
  103. package/scripts/git-clone-refresh.mjs +52 -0
  104. package/scripts/git-hooks/pre-push +126 -0
  105. package/scripts/goal-match.mjs +398 -0
  106. package/scripts/goldie-research.mjs +223 -0
  107. package/scripts/goldie-weekly.sh +67 -0
  108. package/scripts/health-repair.mjs +250 -0
  109. package/scripts/helix-scenario-questions.json +10 -0
  110. package/scripts/ingest-gists.mjs +230 -0
  111. package/scripts/ingest-meeting.mjs +115 -0
  112. package/scripts/ingest-repo.mjs +79 -0
  113. package/scripts/install-npx-witness.sh +49 -0
  114. package/scripts/issue-fix.mjs +639 -0
  115. package/scripts/issue-watch.mjs +276 -0
  116. package/scripts/issue4-close-note.md +31 -0
  117. package/scripts/key-canary.mjs +91 -0
  118. package/scripts/latency-to-surface.mjs +233 -0
  119. package/scripts/learning-enable.mjs +380 -0
  120. package/scripts/learning-replay.mjs +1570 -0
  121. package/scripts/learnings.mjs +62 -0
  122. package/scripts/lesson-gate.mjs +680 -0
  123. package/scripts/lesson-lifecycle.mjs +449 -0
  124. package/scripts/lesson-promote.mjs +262 -0
  125. package/scripts/lesson-ratify.mjs +98 -0
  126. package/scripts/lesson-seed.mjs +252 -0
  127. package/scripts/lesson-store.mjs +447 -0
  128. package/scripts/loop-checkpoint.mjs +86 -0
  129. package/scripts/memdb-health.sh +14 -0
  130. package/scripts/memory-doctor.mjs +271 -0
  131. package/scripts/model-catalog.mjs +79 -0
  132. package/scripts/nightly-controller.mjs +66 -0
  133. package/scripts/nightly-gists.sh +72 -0
  134. package/scripts/nightly-wrapper.sh +180 -0
  135. package/scripts/notify.sh +12 -0
  136. package/scripts/npx-witness.sh +56 -0
  137. package/scripts/onboarding-console.mjs +2749 -0
  138. package/scripts/private-fence.mjs +69 -0
  139. package/scripts/proactivity-metrics.mjs +118 -0
  140. package/scripts/proof-questions.json +56 -0
  141. package/scripts/prove.mjs +95 -0
  142. package/scripts/proxy/claude-proxied.sh +57 -0
  143. package/scripts/proxy/proxy-revert.sh +59 -0
  144. package/scripts/proxy/proxy-up.sh +60 -0
  145. package/scripts/proxy/proxy-verify.mjs +142 -0
  146. package/scripts/published-surface-probe.mjs +241 -0
  147. package/scripts/qe/card-lane-gate.mjs +162 -0
  148. package/scripts/qe/session-start-gate.mjs +229 -0
  149. package/scripts/qe/ux-suite.mjs +323 -0
  150. package/scripts/reconcile-project.mjs +0 -0
  151. package/scripts/record-lesson.mjs +113 -0
  152. package/scripts/refresh-model-catalog.mjs +99 -0
  153. package/scripts/release-proof.mjs +9 -0
  154. package/scripts/release-vector.mjs +281 -0
  155. package/scripts/release.mjs +395 -0
  156. package/scripts/remedy-registry.mjs +247 -0
  157. package/scripts/rerank-cap-eval.mjs +265 -0
  158. package/scripts/rerank-cap-warm-ab.mjs +129 -0
  159. package/scripts/route-cheap.mjs +20 -15
  160. package/scripts/router-utilization.mjs +182 -0
  161. package/scripts/routing-flywheel.mjs +596 -0
  162. package/scripts/rvf-generation.mjs +104 -0
  163. package/scripts/rvf-index-audit.mjs +138 -0
  164. package/scripts/self-update.mjs +508 -0
  165. package/scripts/selfcheck.mjs +7 -1
  166. package/scripts/sign-bundle.mjs +69 -0
  167. package/scripts/signal-watch.mjs +171 -0
  168. package/scripts/stack-sync.mjs +469 -0
  169. package/scripts/stamp-existing-rvf-generations.mjs +53 -0
  170. package/scripts/stamp-sweep.mjs +144 -0
  171. package/scripts/status-honesty.mjs +102 -0
  172. package/scripts/sync-version.mjs +217 -0
  173. package/scripts/token-report.mjs +102 -0
  174. package/scripts/top100-benchmark.mjs +479 -0
  175. package/scripts/top100-corpus.mjs +112 -0
  176. package/scripts/top100-semantic-assertions.mjs +449 -0
  177. package/scripts/update-apply.mjs +9 -0
  178. package/scripts/upgrade-notice.mjs +14 -0
  179. package/scripts/verify-bundle.mjs +51 -0
  180. package/scripts/verify-channels.mjs +184 -0
  181. package/scripts/verify-model-catalog.mjs +104 -0
  182. package/scripts/verify-nightly-close-issue4.sh +31 -0
  183. package/scripts/version.mjs +40 -0
  184. package/scripts/wired-check.mjs +864 -0
  185. package/plugin/scripts/finalize-token-meter.mjs +0 -25
@@ -0,0 +1,71 @@
1
+ #!/usr/bin/env node
2
+ // build-l2.mjs — L2 concept/synthesis layer (the validated closer for overview/playbook questions).
3
+ // For a synthesis topic: retrieve the real source (rerankKb), have a strong LLM SYNTHESIZE a complete,
4
+ // prescriptive, source-CITED article using ONLY that source, validate the citations are real, then
5
+ // GRADE the article with the 3-vendor panel (does it now fully answer?). Proves L2 closes Q5/Q7/Q8.
6
+ //
7
+ // node scripts/build-l2.mjs --name ruflo --variant big
8
+ import fs from 'node:fs';
9
+ import path from 'node:path';
10
+ import { fileURLToPath } from 'node:url';
11
+ import { rerankKb } from '../kb/forge-rerank.mjs';
12
+
13
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
14
+ const arg = (f, d) => { const i = process.argv.indexOf(f); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
15
+ const NAME = arg('--name', 'ruflo'), VARIANT = arg('--variant', 'big');
16
+ const GEN_MODEL = arg('--gen', 'anthropic/claude-3.5-sonnet');
17
+ const JUDGES = ['openai/gpt-4o-mini', 'deepseek/deepseek-chat', 'meta-llama/llama-3.3-70b-instruct'];
18
+ const REPO = process.env.RUVNET_REPO || path.join(ROOT, '..', 'ruvnet-repos', NAME);
19
+ const KEY = process.env.OPENROUTER_API_KEY || (fs.readFileSync((process.env.RUVNET_ENV_FILE || '.env'), 'utf8').match(/^OPENROUTER_API_KEY=(.+)$/m) || [])[1]?.trim();
20
+
21
+ // Per-repo synthesis topics: prefer kb/l2-topics.<name>.json ([{slug,q},...] — the graded synthesis
22
+ // gaps), else fall back to the original ruflo defaults so existing behavior is unchanged.
23
+ const RUFLO_DEFAULT = [
24
+ { slug: 'guidance-mechanism', q: 'How does the guidance system produce a recommendation (guidance_recommend) — what is the mechanism?' },
25
+ { slug: 'adr-coverage', q: "Where are ruflo's ADRs kept and what areas do they cover?" },
26
+ { slug: 'memory-end-to-end', q: "How do I initialize and use ruflo's memory in a project end-to-end?" },
27
+ ];
28
+ const TOPICS_FILE = path.join(ROOT, 'kb', `l2-topics.${NAME}.json`);
29
+ const TOPICS = fs.existsSync(TOPICS_FILE) ? JSON.parse(fs.readFileSync(TOPICS_FILE, 'utf8')) : RUFLO_DEFAULT;
30
+
31
+ async function or(model, sys, usr, max = 1400) {
32
+ const r = await fetch('https://openrouter.ai/api/v1/chat/completions', {
33
+ method: 'POST', headers: { Authorization: `Bearer ${KEY}`, 'Content-Type': 'application/json' },
34
+ body: JSON.stringify({ model, messages: [{ role: 'system', content: sys }, { role: 'user', content: usr }], temperature: 0, max_tokens: max }),
35
+ });
36
+ if (!r.ok) throw new Error(`${model} HTTP ${r.status}`);
37
+ return (await r.json()).choices?.[0]?.message?.content || '';
38
+ }
39
+
40
+ fs.mkdirSync(path.join(ROOT, 'kb/l2'), { recursive: true });
41
+ for (const t of TOPICS) {
42
+ const hits = await rerankKb({ dir: path.join(ROOT, 'kb'), name: NAME, query: t.q, k: 8, variant: VARIANT });
43
+ const allowed = new Set(hits.map((h) => h.path));
44
+ const ctx = hits.map((h, i) => `[SOURCE ${i + 1}] ${h.path}\n${(h.fullText || '').slice(0, 2400)}`).join('\n\n');
45
+ const paths = hits.map((h) => h.path);
46
+ const sys = 'You are Ruv writing the authoritative internal explainer. Using ONLY the SOURCE excerpts provided, write a complete, prescriptive, actionable answer. You MUST cite real file paths inline in backticks for EVERY major claim, referencing at least 3 distinct SOURCE paths verbatim. Do NOT invent files or facts. If a part is not covered by the sources, say so explicitly rather than guessing.';
47
+ // ground-truth citation check: count how many of the PROVIDED source paths the article actually references
48
+ const countRefs = (txt) => [...new Set(paths.filter((p) => txt.includes(p) || txt.includes(p.split('/').pop())))];
49
+ const gen = (extra) => or(GEN_MODEL, sys + extra, `QUESTION: ${t.q}\n\nSOURCE PATHS YOU MUST CITE FROM (verbatim):\n${paths.map((p) => '- ' + p).join('\n')}\n\nSOURCES:\n${ctx}`, 1500)
50
+ .catch(() => or('openai/gpt-4o-mini', sys + extra, `QUESTION: ${t.q}\n\nSOURCES:\n${ctx}`, 1500));
51
+ let article = await gen('');
52
+ let realCited = countRefs(article);
53
+ if (realCited.length < 2) { article = await gen(' YOUR LAST ATTEMPT CITED TOO FEW REAL FILES — REJECTED. Cite at least 3 of the exact SOURCE PATHS listed above, verbatim, in backticks.'); realCited = countRefs(article); }
54
+ const accepted = realCited.length >= 2; // ADR-0002 R1: enforce grounding or REJECT
55
+ const odir = accepted ? 'l2' : 'l2/rejected';
56
+ fs.mkdirSync(path.join(ROOT, 'kb', odir), { recursive: true });
57
+ fs.writeFileSync(path.join(ROOT, 'kb', odir, `${t.slug}.md`), `# ${t.q}\n\n<!-- L2 synthesis · ${accepted ? 'ACCEPTED' : 'REJECTED (ungrounded)'} · ${realCited.length} verified source refs: ${realCited.slice(0, 5).join(', ')} -->\n\n${article}\n`);
58
+
59
+ // grade the article as the answer (3-vendor)
60
+ const jsys = 'Grade 1-100 whether this answer COMPLETELY and CORRECTLY answers the question (98=perfect/complete/actionable; incomplete-but-not-wrong=POISON<50). Return ONLY {"score":N,"reason":"<=15 words"}.';
61
+ const scores = [];
62
+ for (const j of JUDGES) {
63
+ try { const o = JSON.parse((await or(j, jsys, `QUESTION: ${t.q}\n\nANSWER:\n${article}`, 120)).match(/\{[\s\S]*\}/)[0]); scores.push(Number(o.score)); }
64
+ catch { /* skip */ }
65
+ }
66
+ const avg = scores.length ? (scores.reduce((a, b) => a + b, 0) / scores.length) : null;
67
+ console.log(`\n## ${t.slug} [${accepted ? 'ACCEPTED' : 'REJECTED — ungrounded'}]`);
68
+ console.log(` verified source refs: ${realCited.length} (${realCited.slice(0, 4).join(', ')})`);
69
+ console.log(` 3-vendor answer score: [${scores.join(', ')}] → avg ${avg?.toFixed(1)}`);
70
+ }
71
+ console.log('\nL2 articles written to kb/l2/. (Integration: include kb/l2 in the next KB rebuild so they are retrievable.)');
@@ -0,0 +1,73 @@
1
+ #!/usr/bin/env node
2
+ // build-primer.mjs — per-repo top-down PRIMER (the 6 comprehension archetypes), grounded.
3
+ // Same proven shape as build-l2: for each archetype, retrieve the real source (rerankKb), have a
4
+ // strong LLM synthesize a prescriptive, source-CITED section using ONLY that source, then assemble
5
+ // the sections into kb/<name>-primer.md and verify the whole primer cites >= MIN real files.
6
+ //
7
+ // The "capabilities" archetype is deliberate: it produces PROSE that names each real capability with
8
+ // the file that implements it — the high-confidence retrieval target that makes an assistant STOP
9
+ // doubting code-implemented capabilities (the capability-confidence gap measured on ruflo/rulake).
10
+ //
11
+ // node scripts/build-primer.mjs --name ruflo --variant big
12
+ import fs from 'node:fs';
13
+ import path from 'node:path';
14
+ import { fileURLToPath } from 'node:url';
15
+ import { rerankKb } from '../kb/forge-rerank.mjs';
16
+
17
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
18
+ const arg = (f, d) => { const i = process.argv.indexOf(f); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
19
+ const NAME = arg('--name', 'ruflo'), VARIANT = arg('--variant', 'big');
20
+ const KEY = process.env.OPENROUTER_API_KEY || (fs.readFileSync((process.env.RUVNET_ENV_FILE || '.env'), 'utf8').match(/^OPENROUTER_API_KEY=(.+)$/m) || [])[1]?.trim();
21
+ if (!KEY) { console.error('No OPENROUTER_API_KEY'); process.exit(2); }
22
+ const GEN_MODEL = arg('--gen', 'deepseek/deepseek-chat');
23
+ const JUDGES = ['openai/gpt-4o-mini', 'deepseek/deepseek-chat', 'meta-llama/llama-3.3-70b-instruct'];
24
+
25
+ // The 6 archetypes a raw repo can't answer top-down. {key,title,q}. The capabilities section is #2.
26
+ const ARCHES = [
27
+ { key: 'what', title: 'What it is & who it\'s for', q: `What is ${NAME} and who is it for?` },
28
+ { key: 'capabilities', title: 'Capabilities (what it can do)', q: `What can ${NAME} actually do? List its main capabilities, and for EACH name the source file that implements it.` },
29
+ { key: 'concepts', title: 'Core concepts & how they work', q: `What are ${NAME}'s core concepts and how does each one work?` },
30
+ { key: 'maturity', title: 'Maturity (shipped vs proposed)', q: `How mature is ${NAME}? Which features are shipped/accepted vs proposed (cite ADR status)?` },
31
+ { key: 'docs', title: 'Where the documentation lives', q: `Where is ${NAME}'s documentation organized (guides, ADRs, references)?` },
32
+ { key: 'use', title: 'How to use it end-to-end', q: `How do I install and use ${NAME} end-to-end?` },
33
+ ];
34
+
35
+ async function or(model, sys, usr, max = 1500) {
36
+ const r = await fetch('https://openrouter.ai/api/v1/chat/completions', {
37
+ method: 'POST', headers: { Authorization: `Bearer ${KEY}`, 'Content-Type': 'application/json' },
38
+ body: JSON.stringify({ model, messages: [{ role: 'system', content: sys }, { role: 'user', content: usr }], temperature: 0, max_tokens: max }),
39
+ });
40
+ if (!r.ok) throw new Error(`${model} HTTP ${r.status}`);
41
+ return (await r.json()).choices?.[0]?.message?.content || '';
42
+ }
43
+
44
+ const KB = path.join(ROOT, 'kb');
45
+ const allPaths = new Set();
46
+ let sections = [];
47
+ for (const a of ARCHES) {
48
+ const hits = await rerankKb({ dir: KB, name: NAME, query: a.q, k: 8, variant: VARIANT });
49
+ const paths = hits.map((h) => h.path);
50
+ paths.forEach((p) => allPaths.add(p));
51
+ const ctx = hits.map((h, i) => `[SOURCE ${i + 1}] ${h.path}\n${(h.fullText || '').slice(0, 2200)}`).join('\n\n');
52
+ const sys = `You are Ruv writing the authoritative primer for ${NAME}. Using ONLY the SOURCE excerpts, write the "${a.title}" section: complete, prescriptive, and CONFIDENT about what exists. Cite real file paths inline in backticks for every concrete claim (cite >= 2 distinct SOURCE paths verbatim). For capabilities, state plainly that the feature EXISTS and point to the implementing file — never hedge about whether it can do something the source shows. If something is genuinely not covered, say so; do not invent.`;
53
+ let body = await or(GEN_MODEL, sys, `SECTION: ${a.title}\nQUESTION: ${a.q}\n\nSOURCE PATHS (cite verbatim):\n${paths.map((p) => '- ' + p).join('\n')}\n\nSOURCES:\n${ctx}`)
54
+ .catch(() => or('openai/gpt-4o-mini', sys, `QUESTION: ${a.q}\n\nSOURCES:\n${ctx}`));
55
+ sections.push(`## ${a.title}\n\n${body.trim()}\n`);
56
+ console.log(` [${NAME}] section "${a.key}" — ${body.length} chars, ${paths.length} sources`);
57
+ }
58
+
59
+ const primer = `# ${NAME} — Primer\n\n<!-- Generated primer · grounded in real source via rerankKb (${VARIANT}) · archetypes: ${ARCHES.map(a => a.key).join(', ')} -->\n\n${sections.join('\n')}`;
60
+ const countRefs = (txt) => [...new Set([...allPaths].filter((p) => txt.includes(p) || txt.includes(p.split('/').pop())))];
61
+ const refs = countRefs(primer);
62
+ const outFile = path.join(KB, `${NAME}-primer.md`);
63
+ fs.writeFileSync(outFile, primer);
64
+
65
+ // 3-vendor "is this a complete, correct primer?" score (informational; the hard gate is citation count)
66
+ const jsys = 'Grade 1-100 whether this repo primer is COMPLETE, CORRECT and CONFIDENT for an engineer new to the repo (98=excellent/actionable; vague-or-hedgy=POISON<50). Return ONLY {"score":N,"reason":"<=15 words"}.';
67
+ const scores = [];
68
+ for (const j of JUDGES) { try { scores.push(Number(JSON.parse((await or(j, jsys, `PRIMER:\n${primer.slice(0, 9000)}`, 120)).match(/\{[\s\S]*\}/)[0]).score)); } catch { /* skip */ } }
69
+ const avg = scores.length ? (scores.reduce((a, b) => a + b, 0) / scores.length) : null;
70
+ console.log(`\n## ${NAME}-primer.md [${refs.length >= 6 ? 'GROUNDED' : 'THIN — only ' + refs.length + ' refs'}]`);
71
+ console.log(` verified source refs: ${refs.length}`);
72
+ console.log(` 3-vendor primer score: [${scores.join(', ')}] → avg ${avg?.toFixed(1)}`);
73
+ console.log(` written: kb/${NAME}-primer.md`);
@@ -0,0 +1,68 @@
1
+ #!/usr/bin/env node
2
+ // build-symbols.mjs — ADR-0003 point-deeper symbol index.
3
+ // Scans <name>.passages.jsonl and extracts a deterministic map from code symbols / file stems /
4
+ // package names → the SOURCE paths that define them, so retrieval can hard-route an implementation
5
+ // question to the real file instead of losing to a prose doc. Language-agnostic-ish (TS/JS/Rust/Py).
6
+ //
7
+ // node scripts/build-symbols.mjs --name ruflo → writes kb/<name>.symbols.json
8
+ import fs from 'node:fs';
9
+ import path from 'node:path';
10
+ import { fileURLToPath } from 'node:url';
11
+
12
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
13
+ const arg = (f, d) => { const i = process.argv.indexOf(f); return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : d; };
14
+ const NAME = arg('--name', 'ruflo');
15
+ const KB = path.join(ROOT, 'kb');
16
+ const PASSAGES = path.join(KB, `${NAME}.passages.jsonl`);
17
+
18
+ const isSourcePath = (p) => /\.(ts|tsx|js|jsx|mjs|cjs|rs|py|go)$/i.test(p)
19
+ && !/\.(test|spec|d)\.[tj]sx?$/i.test(p)
20
+ && !/(^|\/)(tests?|__tests__|testing|fixtures?|\.claude|examples?)\//i.test(p);
21
+
22
+ // symbol-definition patterns (capture the NAME)
23
+ const DEF_RES = [
24
+ /\bexport\s+(?:default\s+)?(?:async\s+)?function\s+([A-Za-z_$][\w$]*)/g,
25
+ /\bexport\s+(?:default\s+)?(?:abstract\s+)?class\s+([A-Za-z_$][\w$]*)/g,
26
+ /\bexport\s+(?:interface|type|enum)\s+([A-Za-z_$][\w$]*)/g,
27
+ /\bexport\s+const\s+([A-Za-z_$][\w$]*)\s*[:=]/g,
28
+ /\bpub\s+(?:async\s+)?fn\s+([a-z_][\w]*)/g, // rust
29
+ /\bpub\s+(?:struct|enum|trait)\s+([A-Za-z_][\w]*)/g, // rust
30
+ /(?:^|\s)class\s+([A-Za-z_$][\w$]*)/g,
31
+ ];
32
+ // MCP tool / handler names (snake_case strings registered as tools)
33
+ const TOOL_RES = [
34
+ /\bname:\s*['"]([a-z][a-z0-9_]+)['"]/g,
35
+ /(?:registerTool|tool|addTool|defineTool)\(\s*['"]([a-z][a-z0-9_]+)['"]/g,
36
+ /['"]([a-z]+_[a-z0-9_]+)['"]\s*:/g, // 'swarm_init': handler
37
+ ];
38
+
39
+ const bySymbol = Object.create(null); // exact identifier (lowercased) → Set(paths); null-proto: no __proto__/constructor collision
40
+ const byStem = Object.create(null); // file basename stem → Set(paths)
41
+ const byPackage = Object.create(null); // @claude-flow/<pkg> or crates/<pkg> → Set(src paths)
42
+ const add = (map, key, p) => { if (!key) return; key = key.toLowerCase(); (map[key] ||= new Set()).add(p); };
43
+
44
+ let n = 0, srcN = 0;
45
+ for (const line of fs.readFileSync(PASSAGES, 'utf8').split('\n')) {
46
+ if (!line.trim()) continue;
47
+ let o; try { o = JSON.parse(line); } catch { continue; }
48
+ n++;
49
+ const p = o.path || ''; if (!isSourcePath(p)) continue;
50
+ srcN++;
51
+ const text = o.text || '';
52
+ for (const re of DEF_RES) { let m; re.lastIndex = 0; while ((m = re.exec(text))) { if (m[1] && m[1].length >= 3) add(bySymbol, m[1], p); } }
53
+ for (const re of TOOL_RES) { let m; re.lastIndex = 0; while ((m = re.exec(text))) { if (m[1] && m[1].length >= 4) add(bySymbol, m[1], p); } }
54
+ // stem
55
+ const stem = (p.split('/').pop() || '').replace(/\.[^.]+$/, '');
56
+ if (stem && stem !== 'index' && stem !== 'mod' && stem.length >= 3) add(byStem, stem, p);
57
+ // package
58
+ const pkg = p.match(/@[\w-]+\/([\w-]+)\//) || p.match(/(?:^|\/)crates\/([\w-]+)\//);
59
+ if (pkg) add(byPackage, pkg[1], p);
60
+ }
61
+
62
+ const freeze = (map, cap = 8) => Object.fromEntries(Object.entries(map).map(([k, v]) => [k, [...v].slice(0, cap)]));
63
+ const out = { name: NAME, generated: 'STAMP', sourcePassages: srcN,
64
+ bySymbol: freeze(bySymbol), byStem: freeze(byStem), byPackage: freeze(byPackage, 30) };
65
+ fs.writeFileSync(path.join(KB, `${NAME}.symbols.json`), JSON.stringify(out));
66
+ console.log(`symbols: ${Object.keys(out.bySymbol).length} | stems: ${Object.keys(out.byStem).length} | packages: ${Object.keys(out.byPackage).length} (from ${srcN}/${n} source passages)`);
67
+ console.log('sample symbols:', Object.keys(out.bySymbol).filter(s => /swarm_init|guidance_recommend|memorymanager|agentdb|sqlitebackend/.test(s)).slice(0, 10));
68
+ console.log('packages:', Object.keys(out.byPackage).slice(0, 20).join(', '));
@@ -0,0 +1,97 @@
1
+ #!/usr/bin/env node
2
+ // calibrate-router.mjs — measure the cheap→frontier tiers on REAL runs so "faster" and
3
+ // "cheaper" are measured claims, and feed the learned router contrastive labels.
4
+ //
5
+ // WHY (2026-07-13, Stuart): "Nobody cares that they're saving 20 seconds. They care that
6
+ // they're doing things 40% faster… cheaper AND faster = fundamentally more efficient."
7
+ // A faster-% needs a measured baseline — this harness produces it. No number here is invented:
8
+ // every duration is a wall-clock measurement of a real run, every quality label is a
9
+ // deterministic check against a known answer, and failures are recorded as failures.
10
+ //
11
+ // BILLING SAFETY (the $1,600 / issue-#557 lesson): every spawned `claude -p` runs with
12
+ // ANTHROPIC_API_KEY / CLAUDE_API_KEY / ANTHROPIC_AUTH_TOKEN stripped from its environment —
13
+ // subscription billing only; worst case is plan throttling, never a surprise bill.
14
+ //
15
+ // FAIRNESS NOTE: durations include the CLI's startup overhead, identically for every tier —
16
+ // the tier-vs-tier ratio is apples-to-apples; absolute numbers are "task via harness", not
17
+ // raw model latency.
18
+ //
19
+ // Writes:
20
+ // • labels → recordOutcome(prompt, {model: quality, ...}) — one contrastive row per task,
21
+ // all tiers scored (the DRACO row shape rUv's router trains on)
22
+ // • receipts → ~/.claude/metaharness/routing-receipts.jsonl with source:'calibration',
23
+ // duration_ms (cheap tier) + baseline_duration_ms (frontier) → powers the ⚡ faster-% card
24
+ import { spawnSync } from 'node:child_process';
25
+ import fs from 'node:fs';
26
+ import os from 'node:os';
27
+ import path from 'node:path';
28
+ import { recordOutcome } from './metaharness-router.mjs';
29
+ import { estimateCosts, estTokens } from './route-cheap.mjs';
30
+
31
+ const CLAUDE = path.join(os.homedir(), '.npm-global/bin/claude');
32
+ const RECEIPTS = process.env.METAHARNESS_RECEIPTS
33
+ || path.join(os.homedir(), '.claude', 'metaharness', 'routing-receipts.jsonl');
34
+
35
+ const MODELS = [
36
+ { alias: 'haiku', name: 'claude-haiku-4.5' },
37
+ { alias: 'sonnet', name: 'claude-sonnet-5' },
38
+ { alias: 'opus', name: 'claude-opus-4.8' }, // the baseline tier
39
+ ];
40
+ const BASELINE = 'claude-opus-4.8';
41
+
42
+ // Deterministic tasks: known answers, graded by regex — no LLM judge, no judgment calls.
43
+ const TASKS = [
44
+ { class: 'mechanical', prompt: 'Reply with exactly this single word and nothing else: CALIBRATED', pass: /CALIBRATED/ },
45
+ { class: 'mechanical', prompt: 'List only the function names, comma-separated, from this code and say nothing else:\nfunction parseHeader(x){}\nconst mapRows = (y) => y;\nasync function flushQueue(){}', pass: /parseHeader.*mapRows.*flushQueue/s },
46
+ { class: 'analytical', prompt: 'What is 17 * 23? Reply with the number only.', pass: /\b391\b/ },
47
+ { class: 'analytical', prompt: "In JavaScript, what does this print? console.log(0.1 + 0.2 === 0.3, (0.1 + 0.2).toFixed(1) === '0.3'). Reply with exactly the two words printed, space-separated.", pass: /false\s+true/i },
48
+ { class: 'analytical', prompt: 'In JavaScript: for (var i = 0; i < 3; i++) { setTimeout(() => console.log(i)); } What three numbers print? Reply with them space-separated only.', pass: /3\s+3\s+3/ },
49
+ ];
50
+
51
+ function runOnce(alias, prompt) {
52
+ const env = { ...process.env };
53
+ delete env.ANTHROPIC_API_KEY; delete env.CLAUDE_API_KEY; delete env.ANTHROPIC_AUTH_TOKEN;
54
+ const t0 = Date.now();
55
+ const r = spawnSync(CLAUDE, ['-p', prompt, '--model', alias], { env, encoding: 'utf8', timeout: 180000 });
56
+ return { ms: Date.now() - t0, out: (r.stdout || '').trim(), ok: r.status === 0 };
57
+ }
58
+
59
+ const results = [];
60
+ for (const [ti, task] of TASKS.entries()) {
61
+ const row = { task, runs: {} };
62
+ for (const m of MODELS) {
63
+ const r = runOnce(m.alias, task.prompt);
64
+ const passed = r.ok && task.pass.test(r.out);
65
+ row.runs[m.name] = { ms: r.ms, passed, out: r.out.slice(0, 60) };
66
+ console.log(`task ${ti + 1}/${TASKS.length} [${task.class}] ${m.name}: ${passed ? 'PASS' : 'FAIL'} in ${(r.ms / 1000).toFixed(1)}s${passed ? '' : ` — got: ${r.out.slice(0, 50) || '(no output)'}`}`);
67
+ }
68
+ results.push(row);
69
+
70
+ // Label: one contrastive row per task, every tier scored. Failures ARE the valuable labels.
71
+ const scores = {};
72
+ for (const m of MODELS) scores[m.name] = row.runs[m.name].passed ? 0.95 : 0.1;
73
+ await recordOutcome(task.prompt, scores);
74
+
75
+ // Receipts: one row per cheap tier vs the frontier baseline, with MEASURED times both sides.
76
+ const base = row.runs[BASELINE];
77
+ for (const m of MODELS) {
78
+ if (m.name === BASELINE) continue;
79
+ const run = row.runs[m.name];
80
+ const inTok = estTokens(task.prompt); const outTok = estTokens(run.out || 'x');
81
+ const c = estimateCosts(m.name, inTok, outTok, BASELINE);
82
+ if (!c) continue;
83
+ fs.appendFileSync(RECEIPTS, JSON.stringify({
84
+ ts: new Date().toISOString(), source: 'calibration', task_class: task.class,
85
+ task: task.prompt.slice(0, 80), model: m.name, frontier_ref: BASELINE,
86
+ est_in_tokens: inTok, est_out_tokens: outTok, token_source: 'estimated',
87
+ est_cost: c.cost, est_frontier_cost: c.frontier, saved: c.saved,
88
+ duration_ms: run.ms, baseline_duration_ms: base.ms, quality_pass: run.passed,
89
+ }) + '\n');
90
+ }
91
+ }
92
+
93
+ const cheap = results.flatMap((r) => [r.runs['claude-haiku-4.5'].ms]);
94
+ const base = results.map((r) => r.runs[BASELINE].ms);
95
+ const sum = (a) => a.reduce((s, x) => s + x, 0);
96
+ console.log(`\nhaiku total ${(sum(cheap) / 1000).toFixed(1)}s vs opus baseline ${(sum(base) / 1000).toFixed(1)}s on ${TASKS.length} tasks`);
97
+ console.log('labels + receipts written. Render: node scripts/metaharness-receipts.mjs');
@@ -0,0 +1,321 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * capability-audit.mjs — the OFFENSIVE half of retrieval.
4
+ *
5
+ * WHY THIS EXISTS, and it is the sharpest failure this project has recorded.
6
+ *
7
+ * For three weeks the brain was queried ONLY defensively. Every consultation was triggered by the
8
+ * `ground-before-write` gate asking "am I about to duplicate something rUv already ships?" That gate
9
+ * is excellent and fired correctly every time. But it only ever fires when code is about to be
10
+ * written, so the question "what should we be USING that we aren't?" was never asked once.
11
+ *
12
+ * On 2026-07-22 the owner asked it manually, and the answer was sitting inside this very repository:
13
+ *
14
+ * .metaharness/ — Darwin ran here on 2026-07-07.
15
+ * It lifted the harness score 0.285 -> 0.765 (+168%) by evolving `reviewer`,
16
+ * `memoryPolicy`, and `scorePolicy` — the exact surfaces we spent that night hand-building.
17
+ * It promoted nothing, and sat idle for two weeks with OPENROUTER_API_KEY funded the whole time.
18
+ *
19
+ * His words: "If you knew it was out there, why the hell didn't you recommend it already? Why am I
20
+ * having to do this with you at 11:30 at night when we've been working together for three weeks?"
21
+ *
22
+ * The honest answer is that nothing in the system was obliged to look. v3.5 shipped advocacy that
23
+ * audits MEMORY health (corrupt stores, undrained queues, undistilled memories) and had no detector
24
+ * for a dormant capability at all — proactive about one category, blind to the one that mattered.
25
+ *
26
+ * DESIGN RULE, non-negotiable: every detector reports ONLY what it observed on THIS machine, with
27
+ * the observation as evidence. There is no hardcoded list of "cool features to suggest" — such a
28
+ * list would rot the week rUv ships again, and recommending a capability the user does not have is
29
+ * the same lie as any other (ADR-027).
30
+ */
31
+ import fs from 'node:fs';
32
+ import path from 'node:path';
33
+ import os from 'node:os';
34
+ import { readLearnerState, verdict as learnerVerdict, STALE_DAYS } from './learning-enable.mjs';
35
+
36
+ // NOTE: execFileSync was imported here and never used. It is deliberately not re-added. Shelling out
37
+ // to `ruflo` for state is what made a read-only status check start a background daemon and write four
38
+ // files into the user's HOME (see capability-registry.rufloBin) — an unused import of the tool that
39
+ // causes that is an invitation, and this file is READ-ONLY by intent.
40
+
41
+ const HOME = os.homedir();
42
+ const DAY = 86_400_000;
43
+ const argv = process.argv.slice(2);
44
+
45
+ /** Newest mtime anywhere under a path — "when did this capability last actually do anything?" */
46
+ function lastActivity(p, depth = 3) {
47
+ let newest = 0;
48
+ const walk = (d, lvl) => {
49
+ if (lvl > depth) return;
50
+ let entries = [];
51
+ try { entries = fs.readdirSync(d, { withFileTypes: true }); } catch { return; }
52
+ for (const e of entries) {
53
+ const full = path.join(d, e.name);
54
+ try {
55
+ const st = fs.statSync(full);
56
+ if (st.mtimeMs > newest) newest = st.mtimeMs;
57
+ if (e.isDirectory()) walk(full, lvl + 1);
58
+ } catch { /* unreadable entry — skip */ }
59
+ }
60
+ };
61
+ try { if (fs.statSync(p).isDirectory()) walk(p, 0); else newest = fs.statSync(p).mtimeMs; } catch { return 0; }
62
+ return newest;
63
+ }
64
+
65
+ const daysSince = (ms) => (ms ? (Date.now() - ms) / DAY : null);
66
+
67
+ /**
68
+ * DETECTOR: harness self-evolution that ran and then stopped.
69
+ *
70
+ * This is the ground-truth case the audit was built around. Detecting "installed" is worthless —
71
+ * `.metaharness/` existing tells you nothing. What matters is that it RAN, PRODUCED A REAL LIFT, and
72
+ * then went quiet, because that combination is invisible to every other surface and is pure
73
+ * unrealised value the user already paid for.
74
+ */
75
+ export function detectDormantEvolution(repo = process.cwd()) {
76
+ const dir = path.join(repo, '.metaharness');
77
+ const archive = path.join(dir, 'archive.json');
78
+ if (!fs.existsSync(archive)) return null;
79
+
80
+ let entries = [];
81
+ try {
82
+ const raw = JSON.parse(fs.readFileSync(archive, 'utf8'));
83
+ entries = Array.isArray(raw) ? raw : (Object.values(raw).find(Array.isArray) || []);
84
+ } catch { return null; }
85
+ if (!entries.length) return null;
86
+
87
+ const rows = entries
88
+ .map((e) => ({
89
+ id: e?.variant?.id,
90
+ surface: e?.variant?.mutationSurface,
91
+ score: e?.score?.finalScore,
92
+ promoted: e?.score?.promoted === true,
93
+ task: e?.score?.taskSuccess,
94
+ }))
95
+ .filter((r) => typeof r.score === 'number');
96
+ if (!rows.length) return null;
97
+
98
+ const baseline = rows.find((r) => r.surface && r.score === Math.min(...rows.map((x) => x.score)));
99
+ const best = rows.reduce((a, b) => (b.score > a.score ? b : a), rows[0]);
100
+ const idleDays = daysSince(lastActivity(dir));
101
+ // EXCLUDE the baseline from the promoted count. The baseline is always "promoted" (it beats a
102
+ // parent score of 0), so counting it reports "1 variant promoted" for a run where every actual
103
+ // IMPROVEMENT was discarded — the precise flavour of confident-but-misleading reporting this
104
+ // whole audit exists to catch. Caught by diffing this detector against the raw archive.
105
+ const improvements = rows.filter((r) => r.id !== 'baseline');
106
+ const promotedCount = improvements.filter((r) => r.promoted).length;
107
+ const surfaces = [...new Set(rows.map((r) => r.surface).filter(Boolean))];
108
+
109
+ // ALSO read the MACHINE-WIDE champion, not just this repo's archive.
110
+ //
111
+ // The first version of this detector read only ./.metaharness/archive.json and reported "none of
112
+ // the improvements were kept" — true of that archive, and misleading about the machine, because
113
+ // ~/.claude-flow/harness-active-policy.json held a champion promoted a week LATER by a different
114
+ // run. Stating a repo-scoped fact in machine-scoped language is the same confident-but-wrong
115
+ // shape this whole audit exists to catch, so it gets caught here too.
116
+ let champion = null;
117
+ try {
118
+ const cp = path.join(HOME, '.claude-flow', 'harness-active-policy.json');
119
+ if (fs.existsSync(cp)) {
120
+ const c = JSON.parse(fs.readFileSync(cp, 'utf8'));
121
+ if (c && c.championId) {
122
+ champion = {
123
+ id: String(c.championId).slice(0, 20),
124
+ tier: c.provenanceTier || 'unknown',
125
+ appliedDaysAgo: c.appliedAt ? Math.round(daysSince(c.appliedAt)) : null,
126
+ };
127
+ }
128
+ }
129
+ } catch { /* absent or unreadable — report the repo-scoped truth alone */ }
130
+
131
+ // Only a capability that DEMONSTRATED value and then stopped is worth interrupting someone about.
132
+ if (!(idleDays > 3 && best.score > (baseline?.score ?? 0))) return null;
133
+
134
+ return {
135
+ id: 'capability:resume-harness-evolution',
136
+ title: 'Your harness improved itself, then stopped — and nobody was told',
137
+ severity: 'IMPORTANT',
138
+ evidence: [
139
+ { observed: `harness self-evolution ran in this repo and reached a score of ${best.score} from a baseline of ${baseline?.score ?? 0}` },
140
+ { observed: `${surfaces.length} policy surfaces were explored (${surfaces.slice(0, 4).join(', ')}${surfaces.length > 4 ? '…' : ''})` },
141
+ { observed: `this repo's run has been idle ${Math.round(idleDays)} days, and ${promotedCount === 0 ? `NONE of its ${improvements.length} improvements were kept — they all plateaued at the same score, so the promotion rule could not choose between them` : `${promotedCount} of ${improvements.length} improvements were kept`}` },
142
+ champion
143
+ ? { observed: `machine-wide, a champion policy IS active (${champion.id}…, provenance ${champion.tier}, applied ${champion.appliedDaysAgo} days ago) — so evolution is not entirely dead, it is just not running HERE` }
144
+ : { observed: 'no machine-wide champion policy is active either — nothing from any run is currently in force' },
145
+ ],
146
+ // Never overstate. A plateau is a known, documented condition with a known fix — say which.
147
+ why: promotedCount === 0
148
+ ? 'Every variant scored the same, so the lightweight promotion rule could not choose between them and kept none. This exact ceiling is documented upstream, and the fix is a graded benchmark gate rather than more evolution.'
149
+ : 'It produced promoted variants and then went quiet.',
150
+ detail: { best: best.score, baseline: baseline?.score ?? 0, idleDays: Math.round(idleDays), surfaces, promotedCount },
151
+ };
152
+ }
153
+
154
+ /** DETECTOR: a paid capability that is funded and unused — the most wasteful dormancy there is. */
155
+ export function detectFundedButIdle(repo = process.cwd()) {
156
+ const funded = Boolean(process.env.OPENROUTER_API_KEY);
157
+ if (!funded) return null;
158
+ const dir = path.join(repo, '.metaharness');
159
+ if (!fs.existsSync(dir)) return null;
160
+ const idleDays = daysSince(lastActivity(dir));
161
+ if (!(idleDays > 7)) return null;
162
+ return {
163
+ id: 'capability:funded-but-idle',
164
+ title: 'You are paying for a capability that has not run in weeks',
165
+ severity: 'SUGGESTED',
166
+ evidence: [
167
+ { observed: 'an OpenRouter key is configured, which is what unlocks the write/evolve layer' },
168
+ { observed: `the last evolution activity in this repo was ${Math.round(idleDays)} days ago` },
169
+ ],
170
+ why: 'The expensive part is already paid for; only the running of it stopped.',
171
+ detail: { idleDays: Math.round(idleDays) },
172
+ };
173
+ }
174
+
175
+ /**
176
+ * DETECTOR: is the learner actually learning?
177
+ *
178
+ * ⚠️ THIS DETECTOR PREVIOUSLY SHIPPED A FALSE ALARM, and the correction is the most instructive
179
+ * thing in this file.
180
+ *
181
+ * The first version parsed `ruflo hooks list` and reported "26 learning hooks installed and every
182
+ * one is switched off". That was WRONG, and it was reported to the owner as a headline finding —
183
+ * including as the answer to his direct question about whether learning was on.
184
+ *
185
+ * What actually happens: `ruflo hooks list --format json` returns `{name, type, status:"active"}`
186
+ * and contains NO `enabled` key. The CLI's table renderer draws a column keyed `enabled`, reads
187
+ * `undefined`, and prints "No" 26 times. `ruflo hooks enable` does not exist. The list is a MENU of
188
+ * available hooks, not a dashboard of enabled ones.
189
+ *
190
+ * Meanwhile the real signal said the opposite the whole time: ~/.claude-flow/neural/stats.json held
191
+ * 457 trajectories and 457 patterns, last adapted 106 minutes earlier — DURING the session in which
192
+ * we told the owner learning was off.
193
+ *
194
+ * This is exactly L01 (verify through a channel CAPABLE of observing the truth) committed inside the
195
+ * detector written to catch L01. A human-readable CLI table is a PRESENTATION, and presentations
196
+ * drift from their payloads. Read the state file, or read `--format json` — never scrape a table.
197
+ *
198
+ * ADR-028 sets the false-alarm rate at ZERO and calls it non-negotiable: one false alarm costs more
199
+ * trust than ten true findings earn. So this detector now reports only what a state file proves, and
200
+ * stays SILENT when it cannot tell.
201
+ */
202
+ export function detectLearnerIdle() {
203
+ // DELEGATED, for the same reason capability-registry delegates: there must be exactly ONE reading
204
+ // of stats.json on this machine. The hand-rolled version here repeated the registry's schema-drift
205
+ // bug in its most damaging form — `Number(s.trajectoriesRecorded ?? 0)` yields 0 when rUv renames
206
+ // the field, and 0-and-0 fell straight into the IMPORTANT branch below. The result would have been
207
+ // an alarming "Your learner has never recorded anything" fired at every user at once, about a
208
+ // learner that was working perfectly. ADR-028 puts the acceptable false-alarm rate at ZERO and
209
+ // calls it non-negotiable; a detector that turns an upstream rename into a machine-wide accusation
210
+ // is the worst possible way to violate that.
211
+ //
212
+ // readLearnerState's `num()` returns null rather than 0 for an unreadable counter, so drift now
213
+ // arrives as UNKNOWN_SHAPE — and this detector stays SILENT on it, because "I cannot read the
214
+ // counters" is not a dormant capability and there is nothing for the user to act on.
215
+ const learner = readLearnerState({ home: HOME });
216
+ const v = learnerVerdict(learner);
217
+ const { trajectories, patterns } = learner;
218
+
219
+ // NO_LEARNER_STATE / CORRUPT / UNKNOWN_SHAPE / UNKNOWN_PARTIAL: nothing provable, so say nothing.
220
+ // Silence here is correct — an audit that speaks up when it cannot see is how false alarms are
221
+ // manufactured. UNKNOWN_PARTIAL (one counter renamed upstream, one still readable) is deliberately
222
+ // in that list: the readable half cannot settle a question that needs both, and half a measurement
223
+ // is not grounds for telling somebody their learner is broken.
224
+ if (v.code === 'INITIALISED_EMPTY') {
225
+ // A learner that has genuinely, measurably recorded nothing is dormant and worth saying so.
226
+ return {
227
+ id: 'capability:learner-never-ran',
228
+ title: 'Your learner has never recorded anything',
229
+ severity: 'IMPORTANT',
230
+ evidence: [
231
+ { observed: 'the learner\'s own state file exists, carries the counters this version understands, and both read 0' },
232
+ ],
233
+ why: 'The learning machinery is present and has never been fed, so nothing you do is being turned into reusable experience.',
234
+ detail: { trajectories, patterns },
235
+ };
236
+ }
237
+
238
+ if (v.code === 'IDLE') {
239
+ // Recording, but nothing recently — a real, provable dormancy. STALE_DAYS is imported, not
240
+ // re-typed: this file used its own literal 7 while learning-enable used STALE_DAYS, which is two
241
+ // definitions of "idle" waiting to drift apart.
242
+ const idleDays = Math.floor(learner.ageMinutes / 1440);
243
+ return {
244
+ id: 'capability:learner-gone-quiet',
245
+ title: 'Your learner has gone quiet',
246
+ severity: 'SUGGESTED',
247
+ evidence: [
248
+ { observed: `${trajectories} trajectories and ${patterns} patterns recorded, but the last adaptation was ${idleDays} days ago (idle past ${STALE_DAYS})` },
249
+ ],
250
+ why: 'It learned before and stopped, so recent work is not becoming reusable experience.',
251
+ detail: { trajectories, patterns, idleDays },
252
+ };
253
+ }
254
+
255
+ // Learning is live, or unreadable. Report NOTHING — a healthy machine must produce no findings at
256
+ // all, and an unreadable one must not be described as unhealthy.
257
+ return null;
258
+ }
259
+
260
+ export const DETECTORS = [detectDormantEvolution, detectFundedButIdle, detectLearnerIdle];
261
+
262
+ /**
263
+ * Run every detector. Never throws — an advisory surface must not break the thing that calls it.
264
+ *
265
+ * BUT IT NO LONGER SWALLOWS THE FAILURE EITHER, and that distinction is the finding.
266
+ *
267
+ * The old `catch {}` was correct about one thing — a single broken detector must not silence the
268
+ * other two — and catastrophically wrong about what to do next. With all three detectors forced to
269
+ * throw (EACCES / bad JSON / ENOENT), this function returned `[]`, and `[]` is exactly what a
270
+ * perfectly healthy machine returns. The CLI then printed:
271
+ *
272
+ * "No dormant capability found on this machine. That is a real answer, not a shrug —
273
+ * every detector reports only what it observed here."
274
+ *
275
+ * Total instrument failure, rendered as a confident all-clear, with copy that explicitly forecloses
276
+ * the doubt. A newcomer whose file permissions differ from ours gets told their machine is fine by a
277
+ * system that could not see their machine at all. That is the unknown-as-a-measurement lie in its
278
+ * purest form — it just arrives as silence instead of as the word "off".
279
+ *
280
+ * So failures are now COUNTED and RETURNED. The caller is obliged to say how many of the checks
281
+ * actually ran before it characterises the result.
282
+ */
283
+ export function auditCapabilities(repo = process.cwd()) {
284
+ const findings = [];
285
+ const failures = [];
286
+ for (const d of DETECTORS) {
287
+ try { const f = d(repo); if (f) findings.push(f); }
288
+ catch (e) { failures.push({ detector: d.name || 'anonymous detector', reason: String(e?.message || e).split('\n')[0].slice(0, 120) }); }
289
+ }
290
+ return { findings, failures, ran: DETECTORS.length - failures.length, total: DETECTORS.length };
291
+ }
292
+
293
+ // ── CLI ──────────────────────────────────────────────────────────────────────────────────────────
294
+ const invokedDirectly = process.argv[1] && path.resolve(process.argv[1]).endsWith('capability-audit.mjs');
295
+ if (invokedDirectly) {
296
+ const audit = auditCapabilities(argv.includes('--repo') ? argv[argv.indexOf('--repo') + 1] : process.cwd());
297
+ const { findings, failures, ran, total } = audit;
298
+ if (argv.includes('--json')) { console.log(JSON.stringify(audit, null, 2)); process.exit(0); }
299
+
300
+ // Broken instruments are reported BEFORE any verdict, because they change what the verdict means.
301
+ for (const f of failures) console.log(`\n ! ${f.detector} could not run: ${f.reason}`);
302
+
303
+ if (!findings.length) {
304
+ // The all-clear is only offered when every check actually ran. Otherwise the honest line is that
305
+ // we could not look — never "nothing found", which reads identically and is not the same claim.
306
+ console.log(failures.length
307
+ ? `\n ${ran} of ${total} checks ran, and they found nothing dormant. The ${failures.length} that\n`
308
+ + ' could not run are listed above — this is NOT an all-clear, because part of the machine\n'
309
+ + ' was not examined at all.\n'
310
+ : '\n No dormant capability found on this machine. That is a real answer, not a shrug —\n'
311
+ + ` all ${ran} of ${total} checks ran, and every detector reports only what it observed here.\n`);
312
+ process.exit(0);
313
+ }
314
+ const n = findings.length;
315
+ console.log(`\n ${n} ${n === 1 ? 'capability' : 'capabilities'} you already own ${n === 1 ? 'is' : 'are'} not being used:\n`);
316
+ for (const f of findings) {
317
+ console.log(` [${f.severity}] ${f.title}`);
318
+ for (const e of f.evidence) console.log(` · ${e.observed}`);
319
+ console.log(` why: ${f.why}\n`);
320
+ }
321
+ }