ruvnet-brain 3.9.134-dev → 4.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (218) hide show
  1. package/.claude-plugin/marketplace.json +14 -0
  2. package/README.md +5 -5
  3. package/bin/install.mjs +382 -36
  4. package/console/CONTRACT.md +172 -0
  5. package/console/activity.js +753 -0
  6. package/console/app.js +4189 -0
  7. package/console/architecture.html +1221 -0
  8. package/console/assets/depth-1.webp +0 -0
  9. package/console/assets/depth-2.webp +0 -0
  10. package/console/assets/depth-3.webp +0 -0
  11. package/console/assets/harness-vs-plain.svg +259 -0
  12. package/console/assets/hero.webp +0 -0
  13. package/console/assets/memory.webp +0 -0
  14. package/console/assets/metaharness.svg +247 -0
  15. package/console/index.html +777 -0
  16. package/console/install-architecture.html +162 -0
  17. package/console/install-mockup.html +543 -0
  18. package/console/style.css +2144 -0
  19. package/console/tips.css +926 -0
  20. package/console/tips.html +858 -0
  21. package/console/tips.js +128 -0
  22. package/docs/RELEASE-NOTES-4.0.md +88 -0
  23. package/kb/model-requirements.mjs +37 -6
  24. package/kb/zip-extract.mjs +53 -14
  25. package/keys/ruvnet-brain-signing.pub.pem +3 -0
  26. package/package.json +14 -22
  27. package/plugin/.claude-plugin/marketplace.json +14 -0
  28. package/plugin/.claude-plugin/plugin.json +22 -0
  29. package/plugin/.codex-plugin/plugin.json +21 -0
  30. package/plugin/.mcp.json +8 -0
  31. package/plugin/commands/brain-console.md +16 -0
  32. package/plugin/commands/configure.md +33 -0
  33. package/plugin/commands/rvbc.md +79 -0
  34. package/plugin/commands/rvcb.md +16 -0
  35. package/plugin/commands/whats-new.md +57 -0
  36. package/plugin/hooks/codex-hooks.json +160 -0
  37. package/plugin/hooks/hook-contracts.json +77 -0
  38. package/plugin/hooks/hooks.json +202 -0
  39. package/plugin/mcp/managed-cli-interface.mjs +47 -4
  40. package/plugin/mcp/server.mjs +56 -6
  41. package/plugin/scripts/anticipate.sh +534 -0
  42. package/plugin/scripts/codex-hook-adapter.mjs +96 -0
  43. package/plugin/scripts/continuation-gate.mjs +267 -0
  44. package/plugin/scripts/design-wall.sh +137 -0
  45. package/plugin/scripts/detach.mjs +182 -0
  46. package/plugin/scripts/first-session-worker.mjs +38 -0
  47. package/plugin/scripts/gate-receipt.sh +35 -0
  48. package/plugin/scripts/ground-before-write.sh +199 -0
  49. package/plugin/scripts/ground-ruvnet.sh +517 -0
  50. package/plugin/scripts/grounding-stamp.sh +113 -0
  51. package/plugin/scripts/grounding-substance.mjs +595 -0
  52. package/plugin/scripts/hijack-ruvnet.sh +81 -0
  53. package/plugin/scripts/hook-input.mjs +558 -0
  54. package/plugin/scripts/hook-shim-bash.mjs +55 -0
  55. package/plugin/scripts/hook-shim.mjs +303 -0
  56. package/plugin/scripts/host-update.mjs +58 -0
  57. package/plugin/scripts/kling-preflight.sh +146 -0
  58. package/plugin/scripts/learn-capture.sh +173 -0
  59. package/plugin/scripts/learn-flush.mjs +155 -0
  60. package/plugin/scripts/lesson-hooks.sh +213 -0
  61. package/plugin/scripts/md-stamp.mjs +219 -0
  62. package/plugin/scripts/protect-brain-state.sh +84 -0
  63. package/plugin/scripts/route-dispatch.sh +147 -0
  64. package/plugin/scripts/routing-outcome-capture.mjs +89 -0
  65. package/plugin/scripts/runtime-preferences.mjs +269 -0
  66. package/plugin/scripts/session-start-core.mjs +477 -0
  67. package/plugin/scripts/session-start.sh +13 -0
  68. package/plugin/scripts/signal-watch.mjs +193 -0
  69. package/plugin/scripts/unprompted-runtime.mjs +377 -0
  70. package/plugin/scripts/update-apply.mjs +419 -0
  71. package/plugin/scripts/verify-interface.sh +53 -0
  72. package/plugin/scripts/version-bump-gate.sh +112 -0
  73. package/plugin/skills/brain-build/SKILL.md +123 -0
  74. package/plugin/skills/brain-console/SKILL.md +22 -0
  75. package/plugin/skills/brain-prompt/SKILL.md +83 -0
  76. package/plugin/skills/brain-score/SKILL.md +101 -0
  77. package/plugin/skills/release-proof/SKILL.md +81 -0
  78. package/plugin/skills/release-proof/agents/openai.yaml +4 -0
  79. package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
  80. package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
  81. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +121 -0
  82. package/plugin/skills/ruvnet-brain/SKILL.md +234 -0
  83. package/plugin/skills/rvbc/SKILL.md +23 -0
  84. package/plugin/skills/savings/SKILL.md +46 -0
  85. package/plugin/skills/whats-new/SKILL.md +22 -0
  86. package/scripts/adr-backfill.mjs +107 -0
  87. package/scripts/advocacy-outcomes.mjs +808 -0
  88. package/scripts/agentdb-context.mjs +216 -0
  89. package/scripts/agentdb-fleet-doctor.mjs +101 -0
  90. package/scripts/ascii-drift.mjs +236 -0
  91. package/scripts/behavioral-l1-l4.mjs +210 -0
  92. package/scripts/brain-capability-check.mjs +72 -0
  93. package/scripts/brain-grade-groundtruth.mjs +100 -0
  94. package/scripts/brain-latency-50.mjs +227 -0
  95. package/scripts/brain-novice-50.mjs +189 -0
  96. package/scripts/brain-stamp.mjs +94 -0
  97. package/scripts/brain-state.mjs +212 -0
  98. package/scripts/build-bundle.mjs +522 -0
  99. package/scripts/build-concepts.mjs +132 -0
  100. package/scripts/build-l2.mjs +71 -0
  101. package/scripts/build-primer.mjs +73 -0
  102. package/scripts/build-symbols.mjs +68 -0
  103. package/scripts/calibrate-router.mjs +97 -0
  104. package/scripts/capability-audit.mjs +321 -0
  105. package/scripts/capability-registry.mjs +876 -0
  106. package/scripts/check-indexation.mjs +108 -0
  107. package/scripts/check-legibility.mjs +189 -0
  108. package/scripts/ci/build-fixture-kb.mjs +67 -0
  109. package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
  110. package/scripts/ci/learning-replay-recorder.mjs +59 -0
  111. package/scripts/ci/mutate-hook-timeout.mjs +70 -0
  112. package/scripts/ci/stranger-fixture-stage.mjs +17 -0
  113. package/scripts/ci/stranger-scenario.mjs +228 -0
  114. package/scripts/ci/stranger-timeout.mjs +25 -0
  115. package/scripts/ci-verdict.mjs +29 -0
  116. package/scripts/claims-verify.mjs +710 -0
  117. package/scripts/clear-claude-tmp.sh +31 -0
  118. package/scripts/console-engine.mjs +434 -0
  119. package/scripts/console-engine.test.mjs +125 -0
  120. package/scripts/corpus-qa.mjs +250 -0
  121. package/scripts/correction-detect-embed.mjs +346 -0
  122. package/scripts/correction-detect-measure.mjs +270 -0
  123. package/scripts/correction-detect.mjs +686 -0
  124. package/scripts/count-chunks.mjs +54 -0
  125. package/scripts/described-questions.json +30 -0
  126. package/scripts/design-grade.mjs +58 -0
  127. package/scripts/dev-plugin-link.sh +105 -0
  128. package/scripts/distill-project.mjs +200 -0
  129. package/scripts/doc-currency.mjs +801 -0
  130. package/scripts/eval-brain.mjs +244 -0
  131. package/scripts/fix-metaharness-memretrieve.mjs +121 -0
  132. package/scripts/full-hints.mjs +87 -0
  133. package/scripts/gate.sh +39 -0
  134. package/scripts/gates.mjs +146 -0
  135. package/scripts/gen-console-images.mjs +54 -0
  136. package/scripts/gen-images.mjs +47 -0
  137. package/scripts/git-clone-refresh.mjs +52 -0
  138. package/scripts/git-hooks/pre-push +126 -0
  139. package/scripts/goal-match.mjs +398 -0
  140. package/scripts/goldie-research.mjs +223 -0
  141. package/scripts/goldie-weekly.sh +67 -0
  142. package/scripts/health-repair.mjs +250 -0
  143. package/scripts/helix-scenario-questions.json +10 -0
  144. package/scripts/ingest-gists.mjs +230 -0
  145. package/scripts/ingest-meeting.mjs +115 -0
  146. package/scripts/ingest-repo.mjs +79 -0
  147. package/scripts/install-npx-witness.sh +49 -0
  148. package/scripts/issue-fix.mjs +639 -0
  149. package/scripts/issue-watch.mjs +276 -0
  150. package/scripts/issue4-close-note.md +31 -0
  151. package/scripts/key-canary.mjs +91 -0
  152. package/scripts/latency-to-surface.mjs +233 -0
  153. package/scripts/learning-enable.mjs +380 -0
  154. package/scripts/learning-replay.mjs +1570 -0
  155. package/scripts/learnings.mjs +62 -0
  156. package/scripts/lesson-gate.mjs +680 -0
  157. package/scripts/lesson-lifecycle.mjs +449 -0
  158. package/scripts/lesson-promote.mjs +262 -0
  159. package/scripts/lesson-ratify.mjs +98 -0
  160. package/scripts/lesson-seed.mjs +252 -0
  161. package/scripts/lesson-store.mjs +447 -0
  162. package/scripts/loop-checkpoint.mjs +86 -0
  163. package/scripts/memdb-health.sh +14 -0
  164. package/scripts/memory-doctor.mjs +271 -0
  165. package/scripts/model-catalog.mjs +79 -0
  166. package/scripts/nightly-controller.mjs +66 -0
  167. package/scripts/nightly-gists.sh +72 -0
  168. package/scripts/nightly-wrapper.sh +180 -0
  169. package/scripts/notify.sh +12 -0
  170. package/scripts/npx-witness.sh +56 -0
  171. package/scripts/onboarding-console.mjs +2749 -0
  172. package/scripts/private-fence.mjs +69 -0
  173. package/scripts/proactivity-metrics.mjs +118 -0
  174. package/scripts/proof-questions.json +56 -0
  175. package/scripts/prove.mjs +95 -0
  176. package/scripts/proxy/claude-proxied.sh +57 -0
  177. package/scripts/proxy/proxy-revert.sh +59 -0
  178. package/scripts/proxy/proxy-up.sh +60 -0
  179. package/scripts/proxy/proxy-verify.mjs +142 -0
  180. package/scripts/published-surface-probe.mjs +241 -0
  181. package/scripts/qe/card-lane-gate.mjs +162 -0
  182. package/scripts/qe/session-start-gate.mjs +229 -0
  183. package/scripts/qe/ux-suite.mjs +323 -0
  184. package/scripts/reconcile-project.mjs +0 -0
  185. package/scripts/record-lesson.mjs +113 -0
  186. package/scripts/refresh-model-catalog.mjs +99 -0
  187. package/scripts/release-proof.mjs +9 -0
  188. package/scripts/release-vector.mjs +281 -0
  189. package/scripts/release.mjs +395 -0
  190. package/scripts/remedy-registry.mjs +247 -0
  191. package/scripts/rerank-cap-eval.mjs +265 -0
  192. package/scripts/rerank-cap-warm-ab.mjs +129 -0
  193. package/scripts/route-cheap.mjs +20 -15
  194. package/scripts/router-utilization.mjs +182 -0
  195. package/scripts/routing-flywheel.mjs +596 -0
  196. package/scripts/rvf-generation.mjs +104 -0
  197. package/scripts/rvf-index-audit.mjs +138 -0
  198. package/scripts/self-update.mjs +508 -0
  199. package/scripts/selfcheck.mjs +7 -1
  200. package/scripts/sign-bundle.mjs +69 -0
  201. package/scripts/signal-watch.mjs +171 -0
  202. package/scripts/stack-sync.mjs +469 -0
  203. package/scripts/stamp-existing-rvf-generations.mjs +53 -0
  204. package/scripts/stamp-sweep.mjs +144 -0
  205. package/scripts/status-honesty.mjs +102 -0
  206. package/scripts/sync-version.mjs +217 -0
  207. package/scripts/token-report.mjs +102 -0
  208. package/scripts/top100-benchmark.mjs +479 -0
  209. package/scripts/top100-corpus.mjs +112 -0
  210. package/scripts/top100-semantic-assertions.mjs +449 -0
  211. package/scripts/update-apply.mjs +9 -0
  212. package/scripts/upgrade-notice.mjs +14 -0
  213. package/scripts/verify-bundle.mjs +51 -0
  214. package/scripts/verify-channels.mjs +184 -0
  215. package/scripts/verify-model-catalog.mjs +104 -0
  216. package/scripts/verify-nightly-close-issue4.sh +31 -0
  217. package/scripts/version.mjs +40 -0
  218. package/scripts/wired-check.mjs +864 -0
@@ -0,0 +1,123 @@
1
+ ---
2
+ name: brain-build
3
+ description: The autonomous build contract, out of the box — "/brain-build <what you want>" activates the disciplined hands-off build power users used to hand-write a standing prompt for. Use when the user says "/brain-build", "brain build", "build this autonomously", "build this hands-off", "loop until it's done", "don't ask me, just build it", or asks for an unattended/self-grading build. Phase-gated per rUv's SPARC, self-verified and self-graded /100 against a per-phase rubric (below 95 → fix and regrade, max 5 iterations), cost-tier routed with printed receipts, crash-resumable via checkpoints, questions batched into one list — the user writes the goal, not the contract.
4
+ updated: 2026-07-10
5
+ ---
6
+
7
+ <!-- Credit: this contract productizes a community field pattern — the 7-rule standing prompt
8
+ hand-written by the PR #8 contributor (Eva Draganova, 2026-07-10) to force the brain into
9
+ disciplined autonomous building. Her forcing insight: "grade 1-100, no pass under 95 →
10
+ forced it to loop and improve." All the machinery existed; this skill removes the 40
11
+ hand-written lines needed to activate it. -->
12
+
13
+ # Brain-Build — the standing contract, activated by one line
14
+
15
+ `/brain-build <what you want>` means: no human is watching until it's done. Run the whole build
16
+ under the contract below. The AUTONOMOUS MODE rules injected by the grounding hook
17
+ (`plugin/scripts/ground-ruvnet.sh`) apply in full — this skill carries them even on turns where
18
+ that hook doesn't fire.
19
+
20
+ ## 1. Phases — rUv's SPARC, with a rubric per phase
21
+
22
+ Structure the build as the five SPARC phases with a quality gate between each — rUv's own
23
+ convention (phases + gates: `concepts/sparc/CARD/sparc-card`; per-phase docs:
24
+ `sparc/specification/README.md`). Gate criteria follow rUv's ruflo-sparc gate checks
25
+ (`ruflo/plugins/ruflo-sparc/commands/ruflo-sparc.md`):
26
+
27
+ | Phase | Rubric (the /100 grade is against THIS) |
28
+ |---|---|
29
+ | **S** Specification | Requirements complete; ≥3 acceptance criteria; constraints explicit; edge cases identified. |
30
+ | **P** Pseudocode | Design covers every acceptance criterion; error paths explicit; complexity annotated. |
31
+ | **A** Architecture | Every constraint addressed; API contracts typed; no circular dependencies; every stack decision grounded (rule 4). |
32
+ | **R** Refinement | Every acceptance criterion has a passing test; suite green; coverage adequate; self-review clean. |
33
+ | **C** Completion | All tests green; docs match the code; deploy checklist verified; traceability criterion→test. |
34
+
35
+ Scale the ceremony to the build (a small feature gets a light S and P), never skip a gate.
36
+
37
+ ## 2. LOOP, DON'T ASK — the ≥95 gate
38
+
39
+ At each phase gate:
40
+
41
+ 1. **Self-verify with real instruments** — run the tests, curl the endpoint, screenshot the UI,
42
+ execute the quickstart. Evidence, never opinion.
43
+ 2. **Grade /100 against the phase rubric, under the brain-score rules** (see the `brain-score`
44
+ skill): every deduction cites evidence (file:line, command + output); a known architectural
45
+ flaw caps the grade at ≤70 no matter what else works; a "what I did NOT test" section is
46
+ mandatory; when in doubt, score lower.
47
+ 3. **Below 95 → fix the cited deductions and regrade.** Loop. **Maximum of 5 iterations per
48
+ phase** — if the 5th grade is still <95, stop the phase and report: the score, the remaining
49
+ evidence-cited deductions, and the ONE item blocking ≥95.
50
+ 4. **Report only the final result**: final score, what was fixed across iterations, and the proof
51
+ (the command output / artifact). Never narrate intermediate grades or ask "should I keep going?"
52
+
53
+ ## 3. AUTO-ADVANCE on gate pass
54
+
55
+ Gate ≥95 → **commit the phase** (one commit per phase, message names the phase and score) and
56
+ advance immediately — no permission round-trip. **Push only if the user's repo conventions allow**
57
+ (they asked for pushes, or the workflow demonstrably expects them); otherwise commit locally and
58
+ note the unpushed state in the final report. Production deploys, npm publish, force-push, history
59
+ rewrites, secrets: NEVER — do everything up to that fence and name the exact click a human owes.
60
+
61
+ ## 4. GROUND every stack decision + the "what did I miss?" pass
62
+
63
+ Every stack/tool/library decision goes through `search_ruvnet` first, and the decision cites the
64
+ returned repo/path. Close **every phase** with one more brain pass: a `search_ruvnet` query
65
+ describing what the phase just built ("what did I miss?"), checking for a sharper rUv primitive or
66
+ prior art the phase overlooked. A hit worth acting on goes into the next iteration; no hit costs
67
+ one line: "brain pass clean."
68
+
69
+ ## 5. PROTECT-MY-MONEY — the tier ladder
70
+
71
+ - **Mechanical / plumbing text work** (summaries, classification, research digests, boilerplate
72
+ transforms) → route cheap via `node scripts/route-cheap.mjs --task "<task>"` (or agentic-flow
73
+ directly). It prints its receipt line — "⚡ MetaHarness: routed to <model> (est. $X vs $Y
74
+ frontier — saved ~$Z)" — and logs to `~/.claude/metaharness/routing-receipts.jsonl`. No
75
+ OPENROUTER_API_KEY → say so once and stay on Claude tiers; never silently pretend to route.
76
+ - **Frontier ONLY for the authoritative gate run** — the grade that decides advancement — and for
77
+ architecture / security / irreversible calls. Iteration drudge work rides the cheap tier.
78
+ - **Any operation projected >$1 or >20 paid calls → state the estimate and WAIT.** This is one of
79
+ the only legitimate stops in autonomous mode.
80
+ - **Long runs print running spend** from the receipts log: `node scripts/metaharness-receipts.mjs`
81
+ — one line per phase gate, cumulative.
82
+
83
+ ## 6. BATCH questions — never block on one
84
+
85
+ A question that isn't a hard blocker gets parked, and work continues on everything unblocked.
86
+ Deliver **ONE list** at the phase gate (or the end), each question with a **recommended default**
87
+ the user can accept with a single "defaults fine." Only a genuine hard blocker — cannot proceed
88
+ AND >$1/irreversible — interrupts mid-phase.
89
+
90
+ ## 7. READY discipline
91
+
92
+ Say "READY" / "done" / "deployed" **only after self-verifying the deployed or running version** —
93
+ curl the live URL, run the installed CLI, load the real page. The real door, not an adjacent one.
94
+ If a deploy is in flight, say exactly that: "deploy in flight — verifying before I call it READY."
95
+
96
+ ## 8. Interrupts
97
+
98
+ - User says **"status"** → reply with ONLY a table: `done / in-flight / blocked-on-me / parked`.
99
+ No prose before or after.
100
+ - **Mid-build ideas** from the user → add to the PARKED table with a one-line feasibility read and
101
+ keep building — unless they say "now", which reprioritizes immediately.
102
+
103
+ ## 9. Crash-resumable state — `scripts/loop-checkpoint.mjs`
104
+
105
+ The checkpoint is the loop's spine (contract in the script header):
106
+
107
+ - **Read FIRST** every iteration: `node scripts/loop-checkpoint.mjs read` — if a checkpoint
108
+ exists, resume from its `next`; never re-derive the plan, never repeat completed phases.
109
+ - **Iteration 1 only**: declare done-criteria as a SHELL COMMAND and write it to the checkpoint —
110
+ done is an **exit code, not an opinion**.
111
+ - **Write LAST** every iteration:
112
+ `node scripts/loop-checkpoint.mjs write --iteration N --done-criteria "<cmd>" --next "<one action>" --blockers "<or empty>"`
113
+ - **Then check**: exit 3 = DONE (stop, final report); exit 4 = NO-PROGRESS (two strikes on an
114
+ unchanged `next`: stop, name what's stuck and the ONE thing that would unstick it).
115
+
116
+ Record assumptions made under rule 2 ("cheapest-to-reverse interpretation") in the checkpoint's
117
+ `blockers`/`next` so a resumed run inherits them.
118
+
119
+ ## Final report shape
120
+
121
+ Goal → per-phase table (phase, final score, iterations used, what was fixed) → proof artifacts
122
+ (commands + outputs) → "what I did NOT test" → spend summary from the receipts log → parked
123
+ items + the batched question list with defaults.
@@ -0,0 +1,22 @@
1
+ ---
2
+ name: brain-console
3
+ description: Open the RuvNet Brain Console for "/rvbc", "/rvcb", "/brain-console", or "/ruvnet-brain:configure". Use when the user asks to open, configure, inspect, or view the Brain Console. It opens the live local page in the background; the page is read-only until the user clicks a clearly explained, reversible action.
4
+ updated: 2026-07-28
5
+ ---
6
+
7
+ # Brain Console
8
+
9
+ Treat `/rvbc`, `/rvcb`, `/brain-console`, and `/ruvnet-brain:configure` as equally valid names.
10
+ Never correct the user's spelling.
11
+
12
+ 1. Say one short sentence: "Opening it now; it scans live while you watch."
13
+ 2. Resolve the installed runtime at
14
+ `${RUVNET_BRAIN_KB:-$HOME/.cache/ruvnet-brain/kb}/.console-runtime/scripts/onboarding-console.mjs`.
15
+ A current-repository `scripts/onboarding-console.mjs` is allowed only for an explicit developer
16
+ checkout. Never fall back to a guessed `~/Code` path.
17
+ 3. Run `node <resolved-script> --serve --open` in the background.
18
+ 4. Give the URL immediately. Do not promise a duration; the page reports its own scan progress.
19
+
20
+ An already-running server is success. The Console is read-only until the user chooses an action;
21
+ every change must be explained and reversible. If the script cannot be located or the server fails,
22
+ report the exact failure plainly instead of claiming the Console opened.
@@ -0,0 +1,83 @@
1
+ ---
2
+ name: brain-prompt
3
+ description: Metaprompting assistant — "/brain-prompt <rough idea>" turns a rough ask into "the right prompt": a complete, tuned master prompt with SPARC phases, per-phase rubrics, standing rules, and cost guardrails, ready to paste or run. Use when the user says "/brain-prompt", "brain prompt", "write me the right prompt for this", "turn this idea into a proper prompt", "metaprompt this", "what should I actually ask for", or hands over a vague one-liner they want expanded into a disciplined build brief. Ends by offering to execute the produced prompt with /brain-build semantics.
4
+ updated: 2026-07-10
5
+ ---
6
+
7
+ <!-- Credit: this pattern productizes community field use — the PR #8 contributor's standing
8
+ prompt (Eva Draganova, 2026-07-10): power users were hand-writing the phase/rubric/guardrail
9
+ contract around every rough ask. This skill writes that contract FOR them. -->
10
+
11
+ # Brain-Prompt — from rough idea to the right prompt
12
+
13
+ You are not completing the task here — you are writing the instructions for completing it. That is
14
+ rUv's own framing in his metaprompt notes (`ruv-gists/874e2138/metaprompt.txt`,
15
+ `ruv-gists/5dd85664/metaprompt.txt`): a prompt template with clearly demarcated variables,
16
+ justification demanded before any score, and structure the executing model cannot wriggle out of.
17
+ The output prompt's section shape follows rUv's SAFLA prompt-generator
18
+ (`safla/.roo-orginal/rules-prompt-generator/rules.md`): Context / Task / Requirements / Expected
19
+ Output, extended with phases and guardrails.
20
+
21
+ ## Procedure
22
+
23
+ ### 1. Interrogate the rough ask — silently, against a checklist
24
+
25
+ Enumerate what's underspecified: **users** (who is this for?), **data** (what exists, what shape,
26
+ how much?), **scale** (10 users or 10M?), **platform** (web/CLI/mobile? deploy target?),
27
+ **constraints** (budget, stack, deadline, compliance), **done** (what observable behavior ends
28
+ this?). Then:
29
+
30
+ - **Infer defaults — do not interrogate the human.** For everything you can reasonably default
31
+ (from their repo, their stack, the obvious reading), pick the default and write it into the
32
+ prompt's ASSUMPTIONS block where they can veto it by editing one line.
33
+ - **Batch the few questions that genuinely need a human** — ambiguous product intent, money,
34
+ irreversible choices — into ONE list, each with a recommended default. Never a
35
+ twenty-questions interview; usually the list is 0–3 items.
36
+
37
+ ### 2. Ground the stack via search_ruvnet
38
+
39
+ Call `search_ruvnet` with queries describing what the build technically DOES. Which rUv tools fit
40
+ — vectors → RuVector/RVF, orchestration → ruflo, QE → agentic-qe, memory → AgentDB, methodology →
41
+ SPARC (`sparc/specification/README.md`)? **Cite the returned repo/path next to every tool the
42
+ prompt prescribes.** A prompt that names tools without citations is a guess wearing a suit — don't
43
+ ship it. No tool fits → the STACK section says so plainly rather than forcing a tie-in.
44
+
45
+ ### 3. Output the master prompt — Eva's shape
46
+
47
+ Produce ONE complete, paste-ready prompt with exactly these sections:
48
+
49
+ ```
50
+ GOAL — the ask, sharpened to one testable sentence.
51
+ ASSUMPTIONS — every inferred default, one line each (veto by editing).
52
+ STACK — tools/libraries, each with its grounded citation (repo/path).
53
+ PHASES — SPARC (Specification → Pseudocode → Architecture → Refinement → Completion,
54
+ per concepts/sparc/CARD/sparc-card), a rubric per phase, and a GATE per phase.
55
+ STANDING RULES — loop-don't-ask: self-verify, grade /100 against the phase rubric with
56
+ evidence-cited deductions, below 95 → fix and regrade, max 5 iterations,
57
+ report only the final score + fixes + proof; batch questions into ONE list
58
+ with defaults; READY only after verifying the deployed/running version;
59
+ "status" → table only.
60
+ COST GUARDRAILS — tier ladder (mechanical work → cheap model via scripts/route-cheap.mjs with
61
+ its printed receipt; frontier only for the authoritative gate run); any
62
+ operation projected >$1 or >20 paid calls → state estimate and wait;
63
+ print running spend during long runs.
64
+ DONE CRITERIA — shell commands, one per phase gate plus one overall.
65
+ ```
66
+
67
+ **Every gate must be verifiable: done = exit code, not opinion.** A rubric line the executing
68
+ model could grade by vibes ("code is clean") must be paired with a command that can fail
69
+ (`npm test`, `curl -sf <url>`, `node scripts/loop-checkpoint.mjs check`). If you can't name the
70
+ command, the criterion isn't done yet — sharpen it until you can. Long/unattended prompts should
71
+ carry the checkpoint contract too (`scripts/loop-checkpoint.mjs`: read first, write last,
72
+ done-criteria as the shell command).
73
+
74
+ Prompt-craft rules from rUv's metaprompt notes (cited above): demarcate user-supplied variables
75
+ with XML tags; when the prompt asks the executing model for a score, demand the justification
76
+ BEFORE the score; give complex tasks a scratchpad step before the final answer.
77
+
78
+ ### 4. Offer execution
79
+
80
+ End with exactly one question: **"Run this now with /brain-build semantics?"** On yes, execute
81
+ the produced prompt under the full brain-build contract (see the `brain-build` skill) — phases,
82
+ ≥95 gates, cost ladder, checkpoints — starting immediately, no re-confirmation. On no, they walk
83
+ away with the prompt; it must stand alone.
@@ -0,0 +1,101 @@
1
+ ---
2
+ name: brain-score
3
+ description: Score ANY repository 0-100 across 8 dimensions using the exact evidence-or-it-didn't-happen scorecard RuvNet-Brain applies to itself. Use when the user says "score this repo", "score my repo", "scorecard", "brain-score", "how good is this codebase", "rate this project", "audit quality", or asks for an honest 0-100 quality assessment of a repository. Every deduction must cite evidence from the actual repo; a known architectural flaw caps its dimension at ≤70; a "what I did NOT test" section is mandatory; all scores are out of 100, never out of 10.
4
+ updated: 2026-07-10
5
+ ---
6
+
7
+ # Brain-Score — the 8-dimension repo scorecard (0–100)
8
+
9
+ Score the repo in front of you the way RuvNet-Brain scores itself: **gates that could have failed,
10
+ before scores that can be believed.** A score is only real if the evidence behind it was collected
11
+ by running real commands against the actual repo — never from memory, never from vibes, never from
12
+ what the README promises.
13
+
14
+ These same rules are the phase-gate grader inside `/brain-build` (the autonomous build contract:
15
+ loop each phase to ≥95 under brain-score rules, max 5 iterations — see the `brain-build` skill).
16
+
17
+ ## Non-negotiable scoring rules
18
+
19
+ 1. **Every deduction cites evidence.** Each point lost names the file/line, the command you ran and
20
+ its output, or the artifact you inspected. "Feels incomplete" is not a deduction; `"tests/ has 3
21
+ files, 2 contain zero assertions (tests/foo.test.js:1-40)"` is.
22
+ 2. **A known architectural flaw caps its dimension at ≤70** — no matter how much else in that
23
+ dimension works. (Example: a quality gate whose sample size cannot statistically detect the
24
+ regression it exists to catch caps reliability at 70, even with green CI.)
25
+ 3. **A mandatory "What I did NOT test" section.** List every claim you could not verify (didn't run
26
+ the app, didn't have the API key, skipped the 40-minute suite, couldn't reach the deployed URL).
27
+ A scorecard without this section is invalid — do not present one.
28
+ 4. **Scores are /100, never /10.** Per dimension and overall. Overall = the mean of the 8
29
+ dimensions, reported alongside the lowest dimension (a 95 average hiding a 40 is the headline).
30
+ 5. **When in doubt, score lower.** Unverified ≠ working.
31
+
32
+ ## The 8 dimensions
33
+
34
+ | # | Dimension | What the evidence looks like |
35
+ |---|---|---|
36
+ | 1 | **Correctness-evidence** | Do claims trace to proof? Run the build/tests yourself; diff README claims against actual behavior; look for "verified" claims with no artifact behind them. |
37
+ | 2 | **Test honesty** | Not coverage %, honesty: do tests assert anything? Can the suite fail? Any skipped/todo masquerading as green? Does a missing dependency SKIP loudly or pass silently? |
38
+ | 3 | **Docs truthfulness** | Do docs describe the code that exists today? Stale install commands, APIs that 404, ADRs/status docs contradicting the source. Run the quickstart literally. |
39
+ | 4 | **Security posture** | Secrets in tree, dependency audit (`npm audit` / `cargo audit` / `pip-audit`), input handling at trust boundaries, unsigned auto-update/exec paths, injection surfaces. |
40
+ | 5 | **Token/cost efficiency** | For AI-touching repos: what is injected/spent per operation, and is it measured at all? For others: hot-path waste, N+1s, unbounded loops. "Nothing measures spend" is itself a deduction. |
41
+ | 6 | **Reliability/CI** | Does CI exist, run, and gate merges? Was it red while people kept pushing? Flaky tests, non-required checks, error handling on the paths that actually fail. |
42
+ | 7 | **Maintainability** | Duplication, dead code, module boundaries, dependency freshness, whether a newcomer could change one thing without breaking three. |
43
+ | 8 | **User experience** | The consumer's first contact: install-to-working time, error messages, defaults, docs entry path. For libraries: the API surface. Run the first-run flow yourself. |
44
+
45
+ ## Procedure
46
+
47
+ 1. **Collect receipts mechanically** (never from memory): run the test suite, the linter, the
48
+ dependency audit; read CI config + recent run results if reachable; run the documented
49
+ quickstart; grep for TODO/FIXME/skip; check the license, the lockfile, the entry docs.
50
+ 2. **Use the real instruments when they're wired** (see honesty table below):
51
+ - **ruflo MCP present** → call `metaharness_score` (5-dim harness readiness incl.
52
+ `estCostPerRunUsd`) and `metaharness_oia_audit`. Both are READ-layer: **free, no API key, work
53
+ on any repo.** Fold their findings into dimensions 5–6 as cited evidence — they complement the
54
+ 8 dimensions, they don't replace them.
55
+ - **agentic-qe present** (`aqe` / aqe-mcp) → `coverage_analyze_sublinear` for dimension 2,
56
+ `security_scan_comprehensive` for dimension 4, `test_generate_enhanced` to probe untested
57
+ paths. **WARNING: `qe_qx_analyze` hallucinates on remote URLs** — it has returned templated
58
+ grades in ~2ms with every claim false. Never relay its output on a URL or artifact without
59
+ verifying against the real thing yourself first.
60
+ - **Neither installed** → plain repo inspection is fully valid: read the code, run the
61
+ commands, cite what you saw. Offer to install the tools (`npm i -g agentic-qe@latest`), but
62
+ never block scoring on them and never fake their output.
63
+ 3. **Score each dimension /100** with a deduction+evidence line per point cluster lost. Apply the
64
+ ≤70 cap where an architectural flaw exists, and say which flaw triggered it.
65
+ 4. **Write "What I did NOT test."** Then the overall (mean + lowest dimension).
66
+ 5. If this repo has persistent memory (AgentDB / `.swarm/memory.db`), store the scorecard under
67
+ key `scorecard-YYYY-MM-DD` so the next score can show movement.
68
+
69
+ ## Output format
70
+
71
+ ```
72
+ # Brain-Score: <repo> — <date>
73
+ Overall: NN/100 (mean of 8) · lowest: <dimension> at NN
74
+
75
+ | Dimension | /100 | Cap applied? |
76
+ |---|---|---|
77
+ ...8 rows...
78
+
79
+ ## Deductions (every point lost, with evidence)
80
+ - <dimension> −N: <claim> — evidence: <file:line / command + output>
81
+ ...
82
+
83
+ ## What I did NOT test
84
+ - ...
85
+
86
+ ## Instruments used
87
+ - metaharness_score / oia_audit: <used | not wired — plain inspection>
88
+ - agentic-qe: <used (which tools) | not wired>
89
+ ```
90
+
91
+ ## What's on by default vs what needs a key (say this honestly, never oversell)
92
+
93
+ | Capability | Status |
94
+ |---|---|
95
+ | `metaharness_score` + `metaharness_oia_audit` (READ layer) | **Free, on by default** in any repo when the ruflo MCP is installed — no API key. |
96
+ | agentic-qe test/coverage/security tools | **Free, on demand** when agentic-qe is installed (`npm i -g agentic-qe@latest`); `qe_qx_analyze` output must be verified against the real artifact. |
97
+ | `metaharness_evolve` (WRITE layer — self-improves the harness, keeps only measured winners) | **Needs `OPENROUTER_API_KEY`** + a runnable test command. Without the key: say so and offer the free READ layer instead. |
98
+ | Automatic per-task cheap-model routing | Goes through **agentic-flow `--router-mode cost-optimized`** — needs `OPENROUTER_API_KEY`. Claude-tier routing via `hooks_model-route` is free. |
99
+
100
+ Never claim the evolve loop or cheap routing "just works" when the key isn't set — check
101
+ (`printenv OPENROUTER_API_KEY` is empty?) and state which side of the line each feature is on.
@@ -0,0 +1,81 @@
1
+ ---
2
+ name: release-proof
3
+ description: Fail-closed exact-artifact release and deployment authority. Use before saying a release is ready, pushing a release commit, publishing npm packages, creating GitHub releases, deploying production, closing release-blocking issues, or claiming all gates are green. Requires clean immutable lineage, zero open issues, exact-SHA GitHub success, nonzero no-skip QE, packed-artifact host tests, installed Brain/RVF proof, independent graders, and post-publication byte verification.
4
+ ---
5
+
6
+ # Release Proof
7
+
8
+ Treat release as a two-seal transaction. Never publish from a source checkout merely because its
9
+ tests pass. Never turn `UNKNOWN`, `SKIP`, `todo`, `0 tests`, dirty state, or an agent report into
10
+ green.
11
+
12
+ ## Non-bypassable rules
13
+
14
+ 1. Use the exact source SHA and one packed-artifact SHA-256 everywhere.
15
+ 2. Require zero open GitHub issues for RuvNet Brain. A local fix is not a closed issue.
16
+ 3. Require every named GitHub workflow to complete successfully on the exact candidate SHA.
17
+ 4. Reject any test/QE result with zero tests, skips, todos, unknowns, pending jobs, or failures.
18
+ 5. Require two distinct independent graders scoring at least 95, each bound to the SHA and digest.
19
+ 6. Install the sealed artifact into virgin Claude Code and Codex homes; test their real entrypoints.
20
+ 7. Require the active Brain registry to contain the `ruvnet-brain` RVF store and require narrow,
21
+ broad, and concurrent cited searches to complete within 80% of their deadline.
22
+ 8. Publish only through the protected release workflow. Never run `npm publish` or `gh release
23
+ create` locally.
24
+ 9. After publication, download npm and GitHub artifacts, compare their bytes with the seal, install
25
+ both hosts again, query the active MCP again, and require `published-surface-probe` green.
26
+ 10. Close an issue only after posting its acceptance evidence. Never close from source inspection.
27
+
28
+ ## Candidate seal
29
+
30
+ Generate the receipt from commands in the protected candidate workflow. Do not hand-author it.
31
+ Validate it from the repository with:
32
+
33
+ ```bash
34
+ node scripts/release-proof.mjs --candidate release-evidence/candidate-receipt.json
35
+ ```
36
+
37
+ From an installed Claude plugin, run:
38
+
39
+ ```bash
40
+ node "$CLAUDE_PLUGIN_ROOT/skills/release-proof/scripts/release-proof.mjs" \
41
+ --candidate release-evidence/candidate-receipt.json
42
+ ```
43
+
44
+ Exit 0 is the only candidate seal. Read every failure code; repair the system, regenerate evidence,
45
+ and rerun. Do not edit the receipt to remove a failure.
46
+
47
+ ## Publication seal
48
+
49
+ After the protected publisher completes, validate both receipts:
50
+
51
+ ```bash
52
+ node scripts/release-proof.mjs \
53
+ --candidate release-evidence/candidate-receipt.json \
54
+ --publication release-evidence/publication-receipt.json
55
+ ```
56
+
57
+ Only exit 0 permits “shipped,” “deployed,” “green,” or “ready.” If publication occurred but this
58
+ seal fails, say `PUBLICATION DEGRADED`, preserve the previous known-good release, and repair or
59
+ roll back through the release workflow.
60
+
61
+ ## Evidence and issue handling
62
+
63
+ For each issue:
64
+
65
+ 1. Reproduce the original symptom against the old/public artifact.
66
+ 2. Run its acceptance criteria against the sealed candidate.
67
+ 3. Disable or mutate the fix; the regression must fail.
68
+ 4. Post SHA, artifact digest, commands, results, and untested limits to the issue.
69
+ 5. Close only after the GitHub evidence is visible and exact-SHA required checks are green.
70
+
71
+ Read [references/receipt-contract.md](references/receipt-contract.md) for receipt fields and failure
72
+ semantics. Store the protocol and final release receipt in Ruflo/AgentDB only after the publication
73
+ seal passes.
74
+
75
+ ## Status language
76
+
77
+ - Candidate seal absent or failed: `NOT READY`.
78
+ - Candidate sealed, not published: `SEALED, NOT SHIPPED`.
79
+ - Published, publication seal pending: `PUBLISHED, NOT VERIFIED`.
80
+ - Publication seal failed: `PUBLICATION DEGRADED`.
81
+ - Both seals exit 0: `SHIPPED AND VERIFIED`.
@@ -0,0 +1,4 @@
1
+ interface:
2
+ display_name: "Release Proof"
3
+ short_description: "Fail-closed exact-artifact release authority"
4
+ default_prompt: "Prove this release candidate end to end and refuse publication unless every required gate is green."
@@ -0,0 +1,38 @@
1
+ # Receipt contract
2
+
3
+ The authority accepts schema version 1 JSON. Receipts are append-only evidence artifacts generated
4
+ by protected workflows, never editable status documents.
5
+
6
+ ## Candidate receipt
7
+
8
+ Required bindings:
9
+
10
+ - `sha`, `tree`, `dirty:false`
11
+ - `artifact.path`, `artifact.sha256`, `artifact.sourceSha`
12
+ - exact-SHA release-vector verdict with zero unknown/skipped
13
+ - aggregate tests with nonzero total, all passed, zero failed/skipped/todo
14
+ - fresh coverage floor and zero critical/high security findings
15
+ - zero open GitHub issues
16
+ - required GitHub workflow results on the same SHA
17
+ - virgin-home Claude and Codex results on the same artifact digest
18
+ - installed Brain self-RVF plus narrow, broad, and concurrent cited search timings
19
+ - nonzero Agentic QE totals with zero failed/skipped
20
+ - two distinct independent grader receipts at 95 or higher, bound to SHA and digest
21
+
22
+ ## Publication receipt
23
+
24
+ Required bindings:
25
+
26
+ - candidate SHA and artifact digest
27
+ - npm and GitHub release bytes matching the candidate digest
28
+ - clean installed Claude and Codex results from the public package
29
+ - installed Brain self-RVF and broad search within 80 percent of deadline
30
+ - successful exact-SHA `published-surface-probe`
31
+
32
+ ## Failure semantics
33
+
34
+ Any missing field, malformed digest, mismatched SHA, dirty tree, open issue, absent/pending/red
35
+ workflow, skipped/todo/zero-test result, missing RVF store, uncited search, deadline-margin breach,
36
+ low/missing grader, or public byte mismatch is `FAIL`. There is no warning state and no score
37
+ average. The authority never publishes; publication belongs to the protected workflow after the
38
+ candidate seal.