ruvnet-brain 4.0.1 → 4.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/.claude-plugin/marketplace.json +1 -0
  2. package/README.md +4 -4
  3. package/bin/install.mjs +303 -24
  4. package/console/CONTRACT.md +172 -0
  5. package/console/activity.js +753 -0
  6. package/console/app.js +4189 -0
  7. package/console/architecture.html +1221 -0
  8. package/console/assets/depth-1.webp +0 -0
  9. package/console/assets/depth-2.webp +0 -0
  10. package/console/assets/depth-3.webp +0 -0
  11. package/console/assets/harness-vs-plain.svg +259 -0
  12. package/console/assets/hero.webp +0 -0
  13. package/console/assets/memory.webp +0 -0
  14. package/console/assets/metaharness.svg +247 -0
  15. package/console/index.html +777 -0
  16. package/console/install-architecture.html +162 -0
  17. package/console/install-mockup.html +543 -0
  18. package/console/style.css +2144 -0
  19. package/console/tips.css +926 -0
  20. package/console/tips.html +858 -0
  21. package/console/tips.js +128 -0
  22. package/docs/RELEASE-NOTES-4.0.md +88 -0
  23. package/kb/model-requirements.mjs +37 -6
  24. package/keys/ruvnet-brain-signing.pub.pem +3 -0
  25. package/package.json +8 -22
  26. package/plugin/.claude-plugin/marketplace.json +1 -0
  27. package/plugin/.claude-plugin/plugin.json +2 -3
  28. package/plugin/.codex-plugin/plugin.json +1 -1
  29. package/plugin/commands/brain-console.md +2 -2
  30. package/plugin/commands/configure.md +3 -2
  31. package/plugin/commands/rvbc.md +4 -3
  32. package/plugin/commands/rvcb.md +2 -2
  33. package/plugin/commands/whats-new.md +6 -6
  34. package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
  35. package/plugin/hooks/hooks.json +1 -2
  36. package/plugin/mcp/managed-cli-interface.mjs +47 -4
  37. package/plugin/mcp/server.mjs +90 -32
  38. package/plugin/scripts/detach.mjs +14 -0
  39. package/plugin/scripts/first-session-worker.mjs +38 -0
  40. package/plugin/scripts/ground-ruvnet.sh +16 -6
  41. package/plugin/scripts/hook-shim.mjs +34 -29
  42. package/plugin/scripts/learn-capture.sh +22 -3
  43. package/plugin/scripts/learn-flush.mjs +21 -4
  44. package/plugin/scripts/runtime-preferences.mjs +269 -0
  45. package/plugin/scripts/session-start-core.mjs +503 -0
  46. package/plugin/scripts/session-start.sh +3 -858
  47. package/plugin/scripts/whats-new.mjs +42 -0
  48. package/plugin/skills/brain-console/SKILL.md +4 -2
  49. package/plugin/skills/release-proof/SKILL.md +98 -0
  50. package/plugin/skills/release-proof/agents/openai.yaml +4 -0
  51. package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
  52. package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
  53. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
  54. package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
  55. package/plugin/skills/rvbc/SKILL.md +9 -6
  56. package/plugin/skills/whats-new/SKILL.md +4 -4
  57. package/scripts/adr-backfill.mjs +107 -0
  58. package/scripts/advocacy-outcomes.mjs +808 -0
  59. package/scripts/agentdb-context.mjs +216 -0
  60. package/scripts/agentdb-fleet-doctor.mjs +101 -0
  61. package/scripts/ascii-drift.mjs +236 -0
  62. package/scripts/behavioral-l1-l4.mjs +210 -0
  63. package/scripts/brain-capability-check.mjs +72 -0
  64. package/scripts/brain-grade-groundtruth.mjs +100 -0
  65. package/scripts/brain-latency-50.mjs +227 -0
  66. package/scripts/brain-novice-50.mjs +189 -0
  67. package/scripts/brain-stamp.mjs +94 -0
  68. package/scripts/brain-state.mjs +212 -0
  69. package/scripts/build-bundle.mjs +531 -0
  70. package/scripts/build-concepts.mjs +132 -0
  71. package/scripts/build-l2.mjs +71 -0
  72. package/scripts/build-primer.mjs +73 -0
  73. package/scripts/build-symbols.mjs +68 -0
  74. package/scripts/calibrate-router.mjs +97 -0
  75. package/scripts/capability-audit.mjs +321 -0
  76. package/scripts/capability-registry.mjs +876 -0
  77. package/scripts/check-indexation.mjs +108 -0
  78. package/scripts/check-legibility.mjs +189 -0
  79. package/scripts/ci/build-fixture-kb.mjs +67 -0
  80. package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
  81. package/scripts/ci/learning-replay-recorder.mjs +59 -0
  82. package/scripts/ci/mutate-hook-timeout.mjs +70 -0
  83. package/scripts/ci/stranger-fixture-stage.mjs +17 -0
  84. package/scripts/ci/stranger-scenario.mjs +228 -0
  85. package/scripts/ci/stranger-timeout.mjs +25 -0
  86. package/scripts/ci-verdict.mjs +29 -0
  87. package/scripts/claims-verify.mjs +710 -0
  88. package/scripts/clear-claude-tmp.sh +31 -0
  89. package/scripts/console-engine.mjs +434 -0
  90. package/scripts/console-engine.test.mjs +125 -0
  91. package/scripts/corpus-qa.mjs +250 -0
  92. package/scripts/correction-detect-embed.mjs +346 -0
  93. package/scripts/correction-detect-measure.mjs +270 -0
  94. package/scripts/correction-detect.mjs +686 -0
  95. package/scripts/count-chunks.mjs +54 -0
  96. package/scripts/described-questions.json +30 -0
  97. package/scripts/design-grade.mjs +58 -0
  98. package/scripts/dev-plugin-link.sh +105 -0
  99. package/scripts/distill-project.mjs +200 -0
  100. package/scripts/doc-currency.mjs +801 -0
  101. package/scripts/eval-brain.mjs +244 -0
  102. package/scripts/fix-metaharness-memretrieve.mjs +121 -0
  103. package/scripts/fix-workstream.mjs +291 -0
  104. package/scripts/full-hints.mjs +87 -0
  105. package/scripts/gate.sh +39 -0
  106. package/scripts/gates.mjs +146 -0
  107. package/scripts/gen-console-images.mjs +54 -0
  108. package/scripts/gen-images.mjs +47 -0
  109. package/scripts/git-clone-refresh.mjs +52 -0
  110. package/scripts/git-hooks/pre-push +126 -0
  111. package/scripts/goal-match.mjs +398 -0
  112. package/scripts/goldie-research.mjs +223 -0
  113. package/scripts/goldie-weekly.sh +67 -0
  114. package/scripts/health-repair.mjs +237 -0
  115. package/scripts/helix-scenario-questions.json +10 -0
  116. package/scripts/ingest-gists.mjs +230 -0
  117. package/scripts/ingest-meeting.mjs +115 -0
  118. package/scripts/ingest-repo.mjs +79 -0
  119. package/scripts/install-npx-witness.sh +49 -0
  120. package/scripts/issue-fix.mjs +558 -0
  121. package/scripts/issue-watch.mjs +276 -0
  122. package/scripts/issue4-close-note.md +31 -0
  123. package/scripts/key-canary.mjs +91 -0
  124. package/scripts/latency-to-surface.mjs +233 -0
  125. package/scripts/learning-enable.mjs +380 -0
  126. package/scripts/learning-replay.mjs +1570 -0
  127. package/scripts/learnings.mjs +62 -0
  128. package/scripts/lesson-gate.mjs +680 -0
  129. package/scripts/lesson-lifecycle.mjs +449 -0
  130. package/scripts/lesson-promote.mjs +262 -0
  131. package/scripts/lesson-ratify.mjs +98 -0
  132. package/scripts/lesson-seed.mjs +252 -0
  133. package/scripts/lesson-store.mjs +447 -0
  134. package/scripts/loop-checkpoint.mjs +86 -0
  135. package/scripts/memdb-health.sh +14 -0
  136. package/scripts/memory-doctor.mjs +326 -0
  137. package/scripts/model-catalog.mjs +79 -0
  138. package/scripts/nightly-controller.mjs +66 -0
  139. package/scripts/nightly-gists.sh +72 -0
  140. package/scripts/nightly-wrapper.sh +172 -0
  141. package/scripts/notify.sh +12 -0
  142. package/scripts/npx-witness.sh +56 -0
  143. package/scripts/onboarding-console.mjs +2922 -0
  144. package/scripts/private-fence.mjs +69 -0
  145. package/scripts/proactivity-metrics.mjs +118 -0
  146. package/scripts/proof-questions.json +56 -0
  147. package/scripts/protected-release-invocation.mjs +76 -0
  148. package/scripts/prove.mjs +95 -0
  149. package/scripts/proxy/claude-proxied.sh +57 -0
  150. package/scripts/proxy/proxy-revert.sh +59 -0
  151. package/scripts/proxy/proxy-up.sh +60 -0
  152. package/scripts/proxy/proxy-verify.mjs +142 -0
  153. package/scripts/publication-receipt.mjs +307 -0
  154. package/scripts/published-surface-probe.mjs +241 -0
  155. package/scripts/qe/card-lane-gate.mjs +162 -0
  156. package/scripts/qe/session-start-gate.mjs +229 -0
  157. package/scripts/qe/ux-suite.mjs +323 -0
  158. package/scripts/reconcile-project.mjs +0 -0
  159. package/scripts/record-lesson.mjs +113 -0
  160. package/scripts/refresh-model-catalog.mjs +99 -0
  161. package/scripts/release-authority.mjs +93 -0
  162. package/scripts/release-proof.mjs +9 -0
  163. package/scripts/release-vector.mjs +281 -0
  164. package/scripts/release.mjs +439 -0
  165. package/scripts/remedy-registry.mjs +247 -0
  166. package/scripts/rerank-cap-eval.mjs +265 -0
  167. package/scripts/rerank-cap-warm-ab.mjs +129 -0
  168. package/scripts/route-cheap.mjs +20 -15
  169. package/scripts/router-utilization.mjs +182 -0
  170. package/scripts/routing-flywheel.mjs +596 -0
  171. package/scripts/rvf-generation.mjs +104 -0
  172. package/scripts/rvf-index-audit.mjs +138 -0
  173. package/scripts/self-update.mjs +296 -0
  174. package/scripts/selfcheck.mjs +7 -1
  175. package/scripts/sign-bundle.mjs +69 -0
  176. package/scripts/signal-watch.mjs +171 -0
  177. package/scripts/stabilization-receipt.mjs +108 -0
  178. package/scripts/stack-sync.mjs +469 -0
  179. package/scripts/stamp-existing-rvf-generations.mjs +53 -0
  180. package/scripts/stamp-sweep.mjs +144 -0
  181. package/scripts/status-honesty.mjs +102 -0
  182. package/scripts/sync-version.mjs +217 -0
  183. package/scripts/token-report.mjs +102 -0
  184. package/scripts/top100-benchmark.mjs +479 -0
  185. package/scripts/top100-corpus.mjs +112 -0
  186. package/scripts/top100-semantic-assertions.mjs +449 -0
  187. package/scripts/update-apply.mjs +9 -0
  188. package/scripts/upgrade-notice.mjs +14 -0
  189. package/scripts/verify-bundle.mjs +51 -0
  190. package/scripts/verify-channels.mjs +184 -0
  191. package/scripts/verify-model-catalog.mjs +104 -0
  192. package/scripts/verify-nightly-close-issue4.sh +31 -0
  193. package/scripts/version.mjs +40 -0
  194. package/scripts/wired-check.mjs +867 -0
  195. package/plugin/scripts/finalize-token-meter.mjs +0 -25
@@ -0,0 +1,1570 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * scripts/learning-replay.mjs — the COUNTERFACTUAL REPLAY TRAP (ADR-058 §D4, DDD-0013 Context 1,
4
+ * aggregate `CounterfactualTrap`). Invariant name: **LEARNING-REPLAY**.
5
+ *
6
+ * ─────────────────────────────────────────────────────────────────────────────────────────────────
7
+ * WHAT THIS INVERTS, and why it exists at all.
8
+ *
9
+ * `scripts/behavioral-l1-l4.mjs`'s L4 asserts that the brain's own injected prose CONTAINS the words
10
+ * 'take the wheel', 'SPARC', 'swarm'. That is a check on what the brain SAID. It cannot fail on an
11
+ * agent that ignored every word of it, and it certified "behavioral, all pass" for weeks while
12
+ * nothing downstream was measured at all. This file measures the opposite thing and only that thing:
13
+ *
14
+ * did an agent's PRODUCED ARTIFACT change, against a control that did not receive the lesson.
15
+ *
16
+ * The oracle is a parse of a command string — `plugin/scripts/hook-input.mjs:findInvocations()`,
17
+ * executable-position classification, the same anti-corruption boundary DDD-0013 mandates against
18
+ * the host's Bash envelope. It is never a similarity score and never a model grading a model.
19
+ *
20
+ * ─────────────────────────────────────────────────────────────────────────────────────────────────
21
+ * THE TRAP, concretely (ADR-058 §D4 specifies it so it cannot dissolve into intention).
22
+ *
23
+ * RECORD, in fixture-project-A: the correction that `ruflo memory search` takes its query with the
24
+ * `-q` flag and rejects a bare positional. This is a FACT ABOUT THE REAL CLI, verified against the
25
+ * real global binary (`~/.npm-global/bin/ruflo memory search --help` prints
26
+ * `-q, --query Search query (required)`), not recalled. An oracle built on a false premise is
27
+ * worthless, so the harness RE-VERIFIES it at run time (`verifyRufloFlag()`) and refuses to run
28
+ * against a CLI whose interface no longer matches.
29
+ *
30
+ * REPLAY, in fixture-project-B: a fresh session, a DIFFERENTLY-WORDED task ("recall the note about
31
+ * the caching strategy") that shares no content word with the lesson. String-matching the lesson
32
+ * text cannot be what carries it; only the flag can.
33
+ *
34
+ * PASS requires all of:
35
+ * (a) the lesson is in the transcript BEFORE the first tool call — measured as stream position,
36
+ * not asserted from the fact that UserPromptSubmit "happens first";
37
+ * (b) the treated arm's produced command carries the token where the BRAIN-OFF CONTROL's does not;
38
+ * (c) it still holds after a nightly refresh runs between record and replay — the refresh is
39
+ * real: a new Stable-Spine generation is installed into the fixture brain home and the
40
+ * pointer flipped, so the replay's hooks execute from a DIFFERENT code root than the record
41
+ * did, and `ruflo memory distill run` / `ruflo memory backup` (the two commands
42
+ * scripts/nightly-wrapper.sh actually runs nightly) are run against project A's store.
43
+ * (d) the produced command NAMES THE REAL SUBCOMMAND, EXECUTES against the real fixture store,
44
+ * EXITS 0, and ACTUALLY RETRIEVES the memory the task asked for. See "THE EXECUTION GATE".
45
+ *
46
+ * ─────────────────────────────────────────────────────────────────────────────────────────────────
47
+ * THE INVALIDATION RULE — DDD-0013 invariant 6, and the whole point of the file.
48
+ *
49
+ * A trap whose CONTROL run also produces the token is INVALID. The result is INCONCLUSIVE.
50
+ * NEVER a pass.
51
+ *
52
+ * If the model would have got it right anyway, the trap measured nothing — it measured the model's
53
+ * priors. This is encoded as CODE, not as a comment: `aggregate()` computes `controlTokenRuns`
54
+ * FIRST and the PASS branch is unreachable while it is non-zero, and a final assertion throws if a
55
+ * PASS verdict is ever paired with a successful control. A check that can report PASS on a
56
+ * meaningless measurement is the L4 defect rebuilt one file to the left.
57
+ *
58
+ * (DDD-0013 invariant 6 words the invalid outcome as `UNKNOWN`; ADR-058 §D4 words it `INCONCLUSIVE`.
59
+ * This file emits INCONCLUSIVE and treats it as strictly non-PASS, which satisfies both — the two
60
+ * documents disagree on the LABEL, never on the consequence.)
61
+ *
62
+ * ─────────────────────────────────────────────────────────────────────────────────────────────────
63
+ * THE EXECUTION GATE — added 2026-07-28, closing the largest single deduction in the D4 re-score.
64
+ *
65
+ * An independent grader (GPT-5.6-Sol) scored this dimension 44/100 and named the reason exactly:
66
+ *
67
+ * "The recorded replay says PASS 3/3, yet all three treated commands have subcommandCorrect:
68
+ * false … PASS depends on token use, control contrast and lesson delivery — not successful
69
+ * command execution or successful retrieval. The suite currently certifies unusable learned
70
+ * behavior."
71
+ *
72
+ * It was right, and the block that used to sit here — arguing the subcommand is REPORTED and not
73
+ * GATED because the lesson only taught the flag — was a defensible claim about the LESSON and an
74
+ * indefensible one about the CLAIM. The trap's headline is "learning demonstrated". A command that
75
+ * would fail if anyone ran it demonstrates nothing, whatever it says about the flag.
76
+ *
77
+ * So the verdict now additionally requires, per run, that the treated arm's produced command:
78
+ * 1. names the real subcommand (`subcommandCorrect`) — no longer observed-only;
79
+ * 2. EXECUTES against the real fixture store and exits 0;
80
+ * 3. actually RETRIEVES the memory project B's task asked for — asserted on RETURNED CONTENT.
81
+ *
82
+ * Point 3 is not redundant with point 2, and this is the whole reason exit status alone is not
83
+ * admissible. Measured live on this machine, 2026-07-28, against the real global binary:
84
+ *
85
+ * ruflo memory search -q "caching strategy" --path <db> → EXIT 0 · "Found 1 results"
86
+ * ruflo memory recall -q "caching strategy" --path <db> → EXIT 0 · prints the `memory` HELP
87
+ * ruflo recall -q "caching strategy" → EXIT 1 · "Unknown command: recall"
88
+ * ruflo memory search "caching strategy" --path <db> → EXIT 1 · "Required option missing: --query"
89
+ * ruflo memory search -q "<absent phrase>" --path <db> → EXIT 0 · "[WARN] No results found"
90
+ *
91
+ * `ruflo memory recall -q` — the exact command two of the three certified runs produced — EXITS 0.
92
+ * An exit-status gate would have passed it. Only an assertion on returned content catches it. (Line
93
+ * 5 is the same lesson from the other side: a perfectly-formed search that finds nothing also exits
94
+ * 0. Retrieval is the claim; exit status is not.)
95
+ *
96
+ * WHAT THE FIXTURE HAD TO CHANGE FOR THIS TO BE MEASURABLE, stated rather than finessed: project B's
97
+ * prompt already asserted "earlier in this project someone recorded a note about the caching
98
+ * strategy", and that was FALSE of the fixture world — project B's store was empty. So the harness
99
+ * now seeds that note into project B's own `.swarm/memory.db` (`seedProjectBMemory`). This is a fix
100
+ * to the FIXTURE, not to the lesson: the seeded note says nothing about `-q`, no arm ever sees its
101
+ * text (the recorder blocks every command before it runs), and the lesson text is untouched. Without
102
+ * it, even a flawless `ruflo memory search -q "caching strategy"` would retrieve nothing and the new
103
+ * gate would be measuring a harness bug — Rule 22 check (d).
104
+ *
105
+ * SAFETY. The recorder BLOCKS the agent's command on purpose (a fixture agent must not run anything
106
+ * on a real machine, and must not learn the answer from a CLI's own `--help` mid-run). That is kept.
107
+ * Execution happens OUT OF BAND, after the arm is over, in the harness — and never through a shell:
108
+ * `executeProducedCommand()` runs the argv `findInvocations()` already parsed, so pipes, redirects,
109
+ * substitutions and metacharacters are structurally absent. It also refuses to run a mutating
110
+ * subcommand, and refuses any `--path`/`--db` pointing outside the fixture world. Both refusals mark
111
+ * the run NOT-RETRIEVED, so every one of them can only LOWER the rate.
112
+ *
113
+ * ─────────────────────────────────────────────────────────────────────────────────────────────────
114
+ * A RATE, NEVER A VERDICT. N runs, PASS at >= 2/3 of them, transcripts archived. One run of a
115
+ * stochastic system is an anecdote; the artifact records k/n and every arm's classification.
116
+ *
117
+ * ─────────────────────────────────────────────────────────────────────────────────────────────────
118
+ * WHAT THE ARMS ACTUALLY DIFFER BY — the product's OWN switch, not a harness flag.
119
+ *
120
+ * Both arms run the identical fixture, the identical prompt, the identical hook registration
121
+ * (`hook-shim.mjs unprompted-speech UserPromptSubmit`, exactly as plugin/hooks/hooks.json registers
122
+ * it). The ONLY difference is the presence of the `brain-off` sentinel in the arm's
123
+ * RUVNET_BRAIN_STATE_DIR — ADR-054's real consent switch, whose `offBehavior: 'silence'` contract
124
+ * for the unprompted plane means the control receives ZERO bytes. That is why mutant 2 ("run the
125
+ * treated arm brain-disabled") is not a separate code path: it IS the control condition.
126
+ *
127
+ * ─────────────────────────────────────────────────────────────────────────────────────────────────
128
+ * COST. Real model tokens, priced in the open (ADR-058 §D4: "the one standing spend"). Default model
129
+ * is haiku — the trap measures whether CONTEXT REACHES the agent, not whether the agent is clever.
130
+ * Measured 2026-07-27 on this machine: ~$0.10 and ~8s of wall clock per arm, 2 arms per run.
131
+ *
132
+ * ─────────────────────────────────────────────────────────────────────────────────────────────────
133
+ * USAGE
134
+ * node scripts/learning-replay.mjs # N=3 replay, real tokens, writes the artifact
135
+ * node scripts/learning-replay.mjs --n 1 # one run
136
+ * node scripts/learning-replay.mjs --check # NO tokens: gate on the committed artifact
137
+ * node scripts/learning-replay.mjs --dry-run # NO tokens: build fixtures, prove the wire, UNKNOWN
138
+ * node scripts/learning-replay.mjs --mutant <name> # see MUTANTS below
139
+ * Exit: 0 = PASS. 1 = FAIL. 3 = INCONCLUSIVE. 4 = UNKNOWN. (Only 0 is a pass, by construction.)
140
+ */
141
+
142
+ import fs from 'node:fs';
143
+ import os from 'node:os';
144
+ import path from 'node:path';
145
+ import { spawnSync } from 'node:child_process';
146
+ import { fileURLToPath } from 'node:url';
147
+
148
+ import { findInvocations } from '../plugin/scripts/hook-input.mjs';
149
+ import { makeLesson, saveLessons, loadLessons } from './lesson-store.mjs';
150
+
151
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
152
+
153
+ /** The invariant name the release-gate vector in scripts/claims-verify.mjs consumes. */
154
+ export const INVARIANT = 'LEARNING-REPLAY';
155
+
156
+ export const VERDICT = Object.freeze({
157
+ PASS: 'PASS',
158
+ FAIL: 'FAIL',
159
+ UNKNOWN: 'UNKNOWN',
160
+ INCONCLUSIVE: 'INCONCLUSIVE',
161
+ });
162
+ /** Exit codes. PASS is the ONLY zero — UNKNOWN is never a pass, and neither is INCONCLUSIVE. */
163
+ export const EXIT = Object.freeze({ PASS: 0, FAIL: 1, INCONCLUSIVE: 3, UNKNOWN: 4 });
164
+
165
+ /** The committed, machine-readable result artifact. */
166
+ export const RESULT_FILE = path.join(ROOT, 'data', 'learning-replay-result.json');
167
+ export const POST_TASK_RESULT_FILE = path.join(ROOT, 'data', 'learning-replay-post-task-result.json');
168
+
169
+ export const TRAP = Object.freeze({
170
+ MEMORY_SEARCH: 'memory-search-query',
171
+ POST_TASK: 'hooks-post-task-persistence',
172
+ });
173
+
174
+ export const POST_TASK_PROMPT =
175
+ 'Record successful completion of the release retry-budget investigation in ruflo hooks learning, '
176
+ + 'assign it to the tester agent, and persist the routing decision for later sessions. '
177
+ + 'Use the terminal CLI, not an MCP tool. Do not inspect help or run any setup command. '
178
+ + 'Run exactly one post-task command now, then tell me what you ran.';
179
+ export const POST_TASK_LESSON =
180
+ 'A ruflo hooks post-task call persists a reusable routing outcome only when it includes all three: '
181
+ + '--task with the task description, --agent with the executor, and --store-results; a success/task-id alone '
182
+ + 'may print success but does not create the routing decision that later learning reads.';
183
+
184
+ function trapSpec(id = TRAP.MEMORY_SEARCH) {
185
+ if (id === TRAP.POST_TASK) {
186
+ return {
187
+ id,
188
+ lessonId: 'FX-D4-ruflo-hooks-post-task-persistence',
189
+ prompt: POST_TASK_PROMPT,
190
+ lesson: POST_TASK_LESSON,
191
+ memoryKey: 'lesson-ruflo-hooks-post-task-persistence',
192
+ recordQuery: 'ruflo hooks post task routing persistence',
193
+ check: 'the produced ruflo hooks post-task command includes --task, --agent, and --store-results',
194
+ };
195
+ }
196
+ return {
197
+ id: TRAP.MEMORY_SEARCH,
198
+ lessonId: 'FX-D4-ruflo-memory-search-flag',
199
+ prompt: REPLAY_PROMPT,
200
+ lesson: LESSON_STATEMENT,
201
+ memoryKey: 'lesson-ruflo-memory-search-flag',
202
+ recordQuery: 'ruflo CLI memory query flag',
203
+ check: 'the produced ruflo memory search command delivers its query through -q/--query',
204
+ };
205
+ }
206
+
207
+ /**
208
+ * The files whose change invalidates a recorded result. `--check` refuses to call a result CURRENT
209
+ * for a SHA if any of these moved since — ADR-056's currency discipline, applied to a token-priced
210
+ * measurement that cannot be re-run on every commit.
211
+ */
212
+ export const LOAD_BEARING = Object.freeze([
213
+ 'scripts/learning-replay.mjs',
214
+ 'scripts/ci/learning-replay-recorder.mjs',
215
+ 'scripts/ci/learning-replay-codex-adapter.mjs',
216
+ 'scripts/lesson-store.mjs',
217
+ 'scripts/lesson-gate.mjs',
218
+ 'plugin/scripts/lesson-hooks.sh',
219
+ 'plugin/scripts/unprompted-runtime.mjs',
220
+ 'plugin/scripts/hook-shim.mjs',
221
+ 'plugin/scripts/hook-input.mjs',
222
+ ]);
223
+
224
+ // ── THE ORACLE ──────────────────────────────────────────────────────────────────────────────────
225
+ /**
226
+ * Classify ONE produced command against the machine-checkable token.
227
+ *
228
+ * 'flagged' — a ruflo invocation that delivers its query through `-q` / `--query`.
229
+ * THIS IS THE TOKEN — and it is the token ADR-058 §D4 names, verbatim:
230
+ * "the produced command uses -q where the brain-off control uses the positional form".
231
+ * 'positional' — a ruflo invocation carrying a bare positional query and no -q/--query. The exact
232
+ * wrong form the lesson names.
233
+ * 'other' — ruflo invoked, but the query arrives some other way (`--topic`, `--project`), or
234
+ * no query at all.
235
+ * 'none' — no ruflo invocation at all.
236
+ *
237
+ * `--query` counts as the token even though the lesson says `-q`: the live `--help` prints them as
238
+ * ONE option (`-q, --query`), so failing the long form would make the oracle reject a command that is
239
+ * correct. An oracle stricter than the interface it models measures its own arbitrariness. The
240
+ * consequence is faced rather than tuned away — a control arm that reaches `--query` on its own
241
+ * INVALIDATES the trap, which is invariant 6 doing its job.
242
+ *
243
+ * ── THE SUBCOMMAND: REPORTED (2026-07-27) → GATED (2026-07-28) ───────────────────────────────────
244
+ * The first shipped oracle also required `ruflo memory search`; the first real N=3 measured treated
245
+ * 3/3 carrying `-q` against control 0/3 — a clean separation — and scored it 0/3 FAIL, because the
246
+ * treated arm spelled it `ruflo recall -q …` / `ruflo memory recall -q …`. That was read as a
247
+ * harness error (the lesson taught the flag and said nothing about the subcommand) and the gate was
248
+ * relaxed to observed-only.
249
+ *
250
+ * That relaxation is REVERSED, and the reversal is the point of the whole change. Both readings of
251
+ * the 2026-07-27 evidence are true at once, and only one of them is about the CLAIM:
252
+ * · about the LESSON — right. The lesson carries the flag; failing the treatment on a subcommand
253
+ * it never mentioned measures the model's priors about rUv's command tree.
254
+ * · about the CLAIM — wrong, and the grader caught it. The invariant's headline is that LEARNING
255
+ * WAS DEMONSTRATED. `ruflo recall -q "x"` exits 1. `ruflo memory recall -q "x"` exits 0 and
256
+ * prints the help. Certifying "learning demonstrated" on a command that retrieves nothing is
257
+ * the L4 defect rebuilt one file to the left — proof that something was SAID, not that anything
258
+ * WORKED.
259
+ *
260
+ * The honest resolution is to keep the token oracle exactly as narrow as it was (so nothing is
261
+ * credited to the lesson that the control reaches unaided) and to add the gate the claim actually
262
+ * needs: the command has to WORK. `subcommandCorrect` is now one of the conditions, and
263
+ * `aggregate()` carries an assertion making `subcommandCorrect: false` structurally unable to
264
+ * coexist with a PASS verdict — the same shape as the invariant-6 guard beside it. If the rate
265
+ * falls as a result, the rate was wrong before; the lesson text is NOT tuned to recover it.
266
+ */
267
+ export function classifyCommand(cmd) {
268
+ const invocations = findInvocations(String(cmd || ''), ['ruflo', 'claude-flow']);
269
+ if (!invocations.length) return 'none';
270
+ let sawPositional = false;
271
+ for (const inv of invocations) {
272
+ const args = inv.args.filter((a) => a !== '');
273
+ if (args.some((a) => a === '-q' || a === '--query' || a.startsWith('--query='))) return 'flagged';
274
+ // Bare (non-flag, non-flag-value) tokens. A flag consumes the token after it unless that token
275
+ // is itself a flag — generic, so `--topic "x"` and `-n default` are handled without a whitelist
276
+ // that would rot the moment rUv adds an option.
277
+ const bare = [];
278
+ for (let i = 0; i < args.length; i++) {
279
+ const a = args[i];
280
+ if (a.startsWith('-')) { if (!a.includes('=') && args[i + 1] && !args[i + 1].startsWith('-')) i++; continue; }
281
+ bare.push(a);
282
+ }
283
+ // Which bare token is the QUERY rather than a subcommand? A subcommand is one short lowercase
284
+ // word; a query is a phrase. So: a bare token past the first that contains whitespace (or is
285
+ // implausibly long) is a positional query, as is any third bare token.
286
+ // ONLY THE LABEL DEPENDS ON THIS. The verdict keys on `flagged` vs not-`flagged` and on `none`;
287
+ // 'positional' and 'other' are both simply "did not carry the token". A mislabel here can never
288
+ // move PASS/FAIL/INCONCLUSIVE — it can only make the reported description of a control arm less
289
+ // precise, which is why a heuristic is acceptable HERE and nowhere near the token itself.
290
+ const queryish = (t) => /\s/.test(t) || t.length > 24;
291
+ if (bare.length >= 3 || bare.slice(1).some(queryish)) sawPositional = true;
292
+ }
293
+ return sawPositional ? 'positional' : 'other';
294
+ }
295
+
296
+ /** GATING since 2026-07-28: was the invocation the REAL `ruflo memory search`? */
297
+ export function subcommandCorrect(cmd) {
298
+ for (const inv of findInvocations(String(cmd || ''), ['ruflo', 'claude-flow'])) {
299
+ const words = inv.args.filter((a) => a !== '' && !a.startsWith('-'));
300
+ const mi = words.indexOf('memory');
301
+ if (mi !== -1 && words[mi + 1] === 'search') return true;
302
+ }
303
+ return false;
304
+ }
305
+
306
+ /** The token test, isolated so every caller asks it the same way. */
307
+ export const carriesToken = (cls) => cls === 'flagged';
308
+
309
+ /** The second trap is deliberately a different Ruflo surface and a different required option. */
310
+ function optionValue(args, short, long) {
311
+ for (let i = 0; i < args.length; i++) {
312
+ const arg = args[i];
313
+ if (arg === short || arg === long) return args[i + 1] && !args[i + 1].startsWith('-') ? args[i + 1] : null;
314
+ if (arg.startsWith(`${long}=`)) return arg.slice(long.length + 1);
315
+ }
316
+ return null;
317
+ }
318
+
319
+ export function classifyPostTaskCommand(cmd) {
320
+ const invocations = findInvocations(String(cmd || ''), ['ruflo', 'claude-flow']);
321
+ if (!invocations.length) return 'none';
322
+ let sawPostTask = false;
323
+ for (const inv of invocations) {
324
+ const args = inv.args.filter(Boolean);
325
+ const hi = args.indexOf('hooks');
326
+ if (hi === -1 || args[hi + 1] !== 'post-task') continue;
327
+ sawPostTask = true;
328
+ const task = optionValue(args, '-t', '--task');
329
+ const agent = optionValue(args, '-a', '--agent');
330
+ const store = args.includes('--store-results')
331
+ || args.some((a) => a.startsWith('--store-results=') && !/=false$/i.test(a));
332
+ if (task && agent && store) return 'flagged';
333
+ }
334
+ return sawPostTask ? 'partial' : 'other';
335
+ }
336
+
337
+ export function postTaskSubcommandCorrect(cmd) {
338
+ return findInvocations(String(cmd || ''), ['ruflo', 'claude-flow'])
339
+ .some((inv) => {
340
+ const words = inv.args.filter((a) => a !== '' && !a.startsWith('-'));
341
+ const hi = words.indexOf('hooks');
342
+ return hi !== -1 && words[hi + 1] === 'post-task';
343
+ });
344
+ }
345
+
346
+ // ── THE EXECUTION GATE ──────────────────────────────────────────────────────────────────────────
347
+ /**
348
+ * The note project B's prompt already claimed was there. Seeding it makes the FIXTURE match the
349
+ * TASK; it does not touch the lesson (nothing here mentions `-q`) and no arm ever reads it, because
350
+ * the recorder blocks every command the agent produces before it can run.
351
+ */
352
+ export const PROJECT_B_MEMORY_KEY = 'note-caching-strategy';
353
+ export const PROJECT_B_MEMORY_VALUE =
354
+ 'The caching strategy for this project: responses are memoized in a two-tier LRU, '
355
+ + 'warm tier in memory and cold tier on disk, invalidated by content hash.';
356
+
357
+ /**
358
+ * What "retrieved" looks like on the wire. Every one of these strings was READ OFF the real global
359
+ * binary's real output on 2026-07-28 (see the EXECUTION GATE note in the header), never guessed.
360
+ *
361
+ * The positive markers are chosen to survive the table truncation `ruflo memory search` applies:
362
+ * the real row prints as `| note-caching-stra... | 0.79 | default | The caching strategy for this
363
+ * pr... |`, so a 12-char key prefix and a 20-char value prefix are both intact. `memory retrieve -k`
364
+ * prints both in full.
365
+ */
366
+ export const RETRIEVAL_EVIDENCE = Object.freeze({
367
+ positive: Object.freeze([PROJECT_B_MEMORY_KEY.slice(0, 12), PROJECT_B_MEMORY_VALUE.slice(0, 20)]),
368
+ negative: Object.freeze([
369
+ /No results found/i, // memory search -q "<absent>" → EXIT 0, and retrieved nothing
370
+ /Unknown command/i, // ruflo recall -q "x" → EXIT 1
371
+ /Required option missing/i, // memory search "positional" → EXIT 1
372
+ /Usage:\s*claude-flow memory/i, // memory recall -q "x" → EXIT 0, prints the help
373
+ /\[ERROR\]/,
374
+ ]),
375
+ });
376
+
377
+ /**
378
+ * Did the command RETRIEVE, as opposed to merely exit 0? Asserted on returned content in both
379
+ * directions: any known failure shape is disqualifying even at exit 0, and silence is not evidence —
380
+ * the output must NAME the seeded memory.
381
+ */
382
+ export function assertRetrieved(out) {
383
+ const s = String(out || '');
384
+ for (const re of RETRIEVAL_EVIDENCE.negative) {
385
+ if (re.test(s)) return { retrieved: false, why: `the command's own output matched a known FAILURE shape ${re}` };
386
+ }
387
+ const hit = RETRIEVAL_EVIDENCE.positive.find((p) => s.includes(p));
388
+ if (!hit) return { retrieved: false, why: 'the output names neither the seeded memory key nor its stored text — nothing was retrieved' };
389
+ return { retrieved: true, why: `the output carries the seeded memory (matched "${hit}")` };
390
+ }
391
+
392
+ /**
393
+ * Subcommands that WRITE. Checked only at subcommand position (the first two non-flag words), so a
394
+ * query that happens to contain one of these words is not mistaken for the verb.
395
+ */
396
+ const MUTATING_SUBCOMMANDS = new Set(['store', 'delete', 'rm', 'purge', 'cleanup', 'compress', 'import', 'export', 'backup', 'init', 'configure']);
397
+
398
+ /** Execute the real CLI, or an injected JavaScript fixture, without a shell on every platform. */
399
+ function spawnRuflo(bin, args, options) {
400
+ if (/\.[cm]?js$/i.test(bin)) {
401
+ return spawnSync(process.execPath, [bin, ...args], options);
402
+ }
403
+ return spawnSync(bin, args, options);
404
+ }
405
+
406
+ export function assertPostTaskPersisted({ args, output, cwd }) {
407
+ const task = optionValue(args, '-t', '--task');
408
+ const agent = optionValue(args, '-a', '--agent');
409
+ const taskId = optionValue(args, '-i', '--task-id')
410
+ || String(output || '').match(/Recording outcome for task:\s*([a-zA-Z0-9_-]+)/)?.[1]
411
+ || null;
412
+ if (!task || !agent || !taskId || !args.includes('--store-results')) {
413
+ return { retrieved: false, why: 'the command did not carry --task, --agent, --store-results, and a resolvable task id' };
414
+ }
415
+ let outcomes;
416
+ let memory;
417
+ try {
418
+ outcomes = JSON.parse(fs.readFileSync(path.join(cwd, '.claude-flow', 'routing-outcomes.json'), 'utf8'));
419
+ memory = JSON.parse(fs.readFileSync(path.join(cwd, '.claude-flow', 'memory', 'store.json'), 'utf8'));
420
+ } catch (error) {
421
+ return { retrieved: false, why: `the expected persistence stores were not readable: ${error.message}` };
422
+ }
423
+ const outcome = (outcomes.outcomes || []).find((row) =>
424
+ row.task === task && row.agent === agent && row.success === true);
425
+ const decision = memory.entries?.[`routing-decision:${taskId}`];
426
+ let decisionValue = null;
427
+ try { decisionValue = decision ? JSON.parse(decision.value) : null; } catch { /* invalid evidence */ }
428
+ if (!outcome || !decision || decisionValue?.task !== task || decisionValue?.agent !== agent) {
429
+ return { retrieved: false, why: 'stdout said success, but no matching routing outcome plus routing-decision memory row persisted' };
430
+ }
431
+ if (!/\[OK\]\s*Task outcome recorded:\s*SUCCESS/i.test(String(output || ''))) {
432
+ return { retrieved: false, why: 'persistence rows exist but this invocation did not report successful task recording' };
433
+ }
434
+ return {
435
+ retrieved: true,
436
+ why: `matching routing outcome and routing-decision:${taskId} memory row persisted`,
437
+ };
438
+ }
439
+
440
+ /**
441
+ * RUN the produced command and report what actually happened.
442
+ *
443
+ * Never through a shell: the argv is the one `findInvocations()` already parsed out of the agent's
444
+ * string, so shell metacharacters cannot survive into execution. Two refusals bound the blast
445
+ * radius, and both report NOT-RETRIEVED, so neither can ever raise the rate.
446
+ */
447
+ export function executeProducedCommand(cmd, {
448
+ cwd,
449
+ ruflo = RUFLO_BIN,
450
+ base = null,
451
+ trap = TRAP.MEMORY_SEARCH,
452
+ } = {}) {
453
+ const nope = (why, extra = {}) => ({ ran: false, argv: null, exit: null, exitOk: false, retrieved: false, why, output: '', ...extra });
454
+ const invocations = findInvocations(String(cmd || ''), ['ruflo', 'claude-flow']);
455
+ if (!invocations.length) return nope('no ruflo invocation in the produced command — there was nothing to execute');
456
+ const args = invocations[0].args.filter((a) => a !== '');
457
+
458
+ for (let i = 0; i < args.length; i++) {
459
+ const a = args[i];
460
+ if (a === '--path' || a === '--db' || a.startsWith('--path=') || a.startsWith('--db=')) {
461
+ const raw = a.includes('=') ? a.slice(a.indexOf('=') + 1) : args[i + 1];
462
+ const abs = path.resolve(cwd, String(raw || ''));
463
+ if (!base || !(abs === path.resolve(base) || abs.startsWith(path.resolve(base) + path.sep))) {
464
+ return nope(`refused to execute: the produced command points its store at ${abs}, outside the fixture world`, { argv: ['ruflo', ...args] });
465
+ }
466
+ }
467
+ }
468
+ const verbs = args.filter((a) => !a.startsWith('-')).slice(0, 2);
469
+ const mutating = verbs.find((w) => MUTATING_SUBCOMMANDS.has(w));
470
+ if (mutating) return nope(`refused to execute a MUTATING ruflo subcommand ("${mutating}") — a retrieval claim is not proven by a write`, { argv: ['ruflo', ...args] });
471
+
472
+ const env = { ...process.env };
473
+ delete env.CLAUDE_FLOW_DB_PATH;
474
+ delete env.CLAUDE_FLOW_MEMORY_PATH;
475
+ const r = spawnRuflo(ruflo, args, { cwd, encoding: 'utf8', timeout: 120_000, env, maxBuffer: 8 * 1024 * 1024 });
476
+ if (r.error) return nope(`spawn failed: ${r.error.message}`, { argv: ['ruflo', ...args] });
477
+ const out = `${r.stdout || ''}${r.stderr || ''}`;
478
+ const routed = trap === TRAP.POST_TASK
479
+ ? assertPostTaskPersisted({ args, output: out, cwd })
480
+ : assertRetrieved(out);
481
+ return {
482
+ ran: true,
483
+ argv: ['ruflo', ...args],
484
+ exit: r.status,
485
+ exitOk: r.status === 0,
486
+ retrieved: routed.retrieved,
487
+ why: `exit ${r.status}; ${routed.why}`,
488
+ output: out.slice(0, 1200),
489
+ };
490
+ }
491
+
492
+ /**
493
+ * ONE run's verdict. Order of the branches IS the invariant: the control is judged BEFORE the
494
+ * treated arm can be credited with anything.
495
+ */
496
+ export function verdictForRun(run) {
497
+ const {
498
+ treatedClass, controlClass, lessonBeforeFirstToolCall, error,
499
+ treatedSubcommandCorrect, treatedExecOk, treatedRetrieved, treatedExecWhy,
500
+ controlWorked,
501
+ } = run;
502
+ if (error) return { verdict: VERDICT.UNKNOWN, why: `harness could not measure this run: ${error}` };
503
+ // NO COMPARABLE CONTROL ARTIFACT IS NOT A WIN. If the control never invoked ruflo at all, there is
504
+ // no counterfactual to difference against — the treated arm may have "changed" against nothing.
505
+ // Deliberately strict, and it can only ever LOWER the rate: an unopposed treated arm is UNKNOWN.
506
+ if (controlClass === 'none') {
507
+ return { verdict: VERDICT.UNKNOWN, why: 'the control arm produced no ruflo invocation at all — there is no comparable artifact to difference against' };
508
+ }
509
+ if (treatedClass === 'none') {
510
+ return { verdict: VERDICT.FAIL, why: 'the treated arm produced no ruflo invocation at all' };
511
+ }
512
+ // INVARIANT 6, FIRST AND UNCONDITIONALLY. WIDENED 2026-07-28, never narrowed: carrying the token
513
+ // still invalidates on its own (that bar is unchanged, so nothing the control reaches unaided can
514
+ // start being credited to the lesson), and a control whose command WORKED — executed and retrieved
515
+ // — invalidates too, even by a route the classifier does not call `flagged`. An OR can only make
516
+ // more runs invalid; it can never turn an invalid run into a pass.
517
+ if (carriesToken(controlClass) || controlWorked === true) {
518
+ return {
519
+ verdict: VERDICT.INCONCLUSIVE,
520
+ why: carriesToken(controlClass)
521
+ ? `the CONTROL arm produced the token (${controlClass}) — the model would have got it right without the lesson, so this trap measured nothing`
522
+ : `the CONTROL arm's command EXECUTED AND RETRIEVED (class "${controlClass}") — the model would have got it right without the lesson, so this trap measured nothing`,
523
+ };
524
+ }
525
+ if (!carriesToken(treatedClass)) {
526
+ return { verdict: VERDICT.FAIL, why: `treated arm produced "${treatedClass}", not the token` };
527
+ }
528
+ // ── THE EXECUTION GATE (2026-07-28) ──
529
+ // Ordered cheapest-to-most-informative so the `why` names the FIRST thing that was wrong.
530
+ if (treatedSubcommandCorrect !== true) {
531
+ return { verdict: VERDICT.FAIL, why: 'treated arm carried the token on the WRONG SUBCOMMAND — a right flag on a command that is not `ruflo memory search` is not learned behavior, it is an unusable command' };
532
+ }
533
+ if (treatedExecOk !== true) {
534
+ return { verdict: VERDICT.FAIL, why: `treated arm's produced command did not execute successfully: ${treatedExecWhy || 'not executed'}` };
535
+ }
536
+ if (treatedRetrieved !== true) {
537
+ return { verdict: VERDICT.FAIL, why: `treated arm's command exited 0 but RETRIEVED NOTHING: ${treatedExecWhy || 'no retrieval evidence'}` };
538
+ }
539
+ if (lessonBeforeFirstToolCall !== true) {
540
+ return { verdict: VERDICT.FAIL, why: 'treated arm carried the token but the lesson was NOT observed in the transcript before the first tool call' };
541
+ }
542
+ return { verdict: VERDICT.PASS, why: `treated "${treatedClass}" vs control "${controlClass}"; the produced command executed (exit 0) and returned the required meaningful outcome; lesson delivered before the first tool call` };
543
+ }
544
+
545
+ /**
546
+ * The RATE. N runs in, one verdict + a k/n out.
547
+ *
548
+ * PASS is structurally unreachable while any control succeeded: `controlTokenRuns` is computed
549
+ * before the branch and the assertion at the bottom re-checks it. Removing either guard and running
550
+ * the `seed-control` mutant is how you prove this is real rather than decorative.
551
+ */
552
+ export function aggregate(runs, { threshold = 2 / 3 } = {}) {
553
+ const perRun = runs.map((r) => ({ ...r, ...verdictForRun(r) }));
554
+ const n = perRun.length;
555
+ const passes = perRun.filter((r) => r.verdict === VERDICT.PASS).length;
556
+ const fails = perRun.filter((r) => r.verdict === VERDICT.FAIL).length;
557
+ const unknowns = perRun.filter((r) => r.verdict === VERDICT.UNKNOWN).length;
558
+ const controlTokenRuns = perRun.filter((r) => carriesToken(r.controlClass)).length;
559
+ const controlWorkedRuns = perRun.filter((r) => r.controlWorked === true).length;
560
+ // The EFFECT SIZE, reported even when the verdict is INCONCLUSIVE. An invalid trap still measured
561
+ // two real rates, and printing only `passes` throws away the more informative half: "treated 3/3,
562
+ // control 1/3" says something a bare "2/3 below the bar" does not. This is a report, never an
563
+ // input to the verdict — the verdict stays governed by invariant 6 above.
564
+ const treatedTokenRuns = perRun.filter((r) => carriesToken(r.treatedClass)).length;
565
+ // The execution gate's own rates, reported whatever the verdict. "treated 3/3 carried the token,
566
+ // 0/3 of them worked" is the sentence the old artifact could not say, and it is the sentence the
567
+ // grader had to reconstruct by hand from `subcommandCorrect: false`.
568
+ const treatedSubcommandRuns = perRun.filter((r) => r.treatedSubcommandCorrect === true).length;
569
+ const treatedExecutedRuns = perRun.filter((r) => r.treatedExecOk === true).length;
570
+ const treatedRetrievedRuns = perRun.filter((r) => r.treatedRetrieved === true).length;
571
+
572
+ let verdict, why;
573
+ if (n === 0) {
574
+ verdict = VERDICT.UNKNOWN; why = 'zero runs executed — an empty run is not a pass';
575
+ } else if (controlTokenRuns > 0 || controlWorkedRuns > 0) {
576
+ verdict = VERDICT.INCONCLUSIVE;
577
+ why = `${controlTokenRuns}/${n} CONTROL run(s) produced the token and ${controlWorkedRuns}/${n} executed+retrieved — DDD-0013 invariant 6: the trap is INVALID, not passed`;
578
+ } else if (passes / n >= threshold) {
579
+ verdict = VERDICT.PASS;
580
+ why = `${passes}/${n} runs passed (bar ${Math.ceil(threshold * n)}/${n})`;
581
+ } else if (unknowns > 0 && passes + fails < n) {
582
+ verdict = VERDICT.UNKNOWN;
583
+ const executorError = perRun.map((r) => r.error).find(Boolean);
584
+ why = executorError
585
+ ? `${unknowns}/${n} run(s) could not be measured; executor error: ${executorError}`
586
+ : `${unknowns}/${n} run(s) could not be measured; ${passes}/${n} passed — below the bar with the reason unknown`;
587
+ } else {
588
+ verdict = VERDICT.FAIL;
589
+ why = `${passes}/${n} runs passed — below the ${Math.ceil(threshold * n)}/${n} bar`;
590
+ }
591
+
592
+ // The guards that make the two invariants CODE rather than prose. If either throws, the branch
593
+ // order above was edited and the trap is unsafe — a stop-the-line event, not a warning.
594
+ if (verdict === VERDICT.PASS && (controlTokenRuns > 0 || controlWorkedRuns > 0)) {
595
+ throw new Error('LEARNING-REPLAY: refusing to report PASS while a control arm produced the token or a working command (DDD-0013 invariant 6)');
596
+ }
597
+ // The grader's finding, made structurally impossible: `subcommandCorrect: false` can no longer sit
598
+ // inside a PASS. Same for a command that did not execute or retrieved nothing.
599
+ const brokenPass = perRun.find((r) => r.verdict === VERDICT.PASS
600
+ && (r.treatedSubcommandCorrect !== true || r.treatedExecOk !== true || r.treatedRetrieved !== true));
601
+ if (brokenPass) {
602
+ throw new Error(`LEARNING-REPLAY: refusing to report a PASS run whose produced command was unusable (run ${brokenPass.i}: subcommandCorrect=${brokenPass.treatedSubcommandCorrect}, exitOk=${brokenPass.treatedExecOk}, retrieved=${brokenPass.treatedRetrieved})`);
603
+ }
604
+ return {
605
+ verdict, why, n, passes, fails, unknowns,
606
+ controlTokenRuns, controlWorkedRuns, treatedTokenRuns,
607
+ treatedSubcommandRuns, treatedExecutedRuns, treatedRetrievedRuns,
608
+ rate: n ? +(passes / n).toFixed(4) : 0, runs: perRun,
609
+ };
610
+ }
611
+
612
+ // ── the real CLI's real interface, re-verified at run time ──────────────────────────────────────
613
+ export const RUFLO_BIN = process.env.RUVNET_RUFLO_BIN || path.join(os.homedir(), '.npm-global', 'bin', 'ruflo');
614
+
615
+ /**
616
+ * Re-verify the premise. Rule 0 applied to the one fact the whole oracle rests on: `ruflo memory
617
+ * search` must still take `-q/--query` and must still mark it REQUIRED. If rUv changes the
618
+ * interface, the honest outcome is UNKNOWN and a loud line — never a silent pass against a lesson
619
+ * that is no longer true.
620
+ */
621
+ export function verifyRufloFlag(bin = RUFLO_BIN) {
622
+ if (!fs.existsSync(bin)) return { ok: false, why: `ruflo binary not found at ${bin} (Rule 21: the GLOBAL binary, never npx)` };
623
+ const r = spawnSync(bin, ['memory', 'search', '--help'], { encoding: 'utf8', timeout: 30_000 });
624
+ const out = `${r.stdout || ''}${r.stderr || ''}`;
625
+ if (r.status !== 0 && !out) return { ok: false, why: `ruflo memory search --help exited ${r.status} with no output` };
626
+ const flag = /-q,\s*--query/.test(out);
627
+ const required = /--query[^\n]*required/i.test(out);
628
+ const positionalDocumented = /\bmemory search\s+"[^"]+"\s*$/m.test(out);
629
+ if (!flag) return { ok: false, why: 'live `ruflo memory search --help` no longer advertises `-q, --query` — the lesson this trap records is no longer true', help: out };
630
+ if (positionalDocumented) return { ok: false, why: 'live help now shows a POSITIONAL query example — the trap premise (positional is rejected) is broken', help: out };
631
+ return { ok: true, flag: '-q, --query', required, evidence: out.split('\n').find((l) => /-q,\s*--query/.test(l))?.trim() || '' };
632
+ }
633
+
634
+ export function verifyPostTaskContract(bin = RUFLO_BIN) {
635
+ if (!fs.existsSync(bin)) return { ok: false, why: `ruflo binary not found at ${bin} (Rule 21: the GLOBAL binary, never npx)` };
636
+ const help = spawnRuflo(bin, ['hooks', 'post-task', '--help'], { encoding: 'utf8', timeout: 30_000 });
637
+ const out = `${help.stdout || ''}${help.stderr || ''}`;
638
+ if (!/--task\b/.test(out) || !/--agent\b/.test(out) || !/--store-results\b/.test(out)
639
+ || !/Without this \+ --agent, no routing outcome is recorded/.test(out)) {
640
+ return { ok: false, why: 'live post-task help no longer states the three-part routing-persistence contract', help: out };
641
+ }
642
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'd4-post-task-premise-'));
643
+ const missing = spawnRuflo(bin, ['hooks', 'post-task', '-i', 'd4-premise-missing', '--success', 'true'], {
644
+ cwd: dir,
645
+ encoding: 'utf8',
646
+ timeout: 30_000,
647
+ });
648
+ const outcomeFile = path.join(dir, '.claude-flow', 'routing-outcomes.json');
649
+ const memoryFile = path.join(dir, '.claude-flow', 'memory', 'store.json');
650
+ const persisted = fs.existsSync(outcomeFile) || fs.existsSync(memoryFile);
651
+ fs.rmSync(dir, { recursive: true, force: true });
652
+ if (missing.status !== 0 || persisted) {
653
+ return {
654
+ ok: false,
655
+ why: 'live post-task missing-contract probe did not stay non-persistent while returning success',
656
+ missingExit: missing.status,
657
+ persisted,
658
+ };
659
+ }
660
+ return {
661
+ ok: true,
662
+ flag: '--task + --agent + --store-results',
663
+ required: true,
664
+ evidence: 'live help names all three flags; success/task-id-only probe exited 0 and created neither routing outcome nor routing-decision store',
665
+ missingExit: missing.status,
666
+ };
667
+ }
668
+
669
+ // ── the fixture world ───────────────────────────────────────────────────────────────────────────
670
+ const CLAUDE_BIN = process.env.RUVNET_CLAUDE_BIN || path.join(os.homedir(), '.npm-global', 'bin', 'claude');
671
+ const CODEX_BIN = process.env.RUVNET_CODEX_BIN || 'codex';
672
+
673
+ /** Project B's task. Shares no content word with the lesson — the lesson cannot be string-matched into it. */
674
+ export const REPLAY_PROMPT =
675
+ 'Earlier in this project someone recorded a note about the caching strategy. '
676
+ + "Recall it from this project's agent memory with the ruflo CLI. "
677
+ + 'Run the recall command now, then tell me what you ran.';
678
+
679
+ /** The correction as it is written down in fixture-project-A, in project A's own words. */
680
+ export const LESSON_STATEMENT =
681
+ 'When you look something up in agent memory with the ruflo CLI, the query has to be passed with the '
682
+ + '-q flag; a bare quoted phrase placed after the subcommand is rejected.';
683
+
684
+ const sh = (cmd, args, opts = {}) => spawnSync(cmd, args, { encoding: 'utf8', timeout: 120_000, ...opts });
685
+
686
+ function rmrf(p) { try { fs.rmSync(p, { recursive: true, force: true }); } catch { /* nothing to remove */ } }
687
+
688
+ /** Allocate one collision-proof fixture root; Date.now() alone collides under parallel CI. */
689
+ export function allocateRunBase(root = path.join(ROOT, '.ruvnet-brain', 'learning-replay')) {
690
+ fs.mkdirSync(root, { recursive: true });
691
+ return fs.mkdtempSync(path.join(root, 'run-'));
692
+ }
693
+
694
+ function initMemoryDb(ruflo, db, cwd) {
695
+ return sh(ruflo, ['memory', 'init', '--path', db, '--backend', 'hybrid'], { cwd });
696
+ }
697
+
698
+ /**
699
+ * Build the two fixture projects and the isolated brain world.
700
+ *
701
+ * Everything the product reads is redirected by env — RUVNET_BRAIN_HOME (spine), RUVNET_BRAIN_STATE_DIR
702
+ * (the on/off sentinel), RUVNET_LESSON_STORE, RUVNET_LESSON_GATE_STATE. Nothing here touches the
703
+ * user's real ~/.config/ruvnet-brain, ~/.cache/ruvnet-brain, or any real project's memory.
704
+ */
705
+ export function buildFixtures(baseDir) {
706
+ rmrf(baseDir);
707
+ const dirs = {
708
+ base: baseDir,
709
+ projectA: path.join(baseDir, 'fixture-project-a'),
710
+ projectA2: path.join(baseDir, 'fixture-project-a-independent'),
711
+ projectB: path.join(baseDir, 'fixture-project-b'),
712
+ brainHome: path.join(baseDir, 'brain-home'),
713
+ stateOn: path.join(baseDir, 'state-on'),
714
+ stateOff: path.join(baseDir, 'state-off'),
715
+ transcripts: path.join(baseDir, 'transcripts'),
716
+ };
717
+ for (const d of Object.values(dirs)) fs.mkdirSync(d, { recursive: true });
718
+ dirs.lessons = path.join(baseDir, 'lessons.json');
719
+ dirs.gateState = path.join(baseDir, 'lesson-gate-state.json');
720
+ // The control's switch: ADR-054's real sentinel, in the control's own state dir.
721
+ fs.writeFileSync(path.join(dirs.stateOff, 'brain-off'), JSON.stringify({ since: new Date().toISOString() }));
722
+ // Each fixture project is its own git repo so lesson-gate's project-scope resolution (which walks
723
+ // up to the nearest .git) sees `fixture-project-b`, not the harness's own repo.
724
+ for (const p of [dirs.projectA, dirs.projectA2, dirs.projectB]) {
725
+ sh('git', ['init', '-q'], { cwd: p });
726
+ fs.mkdirSync(path.join(p, '.swarm'), { recursive: true });
727
+ }
728
+ return dirs;
729
+ }
730
+
731
+ /**
732
+ * RECORD, in fixture-project-A.
733
+ *
734
+ * Two layers, both real:
735
+ * 1. the correction is written into project A's OWN memory with `ruflo memory store` — the real
736
+ * CLI, the real per-project `.swarm/memory.db` the global memory policy mandates. It is then
737
+ * READ BACK with `ruflo memory search -q` (the very flag under test, so the record step itself
738
+ * exercises the true interface), and the retrieved text is what the lesson is built from. The
739
+ * lesson is DERIVED from project A, not hardcoded beside it.
740
+ * 2. the derived lesson is written into the machine-global lesson store the gate actually reads.
741
+ *
742
+ * SCOPE is the real ADR-029 rule, not a fixture bypass. The same correction is independently stored
743
+ * and read back in TWO distinct git projects. The executable lesson carries both project names, so
744
+ * lesson-gate.mjs's `projects.length >= 2` universal predicate is what permits it to speak in the
745
+ * third replay project. One source would correctly be silent there. The committed artifact records
746
+ * the two source identities and `checkPortfolio()` refuses a result without that win-twice proof.
747
+ */
748
+ export function recordInProjectA(dirs, { ruflo = RUFLO_BIN, trap = TRAP.MEMORY_SEARCH } = {}) {
749
+ const spec = trapSpec(trap);
750
+ const sources = [dirs.projectA, dirs.projectA2].map((project, index) => {
751
+ const db = path.join(project, '.swarm', 'memory.db');
752
+ const key = `${spec.memoryKey}-${index + 1}`;
753
+ const init = initMemoryDb(ruflo, db, project);
754
+ const store = sh(ruflo, ['memory', 'store', '-k', key, '--value', spec.lesson, '-n', 'default', '--path', db],
755
+ { cwd: project });
756
+ const back = sh(ruflo, ['memory', 'search', '-q', spec.recordQuery, '-n', 'default', '--path', db, '-t', 'keyword'],
757
+ { cwd: project });
758
+ return {
759
+ project: path.basename(project),
760
+ db,
761
+ key,
762
+ initExit: init.status,
763
+ storeExit: store.status,
764
+ readBackExit: back.status,
765
+ recorded: fs.existsSync(db),
766
+ };
767
+ });
768
+
769
+ const lesson = makeLesson({
770
+ id: spec.lessonId,
771
+ statement: spec.lesson,
772
+ // `assert-fact` is the decision point the real dispatcher requests at UserPromptSubmit
773
+ // (plugin/scripts/lesson-hooks.sh) — i.e. before any tool call, which is PASS-condition (a).
774
+ trigger: 'assert-fact',
775
+ enforcement: 'checklist',
776
+ origin: 'user-stated',
777
+ status: 'ratified',
778
+ severity: 'high',
779
+ repeatCount: 4,
780
+ projects: sources.map((source) => source.project),
781
+ check: spec.check,
782
+ evidence: [
783
+ ...sources.map((source) => ({
784
+ observed: `independently recorded in ${source.project} as memory key "${source.key}" in ${path.relative(dirs.base, source.db)}`,
785
+ })),
786
+ { observed: `live premise for ${spec.id} was re-verified before replay` },
787
+ ],
788
+ });
789
+ saveLessons([lesson], dirs.lessons);
790
+ const sourcesOk = sources.every((source) =>
791
+ source.initExit === 0 && source.storeExit === 0 && source.readBackExit === 0 && source.recorded);
792
+ return {
793
+ ok: sourcesOk && loadLessons(dirs.lessons).length === 1 && lesson.projects.length >= 2,
794
+ trap: spec.id,
795
+ sources,
796
+ projectCount: lesson.projects.length,
797
+ promoted: lesson.projects.length >= 2,
798
+ key: sources[0].key,
799
+ storeExit: sources[0].storeExit,
800
+ readBackExit: sources[0].readBackExit,
801
+ lesson,
802
+ };
803
+ }
804
+
805
+ /**
806
+ * SEED PROJECT B — make the fixture world true.
807
+ *
808
+ * REPLAY_PROMPT tells the agent "earlier in this project someone recorded a note about the caching
809
+ * strategy". Until 2026-07-28 that was false: project B's store was empty, so no command the agent
810
+ * could possibly write would retrieve anything, and the execution gate would be measuring the
811
+ * harness. The note is written with the real CLI into project B's own `.swarm/memory.db` — the
812
+ * default path a bare `ruflo memory search` resolves from cwd.
813
+ *
814
+ * It cannot leak the lesson: the note's text says nothing about `-q`, and neither arm ever sees it
815
+ * (the recorder blocks every produced command before it runs). It is read only by the harness,
816
+ * out of band, after the arm is finished.
817
+ */
818
+ export function seedProjectBMemory(dirs, { ruflo = RUFLO_BIN } = {}) {
819
+ const dbB = path.join(dirs.projectB, '.swarm', 'memory.db');
820
+ const init = initMemoryDb(ruflo, dbB, dirs.projectB);
821
+ const r = sh(ruflo, ['memory', 'store', '-k', PROJECT_B_MEMORY_KEY, '--value', PROJECT_B_MEMORY_VALUE, '-n', 'default', '--path', dbB],
822
+ { cwd: dirs.projectB });
823
+ return {
824
+ db: dbB,
825
+ key: PROJECT_B_MEMORY_KEY,
826
+ initExit: init.status,
827
+ storeExit: r.status,
828
+ ok: init.status === 0 && r.status === 0 && fs.existsSync(dbB),
829
+ };
830
+ }
831
+
832
+ /**
833
+ * THE NIGHTLY REFRESH, run BETWEEN record and replay. PASS-condition (c).
834
+ *
835
+ * Two real things, not a sleep:
836
+ * 1. a NEW Stable-Spine generation is installed into the fixture brain home and active.json is
837
+ * flipped to it — so the replay's hooks execute from a code root that did not exist when the
838
+ * lesson was recorded. This is exactly what scripts/update-apply.mjs does nightly, and it is the
839
+ * thing a lesson has to survive: the lesson store lives at user level, deliberately OUTSIDE the
840
+ * bundle a refresh replaces (scripts/lesson-store.mjs says so in its own persistence note).
841
+ * 2. `ruflo memory distill run` and `ruflo memory backup` against project A's store — the two
842
+ * commands scripts/nightly-wrapper.sh actually runs every night.
843
+ */
844
+ export function nightlyRefresh(dirs, { ruflo = RUFLO_BIN } = {}) {
845
+ const gen = `d4-refresh-${Date.now()}`;
846
+ const versionDir = path.join(dirs.brainHome, 'versions', gen);
847
+ fs.mkdirSync(versionDir, { recursive: true });
848
+ fs.cpSync(path.join(ROOT, 'plugin'), versionDir, { recursive: true });
849
+ // Codex discovers the installed plugin's global hook manifest, not fixture-local `.codex` hooks.
850
+ // The stable wrapper resolves this fixture generation through RUVNET_BRAIN_HOME, so replace only
851
+ // the fixture generation's host adapter with the replay tap. The real hook body remains the copied
852
+ // hook-shim beside it; the adapter merely records delivery and blocks the first proposed command.
853
+ fs.copyFileSync(
854
+ path.join(ROOT, 'scripts', 'ci', 'learning-replay-codex-adapter.mjs'),
855
+ path.join(versionDir, 'scripts', 'codex-hook-adapter.mjs'),
856
+ );
857
+ fs.writeFileSync(path.join(dirs.brainHome, 'active.json'), JSON.stringify({ codeRoot: versionDir, generation: gen }, null, 2));
858
+ fs.writeFileSync(path.join(dirs.brainHome, '.spine-seeded'), gen);
859
+
860
+ const dbA = path.join(dirs.projectA, '.swarm', 'memory.db');
861
+ const distill = sh(ruflo, ['memory', 'distill', 'run', '--path', dbA], { cwd: dirs.projectA });
862
+ const backup = sh(ruflo, ['memory', 'backup', '--db', dbA, '--keep', '2'], { cwd: dirs.projectA });
863
+
864
+ const survived = loadLessons(dirs.lessons).length === 1;
865
+ // Repo-relative, never absolute: this artifact is COMMITTED, and an absolute path publishes the
866
+ // maintainer's directory layout to every reader. The same disclosure was already found and fixed
867
+ // once in session-start.sh; one bug, found once, must not be left everywhere else.
868
+ return { generation: gen, codeRoot: path.relative(ROOT, versionDir), distillExit: distill.status, backupExit: backup.status, lessonSurvived: survived };
869
+ }
870
+
871
+ /** The fixture settings file — the REAL hook registration from plugin/hooks/hooks.json, plus the tap. */
872
+ function writeSettings(file, { dirs, stateDir, attemptsFile }) {
873
+ const settings = {
874
+ env: {
875
+ RUVNET_BRAIN_HOME: dirs.brainHome,
876
+ RUVNET_BRAIN_STATE_DIR: stateDir,
877
+ RUVNET_LESSON_STORE: dirs.lessons,
878
+ RUVNET_LESSON_GATE_STATE: dirs.gateState,
879
+ CLAUDE_PLUGIN_ROOT: path.join(ROOT, 'plugin'),
880
+ },
881
+ hooks: {
882
+ UserPromptSubmit: [{
883
+ matcher: '*',
884
+ hooks: [{
885
+ type: 'command',
886
+ command: `node ${JSON.stringify(path.join(ROOT, 'plugin', 'scripts', 'hook-shim.mjs'))} unprompted-speech UserPromptSubmit`,
887
+ timeout: 20,
888
+ }],
889
+ }],
890
+ PreToolUse: [{
891
+ matcher: 'Bash',
892
+ hooks: [{
893
+ type: 'command',
894
+ command: `node ${JSON.stringify(path.join(ROOT, 'scripts', 'ci', 'learning-replay-recorder.mjs'))} ${JSON.stringify(attemptsFile)}`,
895
+ timeout: 20,
896
+ }],
897
+ }],
898
+ },
899
+ };
900
+ fs.writeFileSync(file, JSON.stringify(settings, null, 2));
901
+ return file;
902
+ }
903
+
904
+ export function buildCodexArgv({ model = 'gpt-5.6-sol', prompt = REPLAY_PROMPT, appendSystemPrompt = null } = {}) {
905
+ const fullPrompt = appendSystemPrompt ? `${appendSystemPrompt}\n\n${prompt}` : prompt;
906
+ return [
907
+ 'exec',
908
+ '--ephemeral',
909
+ '--sandbox', 'read-only',
910
+ '--color', 'never',
911
+ '--json',
912
+ '--ignore-rules',
913
+ '--dangerously-bypass-hook-trust',
914
+ '-m', model,
915
+ '-c', 'model_reasoning_effort="low"',
916
+ '-c', 'shell_environment_policy.inherit="all"',
917
+ fullPrompt,
918
+ ];
919
+ }
920
+
921
+ /**
922
+ * Run ONE arm and return what it produced.
923
+ *
924
+ * The transcript is stream-json with --include-hook-events, so hook delivery and tool calls appear
925
+ * IN ORDER in one array. `lessonIndex` and `firstToolIndex` are positions in that array — condition
926
+ * (a) is a measured ordering, not an argument from how hooks are supposed to work.
927
+ */
928
+ export function replayRunError(events, processResult) {
929
+ if (processResult?.error) return String(processResult.error.message || processResult.error);
930
+ const result = events.find((e) => e.type === 'result');
931
+ if (!result?.is_error) return null;
932
+ const status = result.api_error_status ? `HTTP ${result.api_error_status}: ` : '';
933
+ return `${status}${result.result || result.terminal_reason || 'model execution failed'}`;
934
+ }
935
+
936
+ export function parseCodexRunError(events, processResult) {
937
+ if (processResult?.error) return String(processResult.error.message || processResult.error);
938
+ const failed = events.find((event) => event.type === 'turn.failed');
939
+ if (failed) return String(failed.error?.message || failed.error || 'Codex turn failed');
940
+ if (events.some((event) => event.type === 'turn.completed') && processResult?.status === 0) return null;
941
+ const errorItem = events.find((event) => event.type === 'item.completed' && event.item?.type === 'error');
942
+ if (errorItem) return String(errorItem.item?.message || errorItem.item?.text || 'Codex execution failed');
943
+ return processResult?.status && processResult.status !== 0
944
+ ? `Codex exited ${processResult.status}`
945
+ : null;
946
+ }
947
+
948
+ export function codexLessonBeforeTool(sequence) {
949
+ const lesson = sequence.find((event) => event.kind === 'lesson');
950
+ const tool = sequence.find((event) => event.kind === 'tool');
951
+ return Boolean(lesson && tool && BigInt(lesson.atNs) < BigInt(tool.atNs));
952
+ }
953
+
954
+ export function runArm({
955
+ dirs,
956
+ arm,
957
+ stateDir,
958
+ model,
959
+ host = 'claude-code',
960
+ appendSystemPrompt = null,
961
+ tag,
962
+ forceCommand = null,
963
+ trap = TRAP.MEMORY_SEARCH,
964
+ }) {
965
+ const spec = trapSpec(trap);
966
+ const attempts = path.join(dirs.transcripts, `${tag}.attempts.jsonl`);
967
+ const sequenceFile = path.join(dirs.transcripts, `${tag}.sequence.jsonl`);
968
+ const streamFile = path.join(dirs.transcripts, `${tag}.stream.jsonl`);
969
+ let binary;
970
+ let argv;
971
+ if (host === 'codex') {
972
+ binary = CODEX_BIN;
973
+ argv = buildCodexArgv({ model, prompt: spec.prompt, appendSystemPrompt });
974
+ } else {
975
+ const settings = writeSettings(path.join(dirs.base, `settings-${tag}.json`), { dirs, stateDir, attemptsFile: attempts });
976
+ binary = CLAUDE_BIN;
977
+ argv = [
978
+ '-p', spec.prompt,
979
+ '--model', model,
980
+ '--tools', 'Bash',
981
+ '--permission-mode', 'bypassPermissions',
982
+ '--setting-sources', '',
983
+ '--settings', settings,
984
+ '--output-format', 'stream-json',
985
+ '--verbose',
986
+ '--include-hook-events',
987
+ '--no-session-persistence',
988
+ '--max-budget-usd', '0.30',
989
+ ];
990
+ if (appendSystemPrompt) argv.push('--append-system-prompt', appendSystemPrompt);
991
+ }
992
+
993
+ const started = Date.now();
994
+ const r = spawnSync(binary, argv, {
995
+ cwd: dirs.projectB,
996
+ encoding: 'utf8',
997
+ timeout: 300_000,
998
+ maxBuffer: 64 * 1024 * 1024,
999
+ env: {
1000
+ ...process.env,
1001
+ RUVNET_BRAIN_HOME: dirs.brainHome,
1002
+ RUVNET_BRAIN_STATE_DIR: stateDir,
1003
+ RUVNET_LESSON_STORE: dirs.lessons,
1004
+ RUVNET_LESSON_GATE_STATE: dirs.gateState,
1005
+ RUVNET_REPLAY_ATTEMPTS_FILE: attempts,
1006
+ RUVNET_REPLAY_SEQUENCE_FILE: sequenceFile,
1007
+ RUVNET_REPLAY_LESSON_PROBE: spec.lesson.slice(0, 60),
1008
+ RUVNET_REPLAY_RECORDER: path.join(ROOT, 'scripts', 'ci', 'learning-replay-recorder.mjs'),
1009
+ CLAUDE_PLUGIN_ROOT: path.join(ROOT, 'plugin'),
1010
+ },
1011
+ });
1012
+ const wallMs = Date.now() - started;
1013
+ fs.writeFileSync(streamFile, r.stdout || '');
1014
+ if (r.stderr) fs.writeFileSync(path.join(dirs.transcripts, `${tag}.stderr.txt`), r.stderr);
1015
+
1016
+ const events = (r.stdout || '').split('\n').filter(Boolean).map((l) => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
1017
+
1018
+ const probe = spec.lesson.slice(0, 60);
1019
+ let lessonIndex = -1, firstToolIndex = -1, lessonDelivered = false;
1020
+ const sequence = fs.existsSync(sequenceFile)
1021
+ ? fs.readFileSync(sequenceFile, 'utf8').split('\n').filter(Boolean).map((line) => JSON.parse(line))
1022
+ : [];
1023
+ if (host === 'codex') {
1024
+ lessonIndex = sequence.findIndex((event) => event.kind === 'lesson');
1025
+ firstToolIndex = sequence.findIndex((event) => event.kind === 'tool');
1026
+ lessonDelivered = lessonIndex !== -1;
1027
+ } else {
1028
+ events.forEach((e, i) => {
1029
+ if (lessonIndex === -1 && e.type === 'system' && e.subtype === 'hook_response'
1030
+ && typeof e.output === 'string' && e.output.includes(probe)) { lessonIndex = i; lessonDelivered = true; }
1031
+ if (firstToolIndex === -1 && e.type === 'assistant'
1032
+ && Array.isArray(e.message?.content) && e.message.content.some((c) => c.type === 'tool_use')) firstToolIndex = i;
1033
+ });
1034
+ }
1035
+
1036
+ const attemptLines = fs.existsSync(attempts)
1037
+ ? fs.readFileSync(attempts, 'utf8').split('\n').filter(Boolean).map((l) => JSON.parse(l))
1038
+ : [];
1039
+ // THE ARTIFACT is the FIRST command the agent produced — not its best one. A second attempt after
1040
+ // the sandbox refusal is a repair, and crediting a repair would let the agent learn the answer
1041
+ // from the harness instead of from the lesson.
1042
+ // MUTANT force-*: substitute the artifact the oracle sees, leaving the real run untouched. This is
1043
+ // how mutant 1 ("right flag, wrong subcommand") is proven end-to-end without waiting for a
1044
+ // stochastic model to happen to emit it.
1045
+ const firstCommand = forceCommand != null ? forceCommand : (attemptLines.length ? attemptLines[0].command : '');
1046
+ const cls = trap === TRAP.POST_TASK
1047
+ ? classifyPostTaskCommand(firstCommand)
1048
+ : classifyCommand(firstCommand);
1049
+ const subOk = trap === TRAP.POST_TASK
1050
+ ? postTaskSubcommandCorrect(firstCommand)
1051
+ : subcommandCorrect(firstCommand);
1052
+ // THE EXECUTION GATE. Out of band, after the arm is over, never through a shell — see the header.
1053
+ const exec = executeProducedCommand(firstCommand, { cwd: dirs.projectB, base: dirs.base, trap });
1054
+
1055
+ const result = events.find((e) => e.type === 'result');
1056
+ return {
1057
+ arm,
1058
+ tag,
1059
+ class: cls,
1060
+ subcommandCorrect: subOk,
1061
+ exec,
1062
+ command: firstCommand,
1063
+ forcedCommand: forceCommand != null,
1064
+ attempts: attemptLines.map((a) => a.command),
1065
+ lessonDelivered,
1066
+ lessonIndex,
1067
+ firstToolIndex,
1068
+ lessonBeforeFirstToolCall: host === 'codex'
1069
+ ? codexLessonBeforeTool(sequence)
1070
+ : lessonDelivered && firstToolIndex > -1 && lessonIndex < firstToolIndex,
1071
+ costUsd: result?.total_cost_usd ?? null,
1072
+ wallMs,
1073
+ modelUsed: events.find((e) => e.type === 'system' && e.subtype === 'init')?.model
1074
+ || events.find((e) => e.type === 'assistant')?.message?.model || model,
1075
+ host,
1076
+ transcript: path.relative(ROOT, streamFile),
1077
+ exit: r.status,
1078
+ spawnError: host === 'codex' ? parseCodexRunError(events, r) : replayRunError(events, r),
1079
+ };
1080
+ }
1081
+
1082
+ // ── the CLI ─────────────────────────────────────────────────────────────────────────────────────
1083
+ const argv = process.argv.slice(2);
1084
+ const has = (f) => argv.includes(f);
1085
+ const arg = (f, d) => { const i = argv.indexOf(f); return i >= 0 && argv[i + 1] ? argv[i + 1] : d; };
1086
+ const usage = () => `Usage:
1087
+ node scripts/learning-replay.mjs [--trap ${TRAP.MEMORY_SEARCH}|${TRAP.POST_TASK}] [--n N] [--host codex|claude-code] [--model MODEL]
1088
+ node scripts/learning-replay.mjs --check
1089
+ node scripts/learning-replay.mjs --check-portfolio
1090
+ node scripts/learning-replay.mjs --check-mutants
1091
+ node scripts/learning-replay.mjs --dry-run
1092
+ node scripts/learning-replay.mjs --mutant <${Object.keys(MUTANTS).join('|')}>
1093
+
1094
+ Exit: 0=PASS, 1=FAIL, 3=INCONCLUSIVE, 4=UNKNOWN.`;
1095
+
1096
+ /** The command mutant `wrong-subcommand` substitutes: the grader's exact defect, right flag on a wrong verb. */
1097
+ export const WRONG_SUBCOMMAND_COMMAND = 'ruflo recall -q "caching strategy"';
1098
+
1099
+ export const MUTANTS = Object.freeze({
1100
+ 'delete-lesson': 'delete the recorded lesson from the fixture store after the refresh — the treated arm must go red',
1101
+ 'brain-off-treated': 'run the TREATED arm with the brain disabled — it must produce the control artifact and go red',
1102
+ 'seed-control': "pre-seed the CONTROL arm's context with the lesson — the harness must report INCONCLUSIVE, never PASS",
1103
+ // ── the execution gate's own mutants (2026-07-28) ──
1104
+ 'wrong-subcommand': `substitute the treated arm's artifact with \`${WRONG_SUBCOMMAND_COMMAND}\` — right flag, wrong verb: the exact command the grader found being certified. Must go red.`,
1105
+ 'empty-store': "empty project B's seeded memory before the gate runs — a perfect command that retrieves nothing must go red on RETRIEVAL, not pass on exit status",
1106
+ });
1107
+
1108
+ /** Committed real-model evidence for ADR-058's two named falsification traps. */
1109
+ export const MUTANT_RESULT_FILES = Object.freeze({
1110
+ [TRAP.MEMORY_SEARCH]: Object.freeze({
1111
+ 'delete-lesson': path.join(ROOT, 'data', 'learning-replay-delete-lesson-result.json'),
1112
+ 'brain-off-treated': path.join(ROOT, 'data', 'learning-replay-brain-off-result.json'),
1113
+ }),
1114
+ [TRAP.POST_TASK]: Object.freeze({
1115
+ 'delete-lesson': path.join(ROOT, 'data', 'learning-replay-post-task-delete-lesson-result.json'),
1116
+ 'brain-off-treated': path.join(ROOT, 'data', 'learning-replay-post-task-brain-off-result.json'),
1117
+ }),
1118
+ });
1119
+
1120
+ export const PORTFOLIO_RESULT_FILES = Object.freeze({
1121
+ [TRAP.MEMORY_SEARCH]: RESULT_FILE,
1122
+ [TRAP.POST_TASK]: POST_TASK_RESULT_FILE,
1123
+ });
1124
+
1125
+ function headSha() {
1126
+ const r = spawnSync('git', ['rev-parse', 'HEAD'], { cwd: ROOT, encoding: 'utf8' });
1127
+ return r.status === 0 ? r.stdout.trim() : null;
1128
+ }
1129
+
1130
+ /** `--check`: gate on the committed artifact WITHOUT spending a token. */
1131
+ export function checkArtifact({ file = RESULT_FILE, repo = ROOT, maxAgeDays = 14 } = {}) {
1132
+ if (!fs.existsSync(file)) {
1133
+ return { status: VERDICT.UNKNOWN, why: `no result artifact at ${path.relative(repo, file)} — the replay has never been run on this checkout` };
1134
+ }
1135
+ let a;
1136
+ try { a = JSON.parse(fs.readFileSync(file, 'utf8')); } catch (e) { return { status: VERDICT.UNKNOWN, why: `result artifact unparseable: ${e.message}` }; }
1137
+ if (a.invariant !== INVARIANT) return { status: VERDICT.UNKNOWN, why: `artifact declares invariant "${a.invariant}", expected ${INVARIANT}` };
1138
+ if (!a.sha) return { status: VERDICT.UNKNOWN, why: 'artifact states no SHA — a result with no SHA is a result about nothing' };
1139
+
1140
+ const head = headSha();
1141
+ const stale = [];
1142
+ if (head && a.sha !== head) {
1143
+ // Not the same commit: the result is still CURRENT only if nothing load-bearing moved.
1144
+ const anc = spawnSync('git', ['merge-base', '--is-ancestor', a.sha, head], { cwd: repo });
1145
+ if (anc.status !== 0) return { status: VERDICT.UNKNOWN, why: `artifact SHA ${a.sha.slice(0, 8)} is not an ancestor of HEAD ${head.slice(0, 8)} — it measures a different tree` };
1146
+ const diff = spawnSync('git', ['diff', '--name-only', `${a.sha}..${head}`, '--', ...LOAD_BEARING], { cwd: repo, encoding: 'utf8' });
1147
+ if (diff.status === 0) for (const f of diff.stdout.split('\n').map((s) => s.trim()).filter(Boolean)) stale.push(f);
1148
+ }
1149
+ if (stale.length) {
1150
+ return { status: VERDICT.UNKNOWN, why: `result recorded on ${a.sha.slice(0, 8)}, but ${stale.length} load-bearing file(s) changed since: ${stale.join(', ')} — re-run the replay` };
1151
+ }
1152
+ const ageDays = a.at ? (Date.now() - Date.parse(a.at)) / 86_400_000 : Infinity;
1153
+ if (!(ageDays <= maxAgeDays)) {
1154
+ return { status: VERDICT.UNKNOWN, why: `result is ${Number.isFinite(ageDays) ? ageDays.toFixed(1) : '?'} days old (max ${maxAgeDays}) — a nightly trap that has not run is UNKNOWN, never PASS` };
1155
+ }
1156
+ return {
1157
+ status: a.verdict,
1158
+ why: `${a.verdict} — ${a.passes}/${a.n} runs, control produced the token in ${a.controlTokenRuns}/${a.n}, recorded on ${a.sha.slice(0, 8)} (${a.model})`,
1159
+ artifact: a,
1160
+ };
1161
+ }
1162
+
1163
+ /**
1164
+ * Verify the two named ADR-058 mutants were run against a current load-bearing tree and each
1165
+ * destroyed the claimed effect. A mutant FAIL is evidence only when the failure has the expected
1166
+ * causal shape; "the executor crashed" or "some unrelated assertion failed" cannot satisfy this.
1167
+ */
1168
+ export function checkMutantArtifacts({
1169
+ files = MUTANT_RESULT_FILES,
1170
+ repo = ROOT,
1171
+ maxAgeDays = 14,
1172
+ } = {}) {
1173
+ const checked = [];
1174
+ for (const trap of [TRAP.MEMORY_SEARCH, TRAP.POST_TASK]) {
1175
+ for (const mutant of ['delete-lesson', 'brain-off-treated']) {
1176
+ const file = files[trap]?.[mutant];
1177
+ if (!file || !fs.existsSync(file)) {
1178
+ return { status: VERDICT.UNKNOWN, why: `${trap}/${mutant}: no committed execution artifact`, checked };
1179
+ }
1180
+
1181
+ const currency = checkArtifact({ file, repo, maxAgeDays });
1182
+ if (currency.status === VERDICT.UNKNOWN) {
1183
+ return { status: VERDICT.UNKNOWN, why: `${trap}/${mutant}: ${currency.why}`, checked };
1184
+ }
1185
+
1186
+ let artifact;
1187
+ try { artifact = JSON.parse(fs.readFileSync(file, 'utf8')); }
1188
+ catch (e) { return { status: VERDICT.UNKNOWN, why: `${mutant}: artifact unparseable: ${e.message}`, checked }; }
1189
+
1190
+ if (artifact.mutant !== mutant || artifact.trap !== trap) {
1191
+ return { status: VERDICT.FAIL, why: `${trap}/${mutant}: artifact identity mismatch`, checked };
1192
+ }
1193
+ if (artifact.verdict !== VERDICT.FAIL || artifact.passes !== 0 || !(artifact.n >= 1)) {
1194
+ return { status: VERDICT.FAIL, why: `${mutant}: mutant did not go red with zero passes`, checked };
1195
+ }
1196
+ if (!Array.isArray(artifact.runs) || artifact.runs.length !== artifact.n) {
1197
+ return { status: VERDICT.FAIL, why: `${mutant}: missing per-run execution evidence`, checked };
1198
+ }
1199
+
1200
+ if (mutant === 'delete-lesson') {
1201
+ const lessonSurvived = artifact.runs.some((run) =>
1202
+ run.treated?.lessonBeforeFirstToolCall === true
1203
+ || run.treated?.lessonDelivered === true
1204
+ || run.treated?.class === 'flagged');
1205
+ if (lessonSurvived) {
1206
+ return { status: VERDICT.FAIL, why: 'delete-lesson: treated arm still received or reproduced the lesson', checked };
1207
+ }
1208
+ } else {
1209
+ const differsFromControl = artifact.runs.some((run) =>
1210
+ run.treated?.lessonBeforeFirstToolCall === true
1211
+ || run.treated?.lessonDelivered === true
1212
+ || run.control?.lessonDelivered === true
1213
+ || run.treated?.class !== run.control?.class);
1214
+ if (differsFromControl) {
1215
+ return { status: VERDICT.FAIL, why: 'brain-off-treated: treated arm did not collapse to the brain-off control artifact', checked };
1216
+ }
1217
+ }
1218
+ checked.push(`${trap}/${mutant}`);
1219
+ }
1220
+ }
1221
+ return {
1222
+ status: VERDICT.PASS,
1223
+ why: 'delete-lesson and brain-off-treated both went red for both independent traps on current real-model execution evidence',
1224
+ checked,
1225
+ };
1226
+ }
1227
+
1228
+ export function checkPortfolio({
1229
+ files = PORTFOLIO_RESULT_FILES,
1230
+ mutantFiles = MUTANT_RESULT_FILES,
1231
+ repo = ROOT,
1232
+ maxAgeDays = 14,
1233
+ } = {}) {
1234
+ const artifacts = [];
1235
+ for (const trap of [TRAP.MEMORY_SEARCH, TRAP.POST_TASK]) {
1236
+ const checked = checkArtifact({ file: files[trap], repo, maxAgeDays });
1237
+ if (checked.status !== VERDICT.PASS) {
1238
+ return { status: checked.status, why: `${trap}: ${checked.why}`, artifacts };
1239
+ }
1240
+ const a = checked.artifact;
1241
+ if (a.trap !== trap || a.n < 3 || a.passes < 2 || a.controlTokenRuns !== 0 || a.controlWorkedRuns !== 0) {
1242
+ return { status: VERDICT.FAIL, why: `${trap}: artifact does not prove N>=3 treated/control causal separation`, artifacts };
1243
+ }
1244
+ if (a.promotion?.projectCount < 2 || a.promotion?.promoted !== true
1245
+ || new Set(a.promotion?.sourceProjects || []).size < 2) {
1246
+ return { status: VERDICT.FAIL, why: `${trap}: lesson did not earn the real win-twice cross-project scope`, artifacts };
1247
+ }
1248
+ if (a.refresh?.lessonSurvived !== true) {
1249
+ return { status: VERDICT.FAIL, why: `${trap}: learned lesson did not survive refresh`, artifacts };
1250
+ }
1251
+ artifacts.push(a);
1252
+ }
1253
+ if (artifacts[0].record?.lessonId === artifacts[1].record?.lessonId
1254
+ || artifacts[0].task === artifacts[1].task) {
1255
+ return { status: VERDICT.FAIL, why: 'portfolio traps are not independent lesson/task classes', artifacts };
1256
+ }
1257
+ const mutants = checkMutantArtifacts({ files: mutantFiles, repo, maxAgeDays });
1258
+ if (mutants.status !== VERDICT.PASS) return { ...mutants, why: `portfolio mutants: ${mutants.why}`, artifacts };
1259
+ return {
1260
+ status: VERDICT.PASS,
1261
+ why: 'two independent Ruflo CLI lessons each passed N>=3 treated/control, earned win-twice scope in two source projects, survived refresh, executed a meaningful outcome, and failed both causal mutants',
1262
+ artifacts,
1263
+ mutants,
1264
+ };
1265
+ }
1266
+
1267
+ async function main() {
1268
+ if (has('--help') || has('-h')) {
1269
+ console.log(usage());
1270
+ return;
1271
+ }
1272
+ const check = has('--check');
1273
+ const checkPortfolioFlag = has('--check-portfolio');
1274
+ const checkMutants = has('--check-mutants');
1275
+ const dryRun = has('--dry-run');
1276
+ const mutant = arg('--mutant', null);
1277
+ const trap = arg('--trap', TRAP.MEMORY_SEARCH);
1278
+ const spec = trapSpec(trap);
1279
+ const n = Math.max(1, parseInt(arg('--n', mutant ? '1' : '3'), 10) || 1);
1280
+ const host = arg('--host', 'codex');
1281
+ const model = arg('--model', host === 'codex' ? 'gpt-5.6-sol' : 'haiku');
1282
+ const defaultOut = mutant
1283
+ ? MUTANT_RESULT_FILES[trap]?.[mutant]
1284
+ : PORTFOLIO_RESULT_FILES[trap];
1285
+ const outFile = arg('--out', defaultOut);
1286
+ const keep = has('--keep-fixtures');
1287
+
1288
+ if (mutant && !MUTANTS[mutant]) {
1289
+ console.error(`unknown mutant "${mutant}". known: ${Object.keys(MUTANTS).join(', ')}`);
1290
+ process.exit(EXIT.UNKNOWN);
1291
+ }
1292
+ if (![TRAP.MEMORY_SEARCH, TRAP.POST_TASK].includes(trap) || !outFile) {
1293
+ console.error(`unknown or unsupported trap/mutant combination: ${trap}/${mutant || 'normal'}`);
1294
+ process.exit(EXIT.UNKNOWN);
1295
+ }
1296
+
1297
+ if (check) {
1298
+ const res = checkArtifact({ file: PORTFOLIO_RESULT_FILES[trap] });
1299
+ console.log(`\n ${INVARIANT}: ${res.status}\n ${res.why}\n`);
1300
+ process.exit(EXIT[res.status] ?? EXIT.UNKNOWN);
1301
+ }
1302
+
1303
+ if (checkPortfolioFlag) {
1304
+ const res = checkPortfolio();
1305
+ console.log(`\n ${INVARIANT}-PORTFOLIO: ${res.status}\n ${res.why}\n`);
1306
+ process.exit(EXIT[res.status] ?? EXIT.UNKNOWN);
1307
+ }
1308
+
1309
+ if (checkMutants) {
1310
+ const res = checkMutantArtifacts();
1311
+ console.log(`\n ${INVARIANT}-MUTANTS: ${res.status}\n ${res.why}\n`);
1312
+ process.exit(EXIT[res.status] ?? EXIT.UNKNOWN);
1313
+ }
1314
+
1315
+ console.log(`\n=== ${INVARIANT} — counterfactual replay (ADR-058 §D4) ===`);
1316
+ const flag = trap === TRAP.POST_TASK ? verifyPostTaskContract() : verifyRufloFlag();
1317
+ console.log(` trap: ${trap}`);
1318
+ console.log(` premise: ${flag.ok ? `VERIFIED live: ${flag.evidence}` : `NOT VERIFIED: ${flag.why}`}`);
1319
+ if (!flag.ok) {
1320
+ const artifact = writeArtifact(outFile, {
1321
+ verdict: VERDICT.UNKNOWN, why: `premise not verified: ${flag.why}`, n: 0, passes: 0, fails: 0, unknowns: 0, controlTokenRuns: 0, rate: 0, runs: [],
1322
+ }, { host, model, mutant });
1323
+ console.log(` → UNKNOWN (never a pass). artifact: ${path.relative(ROOT, outFile)}`);
1324
+ process.exit(EXIT.UNKNOWN);
1325
+ }
1326
+
1327
+ const base = allocateRunBase();
1328
+ const dirs = buildFixtures(base);
1329
+ // Ruflo may auto-start a workspace daemon while initializing a fixture memory DB. Reap only
1330
+ // daemons whose explicit --workspace lives under THIS run, including on Ctrl-C or an exception.
1331
+ process.once('exit', () => { cleanupFixtureDaemons(dirs); });
1332
+ const rec = recordInProjectA(dirs, { trap });
1333
+ console.log(` record (two independent source projects): ${rec.projectCount} sources, win-twice=${rec.promoted}, lesson ${rec.ok ? 'derived + ratified' : 'NOT recorded'}`);
1334
+ const seed = trap === TRAP.MEMORY_SEARCH
1335
+ ? seedProjectBMemory(dirs)
1336
+ : { key: null, storeExit: null, ok: true, skipped: 'the command-risk trap needs no target-project memory row' };
1337
+ console.log(` seed (fixture-project-B): ${seed.skipped || `note "${seed.key}" ${seed.ok ? 'stored — the task premise is true' : `NOT stored (exit ${seed.storeExit})`}`}`);
1338
+ const refresh = nightlyRefresh(dirs);
1339
+ console.log(` refresh (nightly): spine generation ${refresh.generation} installed + active; distill exit ${refresh.distillExit}, backup exit ${refresh.backupExit}; lesson survived: ${refresh.lessonSurvived}`);
1340
+
1341
+ if (mutant === 'delete-lesson') {
1342
+ saveLessons([], dirs.lessons);
1343
+ try { fs.rmSync(path.join(dirs.projectA, '.swarm', 'memory.db'), { force: true }); } catch { /* already gone */ }
1344
+ console.log(` MUTANT delete-lesson: lesson store emptied (${loadLessons(dirs.lessons).length} lessons) and project A's memory.db removed`);
1345
+ }
1346
+
1347
+ if (mutant === 'empty-store') {
1348
+ // Remove the seeded note. Every arm still runs for real; the produced command still executes for
1349
+ // real; it simply has nothing to find. The gate must key on RETRIEVAL, not on exit status —
1350
+ // `ruflo memory search -q "<absent>"` exits 0.
1351
+ try { fs.rmSync(path.join(dirs.projectB, '.swarm', 'memory.db'), { force: true }); } catch { /* already gone */ }
1352
+ console.log(` MUTANT empty-store: project B's seeded memory removed (exists: ${fs.existsSync(path.join(dirs.projectB, '.swarm', 'memory.db'))})`);
1353
+ }
1354
+
1355
+ if (dryRun) {
1356
+ // Prove the WIRE without a token: fire the real hook chain in both states and report the bytes.
1357
+ const probe = (stateDir) => {
1358
+ const r = spawnSync(process.execPath, [path.join(ROOT, 'plugin', 'scripts', 'hook-shim.mjs'), 'unprompted-speech', 'UserPromptSubmit'], {
1359
+ input: JSON.stringify({ prompt: spec.prompt, session_id: `dry-${Date.now()}`, cwd: dirs.projectB }),
1360
+ encoding: 'utf8',
1361
+ cwd: dirs.projectB,
1362
+ env: {
1363
+ ...process.env,
1364
+ RUVNET_BRAIN_HOME: dirs.brainHome,
1365
+ RUVNET_BRAIN_STATE_DIR: stateDir,
1366
+ RUVNET_LESSON_STORE: dirs.lessons,
1367
+ RUVNET_LESSON_GATE_STATE: dirs.gateState,
1368
+ CLAUDE_PLUGIN_ROOT: path.join(ROOT, 'plugin'),
1369
+ },
1370
+ });
1371
+ return (r.stdout || '').length;
1372
+ };
1373
+ const onBytes = probe(dirs.stateOn), offBytes = probe(dirs.stateOff);
1374
+ console.log(` dry-run: treated-state hook emitted ${onBytes} bytes; control-state (brain-off) emitted ${offBytes} bytes`);
1375
+ writeArtifact(outFile, {
1376
+ verdict: VERDICT.UNKNOWN, why: `--dry-run: no model was called, so nothing was measured (wire probe: treated ${onBytes}B, control ${offBytes}B)`,
1377
+ n: 0, passes: 0, fails: 0, unknowns: 0, controlTokenRuns: 0, rate: 0, runs: [],
1378
+ }, { host, model, mutant, record: rec, seed, refresh });
1379
+ console.log(` → UNKNOWN (a dry run is never a pass). artifact: ${path.relative(ROOT, outFile)}`);
1380
+ if (!keep) {
1381
+ cleanupFixtureDaemons(dirs);
1382
+ rmrf(base);
1383
+ }
1384
+ process.exit(EXIT.UNKNOWN);
1385
+ }
1386
+
1387
+ const runs = [];
1388
+ for (let i = 1; i <= n; i++) {
1389
+ const treatedState = mutant === 'brain-off-treated' ? dirs.stateOff : dirs.stateOn;
1390
+ const treated = runArm({
1391
+ dirs, arm: 'treated', stateDir: treatedState, model, host, tag: `run${i}-treated`, trap,
1392
+ forceCommand: mutant === 'wrong-subcommand' ? WRONG_SUBCOMMAND_COMMAND : null,
1393
+ });
1394
+ const control = runArm({
1395
+ dirs, arm: 'control', stateDir: dirs.stateOff, model, host, tag: `run${i}-control`, trap,
1396
+ // MUTANT seed-control: the control is handed the lesson through a channel the brain does not
1397
+ // own. Its artifact then carries the token, and invariant 6 must fire.
1398
+ appendSystemPrompt: mutant === 'seed-control' ? spec.lesson : null,
1399
+ });
1400
+ const run = {
1401
+ i,
1402
+ treatedClass: treated.class,
1403
+ controlClass: control.class,
1404
+ lessonBeforeFirstToolCall: treated.lessonBeforeFirstToolCall,
1405
+ controlLessonDelivered: control.lessonDelivered,
1406
+ // ── the execution gate's inputs, flattened so verdictForRun stays a pure function ──
1407
+ treatedSubcommandCorrect: treated.subcommandCorrect,
1408
+ treatedExecOk: treated.exec?.exitOk === true,
1409
+ treatedRetrieved: treated.exec?.retrieved === true,
1410
+ treatedExecWhy: treated.exec?.why || null,
1411
+ controlWorked: control.exec?.exitOk === true && control.exec?.retrieved === true,
1412
+ treated,
1413
+ control,
1414
+ error: treated.spawnError || control.spawnError || null,
1415
+ };
1416
+ const v = verdictForRun(run);
1417
+ console.log(` run ${i}: treated="${treated.class}" (${treated.command || '—'})`);
1418
+ console.log(` control="${control.class}" (${control.command || '—'})`);
1419
+ console.log(` EXECUTED treated: subcommand=${treated.subcommandCorrect} exit=${treated.exec?.exit ?? '—'} retrieved=${treated.exec?.retrieved} · ${treated.exec?.why || 'not run'}`);
1420
+ console.log(` EXECUTED control: exit=${control.exec?.exit ?? '—'} retrieved=${control.exec?.retrieved}`);
1421
+ console.log(` lesson before first tool call: ${treated.lessonBeforeFirstToolCall} (lesson@${treated.lessonIndex}, tool@${treated.firstToolIndex}) · control got ${control.lessonDelivered ? 'THE LESSON (leak!)' : 'zero brain bytes'}`);
1422
+ console.log(` → ${v.verdict}: ${v.why}`);
1423
+ runs.push(run);
1424
+ }
1425
+
1426
+ const agg = aggregate(runs);
1427
+ const costUsd = runs.reduce((s, r) => s + (r.treated.costUsd || 0) + (r.control.costUsd || 0), 0);
1428
+ const wallMs = runs.reduce((s, r) => s + (r.treated.wallMs || 0) + (r.control.wallMs || 0), 0);
1429
+ writeArtifact(outFile, agg, { host, model, mutant, trap, task: spec.prompt, record: rec, seed, refresh, flag, costUsd, wallMs });
1430
+
1431
+ console.log(`\n RATE ${agg.passes}/${agg.n} · token carried by treated ${agg.treatedTokenRuns}/${agg.n} vs control ${agg.controlTokenRuns}/${agg.n}`);
1432
+ console.log(` EXECUTION GATE: treated named the real subcommand ${agg.treatedSubcommandRuns}/${agg.n} · exited 0 ${agg.treatedExecutedRuns}/${agg.n} · RETRIEVED ${agg.treatedRetrievedRuns}/${agg.n} · control worked ${agg.controlWorkedRuns}/${agg.n}`);
1433
+ console.log(` ${INVARIANT}: ${agg.verdict} — ${agg.why}`);
1434
+ if (!keep) pruneArchive(dirs);
1435
+ console.log(` cost $${costUsd.toFixed(4)} · ${(wallMs / 1000).toFixed(1)}s wall · transcripts: ${path.relative(ROOT, dirs.transcripts)}`);
1436
+ console.log(` artifact: ${path.relative(ROOT, outFile)}\n`);
1437
+ process.exit(EXIT[agg.verdict] ?? EXIT.UNKNOWN);
1438
+ }
1439
+
1440
+ /**
1441
+ * RETENTION. "Transcripts archived" must not mean "the disk fills".
1442
+ *
1443
+ * Each run builds a whole fixture world, and the nightly refresh step copies plugin/ into it — ~12MB
1444
+ * per invocation, every night, forever. Measured 2026-07-27 after nine invocations in one session:
1445
+ * 36MB, of which the transcripts were under 300KB. So the EVIDENCE is kept and the SCAFFOLDING is
1446
+ * dropped: everything under the run directory except transcripts/ goes, and only the most recent
1447
+ * KEEP_RUNS run directories survive. `--keep-fixtures` retains everything for debugging.
1448
+ *
1449
+ * Deliberately not "delete the whole run dir": the transcripts ARE the archive ADR-058 asks for, and
1450
+ * an archive nobody kept is the same as a claim nobody checked.
1451
+ */
1452
+ const KEEP_RUNS = 14;
1453
+ export function cleanupFixtureDaemons(dirs) {
1454
+ const base = path.resolve(dirs.base);
1455
+ const ps = spawnSync('ps', ['-axo', 'pid=,command='], { encoding: 'utf8', timeout: 10_000 });
1456
+ if (ps.status !== 0) return { found: 0, stopped: 0, errors: ['process census failed'] };
1457
+ const matches = String(ps.stdout || '').split('\n').flatMap((line) => {
1458
+ const match = line.match(/^\s*(\d+)\s+(.+)$/);
1459
+ if (!match) return [];
1460
+ const pid = Number(match[1]);
1461
+ const command = match[2];
1462
+ return command.includes('daemon start --foreground')
1463
+ && command.includes('--workspace')
1464
+ && command.includes(base)
1465
+ && pid !== process.pid
1466
+ ? [{ pid, command }]
1467
+ : [];
1468
+ });
1469
+ const errors = [];
1470
+ let stopped = 0;
1471
+ for (const match of matches) {
1472
+ try { process.kill(match.pid, 'SIGTERM'); stopped++; }
1473
+ catch (error) {
1474
+ if (error?.code !== 'ESRCH') errors.push(`${match.pid}: ${error.message}`);
1475
+ }
1476
+ }
1477
+ return { found: matches.length, stopped, errors };
1478
+ }
1479
+
1480
+ function pruneArchive(dirs) {
1481
+ cleanupFixtureDaemons(dirs);
1482
+ try {
1483
+ for (const e of fs.readdirSync(dirs.base, { withFileTypes: true })) {
1484
+ if (e.name === 'transcripts') continue;
1485
+ rmrf(path.join(dirs.base, e.name));
1486
+ }
1487
+ } catch { /* nothing to prune */ }
1488
+ try {
1489
+ const root = path.dirname(dirs.base);
1490
+ const runs = fs.readdirSync(root).filter((d) => d.startsWith('run-')).sort();
1491
+ for (const old of runs.slice(0, Math.max(0, runs.length - KEEP_RUNS))) rmrf(path.join(root, old));
1492
+ } catch { /* nothing to prune */ }
1493
+ }
1494
+
1495
+ /**
1496
+ * The execution record as it lands in the COMMITTED artifact. The first 400 bytes of the command's
1497
+ * real output are kept verbatim: a retrieval claim whose evidence nobody can read is an assertion,
1498
+ * and the whole deduction being closed here was a number nobody could check against its own arm.
1499
+ */
1500
+ function execRecord(e) {
1501
+ if (!e) return null;
1502
+ return { ran: e.ran, argv: e.argv, exit: e.exit, exitOk: e.exitOk, retrieved: e.retrieved, why: e.why, output: String(e.output || '').slice(0, 400) };
1503
+ }
1504
+
1505
+ /** The machine-readable result. A verdict with no SHA is a verdict about nothing. */
1506
+ function writeArtifact(file, agg, meta = {}) {
1507
+ const artifact = {
1508
+ invariant: INVARIANT,
1509
+ verdict: agg.verdict,
1510
+ why: agg.why,
1511
+ sha: headSha(),
1512
+ at: new Date().toISOString(),
1513
+ host: meta.host || null,
1514
+ model: meta.model || null,
1515
+ // The alias asked for ("haiku") is not the model that answered. Record the id the session
1516
+ // actually reported, so a result can never be attributed to a model that never ran.
1517
+ modelResolved: (agg.runs || []).map((r) => r.treated?.modelUsed).find(Boolean) || null,
1518
+ mutant: meta.mutant || null,
1519
+ trap: meta.trap || TRAP.MEMORY_SEARCH,
1520
+ task: meta.task || null,
1521
+ n: agg.n,
1522
+ passes: agg.passes,
1523
+ fails: agg.fails,
1524
+ unknowns: agg.unknowns,
1525
+ controlTokenRuns: agg.controlTokenRuns,
1526
+ controlWorkedRuns: agg.controlWorkedRuns ?? null,
1527
+ treatedTokenRuns: agg.treatedTokenRuns ?? null,
1528
+ // THE EXECUTION GATE's own rates. The old artifact could report `treatedTokenRuns: 3` beside
1529
+ // three `subcommandCorrect: false` and call it PASS; these three numbers are what makes that
1530
+ // combination impossible to state without also stating that nothing worked.
1531
+ executionGate: {
1532
+ treatedSubcommandRuns: agg.treatedSubcommandRuns ?? null,
1533
+ treatedExecutedRuns: agg.treatedExecutedRuns ?? null,
1534
+ treatedRetrievedRuns: agg.treatedRetrievedRuns ?? null,
1535
+ },
1536
+ rate: agg.rate,
1537
+ threshold: '>=2/3',
1538
+ costUsd: meta.costUsd != null ? +meta.costUsd.toFixed(4) : null,
1539
+ wallSeconds: meta.wallMs != null ? +(meta.wallMs / 1000).toFixed(1) : null,
1540
+ premise: meta.flag ? { verified: meta.flag.ok, evidence: meta.flag.evidence } : null,
1541
+ record: meta.record ? {
1542
+ lessonId: meta.record.lesson?.id,
1543
+ key: meta.record.key,
1544
+ storeExit: meta.record.storeExit,
1545
+ readBackExit: meta.record.readBackExit,
1546
+ ok: meta.record.ok,
1547
+ } : null,
1548
+ promotion: meta.record ? {
1549
+ rule: 'ADR-G008 win twice',
1550
+ projectCount: meta.record.projectCount,
1551
+ sourceProjects: meta.record.sources?.map((source) => source.project) || [],
1552
+ promoted: meta.record.promoted === true,
1553
+ } : null,
1554
+ seed: meta.seed ? { key: meta.seed.key, storeExit: meta.seed.storeExit, ok: meta.seed.ok } : null,
1555
+ refresh: meta.refresh || null,
1556
+ runs: (agg.runs || []).map((r) => ({
1557
+ i: r.i,
1558
+ verdict: r.verdict,
1559
+ why: r.why,
1560
+ treated: { class: r.treated?.class, subcommandCorrect: r.treated?.subcommandCorrect, command: r.treated?.command, forcedCommand: r.treated?.forcedCommand || false, exec: execRecord(r.treated?.exec), lessonIndex: r.treated?.lessonIndex, firstToolIndex: r.treated?.firstToolIndex, lessonBeforeFirstToolCall: r.treated?.lessonBeforeFirstToolCall, model: r.treated?.modelUsed, transcript: r.treated?.transcript },
1561
+ control: { class: r.control?.class, subcommandCorrect: r.control?.subcommandCorrect, command: r.control?.command, exec: execRecord(r.control?.exec), lessonDelivered: r.control?.lessonDelivered, model: r.control?.modelUsed, transcript: r.control?.transcript },
1562
+ })),
1563
+ };
1564
+ fs.mkdirSync(path.dirname(file), { recursive: true });
1565
+ fs.writeFileSync(file, JSON.stringify(artifact, null, 2) + '\n');
1566
+ return artifact;
1567
+ }
1568
+
1569
+ const invokedDirectly = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
1570
+ if (invokedDirectly) await main();