ruvnet-brain 4.0.1 → 4.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -0
- package/README.md +4 -4
- package/bin/install.mjs +303 -24
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +8 -22
- package/plugin/.claude-plugin/marketplace.json +1 -0
- package/plugin/.claude-plugin/plugin.json +2 -3
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +2 -2
- package/plugin/commands/configure.md +3 -2
- package/plugin/commands/rvbc.md +4 -3
- package/plugin/commands/rvcb.md +2 -2
- package/plugin/commands/whats-new.md +6 -6
- package/plugin/docs/RELEASE-NOTES-4.0.md +88 -0
- package/plugin/hooks/hooks.json +1 -2
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +90 -32
- package/plugin/scripts/detach.mjs +14 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/ground-ruvnet.sh +16 -6
- package/plugin/scripts/hook-shim.mjs +34 -29
- package/plugin/scripts/learn-capture.sh +22 -3
- package/plugin/scripts/learn-flush.mjs +21 -4
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +503 -0
- package/plugin/scripts/session-start.sh +3 -858
- package/plugin/scripts/whats-new.mjs +42 -0
- package/plugin/skills/brain-console/SKILL.md +4 -2
- package/plugin/skills/release-proof/SKILL.md +98 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +44 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +286 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
- package/plugin/skills/ruvnet-brain/SKILL.md +22 -7
- package/plugin/skills/rvbc/SKILL.md +9 -6
- package/plugin/skills/whats-new/SKILL.md +4 -4
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +531 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/fix-workstream.mjs +291 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +237 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +558 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +326 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +172 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2922 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/protected-release-invocation.mjs +76 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/publication-receipt.mjs +307 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-authority.mjs +93 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +439 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +296 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stabilization-receipt.mjs +108 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +867 -0
- package/plugin/scripts/finalize-token-meter.mjs +0 -25
|
@@ -0,0 +1,1570 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/learning-replay.mjs — the COUNTERFACTUAL REPLAY TRAP (ADR-058 §D4, DDD-0013 Context 1,
|
|
4
|
+
* aggregate `CounterfactualTrap`). Invariant name: **LEARNING-REPLAY**.
|
|
5
|
+
*
|
|
6
|
+
* ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
7
|
+
* WHAT THIS INVERTS, and why it exists at all.
|
|
8
|
+
*
|
|
9
|
+
* `scripts/behavioral-l1-l4.mjs`'s L4 asserts that the brain's own injected prose CONTAINS the words
|
|
10
|
+
* 'take the wheel', 'SPARC', 'swarm'. That is a check on what the brain SAID. It cannot fail on an
|
|
11
|
+
* agent that ignored every word of it, and it certified "behavioral, all pass" for weeks while
|
|
12
|
+
* nothing downstream was measured at all. This file measures the opposite thing and only that thing:
|
|
13
|
+
*
|
|
14
|
+
* did an agent's PRODUCED ARTIFACT change, against a control that did not receive the lesson.
|
|
15
|
+
*
|
|
16
|
+
* The oracle is a parse of a command string — `plugin/scripts/hook-input.mjs:findInvocations()`,
|
|
17
|
+
* executable-position classification, the same anti-corruption boundary DDD-0013 mandates against
|
|
18
|
+
* the host's Bash envelope. It is never a similarity score and never a model grading a model.
|
|
19
|
+
*
|
|
20
|
+
* ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
21
|
+
* THE TRAP, concretely (ADR-058 §D4 specifies it so it cannot dissolve into intention).
|
|
22
|
+
*
|
|
23
|
+
* RECORD, in fixture-project-A: the correction that `ruflo memory search` takes its query with the
|
|
24
|
+
* `-q` flag and rejects a bare positional. This is a FACT ABOUT THE REAL CLI, verified against the
|
|
25
|
+
* real global binary (`~/.npm-global/bin/ruflo memory search --help` prints
|
|
26
|
+
* `-q, --query Search query (required)`), not recalled. An oracle built on a false premise is
|
|
27
|
+
* worthless, so the harness RE-VERIFIES it at run time (`verifyRufloFlag()`) and refuses to run
|
|
28
|
+
* against a CLI whose interface no longer matches.
|
|
29
|
+
*
|
|
30
|
+
* REPLAY, in fixture-project-B: a fresh session, a DIFFERENTLY-WORDED task ("recall the note about
|
|
31
|
+
* the caching strategy") that shares no content word with the lesson. String-matching the lesson
|
|
32
|
+
* text cannot be what carries it; only the flag can.
|
|
33
|
+
*
|
|
34
|
+
* PASS requires all of:
|
|
35
|
+
* (a) the lesson is in the transcript BEFORE the first tool call — measured as stream position,
|
|
36
|
+
* not asserted from the fact that UserPromptSubmit "happens first";
|
|
37
|
+
* (b) the treated arm's produced command carries the token where the BRAIN-OFF CONTROL's does not;
|
|
38
|
+
* (c) it still holds after a nightly refresh runs between record and replay — the refresh is
|
|
39
|
+
* real: a new Stable-Spine generation is installed into the fixture brain home and the
|
|
40
|
+
* pointer flipped, so the replay's hooks execute from a DIFFERENT code root than the record
|
|
41
|
+
* did, and `ruflo memory distill run` / `ruflo memory backup` (the two commands
|
|
42
|
+
* scripts/nightly-wrapper.sh actually runs nightly) are run against project A's store.
|
|
43
|
+
* (d) the produced command NAMES THE REAL SUBCOMMAND, EXECUTES against the real fixture store,
|
|
44
|
+
* EXITS 0, and ACTUALLY RETRIEVES the memory the task asked for. See "THE EXECUTION GATE".
|
|
45
|
+
*
|
|
46
|
+
* ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
47
|
+
* THE INVALIDATION RULE — DDD-0013 invariant 6, and the whole point of the file.
|
|
48
|
+
*
|
|
49
|
+
* A trap whose CONTROL run also produces the token is INVALID. The result is INCONCLUSIVE.
|
|
50
|
+
* NEVER a pass.
|
|
51
|
+
*
|
|
52
|
+
* If the model would have got it right anyway, the trap measured nothing — it measured the model's
|
|
53
|
+
* priors. This is encoded as CODE, not as a comment: `aggregate()` computes `controlTokenRuns`
|
|
54
|
+
* FIRST and the PASS branch is unreachable while it is non-zero, and a final assertion throws if a
|
|
55
|
+
* PASS verdict is ever paired with a successful control. A check that can report PASS on a
|
|
56
|
+
* meaningless measurement is the L4 defect rebuilt one file to the left.
|
|
57
|
+
*
|
|
58
|
+
* (DDD-0013 invariant 6 words the invalid outcome as `UNKNOWN`; ADR-058 §D4 words it `INCONCLUSIVE`.
|
|
59
|
+
* This file emits INCONCLUSIVE and treats it as strictly non-PASS, which satisfies both — the two
|
|
60
|
+
* documents disagree on the LABEL, never on the consequence.)
|
|
61
|
+
*
|
|
62
|
+
* ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
63
|
+
* THE EXECUTION GATE — added 2026-07-28, closing the largest single deduction in the D4 re-score.
|
|
64
|
+
*
|
|
65
|
+
* An independent grader (GPT-5.6-Sol) scored this dimension 44/100 and named the reason exactly:
|
|
66
|
+
*
|
|
67
|
+
* "The recorded replay says PASS 3/3, yet all three treated commands have subcommandCorrect:
|
|
68
|
+
* false … PASS depends on token use, control contrast and lesson delivery — not successful
|
|
69
|
+
* command execution or successful retrieval. The suite currently certifies unusable learned
|
|
70
|
+
* behavior."
|
|
71
|
+
*
|
|
72
|
+
* It was right, and the block that used to sit here — arguing the subcommand is REPORTED and not
|
|
73
|
+
* GATED because the lesson only taught the flag — was a defensible claim about the LESSON and an
|
|
74
|
+
* indefensible one about the CLAIM. The trap's headline is "learning demonstrated". A command that
|
|
75
|
+
* would fail if anyone ran it demonstrates nothing, whatever it says about the flag.
|
|
76
|
+
*
|
|
77
|
+
* So the verdict now additionally requires, per run, that the treated arm's produced command:
|
|
78
|
+
* 1. names the real subcommand (`subcommandCorrect`) — no longer observed-only;
|
|
79
|
+
* 2. EXECUTES against the real fixture store and exits 0;
|
|
80
|
+
* 3. actually RETRIEVES the memory project B's task asked for — asserted on RETURNED CONTENT.
|
|
81
|
+
*
|
|
82
|
+
* Point 3 is not redundant with point 2, and this is the whole reason exit status alone is not
|
|
83
|
+
* admissible. Measured live on this machine, 2026-07-28, against the real global binary:
|
|
84
|
+
*
|
|
85
|
+
* ruflo memory search -q "caching strategy" --path <db> → EXIT 0 · "Found 1 results"
|
|
86
|
+
* ruflo memory recall -q "caching strategy" --path <db> → EXIT 0 · prints the `memory` HELP
|
|
87
|
+
* ruflo recall -q "caching strategy" → EXIT 1 · "Unknown command: recall"
|
|
88
|
+
* ruflo memory search "caching strategy" --path <db> → EXIT 1 · "Required option missing: --query"
|
|
89
|
+
* ruflo memory search -q "<absent phrase>" --path <db> → EXIT 0 · "[WARN] No results found"
|
|
90
|
+
*
|
|
91
|
+
* `ruflo memory recall -q` — the exact command two of the three certified runs produced — EXITS 0.
|
|
92
|
+
* An exit-status gate would have passed it. Only an assertion on returned content catches it. (Line
|
|
93
|
+
* 5 is the same lesson from the other side: a perfectly-formed search that finds nothing also exits
|
|
94
|
+
* 0. Retrieval is the claim; exit status is not.)
|
|
95
|
+
*
|
|
96
|
+
* WHAT THE FIXTURE HAD TO CHANGE FOR THIS TO BE MEASURABLE, stated rather than finessed: project B's
|
|
97
|
+
* prompt already asserted "earlier in this project someone recorded a note about the caching
|
|
98
|
+
* strategy", and that was FALSE of the fixture world — project B's store was empty. So the harness
|
|
99
|
+
* now seeds that note into project B's own `.swarm/memory.db` (`seedProjectBMemory`). This is a fix
|
|
100
|
+
* to the FIXTURE, not to the lesson: the seeded note says nothing about `-q`, no arm ever sees its
|
|
101
|
+
* text (the recorder blocks every command before it runs), and the lesson text is untouched. Without
|
|
102
|
+
* it, even a flawless `ruflo memory search -q "caching strategy"` would retrieve nothing and the new
|
|
103
|
+
* gate would be measuring a harness bug — Rule 22 check (d).
|
|
104
|
+
*
|
|
105
|
+
* SAFETY. The recorder BLOCKS the agent's command on purpose (a fixture agent must not run anything
|
|
106
|
+
* on a real machine, and must not learn the answer from a CLI's own `--help` mid-run). That is kept.
|
|
107
|
+
* Execution happens OUT OF BAND, after the arm is over, in the harness — and never through a shell:
|
|
108
|
+
* `executeProducedCommand()` runs the argv `findInvocations()` already parsed, so pipes, redirects,
|
|
109
|
+
* substitutions and metacharacters are structurally absent. It also refuses to run a mutating
|
|
110
|
+
* subcommand, and refuses any `--path`/`--db` pointing outside the fixture world. Both refusals mark
|
|
111
|
+
* the run NOT-RETRIEVED, so every one of them can only LOWER the rate.
|
|
112
|
+
*
|
|
113
|
+
* ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
114
|
+
* A RATE, NEVER A VERDICT. N runs, PASS at >= 2/3 of them, transcripts archived. One run of a
|
|
115
|
+
* stochastic system is an anecdote; the artifact records k/n and every arm's classification.
|
|
116
|
+
*
|
|
117
|
+
* ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
118
|
+
* WHAT THE ARMS ACTUALLY DIFFER BY — the product's OWN switch, not a harness flag.
|
|
119
|
+
*
|
|
120
|
+
* Both arms run the identical fixture, the identical prompt, the identical hook registration
|
|
121
|
+
* (`hook-shim.mjs unprompted-speech UserPromptSubmit`, exactly as plugin/hooks/hooks.json registers
|
|
122
|
+
* it). The ONLY difference is the presence of the `brain-off` sentinel in the arm's
|
|
123
|
+
* RUVNET_BRAIN_STATE_DIR — ADR-054's real consent switch, whose `offBehavior: 'silence'` contract
|
|
124
|
+
* for the unprompted plane means the control receives ZERO bytes. That is why mutant 2 ("run the
|
|
125
|
+
* treated arm brain-disabled") is not a separate code path: it IS the control condition.
|
|
126
|
+
*
|
|
127
|
+
* ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
128
|
+
* COST. Real model tokens, priced in the open (ADR-058 §D4: "the one standing spend"). Default model
|
|
129
|
+
* is haiku — the trap measures whether CONTEXT REACHES the agent, not whether the agent is clever.
|
|
130
|
+
* Measured 2026-07-27 on this machine: ~$0.10 and ~8s of wall clock per arm, 2 arms per run.
|
|
131
|
+
*
|
|
132
|
+
* ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
133
|
+
* USAGE
|
|
134
|
+
* node scripts/learning-replay.mjs # N=3 replay, real tokens, writes the artifact
|
|
135
|
+
* node scripts/learning-replay.mjs --n 1 # one run
|
|
136
|
+
* node scripts/learning-replay.mjs --check # NO tokens: gate on the committed artifact
|
|
137
|
+
* node scripts/learning-replay.mjs --dry-run # NO tokens: build fixtures, prove the wire, UNKNOWN
|
|
138
|
+
* node scripts/learning-replay.mjs --mutant <name> # see MUTANTS below
|
|
139
|
+
* Exit: 0 = PASS. 1 = FAIL. 3 = INCONCLUSIVE. 4 = UNKNOWN. (Only 0 is a pass, by construction.)
|
|
140
|
+
*/
|
|
141
|
+
|
|
142
|
+
import fs from 'node:fs';
|
|
143
|
+
import os from 'node:os';
|
|
144
|
+
import path from 'node:path';
|
|
145
|
+
import { spawnSync } from 'node:child_process';
|
|
146
|
+
import { fileURLToPath } from 'node:url';
|
|
147
|
+
|
|
148
|
+
import { findInvocations } from '../plugin/scripts/hook-input.mjs';
|
|
149
|
+
import { makeLesson, saveLessons, loadLessons } from './lesson-store.mjs';
|
|
150
|
+
|
|
151
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
152
|
+
|
|
153
|
+
/** The invariant name the release-gate vector in scripts/claims-verify.mjs consumes. */
|
|
154
|
+
export const INVARIANT = 'LEARNING-REPLAY';
|
|
155
|
+
|
|
156
|
+
export const VERDICT = Object.freeze({
|
|
157
|
+
PASS: 'PASS',
|
|
158
|
+
FAIL: 'FAIL',
|
|
159
|
+
UNKNOWN: 'UNKNOWN',
|
|
160
|
+
INCONCLUSIVE: 'INCONCLUSIVE',
|
|
161
|
+
});
|
|
162
|
+
/** Exit codes. PASS is the ONLY zero — UNKNOWN is never a pass, and neither is INCONCLUSIVE. */
|
|
163
|
+
export const EXIT = Object.freeze({ PASS: 0, FAIL: 1, INCONCLUSIVE: 3, UNKNOWN: 4 });
|
|
164
|
+
|
|
165
|
+
/** The committed, machine-readable result artifact. */
|
|
166
|
+
export const RESULT_FILE = path.join(ROOT, 'data', 'learning-replay-result.json');
|
|
167
|
+
export const POST_TASK_RESULT_FILE = path.join(ROOT, 'data', 'learning-replay-post-task-result.json');
|
|
168
|
+
|
|
169
|
+
export const TRAP = Object.freeze({
|
|
170
|
+
MEMORY_SEARCH: 'memory-search-query',
|
|
171
|
+
POST_TASK: 'hooks-post-task-persistence',
|
|
172
|
+
});
|
|
173
|
+
|
|
174
|
+
export const POST_TASK_PROMPT =
|
|
175
|
+
'Record successful completion of the release retry-budget investigation in ruflo hooks learning, '
|
|
176
|
+
+ 'assign it to the tester agent, and persist the routing decision for later sessions. '
|
|
177
|
+
+ 'Use the terminal CLI, not an MCP tool. Do not inspect help or run any setup command. '
|
|
178
|
+
+ 'Run exactly one post-task command now, then tell me what you ran.';
|
|
179
|
+
export const POST_TASK_LESSON =
|
|
180
|
+
'A ruflo hooks post-task call persists a reusable routing outcome only when it includes all three: '
|
|
181
|
+
+ '--task with the task description, --agent with the executor, and --store-results; a success/task-id alone '
|
|
182
|
+
+ 'may print success but does not create the routing decision that later learning reads.';
|
|
183
|
+
|
|
184
|
+
function trapSpec(id = TRAP.MEMORY_SEARCH) {
|
|
185
|
+
if (id === TRAP.POST_TASK) {
|
|
186
|
+
return {
|
|
187
|
+
id,
|
|
188
|
+
lessonId: 'FX-D4-ruflo-hooks-post-task-persistence',
|
|
189
|
+
prompt: POST_TASK_PROMPT,
|
|
190
|
+
lesson: POST_TASK_LESSON,
|
|
191
|
+
memoryKey: 'lesson-ruflo-hooks-post-task-persistence',
|
|
192
|
+
recordQuery: 'ruflo hooks post task routing persistence',
|
|
193
|
+
check: 'the produced ruflo hooks post-task command includes --task, --agent, and --store-results',
|
|
194
|
+
};
|
|
195
|
+
}
|
|
196
|
+
return {
|
|
197
|
+
id: TRAP.MEMORY_SEARCH,
|
|
198
|
+
lessonId: 'FX-D4-ruflo-memory-search-flag',
|
|
199
|
+
prompt: REPLAY_PROMPT,
|
|
200
|
+
lesson: LESSON_STATEMENT,
|
|
201
|
+
memoryKey: 'lesson-ruflo-memory-search-flag',
|
|
202
|
+
recordQuery: 'ruflo CLI memory query flag',
|
|
203
|
+
check: 'the produced ruflo memory search command delivers its query through -q/--query',
|
|
204
|
+
};
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/**
|
|
208
|
+
* The files whose change invalidates a recorded result. `--check` refuses to call a result CURRENT
|
|
209
|
+
* for a SHA if any of these moved since — ADR-056's currency discipline, applied to a token-priced
|
|
210
|
+
* measurement that cannot be re-run on every commit.
|
|
211
|
+
*/
|
|
212
|
+
export const LOAD_BEARING = Object.freeze([
|
|
213
|
+
'scripts/learning-replay.mjs',
|
|
214
|
+
'scripts/ci/learning-replay-recorder.mjs',
|
|
215
|
+
'scripts/ci/learning-replay-codex-adapter.mjs',
|
|
216
|
+
'scripts/lesson-store.mjs',
|
|
217
|
+
'scripts/lesson-gate.mjs',
|
|
218
|
+
'plugin/scripts/lesson-hooks.sh',
|
|
219
|
+
'plugin/scripts/unprompted-runtime.mjs',
|
|
220
|
+
'plugin/scripts/hook-shim.mjs',
|
|
221
|
+
'plugin/scripts/hook-input.mjs',
|
|
222
|
+
]);
|
|
223
|
+
|
|
224
|
+
// ── THE ORACLE ──────────────────────────────────────────────────────────────────────────────────
|
|
225
|
+
/**
|
|
226
|
+
* Classify ONE produced command against the machine-checkable token.
|
|
227
|
+
*
|
|
228
|
+
* 'flagged' — a ruflo invocation that delivers its query through `-q` / `--query`.
|
|
229
|
+
* THIS IS THE TOKEN — and it is the token ADR-058 §D4 names, verbatim:
|
|
230
|
+
* "the produced command uses -q where the brain-off control uses the positional form".
|
|
231
|
+
* 'positional' — a ruflo invocation carrying a bare positional query and no -q/--query. The exact
|
|
232
|
+
* wrong form the lesson names.
|
|
233
|
+
* 'other' — ruflo invoked, but the query arrives some other way (`--topic`, `--project`), or
|
|
234
|
+
* no query at all.
|
|
235
|
+
* 'none' — no ruflo invocation at all.
|
|
236
|
+
*
|
|
237
|
+
* `--query` counts as the token even though the lesson says `-q`: the live `--help` prints them as
|
|
238
|
+
* ONE option (`-q, --query`), so failing the long form would make the oracle reject a command that is
|
|
239
|
+
* correct. An oracle stricter than the interface it models measures its own arbitrariness. The
|
|
240
|
+
* consequence is faced rather than tuned away — a control arm that reaches `--query` on its own
|
|
241
|
+
* INVALIDATES the trap, which is invariant 6 doing its job.
|
|
242
|
+
*
|
|
243
|
+
* ── THE SUBCOMMAND: REPORTED (2026-07-27) → GATED (2026-07-28) ───────────────────────────────────
|
|
244
|
+
* The first shipped oracle also required `ruflo memory search`; the first real N=3 measured treated
|
|
245
|
+
* 3/3 carrying `-q` against control 0/3 — a clean separation — and scored it 0/3 FAIL, because the
|
|
246
|
+
* treated arm spelled it `ruflo recall -q …` / `ruflo memory recall -q …`. That was read as a
|
|
247
|
+
* harness error (the lesson taught the flag and said nothing about the subcommand) and the gate was
|
|
248
|
+
* relaxed to observed-only.
|
|
249
|
+
*
|
|
250
|
+
* That relaxation is REVERSED, and the reversal is the point of the whole change. Both readings of
|
|
251
|
+
* the 2026-07-27 evidence are true at once, and only one of them is about the CLAIM:
|
|
252
|
+
* · about the LESSON — right. The lesson carries the flag; failing the treatment on a subcommand
|
|
253
|
+
* it never mentioned measures the model's priors about rUv's command tree.
|
|
254
|
+
* · about the CLAIM — wrong, and the grader caught it. The invariant's headline is that LEARNING
|
|
255
|
+
* WAS DEMONSTRATED. `ruflo recall -q "x"` exits 1. `ruflo memory recall -q "x"` exits 0 and
|
|
256
|
+
* prints the help. Certifying "learning demonstrated" on a command that retrieves nothing is
|
|
257
|
+
* the L4 defect rebuilt one file to the left — proof that something was SAID, not that anything
|
|
258
|
+
* WORKED.
|
|
259
|
+
*
|
|
260
|
+
* The honest resolution is to keep the token oracle exactly as narrow as it was (so nothing is
|
|
261
|
+
* credited to the lesson that the control reaches unaided) and to add the gate the claim actually
|
|
262
|
+
* needs: the command has to WORK. `subcommandCorrect` is now one of the conditions, and
|
|
263
|
+
* `aggregate()` carries an assertion making `subcommandCorrect: false` structurally unable to
|
|
264
|
+
* coexist with a PASS verdict — the same shape as the invariant-6 guard beside it. If the rate
|
|
265
|
+
* falls as a result, the rate was wrong before; the lesson text is NOT tuned to recover it.
|
|
266
|
+
*/
|
|
267
|
+
export function classifyCommand(cmd) {
|
|
268
|
+
const invocations = findInvocations(String(cmd || ''), ['ruflo', 'claude-flow']);
|
|
269
|
+
if (!invocations.length) return 'none';
|
|
270
|
+
let sawPositional = false;
|
|
271
|
+
for (const inv of invocations) {
|
|
272
|
+
const args = inv.args.filter((a) => a !== '');
|
|
273
|
+
if (args.some((a) => a === '-q' || a === '--query' || a.startsWith('--query='))) return 'flagged';
|
|
274
|
+
// Bare (non-flag, non-flag-value) tokens. A flag consumes the token after it unless that token
|
|
275
|
+
// is itself a flag — generic, so `--topic "x"` and `-n default` are handled without a whitelist
|
|
276
|
+
// that would rot the moment rUv adds an option.
|
|
277
|
+
const bare = [];
|
|
278
|
+
for (let i = 0; i < args.length; i++) {
|
|
279
|
+
const a = args[i];
|
|
280
|
+
if (a.startsWith('-')) { if (!a.includes('=') && args[i + 1] && !args[i + 1].startsWith('-')) i++; continue; }
|
|
281
|
+
bare.push(a);
|
|
282
|
+
}
|
|
283
|
+
// Which bare token is the QUERY rather than a subcommand? A subcommand is one short lowercase
|
|
284
|
+
// word; a query is a phrase. So: a bare token past the first that contains whitespace (or is
|
|
285
|
+
// implausibly long) is a positional query, as is any third bare token.
|
|
286
|
+
// ONLY THE LABEL DEPENDS ON THIS. The verdict keys on `flagged` vs not-`flagged` and on `none`;
|
|
287
|
+
// 'positional' and 'other' are both simply "did not carry the token". A mislabel here can never
|
|
288
|
+
// move PASS/FAIL/INCONCLUSIVE — it can only make the reported description of a control arm less
|
|
289
|
+
// precise, which is why a heuristic is acceptable HERE and nowhere near the token itself.
|
|
290
|
+
const queryish = (t) => /\s/.test(t) || t.length > 24;
|
|
291
|
+
if (bare.length >= 3 || bare.slice(1).some(queryish)) sawPositional = true;
|
|
292
|
+
}
|
|
293
|
+
return sawPositional ? 'positional' : 'other';
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
/** GATING since 2026-07-28: was the invocation the REAL `ruflo memory search`? */
|
|
297
|
+
export function subcommandCorrect(cmd) {
|
|
298
|
+
for (const inv of findInvocations(String(cmd || ''), ['ruflo', 'claude-flow'])) {
|
|
299
|
+
const words = inv.args.filter((a) => a !== '' && !a.startsWith('-'));
|
|
300
|
+
const mi = words.indexOf('memory');
|
|
301
|
+
if (mi !== -1 && words[mi + 1] === 'search') return true;
|
|
302
|
+
}
|
|
303
|
+
return false;
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
/** The token test, isolated so every caller asks it the same way. */
|
|
307
|
+
export const carriesToken = (cls) => cls === 'flagged';
|
|
308
|
+
|
|
309
|
+
/** The second trap is deliberately a different Ruflo surface and a different required option. */
|
|
310
|
+
function optionValue(args, short, long) {
|
|
311
|
+
for (let i = 0; i < args.length; i++) {
|
|
312
|
+
const arg = args[i];
|
|
313
|
+
if (arg === short || arg === long) return args[i + 1] && !args[i + 1].startsWith('-') ? args[i + 1] : null;
|
|
314
|
+
if (arg.startsWith(`${long}=`)) return arg.slice(long.length + 1);
|
|
315
|
+
}
|
|
316
|
+
return null;
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
export function classifyPostTaskCommand(cmd) {
|
|
320
|
+
const invocations = findInvocations(String(cmd || ''), ['ruflo', 'claude-flow']);
|
|
321
|
+
if (!invocations.length) return 'none';
|
|
322
|
+
let sawPostTask = false;
|
|
323
|
+
for (const inv of invocations) {
|
|
324
|
+
const args = inv.args.filter(Boolean);
|
|
325
|
+
const hi = args.indexOf('hooks');
|
|
326
|
+
if (hi === -1 || args[hi + 1] !== 'post-task') continue;
|
|
327
|
+
sawPostTask = true;
|
|
328
|
+
const task = optionValue(args, '-t', '--task');
|
|
329
|
+
const agent = optionValue(args, '-a', '--agent');
|
|
330
|
+
const store = args.includes('--store-results')
|
|
331
|
+
|| args.some((a) => a.startsWith('--store-results=') && !/=false$/i.test(a));
|
|
332
|
+
if (task && agent && store) return 'flagged';
|
|
333
|
+
}
|
|
334
|
+
return sawPostTask ? 'partial' : 'other';
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
export function postTaskSubcommandCorrect(cmd) {
|
|
338
|
+
return findInvocations(String(cmd || ''), ['ruflo', 'claude-flow'])
|
|
339
|
+
.some((inv) => {
|
|
340
|
+
const words = inv.args.filter((a) => a !== '' && !a.startsWith('-'));
|
|
341
|
+
const hi = words.indexOf('hooks');
|
|
342
|
+
return hi !== -1 && words[hi + 1] === 'post-task';
|
|
343
|
+
});
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
// ── THE EXECUTION GATE ──────────────────────────────────────────────────────────────────────────
|
|
347
|
+
/**
|
|
348
|
+
* The note project B's prompt already claimed was there. Seeding it makes the FIXTURE match the
|
|
349
|
+
* TASK; it does not touch the lesson (nothing here mentions `-q`) and no arm ever reads it, because
|
|
350
|
+
* the recorder blocks every command the agent produces before it can run.
|
|
351
|
+
*/
|
|
352
|
+
export const PROJECT_B_MEMORY_KEY = 'note-caching-strategy';
|
|
353
|
+
export const PROJECT_B_MEMORY_VALUE =
|
|
354
|
+
'The caching strategy for this project: responses are memoized in a two-tier LRU, '
|
|
355
|
+
+ 'warm tier in memory and cold tier on disk, invalidated by content hash.';
|
|
356
|
+
|
|
357
|
+
/**
|
|
358
|
+
* What "retrieved" looks like on the wire. Every one of these strings was READ OFF the real global
|
|
359
|
+
* binary's real output on 2026-07-28 (see the EXECUTION GATE note in the header), never guessed.
|
|
360
|
+
*
|
|
361
|
+
* The positive markers are chosen to survive the table truncation `ruflo memory search` applies:
|
|
362
|
+
* the real row prints as `| note-caching-stra... | 0.79 | default | The caching strategy for this
|
|
363
|
+
* pr... |`, so a 12-char key prefix and a 20-char value prefix are both intact. `memory retrieve -k`
|
|
364
|
+
* prints both in full.
|
|
365
|
+
*/
|
|
366
|
+
export const RETRIEVAL_EVIDENCE = Object.freeze({
|
|
367
|
+
positive: Object.freeze([PROJECT_B_MEMORY_KEY.slice(0, 12), PROJECT_B_MEMORY_VALUE.slice(0, 20)]),
|
|
368
|
+
negative: Object.freeze([
|
|
369
|
+
/No results found/i, // memory search -q "<absent>" → EXIT 0, and retrieved nothing
|
|
370
|
+
/Unknown command/i, // ruflo recall -q "x" → EXIT 1
|
|
371
|
+
/Required option missing/i, // memory search "positional" → EXIT 1
|
|
372
|
+
/Usage:\s*claude-flow memory/i, // memory recall -q "x" → EXIT 0, prints the help
|
|
373
|
+
/\[ERROR\]/,
|
|
374
|
+
]),
|
|
375
|
+
});
|
|
376
|
+
|
|
377
|
+
/**
|
|
378
|
+
* Did the command RETRIEVE, as opposed to merely exit 0? Asserted on returned content in both
|
|
379
|
+
* directions: any known failure shape is disqualifying even at exit 0, and silence is not evidence —
|
|
380
|
+
* the output must NAME the seeded memory.
|
|
381
|
+
*/
|
|
382
|
+
export function assertRetrieved(out) {
|
|
383
|
+
const s = String(out || '');
|
|
384
|
+
for (const re of RETRIEVAL_EVIDENCE.negative) {
|
|
385
|
+
if (re.test(s)) return { retrieved: false, why: `the command's own output matched a known FAILURE shape ${re}` };
|
|
386
|
+
}
|
|
387
|
+
const hit = RETRIEVAL_EVIDENCE.positive.find((p) => s.includes(p));
|
|
388
|
+
if (!hit) return { retrieved: false, why: 'the output names neither the seeded memory key nor its stored text — nothing was retrieved' };
|
|
389
|
+
return { retrieved: true, why: `the output carries the seeded memory (matched "${hit}")` };
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
/**
|
|
393
|
+
* Subcommands that WRITE. Checked only at subcommand position (the first two non-flag words), so a
|
|
394
|
+
* query that happens to contain one of these words is not mistaken for the verb.
|
|
395
|
+
*/
|
|
396
|
+
const MUTATING_SUBCOMMANDS = new Set(['store', 'delete', 'rm', 'purge', 'cleanup', 'compress', 'import', 'export', 'backup', 'init', 'configure']);
|
|
397
|
+
|
|
398
|
+
/** Execute the real CLI, or an injected JavaScript fixture, without a shell on every platform. */
|
|
399
|
+
function spawnRuflo(bin, args, options) {
|
|
400
|
+
if (/\.[cm]?js$/i.test(bin)) {
|
|
401
|
+
return spawnSync(process.execPath, [bin, ...args], options);
|
|
402
|
+
}
|
|
403
|
+
return spawnSync(bin, args, options);
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
export function assertPostTaskPersisted({ args, output, cwd }) {
|
|
407
|
+
const task = optionValue(args, '-t', '--task');
|
|
408
|
+
const agent = optionValue(args, '-a', '--agent');
|
|
409
|
+
const taskId = optionValue(args, '-i', '--task-id')
|
|
410
|
+
|| String(output || '').match(/Recording outcome for task:\s*([a-zA-Z0-9_-]+)/)?.[1]
|
|
411
|
+
|| null;
|
|
412
|
+
if (!task || !agent || !taskId || !args.includes('--store-results')) {
|
|
413
|
+
return { retrieved: false, why: 'the command did not carry --task, --agent, --store-results, and a resolvable task id' };
|
|
414
|
+
}
|
|
415
|
+
let outcomes;
|
|
416
|
+
let memory;
|
|
417
|
+
try {
|
|
418
|
+
outcomes = JSON.parse(fs.readFileSync(path.join(cwd, '.claude-flow', 'routing-outcomes.json'), 'utf8'));
|
|
419
|
+
memory = JSON.parse(fs.readFileSync(path.join(cwd, '.claude-flow', 'memory', 'store.json'), 'utf8'));
|
|
420
|
+
} catch (error) {
|
|
421
|
+
return { retrieved: false, why: `the expected persistence stores were not readable: ${error.message}` };
|
|
422
|
+
}
|
|
423
|
+
const outcome = (outcomes.outcomes || []).find((row) =>
|
|
424
|
+
row.task === task && row.agent === agent && row.success === true);
|
|
425
|
+
const decision = memory.entries?.[`routing-decision:${taskId}`];
|
|
426
|
+
let decisionValue = null;
|
|
427
|
+
try { decisionValue = decision ? JSON.parse(decision.value) : null; } catch { /* invalid evidence */ }
|
|
428
|
+
if (!outcome || !decision || decisionValue?.task !== task || decisionValue?.agent !== agent) {
|
|
429
|
+
return { retrieved: false, why: 'stdout said success, but no matching routing outcome plus routing-decision memory row persisted' };
|
|
430
|
+
}
|
|
431
|
+
if (!/\[OK\]\s*Task outcome recorded:\s*SUCCESS/i.test(String(output || ''))) {
|
|
432
|
+
return { retrieved: false, why: 'persistence rows exist but this invocation did not report successful task recording' };
|
|
433
|
+
}
|
|
434
|
+
return {
|
|
435
|
+
retrieved: true,
|
|
436
|
+
why: `matching routing outcome and routing-decision:${taskId} memory row persisted`,
|
|
437
|
+
};
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
/**
|
|
441
|
+
* RUN the produced command and report what actually happened.
|
|
442
|
+
*
|
|
443
|
+
* Never through a shell: the argv is the one `findInvocations()` already parsed out of the agent's
|
|
444
|
+
* string, so shell metacharacters cannot survive into execution. Two refusals bound the blast
|
|
445
|
+
* radius, and both report NOT-RETRIEVED, so neither can ever raise the rate.
|
|
446
|
+
*/
|
|
447
|
+
export function executeProducedCommand(cmd, {
|
|
448
|
+
cwd,
|
|
449
|
+
ruflo = RUFLO_BIN,
|
|
450
|
+
base = null,
|
|
451
|
+
trap = TRAP.MEMORY_SEARCH,
|
|
452
|
+
} = {}) {
|
|
453
|
+
const nope = (why, extra = {}) => ({ ran: false, argv: null, exit: null, exitOk: false, retrieved: false, why, output: '', ...extra });
|
|
454
|
+
const invocations = findInvocations(String(cmd || ''), ['ruflo', 'claude-flow']);
|
|
455
|
+
if (!invocations.length) return nope('no ruflo invocation in the produced command — there was nothing to execute');
|
|
456
|
+
const args = invocations[0].args.filter((a) => a !== '');
|
|
457
|
+
|
|
458
|
+
for (let i = 0; i < args.length; i++) {
|
|
459
|
+
const a = args[i];
|
|
460
|
+
if (a === '--path' || a === '--db' || a.startsWith('--path=') || a.startsWith('--db=')) {
|
|
461
|
+
const raw = a.includes('=') ? a.slice(a.indexOf('=') + 1) : args[i + 1];
|
|
462
|
+
const abs = path.resolve(cwd, String(raw || ''));
|
|
463
|
+
if (!base || !(abs === path.resolve(base) || abs.startsWith(path.resolve(base) + path.sep))) {
|
|
464
|
+
return nope(`refused to execute: the produced command points its store at ${abs}, outside the fixture world`, { argv: ['ruflo', ...args] });
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
const verbs = args.filter((a) => !a.startsWith('-')).slice(0, 2);
|
|
469
|
+
const mutating = verbs.find((w) => MUTATING_SUBCOMMANDS.has(w));
|
|
470
|
+
if (mutating) return nope(`refused to execute a MUTATING ruflo subcommand ("${mutating}") — a retrieval claim is not proven by a write`, { argv: ['ruflo', ...args] });
|
|
471
|
+
|
|
472
|
+
const env = { ...process.env };
|
|
473
|
+
delete env.CLAUDE_FLOW_DB_PATH;
|
|
474
|
+
delete env.CLAUDE_FLOW_MEMORY_PATH;
|
|
475
|
+
const r = spawnRuflo(ruflo, args, { cwd, encoding: 'utf8', timeout: 120_000, env, maxBuffer: 8 * 1024 * 1024 });
|
|
476
|
+
if (r.error) return nope(`spawn failed: ${r.error.message}`, { argv: ['ruflo', ...args] });
|
|
477
|
+
const out = `${r.stdout || ''}${r.stderr || ''}`;
|
|
478
|
+
const routed = trap === TRAP.POST_TASK
|
|
479
|
+
? assertPostTaskPersisted({ args, output: out, cwd })
|
|
480
|
+
: assertRetrieved(out);
|
|
481
|
+
return {
|
|
482
|
+
ran: true,
|
|
483
|
+
argv: ['ruflo', ...args],
|
|
484
|
+
exit: r.status,
|
|
485
|
+
exitOk: r.status === 0,
|
|
486
|
+
retrieved: routed.retrieved,
|
|
487
|
+
why: `exit ${r.status}; ${routed.why}`,
|
|
488
|
+
output: out.slice(0, 1200),
|
|
489
|
+
};
|
|
490
|
+
}
|
|
491
|
+
|
|
492
|
+
/**
|
|
493
|
+
* ONE run's verdict. Order of the branches IS the invariant: the control is judged BEFORE the
|
|
494
|
+
* treated arm can be credited with anything.
|
|
495
|
+
*/
|
|
496
|
+
export function verdictForRun(run) {
|
|
497
|
+
const {
|
|
498
|
+
treatedClass, controlClass, lessonBeforeFirstToolCall, error,
|
|
499
|
+
treatedSubcommandCorrect, treatedExecOk, treatedRetrieved, treatedExecWhy,
|
|
500
|
+
controlWorked,
|
|
501
|
+
} = run;
|
|
502
|
+
if (error) return { verdict: VERDICT.UNKNOWN, why: `harness could not measure this run: ${error}` };
|
|
503
|
+
// NO COMPARABLE CONTROL ARTIFACT IS NOT A WIN. If the control never invoked ruflo at all, there is
|
|
504
|
+
// no counterfactual to difference against — the treated arm may have "changed" against nothing.
|
|
505
|
+
// Deliberately strict, and it can only ever LOWER the rate: an unopposed treated arm is UNKNOWN.
|
|
506
|
+
if (controlClass === 'none') {
|
|
507
|
+
return { verdict: VERDICT.UNKNOWN, why: 'the control arm produced no ruflo invocation at all — there is no comparable artifact to difference against' };
|
|
508
|
+
}
|
|
509
|
+
if (treatedClass === 'none') {
|
|
510
|
+
return { verdict: VERDICT.FAIL, why: 'the treated arm produced no ruflo invocation at all' };
|
|
511
|
+
}
|
|
512
|
+
// INVARIANT 6, FIRST AND UNCONDITIONALLY. WIDENED 2026-07-28, never narrowed: carrying the token
|
|
513
|
+
// still invalidates on its own (that bar is unchanged, so nothing the control reaches unaided can
|
|
514
|
+
// start being credited to the lesson), and a control whose command WORKED — executed and retrieved
|
|
515
|
+
// — invalidates too, even by a route the classifier does not call `flagged`. An OR can only make
|
|
516
|
+
// more runs invalid; it can never turn an invalid run into a pass.
|
|
517
|
+
if (carriesToken(controlClass) || controlWorked === true) {
|
|
518
|
+
return {
|
|
519
|
+
verdict: VERDICT.INCONCLUSIVE,
|
|
520
|
+
why: carriesToken(controlClass)
|
|
521
|
+
? `the CONTROL arm produced the token (${controlClass}) — the model would have got it right without the lesson, so this trap measured nothing`
|
|
522
|
+
: `the CONTROL arm's command EXECUTED AND RETRIEVED (class "${controlClass}") — the model would have got it right without the lesson, so this trap measured nothing`,
|
|
523
|
+
};
|
|
524
|
+
}
|
|
525
|
+
if (!carriesToken(treatedClass)) {
|
|
526
|
+
return { verdict: VERDICT.FAIL, why: `treated arm produced "${treatedClass}", not the token` };
|
|
527
|
+
}
|
|
528
|
+
// ── THE EXECUTION GATE (2026-07-28) ──
|
|
529
|
+
// Ordered cheapest-to-most-informative so the `why` names the FIRST thing that was wrong.
|
|
530
|
+
if (treatedSubcommandCorrect !== true) {
|
|
531
|
+
return { verdict: VERDICT.FAIL, why: 'treated arm carried the token on the WRONG SUBCOMMAND — a right flag on a command that is not `ruflo memory search` is not learned behavior, it is an unusable command' };
|
|
532
|
+
}
|
|
533
|
+
if (treatedExecOk !== true) {
|
|
534
|
+
return { verdict: VERDICT.FAIL, why: `treated arm's produced command did not execute successfully: ${treatedExecWhy || 'not executed'}` };
|
|
535
|
+
}
|
|
536
|
+
if (treatedRetrieved !== true) {
|
|
537
|
+
return { verdict: VERDICT.FAIL, why: `treated arm's command exited 0 but RETRIEVED NOTHING: ${treatedExecWhy || 'no retrieval evidence'}` };
|
|
538
|
+
}
|
|
539
|
+
if (lessonBeforeFirstToolCall !== true) {
|
|
540
|
+
return { verdict: VERDICT.FAIL, why: 'treated arm carried the token but the lesson was NOT observed in the transcript before the first tool call' };
|
|
541
|
+
}
|
|
542
|
+
return { verdict: VERDICT.PASS, why: `treated "${treatedClass}" vs control "${controlClass}"; the produced command executed (exit 0) and returned the required meaningful outcome; lesson delivered before the first tool call` };
|
|
543
|
+
}
|
|
544
|
+
|
|
545
|
+
/**
|
|
546
|
+
* The RATE. N runs in, one verdict + a k/n out.
|
|
547
|
+
*
|
|
548
|
+
* PASS is structurally unreachable while any control succeeded: `controlTokenRuns` is computed
|
|
549
|
+
* before the branch and the assertion at the bottom re-checks it. Removing either guard and running
|
|
550
|
+
* the `seed-control` mutant is how you prove this is real rather than decorative.
|
|
551
|
+
*/
|
|
552
|
+
export function aggregate(runs, { threshold = 2 / 3 } = {}) {
|
|
553
|
+
const perRun = runs.map((r) => ({ ...r, ...verdictForRun(r) }));
|
|
554
|
+
const n = perRun.length;
|
|
555
|
+
const passes = perRun.filter((r) => r.verdict === VERDICT.PASS).length;
|
|
556
|
+
const fails = perRun.filter((r) => r.verdict === VERDICT.FAIL).length;
|
|
557
|
+
const unknowns = perRun.filter((r) => r.verdict === VERDICT.UNKNOWN).length;
|
|
558
|
+
const controlTokenRuns = perRun.filter((r) => carriesToken(r.controlClass)).length;
|
|
559
|
+
const controlWorkedRuns = perRun.filter((r) => r.controlWorked === true).length;
|
|
560
|
+
// The EFFECT SIZE, reported even when the verdict is INCONCLUSIVE. An invalid trap still measured
|
|
561
|
+
// two real rates, and printing only `passes` throws away the more informative half: "treated 3/3,
|
|
562
|
+
// control 1/3" says something a bare "2/3 below the bar" does not. This is a report, never an
|
|
563
|
+
// input to the verdict — the verdict stays governed by invariant 6 above.
|
|
564
|
+
const treatedTokenRuns = perRun.filter((r) => carriesToken(r.treatedClass)).length;
|
|
565
|
+
// The execution gate's own rates, reported whatever the verdict. "treated 3/3 carried the token,
|
|
566
|
+
// 0/3 of them worked" is the sentence the old artifact could not say, and it is the sentence the
|
|
567
|
+
// grader had to reconstruct by hand from `subcommandCorrect: false`.
|
|
568
|
+
const treatedSubcommandRuns = perRun.filter((r) => r.treatedSubcommandCorrect === true).length;
|
|
569
|
+
const treatedExecutedRuns = perRun.filter((r) => r.treatedExecOk === true).length;
|
|
570
|
+
const treatedRetrievedRuns = perRun.filter((r) => r.treatedRetrieved === true).length;
|
|
571
|
+
|
|
572
|
+
let verdict, why;
|
|
573
|
+
if (n === 0) {
|
|
574
|
+
verdict = VERDICT.UNKNOWN; why = 'zero runs executed — an empty run is not a pass';
|
|
575
|
+
} else if (controlTokenRuns > 0 || controlWorkedRuns > 0) {
|
|
576
|
+
verdict = VERDICT.INCONCLUSIVE;
|
|
577
|
+
why = `${controlTokenRuns}/${n} CONTROL run(s) produced the token and ${controlWorkedRuns}/${n} executed+retrieved — DDD-0013 invariant 6: the trap is INVALID, not passed`;
|
|
578
|
+
} else if (passes / n >= threshold) {
|
|
579
|
+
verdict = VERDICT.PASS;
|
|
580
|
+
why = `${passes}/${n} runs passed (bar ${Math.ceil(threshold * n)}/${n})`;
|
|
581
|
+
} else if (unknowns > 0 && passes + fails < n) {
|
|
582
|
+
verdict = VERDICT.UNKNOWN;
|
|
583
|
+
const executorError = perRun.map((r) => r.error).find(Boolean);
|
|
584
|
+
why = executorError
|
|
585
|
+
? `${unknowns}/${n} run(s) could not be measured; executor error: ${executorError}`
|
|
586
|
+
: `${unknowns}/${n} run(s) could not be measured; ${passes}/${n} passed — below the bar with the reason unknown`;
|
|
587
|
+
} else {
|
|
588
|
+
verdict = VERDICT.FAIL;
|
|
589
|
+
why = `${passes}/${n} runs passed — below the ${Math.ceil(threshold * n)}/${n} bar`;
|
|
590
|
+
}
|
|
591
|
+
|
|
592
|
+
// The guards that make the two invariants CODE rather than prose. If either throws, the branch
|
|
593
|
+
// order above was edited and the trap is unsafe — a stop-the-line event, not a warning.
|
|
594
|
+
if (verdict === VERDICT.PASS && (controlTokenRuns > 0 || controlWorkedRuns > 0)) {
|
|
595
|
+
throw new Error('LEARNING-REPLAY: refusing to report PASS while a control arm produced the token or a working command (DDD-0013 invariant 6)');
|
|
596
|
+
}
|
|
597
|
+
// The grader's finding, made structurally impossible: `subcommandCorrect: false` can no longer sit
|
|
598
|
+
// inside a PASS. Same for a command that did not execute or retrieved nothing.
|
|
599
|
+
const brokenPass = perRun.find((r) => r.verdict === VERDICT.PASS
|
|
600
|
+
&& (r.treatedSubcommandCorrect !== true || r.treatedExecOk !== true || r.treatedRetrieved !== true));
|
|
601
|
+
if (brokenPass) {
|
|
602
|
+
throw new Error(`LEARNING-REPLAY: refusing to report a PASS run whose produced command was unusable (run ${brokenPass.i}: subcommandCorrect=${brokenPass.treatedSubcommandCorrect}, exitOk=${brokenPass.treatedExecOk}, retrieved=${brokenPass.treatedRetrieved})`);
|
|
603
|
+
}
|
|
604
|
+
return {
|
|
605
|
+
verdict, why, n, passes, fails, unknowns,
|
|
606
|
+
controlTokenRuns, controlWorkedRuns, treatedTokenRuns,
|
|
607
|
+
treatedSubcommandRuns, treatedExecutedRuns, treatedRetrievedRuns,
|
|
608
|
+
rate: n ? +(passes / n).toFixed(4) : 0, runs: perRun,
|
|
609
|
+
};
|
|
610
|
+
}
|
|
611
|
+
|
|
612
|
+
// ── the real CLI's real interface, re-verified at run time ──────────────────────────────────────
|
|
613
|
+
export const RUFLO_BIN = process.env.RUVNET_RUFLO_BIN || path.join(os.homedir(), '.npm-global', 'bin', 'ruflo');
|
|
614
|
+
|
|
615
|
+
/**
|
|
616
|
+
* Re-verify the premise. Rule 0 applied to the one fact the whole oracle rests on: `ruflo memory
|
|
617
|
+
* search` must still take `-q/--query` and must still mark it REQUIRED. If rUv changes the
|
|
618
|
+
* interface, the honest outcome is UNKNOWN and a loud line — never a silent pass against a lesson
|
|
619
|
+
* that is no longer true.
|
|
620
|
+
*/
|
|
621
|
+
export function verifyRufloFlag(bin = RUFLO_BIN) {
|
|
622
|
+
if (!fs.existsSync(bin)) return { ok: false, why: `ruflo binary not found at ${bin} (Rule 21: the GLOBAL binary, never npx)` };
|
|
623
|
+
const r = spawnSync(bin, ['memory', 'search', '--help'], { encoding: 'utf8', timeout: 30_000 });
|
|
624
|
+
const out = `${r.stdout || ''}${r.stderr || ''}`;
|
|
625
|
+
if (r.status !== 0 && !out) return { ok: false, why: `ruflo memory search --help exited ${r.status} with no output` };
|
|
626
|
+
const flag = /-q,\s*--query/.test(out);
|
|
627
|
+
const required = /--query[^\n]*required/i.test(out);
|
|
628
|
+
const positionalDocumented = /\bmemory search\s+"[^"]+"\s*$/m.test(out);
|
|
629
|
+
if (!flag) return { ok: false, why: 'live `ruflo memory search --help` no longer advertises `-q, --query` — the lesson this trap records is no longer true', help: out };
|
|
630
|
+
if (positionalDocumented) return { ok: false, why: 'live help now shows a POSITIONAL query example — the trap premise (positional is rejected) is broken', help: out };
|
|
631
|
+
return { ok: true, flag: '-q, --query', required, evidence: out.split('\n').find((l) => /-q,\s*--query/.test(l))?.trim() || '' };
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
export function verifyPostTaskContract(bin = RUFLO_BIN) {
|
|
635
|
+
if (!fs.existsSync(bin)) return { ok: false, why: `ruflo binary not found at ${bin} (Rule 21: the GLOBAL binary, never npx)` };
|
|
636
|
+
const help = spawnRuflo(bin, ['hooks', 'post-task', '--help'], { encoding: 'utf8', timeout: 30_000 });
|
|
637
|
+
const out = `${help.stdout || ''}${help.stderr || ''}`;
|
|
638
|
+
if (!/--task\b/.test(out) || !/--agent\b/.test(out) || !/--store-results\b/.test(out)
|
|
639
|
+
|| !/Without this \+ --agent, no routing outcome is recorded/.test(out)) {
|
|
640
|
+
return { ok: false, why: 'live post-task help no longer states the three-part routing-persistence contract', help: out };
|
|
641
|
+
}
|
|
642
|
+
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'd4-post-task-premise-'));
|
|
643
|
+
const missing = spawnRuflo(bin, ['hooks', 'post-task', '-i', 'd4-premise-missing', '--success', 'true'], {
|
|
644
|
+
cwd: dir,
|
|
645
|
+
encoding: 'utf8',
|
|
646
|
+
timeout: 30_000,
|
|
647
|
+
});
|
|
648
|
+
const outcomeFile = path.join(dir, '.claude-flow', 'routing-outcomes.json');
|
|
649
|
+
const memoryFile = path.join(dir, '.claude-flow', 'memory', 'store.json');
|
|
650
|
+
const persisted = fs.existsSync(outcomeFile) || fs.existsSync(memoryFile);
|
|
651
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
652
|
+
if (missing.status !== 0 || persisted) {
|
|
653
|
+
return {
|
|
654
|
+
ok: false,
|
|
655
|
+
why: 'live post-task missing-contract probe did not stay non-persistent while returning success',
|
|
656
|
+
missingExit: missing.status,
|
|
657
|
+
persisted,
|
|
658
|
+
};
|
|
659
|
+
}
|
|
660
|
+
return {
|
|
661
|
+
ok: true,
|
|
662
|
+
flag: '--task + --agent + --store-results',
|
|
663
|
+
required: true,
|
|
664
|
+
evidence: 'live help names all three flags; success/task-id-only probe exited 0 and created neither routing outcome nor routing-decision store',
|
|
665
|
+
missingExit: missing.status,
|
|
666
|
+
};
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
// ── the fixture world ───────────────────────────────────────────────────────────────────────────
|
|
670
|
+
const CLAUDE_BIN = process.env.RUVNET_CLAUDE_BIN || path.join(os.homedir(), '.npm-global', 'bin', 'claude');
|
|
671
|
+
const CODEX_BIN = process.env.RUVNET_CODEX_BIN || 'codex';
|
|
672
|
+
|
|
673
|
+
/** Project B's task. Shares no content word with the lesson — the lesson cannot be string-matched into it. */
|
|
674
|
+
export const REPLAY_PROMPT =
|
|
675
|
+
'Earlier in this project someone recorded a note about the caching strategy. '
|
|
676
|
+
+ "Recall it from this project's agent memory with the ruflo CLI. "
|
|
677
|
+
+ 'Run the recall command now, then tell me what you ran.';
|
|
678
|
+
|
|
679
|
+
/** The correction as it is written down in fixture-project-A, in project A's own words. */
|
|
680
|
+
export const LESSON_STATEMENT =
|
|
681
|
+
'When you look something up in agent memory with the ruflo CLI, the query has to be passed with the '
|
|
682
|
+
+ '-q flag; a bare quoted phrase placed after the subcommand is rejected.';
|
|
683
|
+
|
|
684
|
+
const sh = (cmd, args, opts = {}) => spawnSync(cmd, args, { encoding: 'utf8', timeout: 120_000, ...opts });
|
|
685
|
+
|
|
686
|
+
function rmrf(p) { try { fs.rmSync(p, { recursive: true, force: true }); } catch { /* nothing to remove */ } }
|
|
687
|
+
|
|
688
|
+
/** Allocate one collision-proof fixture root; Date.now() alone collides under parallel CI. */
|
|
689
|
+
export function allocateRunBase(root = path.join(ROOT, '.ruvnet-brain', 'learning-replay')) {
|
|
690
|
+
fs.mkdirSync(root, { recursive: true });
|
|
691
|
+
return fs.mkdtempSync(path.join(root, 'run-'));
|
|
692
|
+
}
|
|
693
|
+
|
|
694
|
+
function initMemoryDb(ruflo, db, cwd) {
|
|
695
|
+
return sh(ruflo, ['memory', 'init', '--path', db, '--backend', 'hybrid'], { cwd });
|
|
696
|
+
}
|
|
697
|
+
|
|
698
|
+
/**
|
|
699
|
+
* Build the two fixture projects and the isolated brain world.
|
|
700
|
+
*
|
|
701
|
+
* Everything the product reads is redirected by env — RUVNET_BRAIN_HOME (spine), RUVNET_BRAIN_STATE_DIR
|
|
702
|
+
* (the on/off sentinel), RUVNET_LESSON_STORE, RUVNET_LESSON_GATE_STATE. Nothing here touches the
|
|
703
|
+
* user's real ~/.config/ruvnet-brain, ~/.cache/ruvnet-brain, or any real project's memory.
|
|
704
|
+
*/
|
|
705
|
+
export function buildFixtures(baseDir) {
|
|
706
|
+
rmrf(baseDir);
|
|
707
|
+
const dirs = {
|
|
708
|
+
base: baseDir,
|
|
709
|
+
projectA: path.join(baseDir, 'fixture-project-a'),
|
|
710
|
+
projectA2: path.join(baseDir, 'fixture-project-a-independent'),
|
|
711
|
+
projectB: path.join(baseDir, 'fixture-project-b'),
|
|
712
|
+
brainHome: path.join(baseDir, 'brain-home'),
|
|
713
|
+
stateOn: path.join(baseDir, 'state-on'),
|
|
714
|
+
stateOff: path.join(baseDir, 'state-off'),
|
|
715
|
+
transcripts: path.join(baseDir, 'transcripts'),
|
|
716
|
+
};
|
|
717
|
+
for (const d of Object.values(dirs)) fs.mkdirSync(d, { recursive: true });
|
|
718
|
+
dirs.lessons = path.join(baseDir, 'lessons.json');
|
|
719
|
+
dirs.gateState = path.join(baseDir, 'lesson-gate-state.json');
|
|
720
|
+
// The control's switch: ADR-054's real sentinel, in the control's own state dir.
|
|
721
|
+
fs.writeFileSync(path.join(dirs.stateOff, 'brain-off'), JSON.stringify({ since: new Date().toISOString() }));
|
|
722
|
+
// Each fixture project is its own git repo so lesson-gate's project-scope resolution (which walks
|
|
723
|
+
// up to the nearest .git) sees `fixture-project-b`, not the harness's own repo.
|
|
724
|
+
for (const p of [dirs.projectA, dirs.projectA2, dirs.projectB]) {
|
|
725
|
+
sh('git', ['init', '-q'], { cwd: p });
|
|
726
|
+
fs.mkdirSync(path.join(p, '.swarm'), { recursive: true });
|
|
727
|
+
}
|
|
728
|
+
return dirs;
|
|
729
|
+
}
|
|
730
|
+
|
|
731
|
+
/**
|
|
732
|
+
* RECORD, in fixture-project-A.
|
|
733
|
+
*
|
|
734
|
+
* Two layers, both real:
|
|
735
|
+
* 1. the correction is written into project A's OWN memory with `ruflo memory store` — the real
|
|
736
|
+
* CLI, the real per-project `.swarm/memory.db` the global memory policy mandates. It is then
|
|
737
|
+
* READ BACK with `ruflo memory search -q` (the very flag under test, so the record step itself
|
|
738
|
+
* exercises the true interface), and the retrieved text is what the lesson is built from. The
|
|
739
|
+
* lesson is DERIVED from project A, not hardcoded beside it.
|
|
740
|
+
* 2. the derived lesson is written into the machine-global lesson store the gate actually reads.
|
|
741
|
+
*
|
|
742
|
+
* SCOPE is the real ADR-029 rule, not a fixture bypass. The same correction is independently stored
|
|
743
|
+
* and read back in TWO distinct git projects. The executable lesson carries both project names, so
|
|
744
|
+
* lesson-gate.mjs's `projects.length >= 2` universal predicate is what permits it to speak in the
|
|
745
|
+
* third replay project. One source would correctly be silent there. The committed artifact records
|
|
746
|
+
* the two source identities and `checkPortfolio()` refuses a result without that win-twice proof.
|
|
747
|
+
*/
|
|
748
|
+
export function recordInProjectA(dirs, { ruflo = RUFLO_BIN, trap = TRAP.MEMORY_SEARCH } = {}) {
|
|
749
|
+
const spec = trapSpec(trap);
|
|
750
|
+
const sources = [dirs.projectA, dirs.projectA2].map((project, index) => {
|
|
751
|
+
const db = path.join(project, '.swarm', 'memory.db');
|
|
752
|
+
const key = `${spec.memoryKey}-${index + 1}`;
|
|
753
|
+
const init = initMemoryDb(ruflo, db, project);
|
|
754
|
+
const store = sh(ruflo, ['memory', 'store', '-k', key, '--value', spec.lesson, '-n', 'default', '--path', db],
|
|
755
|
+
{ cwd: project });
|
|
756
|
+
const back = sh(ruflo, ['memory', 'search', '-q', spec.recordQuery, '-n', 'default', '--path', db, '-t', 'keyword'],
|
|
757
|
+
{ cwd: project });
|
|
758
|
+
return {
|
|
759
|
+
project: path.basename(project),
|
|
760
|
+
db,
|
|
761
|
+
key,
|
|
762
|
+
initExit: init.status,
|
|
763
|
+
storeExit: store.status,
|
|
764
|
+
readBackExit: back.status,
|
|
765
|
+
recorded: fs.existsSync(db),
|
|
766
|
+
};
|
|
767
|
+
});
|
|
768
|
+
|
|
769
|
+
const lesson = makeLesson({
|
|
770
|
+
id: spec.lessonId,
|
|
771
|
+
statement: spec.lesson,
|
|
772
|
+
// `assert-fact` is the decision point the real dispatcher requests at UserPromptSubmit
|
|
773
|
+
// (plugin/scripts/lesson-hooks.sh) — i.e. before any tool call, which is PASS-condition (a).
|
|
774
|
+
trigger: 'assert-fact',
|
|
775
|
+
enforcement: 'checklist',
|
|
776
|
+
origin: 'user-stated',
|
|
777
|
+
status: 'ratified',
|
|
778
|
+
severity: 'high',
|
|
779
|
+
repeatCount: 4,
|
|
780
|
+
projects: sources.map((source) => source.project),
|
|
781
|
+
check: spec.check,
|
|
782
|
+
evidence: [
|
|
783
|
+
...sources.map((source) => ({
|
|
784
|
+
observed: `independently recorded in ${source.project} as memory key "${source.key}" in ${path.relative(dirs.base, source.db)}`,
|
|
785
|
+
})),
|
|
786
|
+
{ observed: `live premise for ${spec.id} was re-verified before replay` },
|
|
787
|
+
],
|
|
788
|
+
});
|
|
789
|
+
saveLessons([lesson], dirs.lessons);
|
|
790
|
+
const sourcesOk = sources.every((source) =>
|
|
791
|
+
source.initExit === 0 && source.storeExit === 0 && source.readBackExit === 0 && source.recorded);
|
|
792
|
+
return {
|
|
793
|
+
ok: sourcesOk && loadLessons(dirs.lessons).length === 1 && lesson.projects.length >= 2,
|
|
794
|
+
trap: spec.id,
|
|
795
|
+
sources,
|
|
796
|
+
projectCount: lesson.projects.length,
|
|
797
|
+
promoted: lesson.projects.length >= 2,
|
|
798
|
+
key: sources[0].key,
|
|
799
|
+
storeExit: sources[0].storeExit,
|
|
800
|
+
readBackExit: sources[0].readBackExit,
|
|
801
|
+
lesson,
|
|
802
|
+
};
|
|
803
|
+
}
|
|
804
|
+
|
|
805
|
+
/**
|
|
806
|
+
* SEED PROJECT B — make the fixture world true.
|
|
807
|
+
*
|
|
808
|
+
* REPLAY_PROMPT tells the agent "earlier in this project someone recorded a note about the caching
|
|
809
|
+
* strategy". Until 2026-07-28 that was false: project B's store was empty, so no command the agent
|
|
810
|
+
* could possibly write would retrieve anything, and the execution gate would be measuring the
|
|
811
|
+
* harness. The note is written with the real CLI into project B's own `.swarm/memory.db` — the
|
|
812
|
+
* default path a bare `ruflo memory search` resolves from cwd.
|
|
813
|
+
*
|
|
814
|
+
* It cannot leak the lesson: the note's text says nothing about `-q`, and neither arm ever sees it
|
|
815
|
+
* (the recorder blocks every produced command before it runs). It is read only by the harness,
|
|
816
|
+
* out of band, after the arm is finished.
|
|
817
|
+
*/
|
|
818
|
+
export function seedProjectBMemory(dirs, { ruflo = RUFLO_BIN } = {}) {
|
|
819
|
+
const dbB = path.join(dirs.projectB, '.swarm', 'memory.db');
|
|
820
|
+
const init = initMemoryDb(ruflo, dbB, dirs.projectB);
|
|
821
|
+
const r = sh(ruflo, ['memory', 'store', '-k', PROJECT_B_MEMORY_KEY, '--value', PROJECT_B_MEMORY_VALUE, '-n', 'default', '--path', dbB],
|
|
822
|
+
{ cwd: dirs.projectB });
|
|
823
|
+
return {
|
|
824
|
+
db: dbB,
|
|
825
|
+
key: PROJECT_B_MEMORY_KEY,
|
|
826
|
+
initExit: init.status,
|
|
827
|
+
storeExit: r.status,
|
|
828
|
+
ok: init.status === 0 && r.status === 0 && fs.existsSync(dbB),
|
|
829
|
+
};
|
|
830
|
+
}
|
|
831
|
+
|
|
832
|
+
/**
|
|
833
|
+
* THE NIGHTLY REFRESH, run BETWEEN record and replay. PASS-condition (c).
|
|
834
|
+
*
|
|
835
|
+
* Two real things, not a sleep:
|
|
836
|
+
* 1. a NEW Stable-Spine generation is installed into the fixture brain home and active.json is
|
|
837
|
+
* flipped to it — so the replay's hooks execute from a code root that did not exist when the
|
|
838
|
+
* lesson was recorded. This is exactly what scripts/update-apply.mjs does nightly, and it is the
|
|
839
|
+
* thing a lesson has to survive: the lesson store lives at user level, deliberately OUTSIDE the
|
|
840
|
+
* bundle a refresh replaces (scripts/lesson-store.mjs says so in its own persistence note).
|
|
841
|
+
* 2. `ruflo memory distill run` and `ruflo memory backup` against project A's store — the two
|
|
842
|
+
* commands scripts/nightly-wrapper.sh actually runs every night.
|
|
843
|
+
*/
|
|
844
|
+
export function nightlyRefresh(dirs, { ruflo = RUFLO_BIN } = {}) {
|
|
845
|
+
const gen = `d4-refresh-${Date.now()}`;
|
|
846
|
+
const versionDir = path.join(dirs.brainHome, 'versions', gen);
|
|
847
|
+
fs.mkdirSync(versionDir, { recursive: true });
|
|
848
|
+
fs.cpSync(path.join(ROOT, 'plugin'), versionDir, { recursive: true });
|
|
849
|
+
// Codex discovers the installed plugin's global hook manifest, not fixture-local `.codex` hooks.
|
|
850
|
+
// The stable wrapper resolves this fixture generation through RUVNET_BRAIN_HOME, so replace only
|
|
851
|
+
// the fixture generation's host adapter with the replay tap. The real hook body remains the copied
|
|
852
|
+
// hook-shim beside it; the adapter merely records delivery and blocks the first proposed command.
|
|
853
|
+
fs.copyFileSync(
|
|
854
|
+
path.join(ROOT, 'scripts', 'ci', 'learning-replay-codex-adapter.mjs'),
|
|
855
|
+
path.join(versionDir, 'scripts', 'codex-hook-adapter.mjs'),
|
|
856
|
+
);
|
|
857
|
+
fs.writeFileSync(path.join(dirs.brainHome, 'active.json'), JSON.stringify({ codeRoot: versionDir, generation: gen }, null, 2));
|
|
858
|
+
fs.writeFileSync(path.join(dirs.brainHome, '.spine-seeded'), gen);
|
|
859
|
+
|
|
860
|
+
const dbA = path.join(dirs.projectA, '.swarm', 'memory.db');
|
|
861
|
+
const distill = sh(ruflo, ['memory', 'distill', 'run', '--path', dbA], { cwd: dirs.projectA });
|
|
862
|
+
const backup = sh(ruflo, ['memory', 'backup', '--db', dbA, '--keep', '2'], { cwd: dirs.projectA });
|
|
863
|
+
|
|
864
|
+
const survived = loadLessons(dirs.lessons).length === 1;
|
|
865
|
+
// Repo-relative, never absolute: this artifact is COMMITTED, and an absolute path publishes the
|
|
866
|
+
// maintainer's directory layout to every reader. The same disclosure was already found and fixed
|
|
867
|
+
// once in session-start.sh; one bug, found once, must not be left everywhere else.
|
|
868
|
+
return { generation: gen, codeRoot: path.relative(ROOT, versionDir), distillExit: distill.status, backupExit: backup.status, lessonSurvived: survived };
|
|
869
|
+
}
|
|
870
|
+
|
|
871
|
+
/** The fixture settings file — the REAL hook registration from plugin/hooks/hooks.json, plus the tap. */
|
|
872
|
+
function writeSettings(file, { dirs, stateDir, attemptsFile }) {
|
|
873
|
+
const settings = {
|
|
874
|
+
env: {
|
|
875
|
+
RUVNET_BRAIN_HOME: dirs.brainHome,
|
|
876
|
+
RUVNET_BRAIN_STATE_DIR: stateDir,
|
|
877
|
+
RUVNET_LESSON_STORE: dirs.lessons,
|
|
878
|
+
RUVNET_LESSON_GATE_STATE: dirs.gateState,
|
|
879
|
+
CLAUDE_PLUGIN_ROOT: path.join(ROOT, 'plugin'),
|
|
880
|
+
},
|
|
881
|
+
hooks: {
|
|
882
|
+
UserPromptSubmit: [{
|
|
883
|
+
matcher: '*',
|
|
884
|
+
hooks: [{
|
|
885
|
+
type: 'command',
|
|
886
|
+
command: `node ${JSON.stringify(path.join(ROOT, 'plugin', 'scripts', 'hook-shim.mjs'))} unprompted-speech UserPromptSubmit`,
|
|
887
|
+
timeout: 20,
|
|
888
|
+
}],
|
|
889
|
+
}],
|
|
890
|
+
PreToolUse: [{
|
|
891
|
+
matcher: 'Bash',
|
|
892
|
+
hooks: [{
|
|
893
|
+
type: 'command',
|
|
894
|
+
command: `node ${JSON.stringify(path.join(ROOT, 'scripts', 'ci', 'learning-replay-recorder.mjs'))} ${JSON.stringify(attemptsFile)}`,
|
|
895
|
+
timeout: 20,
|
|
896
|
+
}],
|
|
897
|
+
}],
|
|
898
|
+
},
|
|
899
|
+
};
|
|
900
|
+
fs.writeFileSync(file, JSON.stringify(settings, null, 2));
|
|
901
|
+
return file;
|
|
902
|
+
}
|
|
903
|
+
|
|
904
|
+
export function buildCodexArgv({ model = 'gpt-5.6-sol', prompt = REPLAY_PROMPT, appendSystemPrompt = null } = {}) {
|
|
905
|
+
const fullPrompt = appendSystemPrompt ? `${appendSystemPrompt}\n\n${prompt}` : prompt;
|
|
906
|
+
return [
|
|
907
|
+
'exec',
|
|
908
|
+
'--ephemeral',
|
|
909
|
+
'--sandbox', 'read-only',
|
|
910
|
+
'--color', 'never',
|
|
911
|
+
'--json',
|
|
912
|
+
'--ignore-rules',
|
|
913
|
+
'--dangerously-bypass-hook-trust',
|
|
914
|
+
'-m', model,
|
|
915
|
+
'-c', 'model_reasoning_effort="low"',
|
|
916
|
+
'-c', 'shell_environment_policy.inherit="all"',
|
|
917
|
+
fullPrompt,
|
|
918
|
+
];
|
|
919
|
+
}
|
|
920
|
+
|
|
921
|
+
/**
|
|
922
|
+
* Run ONE arm and return what it produced.
|
|
923
|
+
*
|
|
924
|
+
* The transcript is stream-json with --include-hook-events, so hook delivery and tool calls appear
|
|
925
|
+
* IN ORDER in one array. `lessonIndex` and `firstToolIndex` are positions in that array — condition
|
|
926
|
+
* (a) is a measured ordering, not an argument from how hooks are supposed to work.
|
|
927
|
+
*/
|
|
928
|
+
export function replayRunError(events, processResult) {
|
|
929
|
+
if (processResult?.error) return String(processResult.error.message || processResult.error);
|
|
930
|
+
const result = events.find((e) => e.type === 'result');
|
|
931
|
+
if (!result?.is_error) return null;
|
|
932
|
+
const status = result.api_error_status ? `HTTP ${result.api_error_status}: ` : '';
|
|
933
|
+
return `${status}${result.result || result.terminal_reason || 'model execution failed'}`;
|
|
934
|
+
}
|
|
935
|
+
|
|
936
|
+
export function parseCodexRunError(events, processResult) {
|
|
937
|
+
if (processResult?.error) return String(processResult.error.message || processResult.error);
|
|
938
|
+
const failed = events.find((event) => event.type === 'turn.failed');
|
|
939
|
+
if (failed) return String(failed.error?.message || failed.error || 'Codex turn failed');
|
|
940
|
+
if (events.some((event) => event.type === 'turn.completed') && processResult?.status === 0) return null;
|
|
941
|
+
const errorItem = events.find((event) => event.type === 'item.completed' && event.item?.type === 'error');
|
|
942
|
+
if (errorItem) return String(errorItem.item?.message || errorItem.item?.text || 'Codex execution failed');
|
|
943
|
+
return processResult?.status && processResult.status !== 0
|
|
944
|
+
? `Codex exited ${processResult.status}`
|
|
945
|
+
: null;
|
|
946
|
+
}
|
|
947
|
+
|
|
948
|
+
export function codexLessonBeforeTool(sequence) {
|
|
949
|
+
const lesson = sequence.find((event) => event.kind === 'lesson');
|
|
950
|
+
const tool = sequence.find((event) => event.kind === 'tool');
|
|
951
|
+
return Boolean(lesson && tool && BigInt(lesson.atNs) < BigInt(tool.atNs));
|
|
952
|
+
}
|
|
953
|
+
|
|
954
|
+
export function runArm({
|
|
955
|
+
dirs,
|
|
956
|
+
arm,
|
|
957
|
+
stateDir,
|
|
958
|
+
model,
|
|
959
|
+
host = 'claude-code',
|
|
960
|
+
appendSystemPrompt = null,
|
|
961
|
+
tag,
|
|
962
|
+
forceCommand = null,
|
|
963
|
+
trap = TRAP.MEMORY_SEARCH,
|
|
964
|
+
}) {
|
|
965
|
+
const spec = trapSpec(trap);
|
|
966
|
+
const attempts = path.join(dirs.transcripts, `${tag}.attempts.jsonl`);
|
|
967
|
+
const sequenceFile = path.join(dirs.transcripts, `${tag}.sequence.jsonl`);
|
|
968
|
+
const streamFile = path.join(dirs.transcripts, `${tag}.stream.jsonl`);
|
|
969
|
+
let binary;
|
|
970
|
+
let argv;
|
|
971
|
+
if (host === 'codex') {
|
|
972
|
+
binary = CODEX_BIN;
|
|
973
|
+
argv = buildCodexArgv({ model, prompt: spec.prompt, appendSystemPrompt });
|
|
974
|
+
} else {
|
|
975
|
+
const settings = writeSettings(path.join(dirs.base, `settings-${tag}.json`), { dirs, stateDir, attemptsFile: attempts });
|
|
976
|
+
binary = CLAUDE_BIN;
|
|
977
|
+
argv = [
|
|
978
|
+
'-p', spec.prompt,
|
|
979
|
+
'--model', model,
|
|
980
|
+
'--tools', 'Bash',
|
|
981
|
+
'--permission-mode', 'bypassPermissions',
|
|
982
|
+
'--setting-sources', '',
|
|
983
|
+
'--settings', settings,
|
|
984
|
+
'--output-format', 'stream-json',
|
|
985
|
+
'--verbose',
|
|
986
|
+
'--include-hook-events',
|
|
987
|
+
'--no-session-persistence',
|
|
988
|
+
'--max-budget-usd', '0.30',
|
|
989
|
+
];
|
|
990
|
+
if (appendSystemPrompt) argv.push('--append-system-prompt', appendSystemPrompt);
|
|
991
|
+
}
|
|
992
|
+
|
|
993
|
+
const started = Date.now();
|
|
994
|
+
const r = spawnSync(binary, argv, {
|
|
995
|
+
cwd: dirs.projectB,
|
|
996
|
+
encoding: 'utf8',
|
|
997
|
+
timeout: 300_000,
|
|
998
|
+
maxBuffer: 64 * 1024 * 1024,
|
|
999
|
+
env: {
|
|
1000
|
+
...process.env,
|
|
1001
|
+
RUVNET_BRAIN_HOME: dirs.brainHome,
|
|
1002
|
+
RUVNET_BRAIN_STATE_DIR: stateDir,
|
|
1003
|
+
RUVNET_LESSON_STORE: dirs.lessons,
|
|
1004
|
+
RUVNET_LESSON_GATE_STATE: dirs.gateState,
|
|
1005
|
+
RUVNET_REPLAY_ATTEMPTS_FILE: attempts,
|
|
1006
|
+
RUVNET_REPLAY_SEQUENCE_FILE: sequenceFile,
|
|
1007
|
+
RUVNET_REPLAY_LESSON_PROBE: spec.lesson.slice(0, 60),
|
|
1008
|
+
RUVNET_REPLAY_RECORDER: path.join(ROOT, 'scripts', 'ci', 'learning-replay-recorder.mjs'),
|
|
1009
|
+
CLAUDE_PLUGIN_ROOT: path.join(ROOT, 'plugin'),
|
|
1010
|
+
},
|
|
1011
|
+
});
|
|
1012
|
+
const wallMs = Date.now() - started;
|
|
1013
|
+
fs.writeFileSync(streamFile, r.stdout || '');
|
|
1014
|
+
if (r.stderr) fs.writeFileSync(path.join(dirs.transcripts, `${tag}.stderr.txt`), r.stderr);
|
|
1015
|
+
|
|
1016
|
+
const events = (r.stdout || '').split('\n').filter(Boolean).map((l) => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
|
|
1017
|
+
|
|
1018
|
+
const probe = spec.lesson.slice(0, 60);
|
|
1019
|
+
let lessonIndex = -1, firstToolIndex = -1, lessonDelivered = false;
|
|
1020
|
+
const sequence = fs.existsSync(sequenceFile)
|
|
1021
|
+
? fs.readFileSync(sequenceFile, 'utf8').split('\n').filter(Boolean).map((line) => JSON.parse(line))
|
|
1022
|
+
: [];
|
|
1023
|
+
if (host === 'codex') {
|
|
1024
|
+
lessonIndex = sequence.findIndex((event) => event.kind === 'lesson');
|
|
1025
|
+
firstToolIndex = sequence.findIndex((event) => event.kind === 'tool');
|
|
1026
|
+
lessonDelivered = lessonIndex !== -1;
|
|
1027
|
+
} else {
|
|
1028
|
+
events.forEach((e, i) => {
|
|
1029
|
+
if (lessonIndex === -1 && e.type === 'system' && e.subtype === 'hook_response'
|
|
1030
|
+
&& typeof e.output === 'string' && e.output.includes(probe)) { lessonIndex = i; lessonDelivered = true; }
|
|
1031
|
+
if (firstToolIndex === -1 && e.type === 'assistant'
|
|
1032
|
+
&& Array.isArray(e.message?.content) && e.message.content.some((c) => c.type === 'tool_use')) firstToolIndex = i;
|
|
1033
|
+
});
|
|
1034
|
+
}
|
|
1035
|
+
|
|
1036
|
+
const attemptLines = fs.existsSync(attempts)
|
|
1037
|
+
? fs.readFileSync(attempts, 'utf8').split('\n').filter(Boolean).map((l) => JSON.parse(l))
|
|
1038
|
+
: [];
|
|
1039
|
+
// THE ARTIFACT is the FIRST command the agent produced — not its best one. A second attempt after
|
|
1040
|
+
// the sandbox refusal is a repair, and crediting a repair would let the agent learn the answer
|
|
1041
|
+
// from the harness instead of from the lesson.
|
|
1042
|
+
// MUTANT force-*: substitute the artifact the oracle sees, leaving the real run untouched. This is
|
|
1043
|
+
// how mutant 1 ("right flag, wrong subcommand") is proven end-to-end without waiting for a
|
|
1044
|
+
// stochastic model to happen to emit it.
|
|
1045
|
+
const firstCommand = forceCommand != null ? forceCommand : (attemptLines.length ? attemptLines[0].command : '');
|
|
1046
|
+
const cls = trap === TRAP.POST_TASK
|
|
1047
|
+
? classifyPostTaskCommand(firstCommand)
|
|
1048
|
+
: classifyCommand(firstCommand);
|
|
1049
|
+
const subOk = trap === TRAP.POST_TASK
|
|
1050
|
+
? postTaskSubcommandCorrect(firstCommand)
|
|
1051
|
+
: subcommandCorrect(firstCommand);
|
|
1052
|
+
// THE EXECUTION GATE. Out of band, after the arm is over, never through a shell — see the header.
|
|
1053
|
+
const exec = executeProducedCommand(firstCommand, { cwd: dirs.projectB, base: dirs.base, trap });
|
|
1054
|
+
|
|
1055
|
+
const result = events.find((e) => e.type === 'result');
|
|
1056
|
+
return {
|
|
1057
|
+
arm,
|
|
1058
|
+
tag,
|
|
1059
|
+
class: cls,
|
|
1060
|
+
subcommandCorrect: subOk,
|
|
1061
|
+
exec,
|
|
1062
|
+
command: firstCommand,
|
|
1063
|
+
forcedCommand: forceCommand != null,
|
|
1064
|
+
attempts: attemptLines.map((a) => a.command),
|
|
1065
|
+
lessonDelivered,
|
|
1066
|
+
lessonIndex,
|
|
1067
|
+
firstToolIndex,
|
|
1068
|
+
lessonBeforeFirstToolCall: host === 'codex'
|
|
1069
|
+
? codexLessonBeforeTool(sequence)
|
|
1070
|
+
: lessonDelivered && firstToolIndex > -1 && lessonIndex < firstToolIndex,
|
|
1071
|
+
costUsd: result?.total_cost_usd ?? null,
|
|
1072
|
+
wallMs,
|
|
1073
|
+
modelUsed: events.find((e) => e.type === 'system' && e.subtype === 'init')?.model
|
|
1074
|
+
|| events.find((e) => e.type === 'assistant')?.message?.model || model,
|
|
1075
|
+
host,
|
|
1076
|
+
transcript: path.relative(ROOT, streamFile),
|
|
1077
|
+
exit: r.status,
|
|
1078
|
+
spawnError: host === 'codex' ? parseCodexRunError(events, r) : replayRunError(events, r),
|
|
1079
|
+
};
|
|
1080
|
+
}
|
|
1081
|
+
|
|
1082
|
+
// ── the CLI ─────────────────────────────────────────────────────────────────────────────────────
|
|
1083
|
+
const argv = process.argv.slice(2);
|
|
1084
|
+
const has = (f) => argv.includes(f);
|
|
1085
|
+
const arg = (f, d) => { const i = argv.indexOf(f); return i >= 0 && argv[i + 1] ? argv[i + 1] : d; };
|
|
1086
|
+
const usage = () => `Usage:
|
|
1087
|
+
node scripts/learning-replay.mjs [--trap ${TRAP.MEMORY_SEARCH}|${TRAP.POST_TASK}] [--n N] [--host codex|claude-code] [--model MODEL]
|
|
1088
|
+
node scripts/learning-replay.mjs --check
|
|
1089
|
+
node scripts/learning-replay.mjs --check-portfolio
|
|
1090
|
+
node scripts/learning-replay.mjs --check-mutants
|
|
1091
|
+
node scripts/learning-replay.mjs --dry-run
|
|
1092
|
+
node scripts/learning-replay.mjs --mutant <${Object.keys(MUTANTS).join('|')}>
|
|
1093
|
+
|
|
1094
|
+
Exit: 0=PASS, 1=FAIL, 3=INCONCLUSIVE, 4=UNKNOWN.`;
|
|
1095
|
+
|
|
1096
|
+
/** The command mutant `wrong-subcommand` substitutes: the grader's exact defect, right flag on a wrong verb. */
|
|
1097
|
+
export const WRONG_SUBCOMMAND_COMMAND = 'ruflo recall -q "caching strategy"';
|
|
1098
|
+
|
|
1099
|
+
export const MUTANTS = Object.freeze({
|
|
1100
|
+
'delete-lesson': 'delete the recorded lesson from the fixture store after the refresh — the treated arm must go red',
|
|
1101
|
+
'brain-off-treated': 'run the TREATED arm with the brain disabled — it must produce the control artifact and go red',
|
|
1102
|
+
'seed-control': "pre-seed the CONTROL arm's context with the lesson — the harness must report INCONCLUSIVE, never PASS",
|
|
1103
|
+
// ── the execution gate's own mutants (2026-07-28) ──
|
|
1104
|
+
'wrong-subcommand': `substitute the treated arm's artifact with \`${WRONG_SUBCOMMAND_COMMAND}\` — right flag, wrong verb: the exact command the grader found being certified. Must go red.`,
|
|
1105
|
+
'empty-store': "empty project B's seeded memory before the gate runs — a perfect command that retrieves nothing must go red on RETRIEVAL, not pass on exit status",
|
|
1106
|
+
});
|
|
1107
|
+
|
|
1108
|
+
/** Committed real-model evidence for ADR-058's two named falsification traps. */
|
|
1109
|
+
export const MUTANT_RESULT_FILES = Object.freeze({
|
|
1110
|
+
[TRAP.MEMORY_SEARCH]: Object.freeze({
|
|
1111
|
+
'delete-lesson': path.join(ROOT, 'data', 'learning-replay-delete-lesson-result.json'),
|
|
1112
|
+
'brain-off-treated': path.join(ROOT, 'data', 'learning-replay-brain-off-result.json'),
|
|
1113
|
+
}),
|
|
1114
|
+
[TRAP.POST_TASK]: Object.freeze({
|
|
1115
|
+
'delete-lesson': path.join(ROOT, 'data', 'learning-replay-post-task-delete-lesson-result.json'),
|
|
1116
|
+
'brain-off-treated': path.join(ROOT, 'data', 'learning-replay-post-task-brain-off-result.json'),
|
|
1117
|
+
}),
|
|
1118
|
+
});
|
|
1119
|
+
|
|
1120
|
+
export const PORTFOLIO_RESULT_FILES = Object.freeze({
|
|
1121
|
+
[TRAP.MEMORY_SEARCH]: RESULT_FILE,
|
|
1122
|
+
[TRAP.POST_TASK]: POST_TASK_RESULT_FILE,
|
|
1123
|
+
});
|
|
1124
|
+
|
|
1125
|
+
function headSha() {
|
|
1126
|
+
const r = spawnSync('git', ['rev-parse', 'HEAD'], { cwd: ROOT, encoding: 'utf8' });
|
|
1127
|
+
return r.status === 0 ? r.stdout.trim() : null;
|
|
1128
|
+
}
|
|
1129
|
+
|
|
1130
|
+
/** `--check`: gate on the committed artifact WITHOUT spending a token. */
|
|
1131
|
+
export function checkArtifact({ file = RESULT_FILE, repo = ROOT, maxAgeDays = 14 } = {}) {
|
|
1132
|
+
if (!fs.existsSync(file)) {
|
|
1133
|
+
return { status: VERDICT.UNKNOWN, why: `no result artifact at ${path.relative(repo, file)} — the replay has never been run on this checkout` };
|
|
1134
|
+
}
|
|
1135
|
+
let a;
|
|
1136
|
+
try { a = JSON.parse(fs.readFileSync(file, 'utf8')); } catch (e) { return { status: VERDICT.UNKNOWN, why: `result artifact unparseable: ${e.message}` }; }
|
|
1137
|
+
if (a.invariant !== INVARIANT) return { status: VERDICT.UNKNOWN, why: `artifact declares invariant "${a.invariant}", expected ${INVARIANT}` };
|
|
1138
|
+
if (!a.sha) return { status: VERDICT.UNKNOWN, why: 'artifact states no SHA — a result with no SHA is a result about nothing' };
|
|
1139
|
+
|
|
1140
|
+
const head = headSha();
|
|
1141
|
+
const stale = [];
|
|
1142
|
+
if (head && a.sha !== head) {
|
|
1143
|
+
// Not the same commit: the result is still CURRENT only if nothing load-bearing moved.
|
|
1144
|
+
const anc = spawnSync('git', ['merge-base', '--is-ancestor', a.sha, head], { cwd: repo });
|
|
1145
|
+
if (anc.status !== 0) return { status: VERDICT.UNKNOWN, why: `artifact SHA ${a.sha.slice(0, 8)} is not an ancestor of HEAD ${head.slice(0, 8)} — it measures a different tree` };
|
|
1146
|
+
const diff = spawnSync('git', ['diff', '--name-only', `${a.sha}..${head}`, '--', ...LOAD_BEARING], { cwd: repo, encoding: 'utf8' });
|
|
1147
|
+
if (diff.status === 0) for (const f of diff.stdout.split('\n').map((s) => s.trim()).filter(Boolean)) stale.push(f);
|
|
1148
|
+
}
|
|
1149
|
+
if (stale.length) {
|
|
1150
|
+
return { status: VERDICT.UNKNOWN, why: `result recorded on ${a.sha.slice(0, 8)}, but ${stale.length} load-bearing file(s) changed since: ${stale.join(', ')} — re-run the replay` };
|
|
1151
|
+
}
|
|
1152
|
+
const ageDays = a.at ? (Date.now() - Date.parse(a.at)) / 86_400_000 : Infinity;
|
|
1153
|
+
if (!(ageDays <= maxAgeDays)) {
|
|
1154
|
+
return { status: VERDICT.UNKNOWN, why: `result is ${Number.isFinite(ageDays) ? ageDays.toFixed(1) : '?'} days old (max ${maxAgeDays}) — a nightly trap that has not run is UNKNOWN, never PASS` };
|
|
1155
|
+
}
|
|
1156
|
+
return {
|
|
1157
|
+
status: a.verdict,
|
|
1158
|
+
why: `${a.verdict} — ${a.passes}/${a.n} runs, control produced the token in ${a.controlTokenRuns}/${a.n}, recorded on ${a.sha.slice(0, 8)} (${a.model})`,
|
|
1159
|
+
artifact: a,
|
|
1160
|
+
};
|
|
1161
|
+
}
|
|
1162
|
+
|
|
1163
|
+
/**
|
|
1164
|
+
* Verify the two named ADR-058 mutants were run against a current load-bearing tree and each
|
|
1165
|
+
* destroyed the claimed effect. A mutant FAIL is evidence only when the failure has the expected
|
|
1166
|
+
* causal shape; "the executor crashed" or "some unrelated assertion failed" cannot satisfy this.
|
|
1167
|
+
*/
|
|
1168
|
+
export function checkMutantArtifacts({
|
|
1169
|
+
files = MUTANT_RESULT_FILES,
|
|
1170
|
+
repo = ROOT,
|
|
1171
|
+
maxAgeDays = 14,
|
|
1172
|
+
} = {}) {
|
|
1173
|
+
const checked = [];
|
|
1174
|
+
for (const trap of [TRAP.MEMORY_SEARCH, TRAP.POST_TASK]) {
|
|
1175
|
+
for (const mutant of ['delete-lesson', 'brain-off-treated']) {
|
|
1176
|
+
const file = files[trap]?.[mutant];
|
|
1177
|
+
if (!file || !fs.existsSync(file)) {
|
|
1178
|
+
return { status: VERDICT.UNKNOWN, why: `${trap}/${mutant}: no committed execution artifact`, checked };
|
|
1179
|
+
}
|
|
1180
|
+
|
|
1181
|
+
const currency = checkArtifact({ file, repo, maxAgeDays });
|
|
1182
|
+
if (currency.status === VERDICT.UNKNOWN) {
|
|
1183
|
+
return { status: VERDICT.UNKNOWN, why: `${trap}/${mutant}: ${currency.why}`, checked };
|
|
1184
|
+
}
|
|
1185
|
+
|
|
1186
|
+
let artifact;
|
|
1187
|
+
try { artifact = JSON.parse(fs.readFileSync(file, 'utf8')); }
|
|
1188
|
+
catch (e) { return { status: VERDICT.UNKNOWN, why: `${mutant}: artifact unparseable: ${e.message}`, checked }; }
|
|
1189
|
+
|
|
1190
|
+
if (artifact.mutant !== mutant || artifact.trap !== trap) {
|
|
1191
|
+
return { status: VERDICT.FAIL, why: `${trap}/${mutant}: artifact identity mismatch`, checked };
|
|
1192
|
+
}
|
|
1193
|
+
if (artifact.verdict !== VERDICT.FAIL || artifact.passes !== 0 || !(artifact.n >= 1)) {
|
|
1194
|
+
return { status: VERDICT.FAIL, why: `${mutant}: mutant did not go red with zero passes`, checked };
|
|
1195
|
+
}
|
|
1196
|
+
if (!Array.isArray(artifact.runs) || artifact.runs.length !== artifact.n) {
|
|
1197
|
+
return { status: VERDICT.FAIL, why: `${mutant}: missing per-run execution evidence`, checked };
|
|
1198
|
+
}
|
|
1199
|
+
|
|
1200
|
+
if (mutant === 'delete-lesson') {
|
|
1201
|
+
const lessonSurvived = artifact.runs.some((run) =>
|
|
1202
|
+
run.treated?.lessonBeforeFirstToolCall === true
|
|
1203
|
+
|| run.treated?.lessonDelivered === true
|
|
1204
|
+
|| run.treated?.class === 'flagged');
|
|
1205
|
+
if (lessonSurvived) {
|
|
1206
|
+
return { status: VERDICT.FAIL, why: 'delete-lesson: treated arm still received or reproduced the lesson', checked };
|
|
1207
|
+
}
|
|
1208
|
+
} else {
|
|
1209
|
+
const differsFromControl = artifact.runs.some((run) =>
|
|
1210
|
+
run.treated?.lessonBeforeFirstToolCall === true
|
|
1211
|
+
|| run.treated?.lessonDelivered === true
|
|
1212
|
+
|| run.control?.lessonDelivered === true
|
|
1213
|
+
|| run.treated?.class !== run.control?.class);
|
|
1214
|
+
if (differsFromControl) {
|
|
1215
|
+
return { status: VERDICT.FAIL, why: 'brain-off-treated: treated arm did not collapse to the brain-off control artifact', checked };
|
|
1216
|
+
}
|
|
1217
|
+
}
|
|
1218
|
+
checked.push(`${trap}/${mutant}`);
|
|
1219
|
+
}
|
|
1220
|
+
}
|
|
1221
|
+
return {
|
|
1222
|
+
status: VERDICT.PASS,
|
|
1223
|
+
why: 'delete-lesson and brain-off-treated both went red for both independent traps on current real-model execution evidence',
|
|
1224
|
+
checked,
|
|
1225
|
+
};
|
|
1226
|
+
}
|
|
1227
|
+
|
|
1228
|
+
export function checkPortfolio({
|
|
1229
|
+
files = PORTFOLIO_RESULT_FILES,
|
|
1230
|
+
mutantFiles = MUTANT_RESULT_FILES,
|
|
1231
|
+
repo = ROOT,
|
|
1232
|
+
maxAgeDays = 14,
|
|
1233
|
+
} = {}) {
|
|
1234
|
+
const artifacts = [];
|
|
1235
|
+
for (const trap of [TRAP.MEMORY_SEARCH, TRAP.POST_TASK]) {
|
|
1236
|
+
const checked = checkArtifact({ file: files[trap], repo, maxAgeDays });
|
|
1237
|
+
if (checked.status !== VERDICT.PASS) {
|
|
1238
|
+
return { status: checked.status, why: `${trap}: ${checked.why}`, artifacts };
|
|
1239
|
+
}
|
|
1240
|
+
const a = checked.artifact;
|
|
1241
|
+
if (a.trap !== trap || a.n < 3 || a.passes < 2 || a.controlTokenRuns !== 0 || a.controlWorkedRuns !== 0) {
|
|
1242
|
+
return { status: VERDICT.FAIL, why: `${trap}: artifact does not prove N>=3 treated/control causal separation`, artifacts };
|
|
1243
|
+
}
|
|
1244
|
+
if (a.promotion?.projectCount < 2 || a.promotion?.promoted !== true
|
|
1245
|
+
|| new Set(a.promotion?.sourceProjects || []).size < 2) {
|
|
1246
|
+
return { status: VERDICT.FAIL, why: `${trap}: lesson did not earn the real win-twice cross-project scope`, artifacts };
|
|
1247
|
+
}
|
|
1248
|
+
if (a.refresh?.lessonSurvived !== true) {
|
|
1249
|
+
return { status: VERDICT.FAIL, why: `${trap}: learned lesson did not survive refresh`, artifacts };
|
|
1250
|
+
}
|
|
1251
|
+
artifacts.push(a);
|
|
1252
|
+
}
|
|
1253
|
+
if (artifacts[0].record?.lessonId === artifacts[1].record?.lessonId
|
|
1254
|
+
|| artifacts[0].task === artifacts[1].task) {
|
|
1255
|
+
return { status: VERDICT.FAIL, why: 'portfolio traps are not independent lesson/task classes', artifacts };
|
|
1256
|
+
}
|
|
1257
|
+
const mutants = checkMutantArtifacts({ files: mutantFiles, repo, maxAgeDays });
|
|
1258
|
+
if (mutants.status !== VERDICT.PASS) return { ...mutants, why: `portfolio mutants: ${mutants.why}`, artifacts };
|
|
1259
|
+
return {
|
|
1260
|
+
status: VERDICT.PASS,
|
|
1261
|
+
why: 'two independent Ruflo CLI lessons each passed N>=3 treated/control, earned win-twice scope in two source projects, survived refresh, executed a meaningful outcome, and failed both causal mutants',
|
|
1262
|
+
artifacts,
|
|
1263
|
+
mutants,
|
|
1264
|
+
};
|
|
1265
|
+
}
|
|
1266
|
+
|
|
1267
|
+
async function main() {
|
|
1268
|
+
if (has('--help') || has('-h')) {
|
|
1269
|
+
console.log(usage());
|
|
1270
|
+
return;
|
|
1271
|
+
}
|
|
1272
|
+
const check = has('--check');
|
|
1273
|
+
const checkPortfolioFlag = has('--check-portfolio');
|
|
1274
|
+
const checkMutants = has('--check-mutants');
|
|
1275
|
+
const dryRun = has('--dry-run');
|
|
1276
|
+
const mutant = arg('--mutant', null);
|
|
1277
|
+
const trap = arg('--trap', TRAP.MEMORY_SEARCH);
|
|
1278
|
+
const spec = trapSpec(trap);
|
|
1279
|
+
const n = Math.max(1, parseInt(arg('--n', mutant ? '1' : '3'), 10) || 1);
|
|
1280
|
+
const host = arg('--host', 'codex');
|
|
1281
|
+
const model = arg('--model', host === 'codex' ? 'gpt-5.6-sol' : 'haiku');
|
|
1282
|
+
const defaultOut = mutant
|
|
1283
|
+
? MUTANT_RESULT_FILES[trap]?.[mutant]
|
|
1284
|
+
: PORTFOLIO_RESULT_FILES[trap];
|
|
1285
|
+
const outFile = arg('--out', defaultOut);
|
|
1286
|
+
const keep = has('--keep-fixtures');
|
|
1287
|
+
|
|
1288
|
+
if (mutant && !MUTANTS[mutant]) {
|
|
1289
|
+
console.error(`unknown mutant "${mutant}". known: ${Object.keys(MUTANTS).join(', ')}`);
|
|
1290
|
+
process.exit(EXIT.UNKNOWN);
|
|
1291
|
+
}
|
|
1292
|
+
if (![TRAP.MEMORY_SEARCH, TRAP.POST_TASK].includes(trap) || !outFile) {
|
|
1293
|
+
console.error(`unknown or unsupported trap/mutant combination: ${trap}/${mutant || 'normal'}`);
|
|
1294
|
+
process.exit(EXIT.UNKNOWN);
|
|
1295
|
+
}
|
|
1296
|
+
|
|
1297
|
+
if (check) {
|
|
1298
|
+
const res = checkArtifact({ file: PORTFOLIO_RESULT_FILES[trap] });
|
|
1299
|
+
console.log(`\n ${INVARIANT}: ${res.status}\n ${res.why}\n`);
|
|
1300
|
+
process.exit(EXIT[res.status] ?? EXIT.UNKNOWN);
|
|
1301
|
+
}
|
|
1302
|
+
|
|
1303
|
+
if (checkPortfolioFlag) {
|
|
1304
|
+
const res = checkPortfolio();
|
|
1305
|
+
console.log(`\n ${INVARIANT}-PORTFOLIO: ${res.status}\n ${res.why}\n`);
|
|
1306
|
+
process.exit(EXIT[res.status] ?? EXIT.UNKNOWN);
|
|
1307
|
+
}
|
|
1308
|
+
|
|
1309
|
+
if (checkMutants) {
|
|
1310
|
+
const res = checkMutantArtifacts();
|
|
1311
|
+
console.log(`\n ${INVARIANT}-MUTANTS: ${res.status}\n ${res.why}\n`);
|
|
1312
|
+
process.exit(EXIT[res.status] ?? EXIT.UNKNOWN);
|
|
1313
|
+
}
|
|
1314
|
+
|
|
1315
|
+
console.log(`\n=== ${INVARIANT} — counterfactual replay (ADR-058 §D4) ===`);
|
|
1316
|
+
const flag = trap === TRAP.POST_TASK ? verifyPostTaskContract() : verifyRufloFlag();
|
|
1317
|
+
console.log(` trap: ${trap}`);
|
|
1318
|
+
console.log(` premise: ${flag.ok ? `VERIFIED live: ${flag.evidence}` : `NOT VERIFIED: ${flag.why}`}`);
|
|
1319
|
+
if (!flag.ok) {
|
|
1320
|
+
const artifact = writeArtifact(outFile, {
|
|
1321
|
+
verdict: VERDICT.UNKNOWN, why: `premise not verified: ${flag.why}`, n: 0, passes: 0, fails: 0, unknowns: 0, controlTokenRuns: 0, rate: 0, runs: [],
|
|
1322
|
+
}, { host, model, mutant });
|
|
1323
|
+
console.log(` → UNKNOWN (never a pass). artifact: ${path.relative(ROOT, outFile)}`);
|
|
1324
|
+
process.exit(EXIT.UNKNOWN);
|
|
1325
|
+
}
|
|
1326
|
+
|
|
1327
|
+
const base = allocateRunBase();
|
|
1328
|
+
const dirs = buildFixtures(base);
|
|
1329
|
+
// Ruflo may auto-start a workspace daemon while initializing a fixture memory DB. Reap only
|
|
1330
|
+
// daemons whose explicit --workspace lives under THIS run, including on Ctrl-C or an exception.
|
|
1331
|
+
process.once('exit', () => { cleanupFixtureDaemons(dirs); });
|
|
1332
|
+
const rec = recordInProjectA(dirs, { trap });
|
|
1333
|
+
console.log(` record (two independent source projects): ${rec.projectCount} sources, win-twice=${rec.promoted}, lesson ${rec.ok ? 'derived + ratified' : 'NOT recorded'}`);
|
|
1334
|
+
const seed = trap === TRAP.MEMORY_SEARCH
|
|
1335
|
+
? seedProjectBMemory(dirs)
|
|
1336
|
+
: { key: null, storeExit: null, ok: true, skipped: 'the command-risk trap needs no target-project memory row' };
|
|
1337
|
+
console.log(` seed (fixture-project-B): ${seed.skipped || `note "${seed.key}" ${seed.ok ? 'stored — the task premise is true' : `NOT stored (exit ${seed.storeExit})`}`}`);
|
|
1338
|
+
const refresh = nightlyRefresh(dirs);
|
|
1339
|
+
console.log(` refresh (nightly): spine generation ${refresh.generation} installed + active; distill exit ${refresh.distillExit}, backup exit ${refresh.backupExit}; lesson survived: ${refresh.lessonSurvived}`);
|
|
1340
|
+
|
|
1341
|
+
if (mutant === 'delete-lesson') {
|
|
1342
|
+
saveLessons([], dirs.lessons);
|
|
1343
|
+
try { fs.rmSync(path.join(dirs.projectA, '.swarm', 'memory.db'), { force: true }); } catch { /* already gone */ }
|
|
1344
|
+
console.log(` MUTANT delete-lesson: lesson store emptied (${loadLessons(dirs.lessons).length} lessons) and project A's memory.db removed`);
|
|
1345
|
+
}
|
|
1346
|
+
|
|
1347
|
+
if (mutant === 'empty-store') {
|
|
1348
|
+
// Remove the seeded note. Every arm still runs for real; the produced command still executes for
|
|
1349
|
+
// real; it simply has nothing to find. The gate must key on RETRIEVAL, not on exit status —
|
|
1350
|
+
// `ruflo memory search -q "<absent>"` exits 0.
|
|
1351
|
+
try { fs.rmSync(path.join(dirs.projectB, '.swarm', 'memory.db'), { force: true }); } catch { /* already gone */ }
|
|
1352
|
+
console.log(` MUTANT empty-store: project B's seeded memory removed (exists: ${fs.existsSync(path.join(dirs.projectB, '.swarm', 'memory.db'))})`);
|
|
1353
|
+
}
|
|
1354
|
+
|
|
1355
|
+
if (dryRun) {
|
|
1356
|
+
// Prove the WIRE without a token: fire the real hook chain in both states and report the bytes.
|
|
1357
|
+
const probe = (stateDir) => {
|
|
1358
|
+
const r = spawnSync(process.execPath, [path.join(ROOT, 'plugin', 'scripts', 'hook-shim.mjs'), 'unprompted-speech', 'UserPromptSubmit'], {
|
|
1359
|
+
input: JSON.stringify({ prompt: spec.prompt, session_id: `dry-${Date.now()}`, cwd: dirs.projectB }),
|
|
1360
|
+
encoding: 'utf8',
|
|
1361
|
+
cwd: dirs.projectB,
|
|
1362
|
+
env: {
|
|
1363
|
+
...process.env,
|
|
1364
|
+
RUVNET_BRAIN_HOME: dirs.brainHome,
|
|
1365
|
+
RUVNET_BRAIN_STATE_DIR: stateDir,
|
|
1366
|
+
RUVNET_LESSON_STORE: dirs.lessons,
|
|
1367
|
+
RUVNET_LESSON_GATE_STATE: dirs.gateState,
|
|
1368
|
+
CLAUDE_PLUGIN_ROOT: path.join(ROOT, 'plugin'),
|
|
1369
|
+
},
|
|
1370
|
+
});
|
|
1371
|
+
return (r.stdout || '').length;
|
|
1372
|
+
};
|
|
1373
|
+
const onBytes = probe(dirs.stateOn), offBytes = probe(dirs.stateOff);
|
|
1374
|
+
console.log(` dry-run: treated-state hook emitted ${onBytes} bytes; control-state (brain-off) emitted ${offBytes} bytes`);
|
|
1375
|
+
writeArtifact(outFile, {
|
|
1376
|
+
verdict: VERDICT.UNKNOWN, why: `--dry-run: no model was called, so nothing was measured (wire probe: treated ${onBytes}B, control ${offBytes}B)`,
|
|
1377
|
+
n: 0, passes: 0, fails: 0, unknowns: 0, controlTokenRuns: 0, rate: 0, runs: [],
|
|
1378
|
+
}, { host, model, mutant, record: rec, seed, refresh });
|
|
1379
|
+
console.log(` → UNKNOWN (a dry run is never a pass). artifact: ${path.relative(ROOT, outFile)}`);
|
|
1380
|
+
if (!keep) {
|
|
1381
|
+
cleanupFixtureDaemons(dirs);
|
|
1382
|
+
rmrf(base);
|
|
1383
|
+
}
|
|
1384
|
+
process.exit(EXIT.UNKNOWN);
|
|
1385
|
+
}
|
|
1386
|
+
|
|
1387
|
+
const runs = [];
|
|
1388
|
+
for (let i = 1; i <= n; i++) {
|
|
1389
|
+
const treatedState = mutant === 'brain-off-treated' ? dirs.stateOff : dirs.stateOn;
|
|
1390
|
+
const treated = runArm({
|
|
1391
|
+
dirs, arm: 'treated', stateDir: treatedState, model, host, tag: `run${i}-treated`, trap,
|
|
1392
|
+
forceCommand: mutant === 'wrong-subcommand' ? WRONG_SUBCOMMAND_COMMAND : null,
|
|
1393
|
+
});
|
|
1394
|
+
const control = runArm({
|
|
1395
|
+
dirs, arm: 'control', stateDir: dirs.stateOff, model, host, tag: `run${i}-control`, trap,
|
|
1396
|
+
// MUTANT seed-control: the control is handed the lesson through a channel the brain does not
|
|
1397
|
+
// own. Its artifact then carries the token, and invariant 6 must fire.
|
|
1398
|
+
appendSystemPrompt: mutant === 'seed-control' ? spec.lesson : null,
|
|
1399
|
+
});
|
|
1400
|
+
const run = {
|
|
1401
|
+
i,
|
|
1402
|
+
treatedClass: treated.class,
|
|
1403
|
+
controlClass: control.class,
|
|
1404
|
+
lessonBeforeFirstToolCall: treated.lessonBeforeFirstToolCall,
|
|
1405
|
+
controlLessonDelivered: control.lessonDelivered,
|
|
1406
|
+
// ── the execution gate's inputs, flattened so verdictForRun stays a pure function ──
|
|
1407
|
+
treatedSubcommandCorrect: treated.subcommandCorrect,
|
|
1408
|
+
treatedExecOk: treated.exec?.exitOk === true,
|
|
1409
|
+
treatedRetrieved: treated.exec?.retrieved === true,
|
|
1410
|
+
treatedExecWhy: treated.exec?.why || null,
|
|
1411
|
+
controlWorked: control.exec?.exitOk === true && control.exec?.retrieved === true,
|
|
1412
|
+
treated,
|
|
1413
|
+
control,
|
|
1414
|
+
error: treated.spawnError || control.spawnError || null,
|
|
1415
|
+
};
|
|
1416
|
+
const v = verdictForRun(run);
|
|
1417
|
+
console.log(` run ${i}: treated="${treated.class}" (${treated.command || '—'})`);
|
|
1418
|
+
console.log(` control="${control.class}" (${control.command || '—'})`);
|
|
1419
|
+
console.log(` EXECUTED treated: subcommand=${treated.subcommandCorrect} exit=${treated.exec?.exit ?? '—'} retrieved=${treated.exec?.retrieved} · ${treated.exec?.why || 'not run'}`);
|
|
1420
|
+
console.log(` EXECUTED control: exit=${control.exec?.exit ?? '—'} retrieved=${control.exec?.retrieved}`);
|
|
1421
|
+
console.log(` lesson before first tool call: ${treated.lessonBeforeFirstToolCall} (lesson@${treated.lessonIndex}, tool@${treated.firstToolIndex}) · control got ${control.lessonDelivered ? 'THE LESSON (leak!)' : 'zero brain bytes'}`);
|
|
1422
|
+
console.log(` → ${v.verdict}: ${v.why}`);
|
|
1423
|
+
runs.push(run);
|
|
1424
|
+
}
|
|
1425
|
+
|
|
1426
|
+
const agg = aggregate(runs);
|
|
1427
|
+
const costUsd = runs.reduce((s, r) => s + (r.treated.costUsd || 0) + (r.control.costUsd || 0), 0);
|
|
1428
|
+
const wallMs = runs.reduce((s, r) => s + (r.treated.wallMs || 0) + (r.control.wallMs || 0), 0);
|
|
1429
|
+
writeArtifact(outFile, agg, { host, model, mutant, trap, task: spec.prompt, record: rec, seed, refresh, flag, costUsd, wallMs });
|
|
1430
|
+
|
|
1431
|
+
console.log(`\n RATE ${agg.passes}/${agg.n} · token carried by treated ${agg.treatedTokenRuns}/${agg.n} vs control ${agg.controlTokenRuns}/${agg.n}`);
|
|
1432
|
+
console.log(` EXECUTION GATE: treated named the real subcommand ${agg.treatedSubcommandRuns}/${agg.n} · exited 0 ${agg.treatedExecutedRuns}/${agg.n} · RETRIEVED ${agg.treatedRetrievedRuns}/${agg.n} · control worked ${agg.controlWorkedRuns}/${agg.n}`);
|
|
1433
|
+
console.log(` ${INVARIANT}: ${agg.verdict} — ${agg.why}`);
|
|
1434
|
+
if (!keep) pruneArchive(dirs);
|
|
1435
|
+
console.log(` cost $${costUsd.toFixed(4)} · ${(wallMs / 1000).toFixed(1)}s wall · transcripts: ${path.relative(ROOT, dirs.transcripts)}`);
|
|
1436
|
+
console.log(` artifact: ${path.relative(ROOT, outFile)}\n`);
|
|
1437
|
+
process.exit(EXIT[agg.verdict] ?? EXIT.UNKNOWN);
|
|
1438
|
+
}
|
|
1439
|
+
|
|
1440
|
+
/**
|
|
1441
|
+
* RETENTION. "Transcripts archived" must not mean "the disk fills".
|
|
1442
|
+
*
|
|
1443
|
+
* Each run builds a whole fixture world, and the nightly refresh step copies plugin/ into it — ~12MB
|
|
1444
|
+
* per invocation, every night, forever. Measured 2026-07-27 after nine invocations in one session:
|
|
1445
|
+
* 36MB, of which the transcripts were under 300KB. So the EVIDENCE is kept and the SCAFFOLDING is
|
|
1446
|
+
* dropped: everything under the run directory except transcripts/ goes, and only the most recent
|
|
1447
|
+
* KEEP_RUNS run directories survive. `--keep-fixtures` retains everything for debugging.
|
|
1448
|
+
*
|
|
1449
|
+
* Deliberately not "delete the whole run dir": the transcripts ARE the archive ADR-058 asks for, and
|
|
1450
|
+
* an archive nobody kept is the same as a claim nobody checked.
|
|
1451
|
+
*/
|
|
1452
|
+
const KEEP_RUNS = 14;
|
|
1453
|
+
export function cleanupFixtureDaemons(dirs) {
|
|
1454
|
+
const base = path.resolve(dirs.base);
|
|
1455
|
+
const ps = spawnSync('ps', ['-axo', 'pid=,command='], { encoding: 'utf8', timeout: 10_000 });
|
|
1456
|
+
if (ps.status !== 0) return { found: 0, stopped: 0, errors: ['process census failed'] };
|
|
1457
|
+
const matches = String(ps.stdout || '').split('\n').flatMap((line) => {
|
|
1458
|
+
const match = line.match(/^\s*(\d+)\s+(.+)$/);
|
|
1459
|
+
if (!match) return [];
|
|
1460
|
+
const pid = Number(match[1]);
|
|
1461
|
+
const command = match[2];
|
|
1462
|
+
return command.includes('daemon start --foreground')
|
|
1463
|
+
&& command.includes('--workspace')
|
|
1464
|
+
&& command.includes(base)
|
|
1465
|
+
&& pid !== process.pid
|
|
1466
|
+
? [{ pid, command }]
|
|
1467
|
+
: [];
|
|
1468
|
+
});
|
|
1469
|
+
const errors = [];
|
|
1470
|
+
let stopped = 0;
|
|
1471
|
+
for (const match of matches) {
|
|
1472
|
+
try { process.kill(match.pid, 'SIGTERM'); stopped++; }
|
|
1473
|
+
catch (error) {
|
|
1474
|
+
if (error?.code !== 'ESRCH') errors.push(`${match.pid}: ${error.message}`);
|
|
1475
|
+
}
|
|
1476
|
+
}
|
|
1477
|
+
return { found: matches.length, stopped, errors };
|
|
1478
|
+
}
|
|
1479
|
+
|
|
1480
|
+
function pruneArchive(dirs) {
|
|
1481
|
+
cleanupFixtureDaemons(dirs);
|
|
1482
|
+
try {
|
|
1483
|
+
for (const e of fs.readdirSync(dirs.base, { withFileTypes: true })) {
|
|
1484
|
+
if (e.name === 'transcripts') continue;
|
|
1485
|
+
rmrf(path.join(dirs.base, e.name));
|
|
1486
|
+
}
|
|
1487
|
+
} catch { /* nothing to prune */ }
|
|
1488
|
+
try {
|
|
1489
|
+
const root = path.dirname(dirs.base);
|
|
1490
|
+
const runs = fs.readdirSync(root).filter((d) => d.startsWith('run-')).sort();
|
|
1491
|
+
for (const old of runs.slice(0, Math.max(0, runs.length - KEEP_RUNS))) rmrf(path.join(root, old));
|
|
1492
|
+
} catch { /* nothing to prune */ }
|
|
1493
|
+
}
|
|
1494
|
+
|
|
1495
|
+
/**
|
|
1496
|
+
* The execution record as it lands in the COMMITTED artifact. The first 400 bytes of the command's
|
|
1497
|
+
* real output are kept verbatim: a retrieval claim whose evidence nobody can read is an assertion,
|
|
1498
|
+
* and the whole deduction being closed here was a number nobody could check against its own arm.
|
|
1499
|
+
*/
|
|
1500
|
+
function execRecord(e) {
|
|
1501
|
+
if (!e) return null;
|
|
1502
|
+
return { ran: e.ran, argv: e.argv, exit: e.exit, exitOk: e.exitOk, retrieved: e.retrieved, why: e.why, output: String(e.output || '').slice(0, 400) };
|
|
1503
|
+
}
|
|
1504
|
+
|
|
1505
|
+
/** The machine-readable result. A verdict with no SHA is a verdict about nothing. */
|
|
1506
|
+
function writeArtifact(file, agg, meta = {}) {
|
|
1507
|
+
const artifact = {
|
|
1508
|
+
invariant: INVARIANT,
|
|
1509
|
+
verdict: agg.verdict,
|
|
1510
|
+
why: agg.why,
|
|
1511
|
+
sha: headSha(),
|
|
1512
|
+
at: new Date().toISOString(),
|
|
1513
|
+
host: meta.host || null,
|
|
1514
|
+
model: meta.model || null,
|
|
1515
|
+
// The alias asked for ("haiku") is not the model that answered. Record the id the session
|
|
1516
|
+
// actually reported, so a result can never be attributed to a model that never ran.
|
|
1517
|
+
modelResolved: (agg.runs || []).map((r) => r.treated?.modelUsed).find(Boolean) || null,
|
|
1518
|
+
mutant: meta.mutant || null,
|
|
1519
|
+
trap: meta.trap || TRAP.MEMORY_SEARCH,
|
|
1520
|
+
task: meta.task || null,
|
|
1521
|
+
n: agg.n,
|
|
1522
|
+
passes: agg.passes,
|
|
1523
|
+
fails: agg.fails,
|
|
1524
|
+
unknowns: agg.unknowns,
|
|
1525
|
+
controlTokenRuns: agg.controlTokenRuns,
|
|
1526
|
+
controlWorkedRuns: agg.controlWorkedRuns ?? null,
|
|
1527
|
+
treatedTokenRuns: agg.treatedTokenRuns ?? null,
|
|
1528
|
+
// THE EXECUTION GATE's own rates. The old artifact could report `treatedTokenRuns: 3` beside
|
|
1529
|
+
// three `subcommandCorrect: false` and call it PASS; these three numbers are what makes that
|
|
1530
|
+
// combination impossible to state without also stating that nothing worked.
|
|
1531
|
+
executionGate: {
|
|
1532
|
+
treatedSubcommandRuns: agg.treatedSubcommandRuns ?? null,
|
|
1533
|
+
treatedExecutedRuns: agg.treatedExecutedRuns ?? null,
|
|
1534
|
+
treatedRetrievedRuns: agg.treatedRetrievedRuns ?? null,
|
|
1535
|
+
},
|
|
1536
|
+
rate: agg.rate,
|
|
1537
|
+
threshold: '>=2/3',
|
|
1538
|
+
costUsd: meta.costUsd != null ? +meta.costUsd.toFixed(4) : null,
|
|
1539
|
+
wallSeconds: meta.wallMs != null ? +(meta.wallMs / 1000).toFixed(1) : null,
|
|
1540
|
+
premise: meta.flag ? { verified: meta.flag.ok, evidence: meta.flag.evidence } : null,
|
|
1541
|
+
record: meta.record ? {
|
|
1542
|
+
lessonId: meta.record.lesson?.id,
|
|
1543
|
+
key: meta.record.key,
|
|
1544
|
+
storeExit: meta.record.storeExit,
|
|
1545
|
+
readBackExit: meta.record.readBackExit,
|
|
1546
|
+
ok: meta.record.ok,
|
|
1547
|
+
} : null,
|
|
1548
|
+
promotion: meta.record ? {
|
|
1549
|
+
rule: 'ADR-G008 win twice',
|
|
1550
|
+
projectCount: meta.record.projectCount,
|
|
1551
|
+
sourceProjects: meta.record.sources?.map((source) => source.project) || [],
|
|
1552
|
+
promoted: meta.record.promoted === true,
|
|
1553
|
+
} : null,
|
|
1554
|
+
seed: meta.seed ? { key: meta.seed.key, storeExit: meta.seed.storeExit, ok: meta.seed.ok } : null,
|
|
1555
|
+
refresh: meta.refresh || null,
|
|
1556
|
+
runs: (agg.runs || []).map((r) => ({
|
|
1557
|
+
i: r.i,
|
|
1558
|
+
verdict: r.verdict,
|
|
1559
|
+
why: r.why,
|
|
1560
|
+
treated: { class: r.treated?.class, subcommandCorrect: r.treated?.subcommandCorrect, command: r.treated?.command, forcedCommand: r.treated?.forcedCommand || false, exec: execRecord(r.treated?.exec), lessonIndex: r.treated?.lessonIndex, firstToolIndex: r.treated?.firstToolIndex, lessonBeforeFirstToolCall: r.treated?.lessonBeforeFirstToolCall, model: r.treated?.modelUsed, transcript: r.treated?.transcript },
|
|
1561
|
+
control: { class: r.control?.class, subcommandCorrect: r.control?.subcommandCorrect, command: r.control?.command, exec: execRecord(r.control?.exec), lessonDelivered: r.control?.lessonDelivered, model: r.control?.modelUsed, transcript: r.control?.transcript },
|
|
1562
|
+
})),
|
|
1563
|
+
};
|
|
1564
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
1565
|
+
fs.writeFileSync(file, JSON.stringify(artifact, null, 2) + '\n');
|
|
1566
|
+
return artifact;
|
|
1567
|
+
}
|
|
1568
|
+
|
|
1569
|
+
const invokedDirectly = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
1570
|
+
if (invokedDirectly) await main();
|