ruvnet-brain 4.0.36 → 4.2.2-dev

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/README.md +28 -5
  2. package/bin/install.mjs +405 -23
  3. package/data/model-catalog.json +104 -15
  4. package/package.json +2 -1
  5. package/plugin/.claude-plugin/plugin.json +2 -2
  6. package/plugin/.codex-plugin/plugin.json +1 -1
  7. package/plugin/commands/brain-console.md +72 -9
  8. package/plugin/commands/configure.md +67 -21
  9. package/plugin/commands/rvcb.md +72 -9
  10. package/plugin/hooks/codex-hooks.json +40 -33
  11. package/plugin/hooks/hook-contracts.json +14 -24
  12. package/plugin/hooks/hooks.json +7 -42
  13. package/plugin/mcp/server.mjs +23 -6
  14. package/plugin/scripts/adr-currency-gate.mjs +150 -0
  15. package/plugin/scripts/capability-registry.mjs +10 -1
  16. package/plugin/scripts/codex-hook-adapter.mjs +121 -19
  17. package/plugin/scripts/codex-hook-wrapper.mjs +61 -4
  18. package/plugin/scripts/continuation-gate.mjs +148 -8
  19. package/plugin/scripts/decision-gate.mjs +428 -0
  20. package/plugin/scripts/decision-outcomes.mjs +231 -0
  21. package/plugin/scripts/degradation-watch.mjs +271 -0
  22. package/plugin/scripts/ground-ruvnet.sh +51 -11
  23. package/plugin/scripts/hijack-ruvnet.sh +11 -3
  24. package/plugin/scripts/hook-registry.mjs +48 -3
  25. package/plugin/scripts/hook-shim.mjs +58 -3
  26. package/plugin/scripts/identifier-preflight.mjs +134 -0
  27. package/plugin/scripts/learn-capture.sh +50 -1
  28. package/plugin/scripts/learn-flush.mjs +5 -5
  29. package/plugin/scripts/lesson-bridge.mjs +343 -0
  30. package/plugin/scripts/lesson-hooks.sh +26 -0
  31. package/plugin/scripts/lesson-promote.mjs +50 -0
  32. package/plugin/scripts/lesson-store.mjs +6 -1
  33. package/plugin/scripts/mcp-readiness.mjs +107 -0
  34. package/plugin/scripts/protect-brain-state.sh +9 -0
  35. package/plugin/scripts/runtime-preferences.mjs +40 -0
  36. package/plugin/scripts/session-snapshot-hook.mjs +15 -6
  37. package/plugin/scripts/spend-guard.mjs +125 -0
  38. package/plugin/scripts/unprompted-runtime.mjs +12 -2
  39. package/plugin/scripts/update-apply.mjs +7 -2
  40. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +20 -5
  41. package/plugin/skills/ruvnet-brain/SKILL.md +3 -3
  42. package/scripts/brain-score.mjs +261 -0
  43. package/scripts/brain-stamp.mjs +5 -1
  44. package/scripts/build-bundle.mjs +38 -4
  45. package/scripts/card-from-source.mjs +114 -0
  46. package/scripts/console-engine.mjs +1 -1
  47. package/scripts/corpus-candidate.mjs +294 -0
  48. package/scripts/corpus-reconcile.mjs +273 -0
  49. package/scripts/corpus-seed-publish.mjs +110 -0
  50. package/scripts/health-repair.mjs +11 -2
  51. package/scripts/ingest-new-repos.mjs +122 -0
  52. package/scripts/ingest-repo.mjs +66 -6
  53. package/scripts/learning-replay-cli.mjs +8 -3
  54. package/scripts/learning-replay-fixture.mjs +25 -6
  55. package/scripts/learning-replay-proof.mjs +38 -0
  56. package/scripts/nightly-wrapper.sh +55 -7
  57. package/scripts/onboarding-console.mjs +21 -1
  58. package/scripts/org-repo-count.mjs +119 -0
  59. package/scripts/rebuild-gists-from-receipts.mjs +246 -0
  60. package/scripts/release-transaction.mjs +30 -2
  61. package/scripts/release.mjs +147 -0
  62. package/scripts/repo-count-detector.mjs +62 -0
  63. package/scripts/restore-local-ingests.mjs +116 -0
  64. package/scripts/rvf-generation.mjs +44 -0
  65. package/scripts/selfcheck.mjs +9 -1
  66. package/scripts/stabilization-receipt.mjs +11 -1
  67. package/scripts/sync-census.mjs +0 -0
  68. package/scripts/sync-commands.mjs +117 -0
@@ -0,0 +1,428 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * decision-gate.mjs — ONE PreToolUse decision, from N policies, with ONE reason.
4
+ *
5
+ * WHY: four independent processes could each refuse the same Write, with no precedence and no shared
6
+ * context, so the user got one arbitrary reason and no hint a second wall stood behind it. The
7
+ * measurement and the full rationale are in docs/adr/0067 — not repeated here.
8
+ *
9
+ * This is ADR-040's speech-chokepoint invariant applied to REFUSAL. One pattern used twice, not two.
10
+ *
11
+ * THE POLICIES ARE UNCHANGED. Each already speaks `exit 0` = allow, `exit 2` + stderr = refuse — a
12
+ * verdict function that was only ever missing a caller. The gate runs each as a CAPTURED child and
13
+ * composes one decision, naming every policy that refused, in declared precedence order.
14
+ *
15
+ * FAIL-OPEN, deliberately: any failure of the GATE ITSELF allows. A gate that blocks because it
16
+ * cannot read a file is one users switch off, and a disabled gate protects nothing.
17
+ */
18
+ import fs from 'node:fs';
19
+ import path from 'node:path';
20
+ import { spawn } from 'node:child_process';
21
+ import { fileURLToPath } from 'node:url';
22
+ // resolveBash ONLY. `skipNoBash` is not a predicate — it is a one-time notice emitter that returns 0
23
+ // and WRITES TO STDERR, which on this hot path is the refusal channel: calling it would have injected
24
+ // an install hint into the middle of a refusal reason, or manufactured stderr on an allow. Read the
25
+ // signature, do not infer it from the name.
26
+ import { resolveBash } from './hook-shim-bash.mjs';
27
+ import { append as appendOutcome, actionKey, recordRefusal, resolve as resolveOutcome, sweepStale } from './decision-outcomes.mjs';
28
+
29
+ const SCRIPTS_DIR = path.dirname(fileURLToPath(import.meta.url));
30
+ const EVENT = process.argv[2] || '';
31
+
32
+ /** Exit codes that mean something to the host. Anything else from a policy is an ERROR, not a refusal. */
33
+ const ALLOW = 0;
34
+ const REFUSE = 2;
35
+
36
+ /**
37
+ * ── THE BUDGET AND THE HOST TIMEOUT ARE ONE NUMBER, NOT TWO ──────────────────────────────────────
38
+ *
39
+ * THE DEFECT, measured by an adversarial audit on 2026-08-13. hooks.json declared `timeout: 5` for
40
+ * both PreToolUse entries; this file enforced an internal 4000ms budget. Those two numbers were never
41
+ * compared anywhere — not in code, not in a test — and 4000ms of policy work plus two node boots does
42
+ * not fit in 5000ms. Fourteen timed runs, in projects with no ruvnet-brain content:
43
+ *
44
+ * 3166 3579 3622 3816 4431 4464 4615 4959 | 5109 5118 5401 5526 5579 5700 ms
45
+ * ^ the 5s manifest timeout
46
+ *
47
+ * SIX OF FOURTEEN were killed by the host — including a plain `ls -la` at 5109ms and a Write at
48
+ * 5401ms. The host renders a killed hook as a FAILED PreToolUse HOOK, so the guard that exists to
49
+ * teach was instead producing the owner's literal complaint: "a ton of hook errors" on opening this
50
+ * plugin in another project. The gate was not refusing anything. It was timing out, and a timeout
51
+ * looks exactly like a broken plugin.
52
+ *
53
+ * The relationship, stated once so it can be asserted (tests/unit/decision-gate.test.mjs derives the
54
+ * manifest value FROM hooks.json — restating 5000 in the test would only re-create the drift):
55
+ *
56
+ * DEFAULT_BUDGET_MS + MIN_HEADROOM_MS ≤ hooks.json PreToolUse timeout × 1000
57
+ *
58
+ * MIN_HEADROOM_MS is not padding. It is the work outside the budget's control: the shim's node boot,
59
+ * this gate's node boot, the outcome-ledger writes, and the SIGKILL teardown of whatever the budget
60
+ * just cancelled.
61
+ *
62
+ * NARROWED BACK 2026-08-19, from 4000+5000 (which required a 10s manifest) to 2000+3000 (which fits
63
+ * the original 5s). The widening above was honest when it was made — those fourteen runs really did
64
+ * straddle the limit. What made it stale was the adr-currency-gate two-pass fix, which stopped the
65
+ * gate reading all 83 ADR bodies on every tool call. Re-measured through the FULL registered path
66
+ * (`hook-shim.mjs decision-gate write`, the way the host actually invokes it), eight runs in a
67
+ * stranger project and six in this repo:
68
+ *
69
+ * stranger: 424 405 405 415 417 416 411 415 ms
70
+ * this repo: 438 416 412 416 412 436 ms
71
+ *
72
+ * ~415ms against a 5000ms ceiling — twelve times the headroom, where before SIX OF FOURTEEN runs
73
+ * were being killed. The 10s ceiling was no longer buying anything, and a ceiling is not free: a
74
+ * sibling test caps PreToolUse at 5s precisely because the USER waits behind this hook on every
75
+ * Write and every Bash, and a 10s stall is the /rvbc hang that already burned people. THE RULE THIS
76
+ * ENCODES: a budget widened by evidence must be narrowed again when the evidence expires. Otherwise
77
+ * every emergency ratchets the ceiling up one notch permanently and nothing ever ratchets it back.
78
+ */
79
+ export const DEFAULT_BUDGET_MS = 2000;
80
+ export const MIN_HEADROOM_MS = 3000;
81
+
82
+ /**
83
+ * THE POLICY REGISTRY — the whole point of this file, and the thing that was previously scattered
84
+ * across seven hooks.json entries with no relationship to each other.
85
+ *
86
+ * ORDER IS PRECEDENCE, most fundamental first, and it is a claim about what matters most when two
87
+ * policies both object:
88
+ *
89
+ * protect-state the user's own consent boundary. It guards the OFF switch itself, so it
90
+ * outranks everything — ADR-054 §3 already says it matters MORE while the
91
+ * brain is off.
92
+ * hijack-ruvnet the managed-memory boundary (ADR-063): a correctness rule about where data
93
+ * goes, ahead of anything about process.
94
+ * ground-before-write don't write RuvNet-product code ungrounded (ADR-0012).
95
+ * design-wall don't ship a visual surface nobody looked at.
96
+ *
97
+ * `unprompted-speech` is LAST and is not really a peer: it is the speech chokepoint, which refuses
98
+ * only for a lesson the user personally opted into blocking. It is included so that Write/Edit and
99
+ * Bash have exactly ONE process that can refuse them — which is the entire invariant — and its
100
+ * allow-path stdout envelope is forwarded untouched.
101
+ */
102
+ const POLICY = (id, file, interpreter = 'bash') => ({ id, file, interpreter });
103
+ const REFUSAL_POLICIES = [
104
+ POLICY('protect-state', 'protect-brain-state.sh'),
105
+ // degradation-watch sits second because it decides whether ANY record this system keeps is real.
106
+ // Measured 2026-08-13: better_sqlite3.node was built for NODE_MODULE_VERSION 141 against a node
107
+ // needing 137, ruflo fell back to sql.js, and nothing persisted for three days while every write
108
+ // printed `[OK] Data stored successfully`. A warning was printed on every one of those writes and
109
+ // read. It could not stop anything, because a warning is text and text is skimmable — so this is
110
+ // a refusal instead. It probes only for commands whose truth DEPENDS on durable memory (a lesson
111
+ // store, a ship), so ordinary Bash pays nothing.
112
+ POLICY('degradation-watch', 'degradation-watch.mjs', 'node'),
113
+ // identifier-preflight is FIRST among the cheap checks and costs one file read: it refuses a
114
+ // command that names a model this machine's CLI does not accept. `codex exec --model gpt-5.6`
115
+ // (correct: gpt-5.6-sol, in ~/.codex/config.toml) printed a 400 and EXITED 0 into a redirected
116
+ // file on 2026-08-13, so a 50-minute audit produced nothing and there was no exit code to catch.
117
+ // It refuses ONLY a positively-known-wrong value and allows every unknown, because a wall that
118
+ // fabricates a reason is one people learn to route around.
119
+ POLICY('identifier-preflight', 'identifier-preflight.mjs', 'node'),
120
+ POLICY('hijack-ruvnet', 'hijack-ruvnet.sh'),
121
+ POLICY('ground-before-write', 'ground-before-write.sh'),
122
+ POLICY('design-wall', 'design-wall.sh'),
123
+ // adr-currency-gate fires on the EDIT, where the pre-push gate fires on the push. Same rule, same
124
+ // machinery (it calls doc-currency.mjs, never a second copy of the logic) — moved to the earliest
125
+ // moment it has enough information. On 2026-08-13 four ADRs went stale together and were caught
126
+ // only at push, after three commits, when the work read as a toll booth. A gate at the end cannot
127
+ // shape the work; it can only penalise it, and it trains running at the wall. This one refuses
128
+ // DEBT, not change: you may edit governed code freely, but not while a document governing it is
129
+ // still unreconciled from the last round.
130
+ POLICY('adr-currency', 'adr-currency-gate.mjs', 'node'),
131
+ // spend-guard refuses an agent FLEET that would inherit metered API keys. The $1,600 of
132
+ // agentic-qe#557: ~374 headless agents billed api.anthropic.com per-token for 11 hours while the
133
+ // Claude Max subscription sat unused. The rule was stored, ratified and severity:high — and
134
+ // delivered as advisory text, which is what gets skimmed. `claude` and `codex` are the seats and
135
+ // are never touched; OPENROUTER is metered and deliberately allowed, because cost-optimal routing
136
+ // exists to spend it and a gate that fires on the feature you configured is the gate you disable.
137
+ POLICY('spend-guard', 'spend-guard.mjs', 'node'),
138
+ ];
139
+ const SPEECH = { id: 'unprompted-speech', file: 'unprompted-runtime.mjs', interpreter: 'node' };
140
+
141
+ /** Which policies apply to which PreToolUse sub-event, mirroring the matchers they replaced. */
142
+ const REGISTRY = {
143
+ 'write': ['protect-state', 'hijack-ruvnet', 'ground-before-write', 'adr-currency'],
144
+ // degradation-watch is bash-only on purpose: the acts it guards — `ruflo memory store`, `git
145
+ // push` — are commands, so the dependency is observable there and nowhere else.
146
+ 'bash': ['protect-state', 'identifier-preflight', 'spend-guard', 'degradation-watch', 'hijack-ruvnet', 'design-wall'],
147
+ };
148
+
149
+ export function policiesFor(event, registry = REGISTRY, all = REFUSAL_POLICIES) {
150
+ const ids = registry[event] || [];
151
+ // Ordered by REFUSAL_POLICIES, never by the registry entry — precedence is a property of the
152
+ // policy, not of where someone happened to list it.
153
+ return all.filter((p) => ids.includes(p.id));
154
+ }
155
+
156
+ /**
157
+ * ── APPLICABILITY: THE CHEAPEST POLICY IS THE ONE NEVER SPAWNED ──────────────────────────────────
158
+ *
159
+ * degradation-watch was spawned for EVERY Bash call and then exited 0 on its own second line — its
160
+ * `dependentEvent()` returns null for anything that is not a ship or a memory store, which is nearly
161
+ * everything. Measured here on 2026-08-14: 63-65ms of node boot bought to learn that `ls -la` is not
162
+ * `git push`, on every single Bash tool call, on the machine where a node boot costs 60ms. On the
163
+ * audit's machine that same boot is ~300ms.
164
+ *
165
+ * The predicate is IMPORTED, never restated. Copying the DEPENDENT_COMMANDS regexes up here would
166
+ * make two answers to one question and guarantee they drift — the same reason adr-currency-gate calls
167
+ * doc-currency.mjs instead of carrying a second copy of the logic.
168
+ *
169
+ * FAIL TOWARD RUNNING THE POLICY. If the import fails, or the predicate throws, the policy is
170
+ * spawned exactly as before: this is a latency optimisation and it may never become a way to silently
171
+ * disable a guard.
172
+ */
173
+ let dependentEvent = null; // set from degradation-watch.mjs at startup; null → spawn it as before
174
+ /**
175
+ * Resolved once per invocation by the runtime block below; null means no bash on this host.
176
+ *
177
+ * Declared HERE, above that block, and not next to runPolicy() where it reads more naturally: the
178
+ * `if (isMain())` block runs during module evaluation, so a `let` declared after it sits in the
179
+ * temporal dead zone and the assignment throws — the identical mistake `speechEventFor` was already
180
+ * a hoisted `function` to avoid, recorded a few lines further down.
181
+ */
182
+ let BASH = null;
183
+ const APPLICABILITY = {
184
+ 'degradation-watch': (input) => (dependentEvent ? Boolean(dependentEvent(input.command)) : true),
185
+ };
186
+
187
+ /** Returns a skip reason, or null if the policy must be consulted. */
188
+ export function skipReason(policy, toolInput, table = APPLICABILITY) {
189
+ const test = table[policy.id];
190
+ if (!test) return null;
191
+ try { return test(toolInput || {}) ? null : 'not-applicable'; } catch { return null; }
192
+ }
193
+
194
+ /**
195
+ * Compose one decision from many verdicts.
196
+ *
197
+ * Pure and exported so the precedence rule is testable without spawning anything — the rule is the
198
+ * product here, and a rule only provable by running four bash scripts is a rule nobody re-checks.
199
+ */
200
+ export function decide(verdicts) {
201
+ const refusals = verdicts.filter((v) => v.code === REFUSE);
202
+ if (!refusals.length) return { allow: true, refusals: [] };
203
+ const [first, ...also] = refusals;
204
+ // The winning policy's own words are the message — it wrote them for this moment, and replacing
205
+ // them with a summary of our own would lose the specific instruction the user needs.
206
+ let reason = (first.stderr || '').trim() || `refused by ${first.id} (no reason given)`;
207
+ if (also.length) {
208
+ // NAMING THE OTHERS IS THE POINT. Under four racing hooks the user fixed the first refusal, ran
209
+ // the command again, and hit the second — with no way to know it was there. One round-trip per
210
+ // wall is how a guard becomes something people route around.
211
+ reason += `\n\n Also refusing this action (fix these too, or they will stop you next):\n`
212
+ + also.map((v) => ` · ${v.id}: ${firstLine(v.stderr) || 'no reason given'}`).join('\n');
213
+ }
214
+ return { allow: false, refusals: refusals.map((v) => v.id), reason };
215
+ }
216
+
217
+ const firstLine = (s) => String(s || '').trim().split('\n').map((l) => l.trim()).filter(Boolean)[0] || '';
218
+
219
+ /** Payload accessors — tolerant, because a malformed payload must degrade to "no measurement". */
220
+ function parsed(payload) { try { const o = JSON.parse(payload); return o && typeof o === 'object' ? o : {}; } catch { return {}; } }
221
+ function sessionOf(payload) { return String(parsed(payload).session_id || ''); }
222
+ function payloadTool(payload) { return String(parsed(payload).tool_name || ''); }
223
+ function payloadInput(payload) { return parsed(payload).tool_input || {}; }
224
+
225
+ /**
226
+ * The sub-event token unprompted-runtime switches on.
227
+ *
228
+ * A hoisted `function`, not a `const` arrow: the `if (isMain())` block below runs DURING module
229
+ * evaluation, so an arrow declared after it sits in the temporal dead zone — it threw
230
+ * `Cannot access 'speechEventFor' before initialization` on this gate's first live refusal.
231
+ */
232
+ function speechEventFor(event) { return event === 'bash' ? 'PreToolUse-bash' : 'PreToolUse-write'; }
233
+
234
+ // ── Runtime ──────────────────────────────────────────────────────────────────────────────────────
235
+
236
+
237
+ if (isMain()) {
238
+ const started = Date.now();
239
+ const payload = readPayload();
240
+ const selected = policiesFor(EVENT);
241
+ // An unknown event is not an occasion to refuse anything. Same rule as unprompted-runtime's
242
+ // "never speak on a guess", pointed at the other decision.
243
+ if (!selected.length && EVENT !== 'write' && EVENT !== 'bash') process.exit(ALLOW);
244
+
245
+ const budgetMs = Number(process.env.RUVNET_DECISION_BUDGET_MS) || DEFAULT_BUDGET_MS;
246
+ const deadline = started + budgetMs;
247
+ // Resolved ONCE. On win32 resolveBash() can shell out to `where.exe`; four bash policies meant up
248
+ // to four of those per tool call, for an answer that cannot change mid-invocation.
249
+ BASH = resolveBash();
250
+ // Best-effort, and deliberately not a static import: a missing or broken degradation-watch.mjs
251
+ // must cost us the optimisation, not the whole gate. `runPolicy` already tolerates a missing
252
+ // policy file; a top-level `import` of it would have made that tolerance a lie.
253
+ try { ({ dependentEvent } = await import('./degradation-watch.mjs')); } catch { dependentEvent = null; }
254
+
255
+ const trace = []; // one row per policy — surfaced by RUVNET_DECISION_TRACE=1
256
+ const unconsulted = []; // policies the budget cost us. NEVER silent; see reportBudget().
257
+ const toolInput = payloadInput(payload);
258
+ const consulted = [];
259
+ for (const p of selected) {
260
+ const why = skipReason(p, toolInput);
261
+ if (why) trace.push({ id: p.id, ms: 0, skipped: why });
262
+ else consulted.push(p);
263
+ }
264
+
265
+ // ── PARALLEL, and the reason is arithmetic ─────────────────────────────────────────────────────
266
+ // Sequentially the gate's wall time was SUM(policies); run together it is MAX(policies). Measured
267
+ // on a `git push` payload by the 2026-08-13 audit: 182 + 345 + 2145..3990 + 629 + 743 ≈ 4.0-5.9s
268
+ // sequential, against a 5s host timeout. Nothing about these policies wanted to be sequential —
269
+ // they share no state, write no files (only stderr), and `decide()` re-sorts the verdicts into
270
+ // REFUSAL_POLICIES precedence order regardless of which finished first. Every selected policy ran
271
+ // on every call before this change too: the gate never short-circuited on the first refusal,
272
+ // because naming EVERY wall is the whole point of ADR-067.
273
+ const results = await Promise.all(consulted.map((p) => runPolicy(p, payload, deadline, undefined, trace)));
274
+ const verdicts = results.filter((r) => typeof r.code === 'number');
275
+ for (const r of results) if (r.skipped === 'budget') unconsulted.push(r.id);
276
+
277
+ const decision = decide(verdicts);
278
+
279
+ // ── OBEDIENCE MEASUREMENT (ADR-067 §outcomes) ──────────────────────────────────────────────────
280
+ // This gate is the ONLY thing that sees every Write/Edit/Bash, so it can close the loop with no new
281
+ // hook: resolve first (did this call retry something we refused?), then open a new debt if we are
282
+ // about to refuse. Order matters — resolving after recording would close the debt we just opened.
283
+ // Entirely best-effort: measurement may never affect the verdict, so it runs after `decide`.
284
+ const session = sessionOf(payload);
285
+ try {
286
+ const key = actionKey(payloadTool(payload), toolInput);
287
+ const ts = Date.now();
288
+ if (session && key) {
289
+ sweepStale({ session, ts }); // debts from dead sessions become `abandoned`, never vanish
290
+ resolveOutcome({ session, key, allowed: decision.allow, ts });
291
+ if (!decision.allow) recordRefusal({ session, key, policies: decision.refusals, ts });
292
+ }
293
+ } catch { /* a ledger must never break a tool call */ }
294
+
295
+ if (!decision.allow) {
296
+ reportBudget({ session, unconsulted, trace, started, budgetMs });
297
+ process.stderr.write(`${decision.reason}\n`);
298
+ process.exit(REFUSE);
299
+ }
300
+
301
+ // Nothing refused: run the speech chokepoint and forward its envelope verbatim. It owns its own
302
+ // per-channel policy; this gate does not inspect or re-decide anything it says.
303
+ //
304
+ // SEQUENTIAL ON PURPOSE, unlike the batch above. unprompted-runtime.mjs records an OFFERED row in
305
+ // the advocacy ledger when it delivers (its line ~372), so starting it in parallel and discarding
306
+ // its stdout after a refusal would book an offer the user never saw — inflating the denominator
307
+ // this project has a CI gate against fabricating. A saved ~280ms is not worth a fabricated number.
308
+ if (deadline - Date.now() <= 0) {
309
+ unconsulted.push(SPEECH.id);
310
+ } else {
311
+ const speech = await runPolicy(SPEECH, payload, deadline, speechEventFor(EVENT), trace);
312
+ if (speech.skipped === 'budget') unconsulted.push(SPEECH.id);
313
+ if (speech.code === REFUSE) {
314
+ reportBudget({ session, unconsulted, trace, started, budgetMs });
315
+ process.stderr.write(`${(speech.stderr || '').trim()}\n`);
316
+ process.exit(REFUSE);
317
+ }
318
+ reportBudget({ session, unconsulted, trace, started, budgetMs });
319
+ if (speech.stdout?.trim()) process.stdout.write(speech.stdout);
320
+ process.exit(ALLOW);
321
+ }
322
+ reportBudget({ session, unconsulted, trace, started, budgetMs });
323
+ process.exit(ALLOW);
324
+ }
325
+
326
+ /**
327
+ * ── A BUDGET THAT CAN BE EXCEEDED SILENTLY FAILS OPEN WITHOUT SAYING SO ──────────────────────────
328
+ *
329
+ * The old loop did `break` when the budget ran out. Every remaining policy AND the speech chokepoint
330
+ * were then skipped with no record anywhere — the gate allowed, and nothing distinguished "five
331
+ * policies agreed this was fine" from "we ran out of time and stopped asking". That is this repo's
332
+ * signature defect wearing a different hat: silence standing in for a measurement. degradation-watch
333
+ * exists because a warning was printed and skimmed; this exists because nothing was printed at all.
334
+ *
335
+ * TWO CHANNELS, because each fails differently:
336
+ * · the outcome ledger (~/.config/ruvnet-brain/, survives `--update`) so a trip COMPOUNDS into
337
+ * evidence instead of scrolling past. `kind: 'budget-exceeded'` is inert in report()'s buckets —
338
+ * it counts neither as a refusal nor as a resolution, so it cannot move the obedience rate.
339
+ * · one stderr line, unconditionally, allow or refuse. Yes, that puts bytes on stderr during an
340
+ * exit-0 allow, which tests/unit/decision-gate.test.mjs asserts never happens on an ordinary
341
+ * write. That assertion is now a SECOND tripwire and is meant to be: after the parallel batch and
342
+ * the applicability skip, an ordinary write measures ~360ms against a 4000ms budget, so a trip
343
+ * there is not noise to be tolerated — it is the defect, and the suite should go red for it.
344
+ */
345
+ function reportBudget({ session, unconsulted, trace, started, budgetMs }) {
346
+ const elapsed = Date.now() - started;
347
+ if (process.env.RUVNET_DECISION_TRACE === '1') {
348
+ process.stderr.write(`[decision-gate] ${EVENT} ${elapsed}ms budget=${budgetMs}ms ${JSON.stringify(trace)}\n`);
349
+ }
350
+ if (!unconsulted.length) return;
351
+ try {
352
+ appendOutcome({ kind: 'budget-exceeded', event: EVENT, session, unconsulted, elapsedMs: elapsed, budgetMs, ts: Date.now() });
353
+ } catch { /* a ledger must never break a tool call */ }
354
+ process.stderr.write(
355
+ `[decision-gate] ${budgetMs}ms budget exhausted after ${elapsed}ms — ALLOWED WITHOUT CONSULTING: `
356
+ + `${unconsulted.join(', ')}. These policies did not vote; this allow is a timeout, not a verdict.\n`,
357
+ );
358
+ }
359
+
360
+ /**
361
+ * Run one policy as a CAPTURED child. Never lets its bytes touch the real streams.
362
+ *
363
+ * Always resolves, never rejects, and always to an object — `{ id, code }` for a real verdict, or
364
+ * `{ id, skipped }` for anything else. The old version returned bare `null` for a missing file, a
365
+ * missing bash, a crash AND a timeout alike, which is precisely why a blown budget could not be
366
+ * reported: by the time the caller saw the result, the reason was gone.
367
+ */
368
+ function runPolicy(p, payload, deadline, extraArg, trace) {
369
+ const t0 = Date.now();
370
+ const done = (r) => {
371
+ trace?.push({ id: p.id, ms: Date.now() - t0, ...(r.skipped ? { skipped: r.skipped } : { code: r.code }) });
372
+ return r;
373
+ };
374
+ const file = path.join(SCRIPTS_DIR, p.file);
375
+ if (!fs.existsSync(file)) return Promise.resolve(done({ id: p.id, skipped: 'missing' }));
376
+ let cmd; const args = [file];
377
+ if (p.interpreter === 'bash') {
378
+ if (!BASH) return Promise.resolve(done({ id: p.id, skipped: 'no-bash' })); // this policy cannot speak here
379
+ cmd = BASH;
380
+ } else {
381
+ cmd = process.execPath;
382
+ }
383
+ if (extraArg) args.push(extraArg);
384
+ const left = deadline - Date.now();
385
+ if (left <= 0) return Promise.resolve(done({ id: p.id, skipped: 'budget' }));
386
+
387
+ return new Promise((resolve) => {
388
+ let settled = false;
389
+ let timer = null;
390
+ const finish = (r) => { if (settled) return; settled = true; clearTimeout(timer); resolve(done(r)); };
391
+ let child;
392
+ try {
393
+ child = spawn(cmd, args, { stdio: ['pipe', 'pipe', 'pipe'], env: { ...process.env, RUVNET_DECISION_GATE: '1' } });
394
+ } catch { return finish({ id: p.id, skipped: 'spawn' }); }
395
+ // SIGKILL, not SIGTERM: a bash policy that has spawned its own child (jq, node, ruflo) can sit in
396
+ // a TERM handler, and the host's own kill is what we are racing. The whole batch shares ONE
397
+ // deadline, so a single slow policy cancels only the time it actually consumed.
398
+ timer = setTimeout(() => { try { child.kill('SIGKILL'); } catch { /* already gone */ } finish({ id: p.id, skipped: 'budget' }); }, left);
399
+ let stdout = ''; let stderr = ''; let bytes = 0;
400
+ const MAX = 1 << 20; // same ceiling spawnSync's maxBuffer enforced; a policy is not a data source
401
+ child.stdout.on('data', (d) => { if (bytes < MAX) { stdout += d; bytes += d.length; } });
402
+ child.stderr.on('data', (d) => { if (bytes < MAX) { stderr += d; bytes += d.length; } });
403
+ child.on('error', () => finish({ id: p.id, skipped: 'spawn' }));
404
+ // EPIPE when a policy exits before reading its payload (degradation-watch's fast path does).
405
+ // Unhandled, that error event would take the whole gate down and turn an allow into a hook error.
406
+ child.stdin.on('error', () => { /* the child did not want the payload; that is not a failure */ });
407
+ child.on('close', (code) => {
408
+ // A spawn failure, a timeout, or any code other than 0/2 is an ERROR — and an error here must
409
+ // never be mistaken for a refusal. That distinction is the one lesson-gate.mjs had to learn twice.
410
+ if (code !== ALLOW && code !== REFUSE) return finish({ id: p.id, skipped: `exit:${code}` });
411
+ finish({ id: p.id, code, stderr, stdout });
412
+ });
413
+ try { child.stdin.end(payload); } catch { /* handled by the stdin error listener above */ }
414
+ });
415
+ }
416
+
417
+ function readPayload() {
418
+ if (process.stdin.isTTY) return '';
419
+ try { return fs.readFileSync(0, 'utf8'); } catch { return ''; }
420
+ }
421
+
422
+ /** Never able to crash a caller that merely imported this (see tests/unit/entrypoint-guard-safety). */
423
+ function isMain() {
424
+ try {
425
+ if (!process.argv[1]) return false;
426
+ return fs.realpathSync(process.argv[1]) === fs.realpathSync(fileURLToPath(import.meta.url));
427
+ } catch { return false; }
428
+ }
@@ -0,0 +1,231 @@
1
+ /**
2
+ * decision-outcomes.mjs — did the refusal TEACH, or just stop them?
3
+ *
4
+ * ─────────────────────────────────────────────────────────────────────────────────────────────────
5
+ * THE GAP THIS CLOSES. ADR-066's own honesty boundary says it plainly: "Delivery is proven;
6
+ * *obedience is not measured*." The brain could show a lesson, refuse an action, and never once know
7
+ * whether either changed anything. And `lesson-stamps-prove-ceremony-not-obedience` — itself one of
8
+ * the bridged lessons — names exactly that failure: a stamp proves the ritual ran, not that its
9
+ * answer was obeyed.
10
+ *
11
+ * A system that cannot see its own outcomes cannot improve, and one that reports a number it cannot
12
+ * source is doing the fabrication this repo has a CI gate against.
13
+ *
14
+ * ─────────────────────────────────────────────────────────────────────────────────────────────────
15
+ * WHAT IS ACTUALLY OBSERVABLE, and this is the honest part.
16
+ *
17
+ * "Did the model obey an advisory?" is NOT observable from a hook. The advisory reaches the model's
18
+ * context and what it does next is unconstrained prose. Claiming to measure that would be the
19
+ * inflated-score failure mode.
20
+ *
21
+ * What IS observable, exactly, from the one gate that sees every Write/Edit/Bash: **what happened
22
+ * after a REFUSAL**. Three outcomes, mutually exclusive and jointly exhaustive:
23
+ *
24
+ * corrected the same target was attempted again and was ALLOWED — the reason landed, the user
25
+ * or model fixed the thing and proceeded. This is the outcome the gate exists to cause.
26
+ * repeated the same target was attempted again and refused again — the reason did NOT land.
27
+ * A guard that produces this is teaching nothing and is on its way to being switched off.
28
+ * abandoned never attempted again in this session. Ambiguous ON PURPOSE and never counted as a
29
+ * win: it is equally "they understood and stopped" and "they gave up and worked around
30
+ * us", and this ledger must not pick the flattering reading.
31
+ *
32
+ * THE ONE WAY TO FABRICATE THIS, named so a reviewer can check for it: record only `corrected`. The
33
+ * rate is corrected ÷ (corrected + repeated + abandoned), so a caller that forgets the other two
34
+ * reports a perfect score. The invariant is therefore NOT "record outcomes" but **every refusal
35
+ * produces exactly one record**, and `abandoned` is what an unresolved refusal becomes at session
36
+ * end. Same invariant, same reasoning, as advocacy-outcomes.mjs — one pattern, not two.
37
+ *
38
+ * SCOPE, stated rather than implied: this measures the BLOCKING path only. Advisory lessons are
39
+ * counted as `surfaced` for coverage and are NEVER scored, because their effect is not observable
40
+ * here. A coverage number that hides what it cannot see is the same lie as a truncated list.
41
+ *
42
+ * STORAGE: ~/.config/ruvnet-brain/, user-level, deliberately OUTSIDE ~/.cache/ruvnet-brain/ which
43
+ * `--update` replaces wholesale — an outcome destroyed by the next release never compounds.
44
+ * Append-only JSONL, bounded, node builtins only, and every write is best-effort: a ledger that
45
+ * could break a tool call would be worse than no ledger.
46
+ */
47
+ import fs from 'node:fs';
48
+ import os from 'node:os';
49
+ import path from 'node:path';
50
+
51
+ export const CONFIG_ROOT = process.env.RUVNET_CONFIG_ROOT
52
+ || path.join(os.homedir(), '.config', 'ruvnet-brain');
53
+ export const LEDGER = process.env.RUVNET_DECISION_LEDGER
54
+ || path.join(CONFIG_ROOT, 'decision-outcomes.jsonl');
55
+ /** Pending refusals awaiting an outcome. Separate from the ledger so a resolve is one small write. */
56
+ export const PENDING = process.env.RUVNET_DECISION_PENDING
57
+ || path.join(CONFIG_ROOT, 'decision-pending.json');
58
+
59
+ const MAX_LEDGER_BYTES = 1 << 20; // ~1MB, then the oldest half is dropped
60
+ const MAX_PENDING = 200;
61
+
62
+ /**
63
+ * The identity of "the same action being retried".
64
+ *
65
+ * Deliberately COARSE: tool + target, never the full input. A model that fixes a refusal usually
66
+ * changes the content while keeping the target, and keying on content would score every correction
67
+ * as a brand-new action and make `repeated` unreachable — a metric that can only produce good news.
68
+ */
69
+ export function actionKey(toolName, toolInput) {
70
+ const t = String(toolName || '').toLowerCase();
71
+ const i = toolInput || {};
72
+ if (t === 'bash') {
73
+ // First two words of the command: `git commit -m …` and `git commit -m …else` are one action.
74
+ return `bash:${String(i.command || '').trim().split(/\s+/).slice(0, 2).join(' ')}`;
75
+ }
76
+ return `${t}:${String(i.file_path || i.path || '').trim()}`;
77
+ }
78
+
79
+ const readJson = (file, fallback) => {
80
+ try { return JSON.parse(fs.readFileSync(file, 'utf8')); } catch { return fallback; }
81
+ };
82
+ const writeJson = (file, value) => {
83
+ try {
84
+ fs.mkdirSync(path.dirname(file), { recursive: true });
85
+ fs.writeFileSync(file, JSON.stringify(value));
86
+ return true;
87
+ } catch { return false; }
88
+ };
89
+
90
+ /** Append one outcome. Never throws; a full disk must not break a tool call. */
91
+ export function append(record, file = LEDGER) {
92
+ try {
93
+ fs.mkdirSync(path.dirname(file), { recursive: true });
94
+ // Bound the file BEFORE writing, so it can never grow without limit on a long-lived machine.
95
+ try {
96
+ if (fs.statSync(file).size > MAX_LEDGER_BYTES) {
97
+ const lines = fs.readFileSync(file, 'utf8').split('\n').filter(Boolean);
98
+ fs.writeFileSync(file, `${lines.slice(Math.floor(lines.length / 2)).join('\n')}\n`);
99
+ }
100
+ } catch { /* no file yet */ }
101
+ fs.appendFileSync(file, `${JSON.stringify(record)}\n`);
102
+ return true;
103
+ } catch { return false; }
104
+ }
105
+
106
+ /**
107
+ * Record that a refusal happened, and open a debt that must resolve to exactly one outcome.
108
+ * `ts` is passed in rather than read from the clock so the caller owns time and tests are hermetic.
109
+ */
110
+ export function recordRefusal({ session, key, policies, ts }, files = {}) {
111
+ const pendingFile = files.pending || PENDING;
112
+ const ledger = files.ledger || LEDGER;
113
+ append({ kind: 'refused', session, key, policies, ts }, ledger);
114
+ const pending = readJson(pendingFile, {});
115
+ pending[`${session}\u0000${key}`] = { session, key, policies, ts };
116
+ // Bound it: on overflow the OLDEST debts are resolved as `abandoned` rather than silently dropped,
117
+ // because a dropped debt is a miss that never appears in the denominator.
118
+ const entries = Object.entries(pending);
119
+ if (entries.length > MAX_PENDING) {
120
+ const sorted = entries.sort((a, b) => (a[1].ts || 0) - (b[1].ts || 0));
121
+ for (const [k, v] of sorted.slice(0, entries.length - MAX_PENDING)) {
122
+ append({ kind: 'abandoned', session: v.session, key: v.key, policies: v.policies, ts }, ledger);
123
+ delete pending[k];
124
+ }
125
+ }
126
+ writeJson(pendingFile, pending);
127
+ }
128
+
129
+ /**
130
+ * A tool call is happening. If it retries something we refused, close that debt.
131
+ * Called by the gate BEFORE it decides, so `allowed` is the verdict it is about to return.
132
+ */
133
+ export function resolve({ session, key, allowed, ts }, files = {}) {
134
+ const pendingFile = files.pending || PENDING;
135
+ const ledger = files.ledger || LEDGER;
136
+ const pending = readJson(pendingFile, {});
137
+ const id = `${session}\u0000${key}`;
138
+ const debt = pending[id];
139
+ if (!debt) return null;
140
+ const kind = allowed ? 'corrected' : 'repeated';
141
+ append({ kind, session, key, policies: debt.policies, ts, afterMs: ts - (debt.ts || ts) }, ledger);
142
+ delete pending[id];
143
+ writeJson(pendingFile, pending);
144
+ return kind;
145
+ }
146
+
147
+ /**
148
+ * Close debts that will never resolve, as `abandoned`.
149
+ *
150
+ * WITHOUT THIS THE METRIC IS A LIE BY OMISSION. A refusal the model simply walked away from is the
151
+ * single most likely outcome — and an unresolved debt sits in `open`, outside the denominator, so
152
+ * `correctedRate` would be computed only over the actions someone bothered to retry. That is the
153
+ * "record only the wins" fabrication this file's header names, arriving through the back door.
154
+ *
155
+ * Swept from the gate itself rather than a SessionEnd hook: SessionEnd is not guaranteed to fire (a
156
+ * crash, a kill, a compact all skip it), and this repo has already paid for a queue that only drained
157
+ * on a graceful exit — ADR-027's heartbeat flush, where 1,884 events accumulated over days. Activity
158
+ * is the trigger, not politeness.
159
+ */
160
+ export function sweepStale({ session, ts, maxAgeMs = 6 * 60 * 60_000 }, files = {}) {
161
+ const pendingFile = files.pending || PENDING;
162
+ const ledger = files.ledger || LEDGER;
163
+ const pending = readJson(pendingFile, {});
164
+ let closed = 0;
165
+ for (const [k, v] of Object.entries(pending)) {
166
+ // A different session is over as far as this process can tell; the current session's debts stay
167
+ // open until they resolve or age out, because a retry three calls later is still a real outcome.
168
+ const stale = v.session !== session || (ts - (v.ts || ts)) > maxAgeMs;
169
+ if (!stale) continue;
170
+ append({ kind: 'abandoned', session: v.session, key: v.key, policies: v.policies, ts }, ledger);
171
+ delete pending[k];
172
+ closed += 1;
173
+ }
174
+ if (closed) writeJson(pendingFile, pending);
175
+ return closed;
176
+ }
177
+
178
+ /** Close every open debt for a session as `abandoned`. Called at SessionEnd. */
179
+ export function abandonSession(session, ts, files = {}) {
180
+ const pendingFile = files.pending || PENDING;
181
+ const ledger = files.ledger || LEDGER;
182
+ const pending = readJson(pendingFile, {});
183
+ let closed = 0;
184
+ for (const [k, v] of Object.entries(pending)) {
185
+ if (v.session !== session) continue;
186
+ append({ kind: 'abandoned', session, key: v.key, policies: v.policies, ts }, ledger);
187
+ delete pending[k];
188
+ closed += 1;
189
+ }
190
+ if (closed) writeJson(pendingFile, pending);
191
+ return closed;
192
+ }
193
+
194
+ /**
195
+ * The report. Every refusal must appear in exactly one terminal bucket, and `open` is stated
196
+ * separately so a half-finished session cannot inflate the rate by shrinking the denominator.
197
+ */
198
+ export function report(files = {}) {
199
+ const ledger = files.ledger || LEDGER;
200
+ const pendingFile = files.pending || PENDING;
201
+ let lines = [];
202
+ try { lines = fs.readFileSync(ledger, 'utf8').split('\n').filter(Boolean); } catch { /* none yet */ }
203
+ const rows = lines.map((l) => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
204
+ const count = (k) => rows.filter((r) => r.kind === k).length;
205
+ const refused = count('refused');
206
+ const corrected = count('corrected');
207
+ const repeated = count('repeated');
208
+ const abandoned = count('abandoned');
209
+ const resolved = corrected + repeated + abandoned;
210
+ const open = Object.keys(readJson(pendingFile, {})).length;
211
+ const byPolicy = {};
212
+ for (const r of rows) {
213
+ if (!['corrected', 'repeated', 'abandoned'].includes(r.kind)) continue;
214
+ for (const p of r.policies || []) {
215
+ byPolicy[p] = byPolicy[p] || { corrected: 0, repeated: 0, abandoned: 0 };
216
+ byPolicy[p][r.kind] += 1;
217
+ }
218
+ }
219
+ return {
220
+ refused,
221
+ resolved,
222
+ open,
223
+ corrected,
224
+ repeated,
225
+ abandoned,
226
+ // NULL, not 0, when nothing has resolved. A rate printed as 0% on an empty ledger is a claim the
227
+ // guards are failing; the truth is that nothing has been measured yet.
228
+ correctedRate: resolved ? +(corrected / resolved).toFixed(3) : null,
229
+ byPolicy,
230
+ };
231
+ }