ruvnet-brain 4.3.37 → 4.3.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +2 -2
  2. package/bin/install.mjs +32 -9
  3. package/kb/forge-update.mjs +1884 -0
  4. package/package.json +3 -1
  5. package/plugin/.claude-plugin/plugin.json +1 -1
  6. package/plugin/.codex-plugin/plugin.json +1 -1
  7. package/plugin/hooks/codex-hooks.json +4 -4
  8. package/plugin/hooks/hook-contracts.json +7 -7
  9. package/plugin/hooks/hooks.json +2 -2
  10. package/plugin/scripts/capability-claim-evidence.mjs +11 -2
  11. package/plugin/scripts/capacity-aware-parallel-work.mjs +71 -15
  12. package/plugin/scripts/codex-hook-wrapper.mjs +3 -0
  13. package/plugin/scripts/completion-claim-evidence.mjs +262 -0
  14. package/plugin/scripts/continuation-gate.mjs +105 -21
  15. package/plugin/scripts/continuation-objective.mjs +15 -0
  16. package/plugin/scripts/decision-gate.mjs +22 -2
  17. package/plugin/scripts/duplicate-gate.mjs +503 -0
  18. package/plugin/scripts/grounding-turn-evidence.mjs +339 -0
  19. package/plugin/scripts/grounding-turn-gate.mjs +77 -25
  20. package/plugin/scripts/grounding-turn-mark.mjs +61 -14
  21. package/plugin/scripts/hook-input.mjs +15 -0
  22. package/plugin/scripts/host-update.mjs +45 -0
  23. package/plugin/scripts/nightly-scheduler.mjs +34 -0
  24. package/plugin/scripts/session-start-budget.mjs +1 -0
  25. package/plugin/scripts/session-start-core.mjs +15 -2
  26. package/plugin/scripts/session-start-health.mjs +101 -1
  27. package/plugin/scripts/session-start-update-plane.mjs +71 -1
  28. package/scripts/approved-runtime.mjs +1 -1
  29. package/scripts/completion-claim-replay.mjs +114 -0
  30. package/scripts/corpus-canary.mjs +396 -0
  31. package/scripts/corpus-dispatch-decision.mjs +2 -2
  32. package/scripts/corpus-promotion.mjs +49 -0
  33. package/scripts/corpus-reconcile.mjs +27 -3
  34. package/scripts/corpus-watchdog.mjs +45 -6
  35. package/scripts/duplicate-gate-replay.mjs +98 -0
  36. package/scripts/grounding-turn-replay.mjs +131 -0
  37. package/scripts/nightly-watchdog.mjs +3 -3
  38. package/scripts/protected-release-invocation.mjs +1 -1
  39. package/scripts/release.mjs +180 -77
  40. package/scripts/single-source-check.mjs +12 -6
  41. package/scripts/wired-check.mjs +2 -0
@@ -0,0 +1,339 @@
1
+ /**
2
+ * grounding-turn-evidence.mjs — the pure half of ADR-0030 decision-point gate #1: "before asserting
3
+ * what a tool or platform can or cannot do, did you CHECK a relevant source this turn, or are you
4
+ * recalling?" Plus gates #2 (architecture needs >= 3 options) and #3 (relayed numbers need a
5
+ * re-check), which run in SHADOW mode only: they are measured and logged, never delivered.
6
+ *
7
+ * WHY (owner, 2026-09-30): "a nudge without enforcement is worthless." The concrete incident: an
8
+ * answer asserted "No hook can change the model of the current turn" from a WebFetch page summary
9
+ * written by a small model, and a design was built on it. The existing grounding-turn-gate only
10
+ * checked that SOME search_ruvnet stamp was newer than the turn marker — presence, not relevance or
11
+ * order — armed only on rUv-stack prompts, and could not see WebFetch at all.
12
+ *
13
+ * DESIGN, deterministic and local (rUv ADR-G004 rejects LLM gate evaluation; no model is called):
14
+ * 1. UserPromptSubmit (grounding-turn-mark.mjs) arms the turn only when the prompt ASKS for a
15
+ * capability / feasibility / architecture judgement AND names a subject (classifyPrompt). The
16
+ * caller replayed a naive output-only regex at 13.5% of turns and an order-aware one at 0.93%,
17
+ * mostly on harmless hedges; arming at prompt time is what keeps unarmed turns out of scope.
18
+ * 2. The sources read this turn come from the host transcript (turnSources) — the ordered,
19
+ * complete record of every tool call and its result, including search_ruvnet's returned paths.
20
+ * A WebFetch body is a small model's summary of the page, so it is WEAK evidence; so is a
21
+ * subagent's relayed report and a WebSearch snippet list.
22
+ * 3. At Stop, each capability claim in the final answer (capability-claim-evidence.mjs's own
23
+ * extractor, widened with this turn's vocabulary) needs one STRONG source whose path, URL,
24
+ * command or query names the claim's subject and that was read AFTER the last weak source about
25
+ * that subject (auditAssertions). Otherwise: one correction.
26
+ *
27
+ * Claims continuation-gate.mjs already audits (the RUVNET_TOOL behaviour class) are skipped here —
28
+ * one correction per claim, never two gates arguing over the same sentence.
29
+ */
30
+ import fs from 'node:fs';
31
+ import os from 'node:os';
32
+ import path from 'node:path';
33
+ import { fileURLToPath } from 'node:url';
34
+ import { RUVNET_GATE1_TERMS } from './ruvnet-gate1-pattern.mjs';
35
+ import { extractClaims } from './capability-claim-evidence.mjs';
36
+ import { currentTurnRecords, strippedProse } from './completion-claim-evidence.mjs';
37
+
38
+ const SCRIPTS_DIR = path.dirname(fileURLToPath(import.meta.url));
39
+
40
+ // ── vocabulary ─────────────────────────────────────────────────────────────────────────────────────
41
+ /** Platform nouns a capability question is usually about, beyond the product names. */
42
+ const PLATFORM_NOUNS = ['hook', 'hooks', 'mcp', 'plugin', 'plugins', 'skill', 'skills', 'subagent', 'subagents',
43
+ 'codex', 'claude code', 'cli', 'sdk', 'api', 'npm', 'github actions', 'workflow', 'launchd', 'cron',
44
+ 'vercel', 'webfetch', 'websearch', 'hnsw', 'onnx', 'sqlite', 'rvf', 'metaharness', 'ruvllm', 'ruview',
45
+ 'agentdb', 'ruflo', 'ruvector', 'aidefence', 'agentic-flow', 'agentic-qe', 'claude flow', 'agent browser'];
46
+ const WEAK_WORDS = new Set(['the', 'and', 'for', 'with', 'this', 'that', 'brain', 'repo', 'store', 'docs', 'data',
47
+ 'test', 'tests', 'main', 'core', 'app', 'web', 'site', 'code', 'tool', 'tools', 'model', 'models', 'agent',
48
+ 'agents', 'file', 'files', 'flow', 'swarm', 'ruv', 'sparc', 'qe']);
49
+
50
+ const brainHome = (env) => env.RUVNET_BRAIN_HOME || path.join(env.HOME || env.USERPROFILE || os.homedir(), '.cache', 'ruvnet-brain');
51
+
52
+ /** Subject phrases: Gate-1 terms, platform nouns, repo aliases and installed store names. Never throws. */
53
+ export function loadVocabulary({ env = process.env } = {}) {
54
+ const out = new Set([...RUVNET_GATE1_TERMS, ...PLATFORM_NOUNS]);
55
+ const kbDirs = [env.RUVNET_KB_DIR, path.join(brainHome(env), 'kb'), path.join(SCRIPTS_DIR, '..', '..', 'kb')].filter(Boolean);
56
+ for (const dir of kbDirs) {
57
+ try {
58
+ const aliases = JSON.parse(fs.readFileSync(path.join(dir, 'repo-aliases.json'), 'utf8'));
59
+ for (const [repo, list] of Object.entries(aliases || {})) { out.add(repo); for (const a of list || []) out.add(a); }
60
+ } catch { /* absent in this location */ }
61
+ try {
62
+ // Installed store names — but a plain single word ("support", "concepts", "marketing") is an
63
+ // English word far more often than a product; measured, it produced "doesn't support" claims.
64
+ for (const name of fs.readdirSync(dir)) {
65
+ if (!name.endsWith('.meta.json')) continue;
66
+ const store = name.slice(0, -'.meta.json'.length);
67
+ if (/[-_.0-9]/.test(store) || /^(?:ru|rv|agentic|cognitum)/i.test(store)) out.add(store);
68
+ }
69
+ } catch { /* absent in this location */ }
70
+ }
71
+ return [...out].map((v) => String(v).toLowerCase().trim())
72
+ .filter((v) => v.length >= 3 && !WEAK_WORDS.has(v) && /[a-z]/.test(v));
73
+ }
74
+
75
+ // ── tokens ─────────────────────────────────────────────────────────────────────────────────────────
76
+ const singular = (t) => (t.length > 4 && t.endsWith('s') && !t.endsWith('ss') ? t.slice(0, -1) : t);
77
+ /** Lowercased word tokens: each hyphen/dot/underscore compound (separators normalised to '-') and its
78
+ * parts, singularised — so `code.claude.com/docs/en/hooks` yields claude, code, hook, … */
79
+ export function tokenSet(text) {
80
+ const set = new Set();
81
+ const raw = String(text || '');
82
+ const s = `${raw.replace(/([a-z0-9])([A-Z])/g, '$1 $2')} ${raw}`.toLowerCase();
83
+ for (const compound of s.match(/[a-z0-9]+(?:[-_.][a-z0-9]+)*/g) || []) {
84
+ const parts = compound.split(/[-_.]/).filter(Boolean);
85
+ set.add(singular(parts.join('-')));
86
+ for (const part of parts) set.add(singular(part));
87
+ }
88
+ return set;
89
+ }
90
+ /** A subject phrase's words; a source binds the subject only when it names every one of them. */
91
+ const subjectWords = (subject) => String(subject || '').toLowerCase().split(/\s+/)
92
+ .map((w) => singular(w.split(/[-_.]/).filter(Boolean).join('-'))).filter((w) => w.length >= 2);
93
+
94
+ // ── prompt-time classification (UserPromptSubmit) ─────────────────────────────────────────────────
95
+ const CAPABILITY_ASK = new RegExp([
96
+ String.raw`\b(?:can|could|does|do|is|are|will|would)\s+(?:it|we|you|i|there|(?:a|an|the|any|this|that|my|our)\s+[\w-]+|[\w-]+)\s+(?:[\w-]+\s+){0,3}?(?:do|support|handle|work|run|use|call|change|switch|make|force|read|write|access|detect|block|enforce|intercept|see|know|override|rewrite|route|pick|select|stop|prevent|allow|expose|return|send|trigger|fire)\b`,
97
+ String.raw`\bis\s+(?:it|there)\s+(?:possible|a\s+way|any\s+way)\b`, String.raw`\bpossible\s+to\b`, String.raw`\bable\s+to\b`,
98
+ String.raw`\bcapab(?:le|ility|ilities)\b`, String.raw`\bfeasib(?:le|ility)\b`, String.raw`\blimitations?\b`,
99
+ String.raw`\bwhat\s+(?:can|does|do)\b`, String.raw`\bhow\s+(?:does|do|can|could|would|should)\b`,
100
+ String.raw`\bwhether\b`, String.raw`\bsupports?\b`,
101
+ ].join('|'), 'i');
102
+ const ARCHITECTURE_ASK = /\barchitect(?:ure|ural)?\b|\bdesign\b|\bapproach(?:es)?\b|\bbest\s+way\b|\bshould\s+(?:we|i)\s+(?:use|build|go|pick|choose)\b|\brecommend|\bwhich\s+(?:tool|library|option|approach)\b|\btrade-?offs?\b|\bfigure\s+out\s+(?:the\s+)?(?:best|how)\b/i;
103
+ const CODE_ISH = /`([^`\n]{2,60})`|\b([A-Za-z][A-Za-z0-9]*(?:[-_.][A-Za-z0-9]+)+|[a-z]+[A-Z][A-Za-z0-9]+|[A-Z][a-z]+[A-Z][A-Za-z0-9]*)\b/g;
104
+
105
+ /** Vocabulary phrases and code-ish identifiers the text names. */
106
+ export function subjectsIn(text, vocab) {
107
+ const s = String(text || '');
108
+ const lower = s.toLowerCase();
109
+ const found = new Set();
110
+ for (const v of vocab) {
111
+ const re = new RegExp(`(?<![a-z0-9-])${v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}(?![a-z0-9-])`, 'i');
112
+ if (lower.includes(v) && re.test(s)) found.add(v);
113
+ }
114
+ for (const m of s.replace(/<\/?[A-Za-z_][\w-]*[^>]*>/g, ' ').matchAll(CODE_ISH)) {
115
+ const t = (m[1] || m[2] || '').trim();
116
+ if (t.length >= 4 && t.length <= 60 && !/^\d|\.(?:md|json|txt|png|jpg)$|^https?:|^e\.g\.?$|^i\.e\.?$/i.test(t) && !/\s/.test(t)) found.add(t.toLowerCase());
117
+ if (found.size >= 24) break;
118
+ }
119
+ return [...found].slice(0, 24);
120
+ }
121
+
122
+ /** Arm the turn when the prompt asks a capability/feasibility/architecture question about a subject. */
123
+ export function classifyPrompt(text, vocab) {
124
+ const t = String(text || '');
125
+ const subjects = subjectsIn(t, vocab);
126
+ const architecture = ARCHITECTURE_ASK.test(t) && subjects.length > 0;
127
+ const capability = CAPABILITY_ASK.test(t) && subjects.length > 0;
128
+ return { assert: capability || architecture, architecture, subjects };
129
+ }
130
+
131
+ // ── the turn's sources (Stop, from the host transcript) ───────────────────────────────────────────
132
+ const textOf = (c) => (typeof c === 'string' ? c : Array.isArray(c)
133
+ ? c.map((x) => (typeof x === 'string' ? x : x?.type === 'text' ? x.text : '')).join('\n') : '');
134
+ const NOT_A_SOURCE = /^(?:Edit|Write|MultiEdit|NotebookEdit|TodoWrite|ToolSearch|AskUserQuestion|ExitPlanMode|SendMessage|TaskStop|Monitor|Skill|Artifact.*|EnterWorktree|ExitWorktree)$/;
135
+ const MCP_MUTATING = /__(?:create|update|delete|remove|publish|deploy|push|write|send|set|merge|upload|patch|put|post|add|rename|move|approve|promote|rollback|cancel|buy|store|edit|import|reset|stop|terminate|spawn|execute)[a-z_-]*$/i;
136
+
137
+ /** One tool call as evidence: what it looked at (`text`, used for binding) and how much to trust it. */
138
+ export function sourceOf(name, input = {}, result = '') {
139
+ const n = String(name || '');
140
+ const r = String(result || '');
141
+ if (/(?:^|__)search_ruvnet$/.test(n)) {
142
+ const ok = /Searched \d+ RuvNet repos/.test(r) && !/^\s*(?:search_ruvnet error:|.{0,200}RUVNET BRAIN IS DOWN|.{0,200}RuvNet Brain is disabled)/s.test(r);
143
+ const paths = [...r.matchAll(/^path : (\S+)/gm)].map((m) => m[1]).slice(0, 20);
144
+ return { kind: 'search_ruvnet', ref: String(input.query || ''), strength: ok ? 'strong' : 'failed', ok, text: [input.query, ...paths].join(' ') };
145
+ }
146
+ if (n === 'WebFetch') return { kind: 'WebFetch', ref: String(input.url || ''), strength: 'weak', why: 'summarised-by-small-model', text: String(input.url || '') };
147
+ if (n === 'WebSearch') {
148
+ const urls = [...r.matchAll(/https?:\/\/[^\s)"'\]]+/g)].map((m) => m[0]).slice(0, 10);
149
+ return { kind: 'WebSearch', ref: String(input.query || ''), strength: 'weak', why: 'search-snippets', text: [input.query, ...urls].join(' ') };
150
+ }
151
+ if (n === 'Agent' || n === 'Task') {
152
+ return { kind: n, ref: String(input.description || input.subagent_type || ''), strength: 'weak', why: 'relayed-by-subagent',
153
+ text: `${input.description || ''} ${String(input.prompt || '').slice(0, 400)}`, result: r.slice(0, 20000) };
154
+ }
155
+ if (n === 'Read' || n === 'NotebookRead') return { kind: 'Read', ref: String(input.file_path || input.notebook_path || ''), strength: 'strong', text: String(input.file_path || input.notebook_path || ''), result: r.slice(0, 20000) };
156
+ if (n === 'Grep' || n === 'Glob') {
157
+ const ref = [input.pattern, input.path, input.glob].filter(Boolean).join(' ');
158
+ return { kind: n, ref, strength: 'strong', text: ref, result: r.slice(0, 20000) };
159
+ }
160
+ if (n === 'Bash') return { kind: 'Bash', ref: String(input.description || input.command || '').slice(0, 120), strength: 'strong', text: String(input.command || '').slice(0, 2000), result: r.slice(0, 20000) };
161
+ if (n.startsWith('mcp__') && !MCP_MUTATING.test(n)) {
162
+ return { kind: 'mcp', ref: n.split('__').pop(), strength: 'strong', text: `${n} ${JSON.stringify(input).slice(0, 1000)}`, result: r.slice(0, 20000) };
163
+ }
164
+ if (NOT_A_SOURCE.test(n)) return null;
165
+ return null;
166
+ }
167
+
168
+ /** Every source read this turn, in order, from a Claude JSONL transcript's lines. */
169
+ export function turnSources(lines) {
170
+ const { boundaryFound, prompt, recs } = currentTurnRecords(lines);
171
+ const results = new Map();
172
+ for (const o of recs) {
173
+ const c = o?.message?.content;
174
+ if (Array.isArray(c)) for (const r of c) if (r?.type === 'tool_result' && r.tool_use_id) results.set(r.tool_use_id, textOf(r.content));
175
+ }
176
+ const sources = [];
177
+ for (const o of recs) {
178
+ const c = o?.message?.content;
179
+ if (o?.type !== 'assistant' || !Array.isArray(c)) continue;
180
+ for (const u of c) {
181
+ if (u?.type !== 'tool_use') continue;
182
+ const s = sourceOf(u.name, u.input || {}, results.get(u.id) || '');
183
+ if (s) sources.push({ ...s, order: sources.length });
184
+ }
185
+ }
186
+ return { boundaryFound, prompt, sources };
187
+ }
188
+
189
+ /** Did a search_ruvnet call this turn return a real grounded answer? (Gate 1, from the transcript.) */
190
+ export const searchedThisTurn = (sources) => sources.some((s) => s.kind === 'search_ruvnet' && s.ok);
191
+
192
+ // ── Stop-time audit ────────────────────────────────────────────────────────────────────────────────
193
+ const HEDGE = /\?|\b(?:might|may|maybe|perhaps|probably|possibly|likely|unlikely|apparently|seems?|i\s+think|i\s+believe|i\s+suspect|i\s+(?:could|did)\s*n[o']?t\s+(?:confirm|verify|check)|not\s+sure|unsure|unverified|unconfirmed|not\s+verified|assum(?:e|ed|ing)|if|unless|whether|would|should|once|when)\b/i;
194
+ const NEGATIVE = /\b(?:no\s+[\w-]+\s+(?:can|could|will)|cannot|can(?:'|’)t|can\s+not|is\s*n(?:'|’)?t\s+(?:possible|supported|able)|not\s+possible|impossible|does\s*n(?:'|’)?t\s+(?:support|allow|expose|provide|exist|let|offer)|does\s+not\s+(?:support|allow|expose|provide|exist|let|offer)|there(?:'|’)?s\s+no\s+(?:way|api|hook|setting|option)|there\s+is\s+no\s+(?:way|api|hook|setting|option)|has\s+no\s+(?:way|api|hook|setting|option)|only\s+(?:supports?|allows?|exposes?))\b/i;
195
+
196
+ /** Not an assertion: a hedge or question, a table cell, a "Label: description" status line, or an
197
+ * instruction introducing a command block ("Run this …:"). */
198
+ const notAClaim = (s) => HEDGE.test(s) || s.includes('|') || /^[\w\s/-]{1,40}:\s/.test(s) || /:\s*$/.test(s);
199
+
200
+ function sentences(message) {
201
+ return strippedProse(message).split(/(?<=[.!?])\s+|\n+/).map((s) => s.replace(/^[\s\-*•#>]+|\*\*/g, '').trim())
202
+ .filter((s) => s && s.length <= 400);
203
+ }
204
+
205
+ /** Capability claims about a vocabulary subject, excluding hedges and the sentences continuation-gate owns. */
206
+ export function capabilityClaims(rawMessage, tools) {
207
+ // Headings name a topic, they do not assert one; emphasis markers break sentence splitting.
208
+ const message = String(rawMessage || '').replace(/^\s*#{1,6}\s.*$/gm, ' ').replace(/\*\*|__/g, '');
209
+ const owned = new Set(extractClaims(message).map((c) => c.text));
210
+ const out = new Map();
211
+ const subjectRe = tools.length ? new RegExp(`(?<![a-z0-9-])(${tools.map((t) => t.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')).sort((a, b) => b.length - a.length).join('|')})(?![a-z0-9-])`, 'i') : null;
212
+ for (const claim of extractClaims(message, { tools })) {
213
+ if (claim.class !== 'behavior' || owned.has(claim.text)) continue;
214
+ const text = claim.text.replace(/^[\s\-*•#>]+|\*\*/g, '').trim();
215
+ if (!notAClaim(text)) out.set(text, { text, subject: claim.tool.toLowerCase() });
216
+ }
217
+ if (subjectRe) {
218
+ for (const s of sentences(message)) {
219
+ const neg = NEGATIVE.exec(s);
220
+ if (out.has(s) || owned.has(s) || notAClaim(s) || !neg) continue;
221
+ // The subject must be what the negation is about: named before it, within a short clause.
222
+ const m = subjectRe.exec(s);
223
+ if (m && m.index <= neg.index + 12 && neg.index - m.index <= 40) out.set(s, { text: s, subject: m[1].toLowerCase() });
224
+ }
225
+ }
226
+ return [...out.values()];
227
+ }
228
+
229
+ /**
230
+ * Sources that name every word of the subject. A STRONG source may also bind through what it
231
+ * returned (a file read, a command's output, a search's hits); a weak one only through what it
232
+ * pointed at — a summary's own wording is exactly what is not trusted. A word of 5+ characters also
233
+ * matches inside a longer compound (`displaylink` in `DisplayLinkUserAgent`).
234
+ */
235
+ export function bindingSources(subject, sources) {
236
+ const words = subjectWords(subject);
237
+ if (!words.length) return [];
238
+ const has = (tokens, w) => tokens.has(w) || (w.length >= 5 && [...tokens].some((t) => t.length > w.length && t.includes(w)));
239
+ return sources.filter((s) => {
240
+ const t = tokenSet(s.strength === 'strong' ? `${s.text} ${s.ref} ${s.result || ''}` : s.text);
241
+ return words.every((w) => has(t, w));
242
+ });
243
+ }
244
+
245
+ /**
246
+ * The audit. `sources` null = this host's sources are unknown (Codex rollout not parsed): only claims
247
+ * a stamp term can bind are judged, the rest are UNKNOWN and never blocked.
248
+ */
249
+ /** Proper nouns / identifiers the ANSWER uses as a sentence subject (e.g. "Thunderbolt can't …"). */
250
+ const ANSWER_SUBJECT = /(?<=[a-z,;:]\s)([A-Z][A-Za-z0-9]*(?:[-.][A-Za-z0-9]+)*)(?=\s+(?:can(?:not|'t|’t)?|does(?:n't|n’t|\s+not)?|supports?|only|is\s*n(?:'|’)?t|has\s+no|won't|will\s+not)\b)|^([A-Z][A-Za-z0-9]*(?:[-.][A-Za-z0-9]+)+|[A-Z][a-z]+[A-Z][A-Za-z0-9]*)(?=\s+(?:can|does|supports?|only|is\s*n))/gm;
251
+ const NOT_SUBJECT = new Set(['i', 'it', 'this', 'that', 'there', 'they', 'we', 'you', 'he', 'she', 'which', 'what', 'who', 'nothing', 'none', 'one']);
252
+ export function answerSubjects(message) {
253
+ const out = new Set();
254
+ for (const m of strippedProse(message).matchAll(ANSWER_SUBJECT)) {
255
+ const t = (m[1] || m[2] || '').toLowerCase();
256
+ if (t.length >= 3 && !NOT_SUBJECT.has(t)) out.add(t);
257
+ }
258
+ return [...out];
259
+ }
260
+
261
+ export function auditAssertions({ message, subjects = [], vocab = [], sources = null, stampTerms = [] }) {
262
+ const tools = [...new Set([...vocab, ...subjects, ...answerSubjects(message)])].filter((t) => t.length >= 3);
263
+ const claims = capabilityClaims(message, tools);
264
+ const findings = [];
265
+ const unknown = [];
266
+ for (const claim of claims) {
267
+ if (sources === null) {
268
+ const words = subjectWords(claim.subject);
269
+ const rUv = words.some((w) => RUVNET_GATE1_TERMS.includes(w));
270
+ if (!rUv) { unknown.push(claim); continue; }
271
+ if (!words.every((w) => stampTerms.includes(w))) findings.push({ ...claim, reason: 'no search_ruvnet stamp for this subject this turn', read: stampTerms.map((t) => `search_ruvnet stamp: ${t}`) });
272
+ continue;
273
+ }
274
+ const binding = bindingSources(claim.subject, sources.filter((s) => s.strength !== 'failed'));
275
+ const lastWeak = Math.max(-1, ...binding.filter((s) => s.strength === 'weak').map((s) => s.order));
276
+ const strongAfter = binding.some((s) => s.strength === 'strong' && s.order > lastWeak);
277
+ if (!strongAfter) {
278
+ findings.push({ ...claim, reason: lastWeak >= 0 ? 'the only sources about it this turn are weak (summarised or relayed)' : 'no source about it was read this turn',
279
+ read: describeSources(lastWeak >= 0 ? binding : sources) });
280
+ }
281
+ }
282
+ return { claims, findings, unknown };
283
+ }
284
+
285
+ export function describeSources(sources, max = 4) {
286
+ if (!sources.length) return ['nothing'];
287
+ const shown = sources.slice(-max).map((s) => `${s.kind} ${JSON.stringify(String(s.ref).slice(0, 70))}${s.strength === 'weak' ? ` [${s.why} = weak evidence]` : ''}`);
288
+ return sources.length > max ? [`${sources.length - max} earlier`, ...shown] : shown;
289
+ }
290
+
291
+ export function correctionText(findings) {
292
+ const f = findings[0];
293
+ const lines = [
294
+ `You asserted "${f.text.slice(0, 200)}" about ${f.subject}; no relevant source was read this turn`
295
+ + ` (${f.reason}; read: ${f.read.join('; ')}).`,
296
+ ...findings.slice(1, 3).map((x) => `Also unsourced: "${x.text.slice(0, 160)}" about ${x.subject}.`),
297
+ 'Check the real source now (read the file, run the command with --help, search_ruvnet, or fetch the',
298
+ 'raw page with curl — a WebFetch body is a small model\'s summary) or restate each claim as UNVERIFIED.',
299
+ 'ADR-0030 decision point #1: "Did you CHECK, or are you recalling? Name the source."',
300
+ ];
301
+ return lines.join('\n');
302
+ }
303
+
304
+ // ── shadow gates #2 and #3 (logged, never delivered) ──────────────────────────────────────────────
305
+ const RECOMMENDS = /\b(?:I(?:'d|’d|\s+would)?\s+(?:recommend|propose|suggest)|my\s+recommendation|recommended\s+(?:approach|option|design)|the\s+design\s+I(?:'d|’d|\s+would)\s+propose|I(?:'d|’d)\s+(?:go|build)\s+with|go\s+with\s+option)\b/i;
306
+ const OPTION_MARK = /(?:^|\n)\s*(?:#{1,4}\s*|[-*•]\s*|\*\*)?(?:option|alternative|approach)\s*(?:[A-Z1-9]|one|two|three|four)\b|\b(?:option|alternative)\s+(?:[A-D1-4])\b/gi;
307
+ export function architectureShadow({ architecture, message }) {
308
+ if (!architecture || !RECOMMENDS.test(strippedProse(message))) return null;
309
+ const options = new Set([...String(message).matchAll(OPTION_MARK)].map((m) => m[0].trim().toLowerCase().replace(/[^a-z0-9 ]/g, ''))).size;
310
+ return options >= 3 ? null : { gate: 'adr-0030-2-architecture-options', options, wouldBlock: true };
311
+ }
312
+
313
+ const NUMBER = /(?<![\w.])\d+(?:[.,]\d+)?(?:\s?%|\/\d+)?(?![\w])/g;
314
+ export function relayShadow({ message, sources }) {
315
+ if (!sources?.length) return null;
316
+ const agentIdx = sources.filter((s) => s.kind === 'Agent' || s.kind === 'Task');
317
+ if (!agentIdx.length) return null;
318
+ const nums = [...new Set((strippedProse(message).match(NUMBER) || []).map((n) => n.replace(/\s/g, '')))]
319
+ .filter((n) => /[.%/]/.test(n) || n.replace(/\D/g, '').length >= 3).filter((n) => !/^(?:19|20)\d\d$/.test(n));
320
+ const relayed = nums.filter((n) => {
321
+ const from = agentIdx.find((a) => String(a.result || '').includes(n));
322
+ if (!from) return false;
323
+ return !sources.some((s) => s.order > from.order && s.kind !== 'Agent' && s.kind !== 'Task' && String(s.result || '').includes(n));
324
+ });
325
+ return relayed.length ? { gate: 'adr-0030-3-relayed-number', numbers: relayed.slice(0, 8), wouldBlock: true } : null;
326
+ }
327
+
328
+ /** Append one shadow row, bounded. Never throws. */
329
+ export function logShadow(row, { env = process.env } = {}) {
330
+ try {
331
+ const file = env.RUVNET_ASSERTION_SHADOW_LOG || path.join(brainHome(env), 'assertion-gate-shadow.jsonl');
332
+ fs.mkdirSync(path.dirname(file), { recursive: true });
333
+ fs.appendFileSync(file, `${JSON.stringify(row)}\n`);
334
+ if (fs.statSync(file).size > 512 * 1024) {
335
+ const keep = fs.readFileSync(file, 'utf8').split('\n').filter(Boolean).slice(-500);
336
+ fs.writeFileSync(file, `${keep.join('\n')}\n`);
337
+ }
338
+ } catch { /* shadow measurement never breaks a turn */ }
339
+ }
@@ -58,6 +58,20 @@
58
58
  * interrupted/cancelled turn is never forced. The marker is consumed (deleted) whether or not it
59
59
  * fires, so a genuinely abandoned marker cannot pressure some unrelated later turn.
60
60
  *
61
+ * 2026-09-30 — ADR-0030 DECISION POINT #1, AND THE FALSE ALARM. Two changes, one registration:
62
+ * - On Claude the "was search_ruvnet called" question is answered from the TRANSCRIPT (the ordered
63
+ * record of every tool call and result, grounding-turn-evidence.mjs turnSources), not from stamp
64
+ * mtimes. The stamp was a lossy proxy: 7 real false alarms were measured, 3 from a queued
65
+ * mid-turn prompt re-dating the marker (fixed in grounding-turn-mark.mjs), 3 pre-H1 vocabulary
66
+ * misses, 1 successful search whose stamp never minted. Codex's rollout is not parsed anywhere in
67
+ * this repo, so Codex keeps the stamp evidence.
68
+ * - When the marker says the prompt asked a capability/feasibility/architecture question, every
69
+ * capability claim in the final answer needs a RELEVANT, STRONG source read this turn after the
70
+ * last weak one (auditAssertions). A WebFetch body is a small model's summary: weak.
71
+ * Gates #2/#3 of ADR-0030 run in shadow (logShadow) — measured, never delivered.
72
+ * At most ONE correction per stop episode: both checks compose into one message, and
73
+ * stop_hook_active silences the continued stop.
74
+ *
61
75
  * FAILS OPEN ALWAYS. Exit 0 unconditionally — a gate that breaks a turn's completion because a
62
76
  * cache directory was unreadable would be disabled within a day.
63
77
  */
@@ -65,8 +79,13 @@ import fs from 'node:fs';
65
79
  import os from 'node:os';
66
80
  import path from 'node:path';
67
81
  import { fileURLToPath } from 'node:url';
68
- import { readStdinBounded } from './hook-input.mjs';
69
- import { markerPathFor } from './grounding-turn-mark.mjs';
82
+ import { readStopHookInput } from './hook-input.mjs';
83
+ import { markerPathFor, readMarker } from './grounding-turn-mark.mjs';
84
+ import { readSettledTranscript } from './turn-outcome-capture.mjs';
85
+ import {
86
+ architectureShadow, auditAssertions, correctionText, describeSources, loadVocabulary, logShadow,
87
+ relayShadow, searchedThisTurn, turnSources,
88
+ } from './grounding-turn-evidence.mjs';
70
89
 
71
90
  const HOME = os.homedir();
72
91
  const EXIT_ALLOW = 0;
@@ -94,6 +113,14 @@ export function newestGroundingStampMs(dir = GROUNDED_DIR) {
94
113
  return newest;
95
114
  }
96
115
 
116
+ /** Product terms whose stamp was minted at or after `sinceMs` (Codex's only view of this turn's searches). */
117
+ export function stampTermsSince(sinceMs, dir = GROUNDED_DIR) {
118
+ try {
119
+ return fs.readdirSync(dir).filter((name) => !name.startsWith('.')
120
+ && fs.statSync(path.join(dir, name)).mtimeMs >= sinceMs - SKEW_MS);
121
+ } catch { return []; }
122
+ }
123
+
97
124
  /** A little slack for filesystem mtime granularity (some filesystems round to whole seconds), so a
98
125
  * stamp written the same wall-clock second as the marker is never wrongly judged "before" it. */
99
126
  const SKEW_MS = 1500;
@@ -106,18 +133,54 @@ export function wasGroundedSince(markerMs, newestStampMs) {
106
133
  return newestStampMs >= markerMs - SKEW_MS;
107
134
  }
108
135
 
109
- async function readHookInput() {
110
- // Mirrors continuation-gate.mjs's own three-way source classification exactly (ADR-043 /
111
- // Fable #1): only a payload we actually parsed off stdin may ever force a continuation.
112
- if (process.stdin.isTTY) return { __source: 'tty' };
136
+ /**
137
+ * The whole Stop decision for one armed turn: the correction text, or null. Exported so tests can
138
+ * drive it with a synthetic transcript. Every failure inside returns null (fail open).
139
+ */
140
+ export function decide({ hookInput, marker, markerMs, env = process.env, read = readSettledTranscript }) {
113
141
  try {
114
- const raw = (await readStdinBounded()).toString('utf8');
115
- return { ...JSON.parse(raw || '{}'), __source: 'stdin' };
116
- } catch { return { __source: 'unreadable' }; }
142
+ const host = env.RUVNET_HOOK_HOST === 'codex' ? 'codex' : 'claude';
143
+ const tp = hookInput.transcript_path;
144
+ let turn = null;
145
+ if (host === 'claude' && typeof tp === 'string' && /\.jsonl$/i.test(tp)) {
146
+ try { turn = turnSources(read(tp, { maxMs: 0 })); } catch { turn = null; }
147
+ }
148
+ const sources = turn ? turn.sources : null;
149
+ const message = String(hookInput.last_assistant_message || '');
150
+
151
+ let assertion = null;
152
+ if (marker.assert && message) {
153
+ const vocab = loadVocabulary({ env });
154
+ const audit = auditAssertions({ message, subjects: marker.subjects, vocab, sources,
155
+ stampTerms: sources ? [] : stampTermsSince(markerMs) });
156
+ if (audit.findings.length) assertion = audit.findings;
157
+ const shadow = [architectureShadow({ architecture: marker.architecture, message }), relayShadow({ message, sources })].filter(Boolean);
158
+ for (const row of shadow) logShadow({ ...row, at: new Date().toISOString(), session: hookInput.session_id, host });
159
+ }
160
+
161
+ const grounded = marker.gate1 === false ? true
162
+ : sources ? searchedThisTurn(sources) : wasGroundedSince(markerMs, newestGroundingStampMs());
163
+ if (assertion) {
164
+ return correctionText(assertion) + (grounded ? '' : '\nThis turn also touched the rUv stack and no successful search_ruvnet call was recorded: call it with the product term(s).');
165
+ }
166
+ if (grounded) return null;
167
+ return [
168
+ 'This turn touched the RuvNet / rUv stack and ground-ruvnet\'s directive required calling the',
169
+ 'search_ruvnet MCP tool before asserting what any RuvNet tool can/cannot do — but no successful',
170
+ sources ? `search_ruvnet call is in this turn's transcript (read this turn: ${describeSources(sources).join('; ')}).`
171
+ : 'search_ruvnet call was recorded this turn (checked against the grounding-stamp evidence).',
172
+ '',
173
+ 'Do NOT end the turn on an ungrounded rUv-domain answer. Call `search_ruvnet` now with the',
174
+ 'relevant product term(s) in the query, ground your answer in the cited source paths it returns,',
175
+ 'and correct anything you already asserted from memory. Training priors on the rUv stack are',
176
+ 'stale by construction (ADR-0012) — this is not a formality.',
177
+ ].join('\n');
178
+ } catch { return null; }
117
179
  }
118
180
 
181
+
119
182
  async function main() {
120
- const hookInput = await readHookInput();
183
+ const hookInput = await readStopHookInput();
121
184
  if (hookInput.__source !== 'stdin') process.exit(EXIT_ALLOW);
122
185
  if (hookInput.stop_hook_active) process.exit(EXIT_ALLOW);
123
186
  if (hookInput.hook_event_name !== 'Stop' || hookInput.interrupted || hookInput.cancelled) {
@@ -135,27 +198,16 @@ async function main() {
135
198
  // unrelated turn (same reasoning as continuation-gate.mjs's cooldown lock, applied here as a
136
199
  // single-use marker instead of a timed window, because "did this turn ground itself" has no
137
200
  // meaningful reading beyond the one turn it was written for).
201
+ const armed = readMarker(marker) || { gate1: true, subjects: [] };
138
202
  try { fs.unlinkSync(marker); } catch { /* a marker that vanished between stat and unlink already told us what we needed */ }
139
203
 
140
- const grounded = wasGroundedSince(markerStat.mtimeMs, newestGroundingStampMs());
141
- if (grounded) process.exit(EXIT_ALLOW);
142
-
143
- const lines = [
144
- 'This turn touched the RuvNet / rUv stack and ground-ruvnet\'s directive required calling the',
145
- 'search_ruvnet MCP tool before asserting what any RuvNet tool can/cannot do — but no successful',
146
- 'search_ruvnet call was recorded this turn (checked against the same grounding-stamp evidence',
147
- 'ground-before-write.sh already trusts).',
148
- '',
149
- 'Do NOT end the turn on an ungrounded rUv-domain answer. Call `search_ruvnet` now with the',
150
- 'relevant product term(s) in the query, ground your answer in the cited source paths it returns,',
151
- 'and correct anything you already asserted from memory. Training priors on the rUv stack are',
152
- 'stale by construction (ADR-0012) — this is not a formality.',
153
- ];
204
+ const text = decide({ hookInput, marker: armed, markerMs: markerStat.mtimeMs });
205
+ if (!text) process.exit(EXIT_ALLOW);
154
206
 
155
207
  process.stdout.write(JSON.stringify({
156
208
  hookSpecificOutput: {
157
209
  hookEventName: 'Stop',
158
- additionalContext: lines.join('\n'),
210
+ additionalContext: text,
159
211
  },
160
212
  }));
161
213
  process.exit(EXIT_ALLOW);
@@ -24,6 +24,22 @@
24
24
  * one). Its CONTENT is a JSON blob for a human reading the cache, but the Stop-time gate only ever
25
25
  * trusts the mtime.
26
26
  *
27
+ * 2026-09-30 — TWO ARMS, ONE MARKER (ADR-0030 decision point #1). The marker now also records, as
28
+ * JSON content, whether the prompt ASKS for a capability / feasibility / architecture judgement about
29
+ * a subject (grounding-turn-evidence.mjs classifyPrompt) and which subjects it named, so the Stop gate
30
+ * can require a relevant source for any capability claim the answer makes — on any platform, not
31
+ * only the rUv stack. `gate1` keeps the original meaning (the prompt matched Gate 1).
32
+ *
33
+ * THE FALSE-ALARM FIX. Measured on real transcripts (7 turns where the Stop gate said "no successful
34
+ * search_ruvnet call was recorded" although one had been made): in 3 of them a QUEUED message (a
35
+ * real user message typed mid-turn, or a task notification before H2) fired UserPromptSubmit again
36
+ * AFTER the search and rewrote this marker, moving the turn boundary past the evidence. So an
37
+ * unconsumed marker is now MERGED, never re-dated: its mtime (the boundary) stays at the first arm
38
+ * of the stop episode. A marker older than STALE_MS (an interrupted turn never reaches Stop) is
39
+ * replaced instead. Of the other 4, three were pre-H1 vocabulary misses (fixed by H1) and one was a
40
+ * successful search whose stamp never minted (2026-09-30, cause not recoverable from the transcript);
41
+ * so on Claude the Stop gate now reads the transcript itself (grounding-turn-gate.mjs).
42
+ *
27
43
  * CONTRACT: PostToolUse-shaped hooks in this repo are advisory; this one is too — it can never
28
44
  * block a prompt. It exits 0 unconditionally and writes nothing to stdout Claude/Codex would act
29
45
  * on (UserPromptSubmit's silence contract). A write failure (unwritable cache dir, race, etc.) is
@@ -36,6 +52,7 @@ import path from 'node:path';
36
52
  import { fileURLToPath } from 'node:url';
37
53
  import { readStdinBounded, isHarnessGenerated } from './hook-input.mjs';
38
54
  import { ruvnetGate1Matches } from './ruvnet-gate1-pattern.mjs';
55
+ import { classifyPrompt, loadVocabulary } from './grounding-turn-evidence.mjs';
39
56
 
40
57
  const HOME = os.homedir();
41
58
  export const MARKER_DIR = process.env.RUVNET_GROUNDING_TURN_DIR
@@ -49,17 +66,50 @@ export function markerPathFor(sessionId, dir = MARKER_DIR) {
49
66
  return safe ? path.join(dir, `${safe}.json`) : null;
50
67
  }
51
68
 
52
- /** Exported for the unit test: pure decision, no I/O. */
53
- export function shouldMark(hookInput) {
54
- if (!hookInput || hookInput.hook_event_name !== 'UserPromptSubmit') return false;
55
- if (!hookInput.session_id) return false;
56
- const text = String(hookInput.prompt ?? hookInput.user_prompt ?? hookInput.input ?? '');
69
+ const promptOf = (hookInput) => String(hookInput?.prompt ?? hookInput?.user_prompt ?? hookInput?.input ?? '');
70
+ const eligible = (hookInput) => Boolean(hookInput && hookInput.hook_event_name === 'UserPromptSubmit' && hookInput.session_id
57
71
  // H2: a background task notification, slash-command scaffold, or other harness-authored message
58
72
  // arrives on UserPromptSubmit exactly like real user text — arming the Stop-time grounding gate off
59
73
  // one of these (because it happens to mention a rUv term) would demand a search_ruvnet call to
60
74
  // close out a "turn" nobody had a hand in.
61
- if (isHarnessGenerated(text)) return false;
62
- return ruvnetGate1Matches(text);
75
+ && !isHarnessGenerated(promptOf(hookInput)));
76
+
77
+ /** Exported for the unit test: pure decision, no I/O. Gate 1 (the rUv-stack search requirement). */
78
+ export function shouldMark(hookInput) {
79
+ return eligible(hookInput) && ruvnetGate1Matches(promptOf(hookInput));
80
+ }
81
+
82
+ /** Both arms for one prompt, or null when neither fires. Pure apart from the vocabulary it is given. */
83
+ export function armFor(hookInput, vocab = []) {
84
+ if (!eligible(hookInput)) return null;
85
+ const gate1 = ruvnetGate1Matches(promptOf(hookInput));
86
+ const c = classifyPrompt(promptOf(hookInput), vocab);
87
+ if (!gate1 && !c.assert) return null;
88
+ return { gate1, assert: c.assert, architecture: c.architecture, subjects: c.subjects };
89
+ }
90
+
91
+ export const STALE_MS = 2 * 3600_000;
92
+ /** A marker's JSON, or null. Old markers (no `gate1` field) were only ever written for Gate 1. */
93
+ export function readMarker(file) {
94
+ try {
95
+ const m = JSON.parse(fs.readFileSync(file, 'utf8'));
96
+ return m && typeof m === 'object' ? { gate1: m.gate1 !== false, assert: !!m.assert, architecture: !!m.architecture,
97
+ subjects: Array.isArray(m.subjects) ? m.subjects.map(String) : [], at: m.at } : null;
98
+ } catch { return null; }
99
+ }
100
+
101
+ /** Merge an arm into an unconsumed marker WITHOUT moving its mtime (the turn boundary). */
102
+ export function writeArm(file, arm, meta = {}, now = Date.now()) {
103
+ fs.mkdirSync(path.dirname(file), { recursive: true });
104
+ let st = null;
105
+ try { st = fs.statSync(file); } catch { /* none yet */ }
106
+ const prev = st && now - st.mtimeMs < STALE_MS ? readMarker(file) : null;
107
+ const next = prev ? { ...meta, at: prev.at, gate1: prev.gate1 || arm.gate1, assert: prev.assert || arm.assert,
108
+ architecture: prev.architecture || arm.architecture, subjects: [...new Set([...prev.subjects, ...arm.subjects])].slice(0, 32) }
109
+ : { ...meta, at: new Date(now).toISOString(), ...arm };
110
+ fs.writeFileSync(file, JSON.stringify(next) + '\n');
111
+ if (prev) fs.utimesSync(file, st.atime, st.mtime);
112
+ return next;
63
113
  }
64
114
 
65
115
  async function main() {
@@ -69,17 +119,14 @@ async function main() {
69
119
  hookInput = JSON.parse(raw || '{}');
70
120
  } catch { process.exit(0); }
71
121
 
72
- if (!shouldMark(hookInput)) process.exit(0);
122
+ let arm = null;
123
+ try { arm = armFor(hookInput, loadVocabulary()); } catch { arm = shouldMark(hookInput) ? { gate1: true, assert: false, architecture: false, subjects: [] } : null; }
124
+ if (!arm) process.exit(0);
73
125
 
74
126
  const file = markerPathFor(hookInput.session_id);
75
127
  if (!file) process.exit(0);
76
128
  try {
77
- fs.mkdirSync(path.dirname(file), { recursive: true });
78
- fs.writeFileSync(file, JSON.stringify({
79
- at: new Date().toISOString(),
80
- sessionId: hookInput.session_id,
81
- turnId: hookInput.turn_id || hookInput.prompt_id || null,
82
- }) + '\n');
129
+ writeArm(file, arm, { sessionId: hookInput.session_id, turnId: hookInput.turn_id || hookInput.prompt_id || null });
83
130
  } catch { /* fail-open: no marker means the Stop gate stays silent, never a false block */ }
84
131
  process.exit(0);
85
132
  }
@@ -96,6 +96,21 @@ export function readStdinBounded({ maxBytes = 65536, idleMs = 50, emptyMs = 250
96
96
  });
97
97
  }
98
98
 
99
+ /**
100
+ * The Stop-hook payload with its provenance, shared by every Stop gate (was a verbatim copy in
101
+ * continuation-gate and grounding-turn-gate). Three sources, treated differently by callers:
102
+ * 'tty' (run bare, never force), 'unreadable' (read/parse failed — fs.readFileSync(0) throws EAGAIN
103
+ * intermittently on macOS; laundering that into {} would read as a fresh stop), 'stdin' (parsed — the
104
+ * only source allowed to force). ADR-043.
105
+ */
106
+ export async function readStopHookInput() {
107
+ if (process.stdin.isTTY) return { __source: 'tty' };
108
+ try {
109
+ const raw = (await readStdinBounded()).toString('utf8');
110
+ return { ...JSON.parse(raw || '{}'), __source: 'stdin' };
111
+ } catch { return { __source: 'unreadable' }; }
112
+ }
113
+
99
114
  /** The tool being invoked ("Bash", "Write", …), or "" if absent. */
100
115
  export function toolName(ev) {
101
116
  return ev && typeof ev.tool_name === 'string' ? ev.tool_name : '';