ruvnet-brain 4.3.36 → 4.3.38
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/bin/install.mjs +32 -9
- package/kb/forge-update.mjs +1867 -0
- package/package.json +3 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/hooks/codex-hooks.json +6 -1
- package/plugin/hooks/hook-contracts.json +11 -9
- package/plugin/scripts/capability-claim-evidence.mjs +11 -2
- package/plugin/scripts/capacity-aware-parallel-work.mjs +71 -15
- package/plugin/scripts/completion-claim-evidence.mjs +262 -0
- package/plugin/scripts/continuation-gate.mjs +105 -21
- package/plugin/scripts/continuation-objective.mjs +15 -0
- package/plugin/scripts/continuity-hook-policy.mjs +10 -3
- package/plugin/scripts/decision-gate.mjs +22 -2
- package/plugin/scripts/duplicate-gate.mjs +503 -0
- package/plugin/scripts/grounding-turn-evidence.mjs +339 -0
- package/plugin/scripts/grounding-turn-gate.mjs +77 -25
- package/plugin/scripts/grounding-turn-mark.mjs +61 -14
- package/plugin/scripts/hook-input.mjs +15 -0
- package/plugin/scripts/hook-shim.mjs +3 -1
- package/plugin/scripts/host-update.mjs +45 -0
- package/plugin/scripts/nightly-scheduler.mjs +34 -0
- package/plugin/scripts/session-snapshot-hook.mjs +11 -1
- package/plugin/scripts/session-start-budget.mjs +1 -0
- package/plugin/scripts/session-start-core.mjs +15 -2
- package/plugin/scripts/session-start-health.mjs +101 -1
- package/plugin/scripts/session-start-update-plane.mjs +71 -1
- package/plugin/scripts/turn-outcome-capture.mjs +292 -0
- package/scripts/completion-claim-replay.mjs +114 -0
- package/scripts/corpus-canary.mjs +399 -0
- package/scripts/corpus-dispatch-decision.mjs +2 -2
- package/scripts/corpus-promotion.mjs +49 -0
- package/scripts/corpus-reconcile.mjs +27 -3
- package/scripts/corpus-watchdog.mjs +45 -6
- package/scripts/derive-passage-content-map.mjs +71 -0
- package/scripts/duplicate-gate-replay.mjs +98 -0
- package/scripts/grounding-turn-replay.mjs +131 -0
- package/scripts/nightly-watchdog.mjs +3 -3
- package/scripts/public-verification-inputs.mjs +24 -2
- package/scripts/release-transaction-provider.mjs +29 -6
- package/scripts/release.mjs +177 -74
- package/scripts/retrieval-canary.mjs +41 -4
- package/scripts/retrieval-passage-identity.mjs +58 -0
- package/scripts/sync-census.mjs +0 -0
- package/scripts/wired-check.mjs +6 -0
|
@@ -0,0 +1,339 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* grounding-turn-evidence.mjs — the pure half of ADR-0030 decision-point gate #1: "before asserting
|
|
3
|
+
* what a tool or platform can or cannot do, did you CHECK a relevant source this turn, or are you
|
|
4
|
+
* recalling?" Plus gates #2 (architecture needs >= 3 options) and #3 (relayed numbers need a
|
|
5
|
+
* re-check), which run in SHADOW mode only: they are measured and logged, never delivered.
|
|
6
|
+
*
|
|
7
|
+
* WHY (owner, 2026-09-30): "a nudge without enforcement is worthless." The concrete incident: an
|
|
8
|
+
* answer asserted "No hook can change the model of the current turn" from a WebFetch page summary
|
|
9
|
+
* written by a small model, and a design was built on it. The existing grounding-turn-gate only
|
|
10
|
+
* checked that SOME search_ruvnet stamp was newer than the turn marker — presence, not relevance or
|
|
11
|
+
* order — armed only on rUv-stack prompts, and could not see WebFetch at all.
|
|
12
|
+
*
|
|
13
|
+
* DESIGN, deterministic and local (rUv ADR-G004 rejects LLM gate evaluation; no model is called):
|
|
14
|
+
* 1. UserPromptSubmit (grounding-turn-mark.mjs) arms the turn only when the prompt ASKS for a
|
|
15
|
+
* capability / feasibility / architecture judgement AND names a subject (classifyPrompt). The
|
|
16
|
+
* caller replayed a naive output-only regex at 13.5% of turns and an order-aware one at 0.93%,
|
|
17
|
+
* mostly on harmless hedges; arming at prompt time is what keeps unarmed turns out of scope.
|
|
18
|
+
* 2. The sources read this turn come from the host transcript (turnSources) — the ordered,
|
|
19
|
+
* complete record of every tool call and its result, including search_ruvnet's returned paths.
|
|
20
|
+
* A WebFetch body is a small model's summary of the page, so it is WEAK evidence; so is a
|
|
21
|
+
* subagent's relayed report and a WebSearch snippet list.
|
|
22
|
+
* 3. At Stop, each capability claim in the final answer (capability-claim-evidence.mjs's own
|
|
23
|
+
* extractor, widened with this turn's vocabulary) needs one STRONG source whose path, URL,
|
|
24
|
+
* command or query names the claim's subject and that was read AFTER the last weak source about
|
|
25
|
+
* that subject (auditAssertions). Otherwise: one correction.
|
|
26
|
+
*
|
|
27
|
+
* Claims continuation-gate.mjs already audits (the RUVNET_TOOL behaviour class) are skipped here —
|
|
28
|
+
* one correction per claim, never two gates arguing over the same sentence.
|
|
29
|
+
*/
|
|
30
|
+
import fs from 'node:fs';
|
|
31
|
+
import os from 'node:os';
|
|
32
|
+
import path from 'node:path';
|
|
33
|
+
import { fileURLToPath } from 'node:url';
|
|
34
|
+
import { RUVNET_GATE1_TERMS } from './ruvnet-gate1-pattern.mjs';
|
|
35
|
+
import { extractClaims } from './capability-claim-evidence.mjs';
|
|
36
|
+
import { currentTurnRecords, strippedProse } from './completion-claim-evidence.mjs';
|
|
37
|
+
|
|
38
|
+
const SCRIPTS_DIR = path.dirname(fileURLToPath(import.meta.url));
|
|
39
|
+
|
|
40
|
+
// ── vocabulary ─────────────────────────────────────────────────────────────────────────────────────
|
|
41
|
+
/** Platform nouns a capability question is usually about, beyond the product names. */
|
|
42
|
+
const PLATFORM_NOUNS = ['hook', 'hooks', 'mcp', 'plugin', 'plugins', 'skill', 'skills', 'subagent', 'subagents',
|
|
43
|
+
'codex', 'claude code', 'cli', 'sdk', 'api', 'npm', 'github actions', 'workflow', 'launchd', 'cron',
|
|
44
|
+
'vercel', 'webfetch', 'websearch', 'hnsw', 'onnx', 'sqlite', 'rvf', 'metaharness', 'ruvllm', 'ruview',
|
|
45
|
+
'agentdb', 'ruflo', 'ruvector', 'aidefence', 'agentic-flow', 'agentic-qe', 'claude flow', 'agent browser'];
|
|
46
|
+
const WEAK_WORDS = new Set(['the', 'and', 'for', 'with', 'this', 'that', 'brain', 'repo', 'store', 'docs', 'data',
|
|
47
|
+
'test', 'tests', 'main', 'core', 'app', 'web', 'site', 'code', 'tool', 'tools', 'model', 'models', 'agent',
|
|
48
|
+
'agents', 'file', 'files', 'flow', 'swarm', 'ruv', 'sparc', 'qe']);
|
|
49
|
+
|
|
50
|
+
const brainHome = (env) => env.RUVNET_BRAIN_HOME || path.join(env.HOME || env.USERPROFILE || os.homedir(), '.cache', 'ruvnet-brain');
|
|
51
|
+
|
|
52
|
+
/** Subject phrases: Gate-1 terms, platform nouns, repo aliases and installed store names. Never throws. */
|
|
53
|
+
export function loadVocabulary({ env = process.env } = {}) {
|
|
54
|
+
const out = new Set([...RUVNET_GATE1_TERMS, ...PLATFORM_NOUNS]);
|
|
55
|
+
const kbDirs = [env.RUVNET_KB_DIR, path.join(brainHome(env), 'kb'), path.join(SCRIPTS_DIR, '..', '..', 'kb')].filter(Boolean);
|
|
56
|
+
for (const dir of kbDirs) {
|
|
57
|
+
try {
|
|
58
|
+
const aliases = JSON.parse(fs.readFileSync(path.join(dir, 'repo-aliases.json'), 'utf8'));
|
|
59
|
+
for (const [repo, list] of Object.entries(aliases || {})) { out.add(repo); for (const a of list || []) out.add(a); }
|
|
60
|
+
} catch { /* absent in this location */ }
|
|
61
|
+
try {
|
|
62
|
+
// Installed store names — but a plain single word ("support", "concepts", "marketing") is an
|
|
63
|
+
// English word far more often than a product; measured, it produced "doesn't support" claims.
|
|
64
|
+
for (const name of fs.readdirSync(dir)) {
|
|
65
|
+
if (!name.endsWith('.meta.json')) continue;
|
|
66
|
+
const store = name.slice(0, -'.meta.json'.length);
|
|
67
|
+
if (/[-_.0-9]/.test(store) || /^(?:ru|rv|agentic|cognitum)/i.test(store)) out.add(store);
|
|
68
|
+
}
|
|
69
|
+
} catch { /* absent in this location */ }
|
|
70
|
+
}
|
|
71
|
+
return [...out].map((v) => String(v).toLowerCase().trim())
|
|
72
|
+
.filter((v) => v.length >= 3 && !WEAK_WORDS.has(v) && /[a-z]/.test(v));
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// ── tokens ─────────────────────────────────────────────────────────────────────────────────────────
|
|
76
|
+
const singular = (t) => (t.length > 4 && t.endsWith('s') && !t.endsWith('ss') ? t.slice(0, -1) : t);
|
|
77
|
+
/** Lowercased word tokens: each hyphen/dot/underscore compound (separators normalised to '-') and its
|
|
78
|
+
* parts, singularised — so `code.claude.com/docs/en/hooks` yields claude, code, hook, … */
|
|
79
|
+
export function tokenSet(text) {
|
|
80
|
+
const set = new Set();
|
|
81
|
+
const raw = String(text || '');
|
|
82
|
+
const s = `${raw.replace(/([a-z0-9])([A-Z])/g, '$1 $2')} ${raw}`.toLowerCase();
|
|
83
|
+
for (const compound of s.match(/[a-z0-9]+(?:[-_.][a-z0-9]+)*/g) || []) {
|
|
84
|
+
const parts = compound.split(/[-_.]/).filter(Boolean);
|
|
85
|
+
set.add(singular(parts.join('-')));
|
|
86
|
+
for (const part of parts) set.add(singular(part));
|
|
87
|
+
}
|
|
88
|
+
return set;
|
|
89
|
+
}
|
|
90
|
+
/** A subject phrase's words; a source binds the subject only when it names every one of them. */
|
|
91
|
+
const subjectWords = (subject) => String(subject || '').toLowerCase().split(/\s+/)
|
|
92
|
+
.map((w) => singular(w.split(/[-_.]/).filter(Boolean).join('-'))).filter((w) => w.length >= 2);
|
|
93
|
+
|
|
94
|
+
// ── prompt-time classification (UserPromptSubmit) ─────────────────────────────────────────────────
|
|
95
|
+
const CAPABILITY_ASK = new RegExp([
|
|
96
|
+
String.raw`\b(?:can|could|does|do|is|are|will|would)\s+(?:it|we|you|i|there|(?:a|an|the|any|this|that|my|our)\s+[\w-]+|[\w-]+)\s+(?:[\w-]+\s+){0,3}?(?:do|support|handle|work|run|use|call|change|switch|make|force|read|write|access|detect|block|enforce|intercept|see|know|override|rewrite|route|pick|select|stop|prevent|allow|expose|return|send|trigger|fire)\b`,
|
|
97
|
+
String.raw`\bis\s+(?:it|there)\s+(?:possible|a\s+way|any\s+way)\b`, String.raw`\bpossible\s+to\b`, String.raw`\bable\s+to\b`,
|
|
98
|
+
String.raw`\bcapab(?:le|ility|ilities)\b`, String.raw`\bfeasib(?:le|ility)\b`, String.raw`\blimitations?\b`,
|
|
99
|
+
String.raw`\bwhat\s+(?:can|does|do)\b`, String.raw`\bhow\s+(?:does|do|can|could|would|should)\b`,
|
|
100
|
+
String.raw`\bwhether\b`, String.raw`\bsupports?\b`,
|
|
101
|
+
].join('|'), 'i');
|
|
102
|
+
const ARCHITECTURE_ASK = /\barchitect(?:ure|ural)?\b|\bdesign\b|\bapproach(?:es)?\b|\bbest\s+way\b|\bshould\s+(?:we|i)\s+(?:use|build|go|pick|choose)\b|\brecommend|\bwhich\s+(?:tool|library|option|approach)\b|\btrade-?offs?\b|\bfigure\s+out\s+(?:the\s+)?(?:best|how)\b/i;
|
|
103
|
+
const CODE_ISH = /`([^`\n]{2,60})`|\b([A-Za-z][A-Za-z0-9]*(?:[-_.][A-Za-z0-9]+)+|[a-z]+[A-Z][A-Za-z0-9]+|[A-Z][a-z]+[A-Z][A-Za-z0-9]*)\b/g;
|
|
104
|
+
|
|
105
|
+
/** Vocabulary phrases and code-ish identifiers the text names. */
|
|
106
|
+
export function subjectsIn(text, vocab) {
|
|
107
|
+
const s = String(text || '');
|
|
108
|
+
const lower = s.toLowerCase();
|
|
109
|
+
const found = new Set();
|
|
110
|
+
for (const v of vocab) {
|
|
111
|
+
const re = new RegExp(`(?<![a-z0-9-])${v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}(?![a-z0-9-])`, 'i');
|
|
112
|
+
if (lower.includes(v) && re.test(s)) found.add(v);
|
|
113
|
+
}
|
|
114
|
+
for (const m of s.replace(/<\/?[A-Za-z_][\w-]*[^>]*>/g, ' ').matchAll(CODE_ISH)) {
|
|
115
|
+
const t = (m[1] || m[2] || '').trim();
|
|
116
|
+
if (t.length >= 4 && t.length <= 60 && !/^\d|\.(?:md|json|txt|png|jpg)$|^https?:|^e\.g\.?$|^i\.e\.?$/i.test(t) && !/\s/.test(t)) found.add(t.toLowerCase());
|
|
117
|
+
if (found.size >= 24) break;
|
|
118
|
+
}
|
|
119
|
+
return [...found].slice(0, 24);
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/** Arm the turn when the prompt asks a capability/feasibility/architecture question about a subject. */
|
|
123
|
+
export function classifyPrompt(text, vocab) {
|
|
124
|
+
const t = String(text || '');
|
|
125
|
+
const subjects = subjectsIn(t, vocab);
|
|
126
|
+
const architecture = ARCHITECTURE_ASK.test(t) && subjects.length > 0;
|
|
127
|
+
const capability = CAPABILITY_ASK.test(t) && subjects.length > 0;
|
|
128
|
+
return { assert: capability || architecture, architecture, subjects };
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
// ── the turn's sources (Stop, from the host transcript) ───────────────────────────────────────────
|
|
132
|
+
const textOf = (c) => (typeof c === 'string' ? c : Array.isArray(c)
|
|
133
|
+
? c.map((x) => (typeof x === 'string' ? x : x?.type === 'text' ? x.text : '')).join('\n') : '');
|
|
134
|
+
const NOT_A_SOURCE = /^(?:Edit|Write|MultiEdit|NotebookEdit|TodoWrite|ToolSearch|AskUserQuestion|ExitPlanMode|SendMessage|TaskStop|Monitor|Skill|Artifact.*|EnterWorktree|ExitWorktree)$/;
|
|
135
|
+
const MCP_MUTATING = /__(?:create|update|delete|remove|publish|deploy|push|write|send|set|merge|upload|patch|put|post|add|rename|move|approve|promote|rollback|cancel|buy|store|edit|import|reset|stop|terminate|spawn|execute)[a-z_-]*$/i;
|
|
136
|
+
|
|
137
|
+
/** One tool call as evidence: what it looked at (`text`, used for binding) and how much to trust it. */
|
|
138
|
+
export function sourceOf(name, input = {}, result = '') {
|
|
139
|
+
const n = String(name || '');
|
|
140
|
+
const r = String(result || '');
|
|
141
|
+
if (/(?:^|__)search_ruvnet$/.test(n)) {
|
|
142
|
+
const ok = /Searched \d+ RuvNet repos/.test(r) && !/^\s*(?:search_ruvnet error:|.{0,200}RUVNET BRAIN IS DOWN|.{0,200}RuvNet Brain is disabled)/s.test(r);
|
|
143
|
+
const paths = [...r.matchAll(/^path : (\S+)/gm)].map((m) => m[1]).slice(0, 20);
|
|
144
|
+
return { kind: 'search_ruvnet', ref: String(input.query || ''), strength: ok ? 'strong' : 'failed', ok, text: [input.query, ...paths].join(' ') };
|
|
145
|
+
}
|
|
146
|
+
if (n === 'WebFetch') return { kind: 'WebFetch', ref: String(input.url || ''), strength: 'weak', why: 'summarised-by-small-model', text: String(input.url || '') };
|
|
147
|
+
if (n === 'WebSearch') {
|
|
148
|
+
const urls = [...r.matchAll(/https?:\/\/[^\s)"'\]]+/g)].map((m) => m[0]).slice(0, 10);
|
|
149
|
+
return { kind: 'WebSearch', ref: String(input.query || ''), strength: 'weak', why: 'search-snippets', text: [input.query, ...urls].join(' ') };
|
|
150
|
+
}
|
|
151
|
+
if (n === 'Agent' || n === 'Task') {
|
|
152
|
+
return { kind: n, ref: String(input.description || input.subagent_type || ''), strength: 'weak', why: 'relayed-by-subagent',
|
|
153
|
+
text: `${input.description || ''} ${String(input.prompt || '').slice(0, 400)}`, result: r.slice(0, 20000) };
|
|
154
|
+
}
|
|
155
|
+
if (n === 'Read' || n === 'NotebookRead') return { kind: 'Read', ref: String(input.file_path || input.notebook_path || ''), strength: 'strong', text: String(input.file_path || input.notebook_path || ''), result: r.slice(0, 20000) };
|
|
156
|
+
if (n === 'Grep' || n === 'Glob') {
|
|
157
|
+
const ref = [input.pattern, input.path, input.glob].filter(Boolean).join(' ');
|
|
158
|
+
return { kind: n, ref, strength: 'strong', text: ref, result: r.slice(0, 20000) };
|
|
159
|
+
}
|
|
160
|
+
if (n === 'Bash') return { kind: 'Bash', ref: String(input.description || input.command || '').slice(0, 120), strength: 'strong', text: String(input.command || '').slice(0, 2000), result: r.slice(0, 20000) };
|
|
161
|
+
if (n.startsWith('mcp__') && !MCP_MUTATING.test(n)) {
|
|
162
|
+
return { kind: 'mcp', ref: n.split('__').pop(), strength: 'strong', text: `${n} ${JSON.stringify(input).slice(0, 1000)}`, result: r.slice(0, 20000) };
|
|
163
|
+
}
|
|
164
|
+
if (NOT_A_SOURCE.test(n)) return null;
|
|
165
|
+
return null;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** Every source read this turn, in order, from a Claude JSONL transcript's lines. */
|
|
169
|
+
export function turnSources(lines) {
|
|
170
|
+
const { boundaryFound, prompt, recs } = currentTurnRecords(lines);
|
|
171
|
+
const results = new Map();
|
|
172
|
+
for (const o of recs) {
|
|
173
|
+
const c = o?.message?.content;
|
|
174
|
+
if (Array.isArray(c)) for (const r of c) if (r?.type === 'tool_result' && r.tool_use_id) results.set(r.tool_use_id, textOf(r.content));
|
|
175
|
+
}
|
|
176
|
+
const sources = [];
|
|
177
|
+
for (const o of recs) {
|
|
178
|
+
const c = o?.message?.content;
|
|
179
|
+
if (o?.type !== 'assistant' || !Array.isArray(c)) continue;
|
|
180
|
+
for (const u of c) {
|
|
181
|
+
if (u?.type !== 'tool_use') continue;
|
|
182
|
+
const s = sourceOf(u.name, u.input || {}, results.get(u.id) || '');
|
|
183
|
+
if (s) sources.push({ ...s, order: sources.length });
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
return { boundaryFound, prompt, sources };
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
/** Did a search_ruvnet call this turn return a real grounded answer? (Gate 1, from the transcript.) */
|
|
190
|
+
export const searchedThisTurn = (sources) => sources.some((s) => s.kind === 'search_ruvnet' && s.ok);
|
|
191
|
+
|
|
192
|
+
// ── Stop-time audit ────────────────────────────────────────────────────────────────────────────────
|
|
193
|
+
const HEDGE = /\?|\b(?:might|may|maybe|perhaps|probably|possibly|likely|unlikely|apparently|seems?|i\s+think|i\s+believe|i\s+suspect|i\s+(?:could|did)\s*n[o']?t\s+(?:confirm|verify|check)|not\s+sure|unsure|unverified|unconfirmed|not\s+verified|assum(?:e|ed|ing)|if|unless|whether|would|should|once|when)\b/i;
|
|
194
|
+
const NEGATIVE = /\b(?:no\s+[\w-]+\s+(?:can|could|will)|cannot|can(?:'|’)t|can\s+not|is\s*n(?:'|’)?t\s+(?:possible|supported|able)|not\s+possible|impossible|does\s*n(?:'|’)?t\s+(?:support|allow|expose|provide|exist|let|offer)|does\s+not\s+(?:support|allow|expose|provide|exist|let|offer)|there(?:'|’)?s\s+no\s+(?:way|api|hook|setting|option)|there\s+is\s+no\s+(?:way|api|hook|setting|option)|has\s+no\s+(?:way|api|hook|setting|option)|only\s+(?:supports?|allows?|exposes?))\b/i;
|
|
195
|
+
|
|
196
|
+
/** Not an assertion: a hedge or question, a table cell, a "Label: description" status line, or an
|
|
197
|
+
* instruction introducing a command block ("Run this …:"). */
|
|
198
|
+
const notAClaim = (s) => HEDGE.test(s) || s.includes('|') || /^[\w\s/-]{1,40}:\s/.test(s) || /:\s*$/.test(s);
|
|
199
|
+
|
|
200
|
+
function sentences(message) {
|
|
201
|
+
return strippedProse(message).split(/(?<=[.!?])\s+|\n+/).map((s) => s.replace(/^[\s\-*•#>]+|\*\*/g, '').trim())
|
|
202
|
+
.filter((s) => s && s.length <= 400);
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/** Capability claims about a vocabulary subject, excluding hedges and the sentences continuation-gate owns. */
|
|
206
|
+
export function capabilityClaims(rawMessage, tools) {
|
|
207
|
+
// Headings name a topic, they do not assert one; emphasis markers break sentence splitting.
|
|
208
|
+
const message = String(rawMessage || '').replace(/^\s*#{1,6}\s.*$/gm, ' ').replace(/\*\*|__/g, '');
|
|
209
|
+
const owned = new Set(extractClaims(message).map((c) => c.text));
|
|
210
|
+
const out = new Map();
|
|
211
|
+
const subjectRe = tools.length ? new RegExp(`(?<![a-z0-9-])(${tools.map((t) => t.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')).sort((a, b) => b.length - a.length).join('|')})(?![a-z0-9-])`, 'i') : null;
|
|
212
|
+
for (const claim of extractClaims(message, { tools })) {
|
|
213
|
+
if (claim.class !== 'behavior' || owned.has(claim.text)) continue;
|
|
214
|
+
const text = claim.text.replace(/^[\s\-*•#>]+|\*\*/g, '').trim();
|
|
215
|
+
if (!notAClaim(text)) out.set(text, { text, subject: claim.tool.toLowerCase() });
|
|
216
|
+
}
|
|
217
|
+
if (subjectRe) {
|
|
218
|
+
for (const s of sentences(message)) {
|
|
219
|
+
const neg = NEGATIVE.exec(s);
|
|
220
|
+
if (out.has(s) || owned.has(s) || notAClaim(s) || !neg) continue;
|
|
221
|
+
// The subject must be what the negation is about: named before it, within a short clause.
|
|
222
|
+
const m = subjectRe.exec(s);
|
|
223
|
+
if (m && m.index <= neg.index + 12 && neg.index - m.index <= 40) out.set(s, { text: s, subject: m[1].toLowerCase() });
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
return [...out.values()];
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
/**
|
|
230
|
+
* Sources that name every word of the subject. A STRONG source may also bind through what it
|
|
231
|
+
* returned (a file read, a command's output, a search's hits); a weak one only through what it
|
|
232
|
+
* pointed at — a summary's own wording is exactly what is not trusted. A word of 5+ characters also
|
|
233
|
+
* matches inside a longer compound (`displaylink` in `DisplayLinkUserAgent`).
|
|
234
|
+
*/
|
|
235
|
+
export function bindingSources(subject, sources) {
|
|
236
|
+
const words = subjectWords(subject);
|
|
237
|
+
if (!words.length) return [];
|
|
238
|
+
const has = (tokens, w) => tokens.has(w) || (w.length >= 5 && [...tokens].some((t) => t.length > w.length && t.includes(w)));
|
|
239
|
+
return sources.filter((s) => {
|
|
240
|
+
const t = tokenSet(s.strength === 'strong' ? `${s.text} ${s.ref} ${s.result || ''}` : s.text);
|
|
241
|
+
return words.every((w) => has(t, w));
|
|
242
|
+
});
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
/**
|
|
246
|
+
* The audit. `sources` null = this host's sources are unknown (Codex rollout not parsed): only claims
|
|
247
|
+
* a stamp term can bind are judged, the rest are UNKNOWN and never blocked.
|
|
248
|
+
*/
|
|
249
|
+
/** Proper nouns / identifiers the ANSWER uses as a sentence subject (e.g. "Thunderbolt can't …"). */
|
|
250
|
+
const ANSWER_SUBJECT = /(?<=[a-z,;:]\s)([A-Z][A-Za-z0-9]*(?:[-.][A-Za-z0-9]+)*)(?=\s+(?:can(?:not|'t|’t)?|does(?:n't|n’t|\s+not)?|supports?|only|is\s*n(?:'|’)?t|has\s+no|won't|will\s+not)\b)|^([A-Z][A-Za-z0-9]*(?:[-.][A-Za-z0-9]+)+|[A-Z][a-z]+[A-Z][A-Za-z0-9]*)(?=\s+(?:can|does|supports?|only|is\s*n))/gm;
|
|
251
|
+
const NOT_SUBJECT = new Set(['i', 'it', 'this', 'that', 'there', 'they', 'we', 'you', 'he', 'she', 'which', 'what', 'who', 'nothing', 'none', 'one']);
|
|
252
|
+
export function answerSubjects(message) {
|
|
253
|
+
const out = new Set();
|
|
254
|
+
for (const m of strippedProse(message).matchAll(ANSWER_SUBJECT)) {
|
|
255
|
+
const t = (m[1] || m[2] || '').toLowerCase();
|
|
256
|
+
if (t.length >= 3 && !NOT_SUBJECT.has(t)) out.add(t);
|
|
257
|
+
}
|
|
258
|
+
return [...out];
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
export function auditAssertions({ message, subjects = [], vocab = [], sources = null, stampTerms = [] }) {
|
|
262
|
+
const tools = [...new Set([...vocab, ...subjects, ...answerSubjects(message)])].filter((t) => t.length >= 3);
|
|
263
|
+
const claims = capabilityClaims(message, tools);
|
|
264
|
+
const findings = [];
|
|
265
|
+
const unknown = [];
|
|
266
|
+
for (const claim of claims) {
|
|
267
|
+
if (sources === null) {
|
|
268
|
+
const words = subjectWords(claim.subject);
|
|
269
|
+
const rUv = words.some((w) => RUVNET_GATE1_TERMS.includes(w));
|
|
270
|
+
if (!rUv) { unknown.push(claim); continue; }
|
|
271
|
+
if (!words.every((w) => stampTerms.includes(w))) findings.push({ ...claim, reason: 'no search_ruvnet stamp for this subject this turn', read: stampTerms.map((t) => `search_ruvnet stamp: ${t}`) });
|
|
272
|
+
continue;
|
|
273
|
+
}
|
|
274
|
+
const binding = bindingSources(claim.subject, sources.filter((s) => s.strength !== 'failed'));
|
|
275
|
+
const lastWeak = Math.max(-1, ...binding.filter((s) => s.strength === 'weak').map((s) => s.order));
|
|
276
|
+
const strongAfter = binding.some((s) => s.strength === 'strong' && s.order > lastWeak);
|
|
277
|
+
if (!strongAfter) {
|
|
278
|
+
findings.push({ ...claim, reason: lastWeak >= 0 ? 'the only sources about it this turn are weak (summarised or relayed)' : 'no source about it was read this turn',
|
|
279
|
+
read: describeSources(lastWeak >= 0 ? binding : sources) });
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
return { claims, findings, unknown };
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
export function describeSources(sources, max = 4) {
|
|
286
|
+
if (!sources.length) return ['nothing'];
|
|
287
|
+
const shown = sources.slice(-max).map((s) => `${s.kind} ${JSON.stringify(String(s.ref).slice(0, 70))}${s.strength === 'weak' ? ` [${s.why} = weak evidence]` : ''}`);
|
|
288
|
+
return sources.length > max ? [`${sources.length - max} earlier`, ...shown] : shown;
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
export function correctionText(findings) {
|
|
292
|
+
const f = findings[0];
|
|
293
|
+
const lines = [
|
|
294
|
+
`You asserted "${f.text.slice(0, 200)}" about ${f.subject}; no relevant source was read this turn`
|
|
295
|
+
+ ` (${f.reason}; read: ${f.read.join('; ')}).`,
|
|
296
|
+
...findings.slice(1, 3).map((x) => `Also unsourced: "${x.text.slice(0, 160)}" about ${x.subject}.`),
|
|
297
|
+
'Check the real source now (read the file, run the command with --help, search_ruvnet, or fetch the',
|
|
298
|
+
'raw page with curl — a WebFetch body is a small model\'s summary) or restate each claim as UNVERIFIED.',
|
|
299
|
+
'ADR-0030 decision point #1: "Did you CHECK, or are you recalling? Name the source."',
|
|
300
|
+
];
|
|
301
|
+
return lines.join('\n');
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
// ── shadow gates #2 and #3 (logged, never delivered) ──────────────────────────────────────────────
|
|
305
|
+
const RECOMMENDS = /\b(?:I(?:'d|’d|\s+would)?\s+(?:recommend|propose|suggest)|my\s+recommendation|recommended\s+(?:approach|option|design)|the\s+design\s+I(?:'d|’d|\s+would)\s+propose|I(?:'d|’d)\s+(?:go|build)\s+with|go\s+with\s+option)\b/i;
|
|
306
|
+
const OPTION_MARK = /(?:^|\n)\s*(?:#{1,4}\s*|[-*•]\s*|\*\*)?(?:option|alternative|approach)\s*(?:[A-Z1-9]|one|two|three|four)\b|\b(?:option|alternative)\s+(?:[A-D1-4])\b/gi;
|
|
307
|
+
export function architectureShadow({ architecture, message }) {
|
|
308
|
+
if (!architecture || !RECOMMENDS.test(strippedProse(message))) return null;
|
|
309
|
+
const options = new Set([...String(message).matchAll(OPTION_MARK)].map((m) => m[0].trim().toLowerCase().replace(/[^a-z0-9 ]/g, ''))).size;
|
|
310
|
+
return options >= 3 ? null : { gate: 'adr-0030-2-architecture-options', options, wouldBlock: true };
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
const NUMBER = /(?<![\w.])\d+(?:[.,]\d+)?(?:\s?%|\/\d+)?(?![\w])/g;
|
|
314
|
+
export function relayShadow({ message, sources }) {
|
|
315
|
+
if (!sources?.length) return null;
|
|
316
|
+
const agentIdx = sources.filter((s) => s.kind === 'Agent' || s.kind === 'Task');
|
|
317
|
+
if (!agentIdx.length) return null;
|
|
318
|
+
const nums = [...new Set((strippedProse(message).match(NUMBER) || []).map((n) => n.replace(/\s/g, '')))]
|
|
319
|
+
.filter((n) => /[.%/]/.test(n) || n.replace(/\D/g, '').length >= 3).filter((n) => !/^(?:19|20)\d\d$/.test(n));
|
|
320
|
+
const relayed = nums.filter((n) => {
|
|
321
|
+
const from = agentIdx.find((a) => String(a.result || '').includes(n));
|
|
322
|
+
if (!from) return false;
|
|
323
|
+
return !sources.some((s) => s.order > from.order && s.kind !== 'Agent' && s.kind !== 'Task' && String(s.result || '').includes(n));
|
|
324
|
+
});
|
|
325
|
+
return relayed.length ? { gate: 'adr-0030-3-relayed-number', numbers: relayed.slice(0, 8), wouldBlock: true } : null;
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
/** Append one shadow row, bounded. Never throws. */
|
|
329
|
+
export function logShadow(row, { env = process.env } = {}) {
|
|
330
|
+
try {
|
|
331
|
+
const file = env.RUVNET_ASSERTION_SHADOW_LOG || path.join(brainHome(env), 'assertion-gate-shadow.jsonl');
|
|
332
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
333
|
+
fs.appendFileSync(file, `${JSON.stringify(row)}\n`);
|
|
334
|
+
if (fs.statSync(file).size > 512 * 1024) {
|
|
335
|
+
const keep = fs.readFileSync(file, 'utf8').split('\n').filter(Boolean).slice(-500);
|
|
336
|
+
fs.writeFileSync(file, `${keep.join('\n')}\n`);
|
|
337
|
+
}
|
|
338
|
+
} catch { /* shadow measurement never breaks a turn */ }
|
|
339
|
+
}
|
|
@@ -58,6 +58,20 @@
|
|
|
58
58
|
* interrupted/cancelled turn is never forced. The marker is consumed (deleted) whether or not it
|
|
59
59
|
* fires, so a genuinely abandoned marker cannot pressure some unrelated later turn.
|
|
60
60
|
*
|
|
61
|
+
* 2026-09-30 — ADR-0030 DECISION POINT #1, AND THE FALSE ALARM. Two changes, one registration:
|
|
62
|
+
* - On Claude the "was search_ruvnet called" question is answered from the TRANSCRIPT (the ordered
|
|
63
|
+
* record of every tool call and result, grounding-turn-evidence.mjs turnSources), not from stamp
|
|
64
|
+
* mtimes. The stamp was a lossy proxy: 7 real false alarms were measured, 3 from a queued
|
|
65
|
+
* mid-turn prompt re-dating the marker (fixed in grounding-turn-mark.mjs), 3 pre-H1 vocabulary
|
|
66
|
+
* misses, 1 successful search whose stamp never minted. Codex's rollout is not parsed anywhere in
|
|
67
|
+
* this repo, so Codex keeps the stamp evidence.
|
|
68
|
+
* - When the marker says the prompt asked a capability/feasibility/architecture question, every
|
|
69
|
+
* capability claim in the final answer needs a RELEVANT, STRONG source read this turn after the
|
|
70
|
+
* last weak one (auditAssertions). A WebFetch body is a small model's summary: weak.
|
|
71
|
+
* Gates #2/#3 of ADR-0030 run in shadow (logShadow) — measured, never delivered.
|
|
72
|
+
* At most ONE correction per stop episode: both checks compose into one message, and
|
|
73
|
+
* stop_hook_active silences the continued stop.
|
|
74
|
+
*
|
|
61
75
|
* FAILS OPEN ALWAYS. Exit 0 unconditionally — a gate that breaks a turn's completion because a
|
|
62
76
|
* cache directory was unreadable would be disabled within a day.
|
|
63
77
|
*/
|
|
@@ -65,8 +79,13 @@ import fs from 'node:fs';
|
|
|
65
79
|
import os from 'node:os';
|
|
66
80
|
import path from 'node:path';
|
|
67
81
|
import { fileURLToPath } from 'node:url';
|
|
68
|
-
import {
|
|
69
|
-
import { markerPathFor } from './grounding-turn-mark.mjs';
|
|
82
|
+
import { readStopHookInput } from './hook-input.mjs';
|
|
83
|
+
import { markerPathFor, readMarker } from './grounding-turn-mark.mjs';
|
|
84
|
+
import { readSettledTranscript } from './turn-outcome-capture.mjs';
|
|
85
|
+
import {
|
|
86
|
+
architectureShadow, auditAssertions, correctionText, describeSources, loadVocabulary, logShadow,
|
|
87
|
+
relayShadow, searchedThisTurn, turnSources,
|
|
88
|
+
} from './grounding-turn-evidence.mjs';
|
|
70
89
|
|
|
71
90
|
const HOME = os.homedir();
|
|
72
91
|
const EXIT_ALLOW = 0;
|
|
@@ -94,6 +113,14 @@ export function newestGroundingStampMs(dir = GROUNDED_DIR) {
|
|
|
94
113
|
return newest;
|
|
95
114
|
}
|
|
96
115
|
|
|
116
|
+
/** Product terms whose stamp was minted at or after `sinceMs` (Codex's only view of this turn's searches). */
|
|
117
|
+
export function stampTermsSince(sinceMs, dir = GROUNDED_DIR) {
|
|
118
|
+
try {
|
|
119
|
+
return fs.readdirSync(dir).filter((name) => !name.startsWith('.')
|
|
120
|
+
&& fs.statSync(path.join(dir, name)).mtimeMs >= sinceMs - SKEW_MS);
|
|
121
|
+
} catch { return []; }
|
|
122
|
+
}
|
|
123
|
+
|
|
97
124
|
/** A little slack for filesystem mtime granularity (some filesystems round to whole seconds), so a
|
|
98
125
|
* stamp written the same wall-clock second as the marker is never wrongly judged "before" it. */
|
|
99
126
|
const SKEW_MS = 1500;
|
|
@@ -106,18 +133,54 @@ export function wasGroundedSince(markerMs, newestStampMs) {
|
|
|
106
133
|
return newestStampMs >= markerMs - SKEW_MS;
|
|
107
134
|
}
|
|
108
135
|
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
136
|
+
/**
|
|
137
|
+
* The whole Stop decision for one armed turn: the correction text, or null. Exported so tests can
|
|
138
|
+
* drive it with a synthetic transcript. Every failure inside returns null (fail open).
|
|
139
|
+
*/
|
|
140
|
+
export function decide({ hookInput, marker, markerMs, env = process.env, read = readSettledTranscript }) {
|
|
113
141
|
try {
|
|
114
|
-
const
|
|
115
|
-
|
|
116
|
-
|
|
142
|
+
const host = env.RUVNET_HOOK_HOST === 'codex' ? 'codex' : 'claude';
|
|
143
|
+
const tp = hookInput.transcript_path;
|
|
144
|
+
let turn = null;
|
|
145
|
+
if (host === 'claude' && typeof tp === 'string' && /\.jsonl$/i.test(tp)) {
|
|
146
|
+
try { turn = turnSources(read(tp, { maxMs: 0 })); } catch { turn = null; }
|
|
147
|
+
}
|
|
148
|
+
const sources = turn ? turn.sources : null;
|
|
149
|
+
const message = String(hookInput.last_assistant_message || '');
|
|
150
|
+
|
|
151
|
+
let assertion = null;
|
|
152
|
+
if (marker.assert && message) {
|
|
153
|
+
const vocab = loadVocabulary({ env });
|
|
154
|
+
const audit = auditAssertions({ message, subjects: marker.subjects, vocab, sources,
|
|
155
|
+
stampTerms: sources ? [] : stampTermsSince(markerMs) });
|
|
156
|
+
if (audit.findings.length) assertion = audit.findings;
|
|
157
|
+
const shadow = [architectureShadow({ architecture: marker.architecture, message }), relayShadow({ message, sources })].filter(Boolean);
|
|
158
|
+
for (const row of shadow) logShadow({ ...row, at: new Date().toISOString(), session: hookInput.session_id, host });
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
const grounded = marker.gate1 === false ? true
|
|
162
|
+
: sources ? searchedThisTurn(sources) : wasGroundedSince(markerMs, newestGroundingStampMs());
|
|
163
|
+
if (assertion) {
|
|
164
|
+
return correctionText(assertion) + (grounded ? '' : '\nThis turn also touched the rUv stack and no successful search_ruvnet call was recorded: call it with the product term(s).');
|
|
165
|
+
}
|
|
166
|
+
if (grounded) return null;
|
|
167
|
+
return [
|
|
168
|
+
'This turn touched the RuvNet / rUv stack and ground-ruvnet\'s directive required calling the',
|
|
169
|
+
'search_ruvnet MCP tool before asserting what any RuvNet tool can/cannot do — but no successful',
|
|
170
|
+
sources ? `search_ruvnet call is in this turn's transcript (read this turn: ${describeSources(sources).join('; ')}).`
|
|
171
|
+
: 'search_ruvnet call was recorded this turn (checked against the grounding-stamp evidence).',
|
|
172
|
+
'',
|
|
173
|
+
'Do NOT end the turn on an ungrounded rUv-domain answer. Call `search_ruvnet` now with the',
|
|
174
|
+
'relevant product term(s) in the query, ground your answer in the cited source paths it returns,',
|
|
175
|
+
'and correct anything you already asserted from memory. Training priors on the rUv stack are',
|
|
176
|
+
'stale by construction (ADR-0012) — this is not a formality.',
|
|
177
|
+
].join('\n');
|
|
178
|
+
} catch { return null; }
|
|
117
179
|
}
|
|
118
180
|
|
|
181
|
+
|
|
119
182
|
async function main() {
|
|
120
|
-
const hookInput = await
|
|
183
|
+
const hookInput = await readStopHookInput();
|
|
121
184
|
if (hookInput.__source !== 'stdin') process.exit(EXIT_ALLOW);
|
|
122
185
|
if (hookInput.stop_hook_active) process.exit(EXIT_ALLOW);
|
|
123
186
|
if (hookInput.hook_event_name !== 'Stop' || hookInput.interrupted || hookInput.cancelled) {
|
|
@@ -135,27 +198,16 @@ async function main() {
|
|
|
135
198
|
// unrelated turn (same reasoning as continuation-gate.mjs's cooldown lock, applied here as a
|
|
136
199
|
// single-use marker instead of a timed window, because "did this turn ground itself" has no
|
|
137
200
|
// meaningful reading beyond the one turn it was written for).
|
|
201
|
+
const armed = readMarker(marker) || { gate1: true, subjects: [] };
|
|
138
202
|
try { fs.unlinkSync(marker); } catch { /* a marker that vanished between stat and unlink already told us what we needed */ }
|
|
139
203
|
|
|
140
|
-
const
|
|
141
|
-
if (
|
|
142
|
-
|
|
143
|
-
const lines = [
|
|
144
|
-
'This turn touched the RuvNet / rUv stack and ground-ruvnet\'s directive required calling the',
|
|
145
|
-
'search_ruvnet MCP tool before asserting what any RuvNet tool can/cannot do — but no successful',
|
|
146
|
-
'search_ruvnet call was recorded this turn (checked against the same grounding-stamp evidence',
|
|
147
|
-
'ground-before-write.sh already trusts).',
|
|
148
|
-
'',
|
|
149
|
-
'Do NOT end the turn on an ungrounded rUv-domain answer. Call `search_ruvnet` now with the',
|
|
150
|
-
'relevant product term(s) in the query, ground your answer in the cited source paths it returns,',
|
|
151
|
-
'and correct anything you already asserted from memory. Training priors on the rUv stack are',
|
|
152
|
-
'stale by construction (ADR-0012) — this is not a formality.',
|
|
153
|
-
];
|
|
204
|
+
const text = decide({ hookInput, marker: armed, markerMs: markerStat.mtimeMs });
|
|
205
|
+
if (!text) process.exit(EXIT_ALLOW);
|
|
154
206
|
|
|
155
207
|
process.stdout.write(JSON.stringify({
|
|
156
208
|
hookSpecificOutput: {
|
|
157
209
|
hookEventName: 'Stop',
|
|
158
|
-
additionalContext:
|
|
210
|
+
additionalContext: text,
|
|
159
211
|
},
|
|
160
212
|
}));
|
|
161
213
|
process.exit(EXIT_ALLOW);
|
|
@@ -24,6 +24,22 @@
|
|
|
24
24
|
* one). Its CONTENT is a JSON blob for a human reading the cache, but the Stop-time gate only ever
|
|
25
25
|
* trusts the mtime.
|
|
26
26
|
*
|
|
27
|
+
* 2026-09-30 — TWO ARMS, ONE MARKER (ADR-0030 decision point #1). The marker now also records, as
|
|
28
|
+
* JSON content, whether the prompt ASKS for a capability / feasibility / architecture judgement about
|
|
29
|
+
* a subject (grounding-turn-evidence.mjs classifyPrompt) and which subjects it named, so the Stop gate
|
|
30
|
+
* can require a relevant source for any capability claim the answer makes — on any platform, not
|
|
31
|
+
* only the rUv stack. `gate1` keeps the original meaning (the prompt matched Gate 1).
|
|
32
|
+
*
|
|
33
|
+
* THE FALSE-ALARM FIX. Measured on real transcripts (7 turns where the Stop gate said "no successful
|
|
34
|
+
* search_ruvnet call was recorded" although one had been made): in 3 of them a QUEUED message (a
|
|
35
|
+
* real user message typed mid-turn, or a task notification before H2) fired UserPromptSubmit again
|
|
36
|
+
* AFTER the search and rewrote this marker, moving the turn boundary past the evidence. So an
|
|
37
|
+
* unconsumed marker is now MERGED, never re-dated: its mtime (the boundary) stays at the first arm
|
|
38
|
+
* of the stop episode. A marker older than STALE_MS (an interrupted turn never reaches Stop) is
|
|
39
|
+
* replaced instead. Of the other 4, three were pre-H1 vocabulary misses (fixed by H1) and one was a
|
|
40
|
+
* successful search whose stamp never minted (2026-09-30, cause not recoverable from the transcript);
|
|
41
|
+
* so on Claude the Stop gate now reads the transcript itself (grounding-turn-gate.mjs).
|
|
42
|
+
*
|
|
27
43
|
* CONTRACT: PostToolUse-shaped hooks in this repo are advisory; this one is too — it can never
|
|
28
44
|
* block a prompt. It exits 0 unconditionally and writes nothing to stdout Claude/Codex would act
|
|
29
45
|
* on (UserPromptSubmit's silence contract). A write failure (unwritable cache dir, race, etc.) is
|
|
@@ -36,6 +52,7 @@ import path from 'node:path';
|
|
|
36
52
|
import { fileURLToPath } from 'node:url';
|
|
37
53
|
import { readStdinBounded, isHarnessGenerated } from './hook-input.mjs';
|
|
38
54
|
import { ruvnetGate1Matches } from './ruvnet-gate1-pattern.mjs';
|
|
55
|
+
import { classifyPrompt, loadVocabulary } from './grounding-turn-evidence.mjs';
|
|
39
56
|
|
|
40
57
|
const HOME = os.homedir();
|
|
41
58
|
export const MARKER_DIR = process.env.RUVNET_GROUNDING_TURN_DIR
|
|
@@ -49,17 +66,50 @@ export function markerPathFor(sessionId, dir = MARKER_DIR) {
|
|
|
49
66
|
return safe ? path.join(dir, `${safe}.json`) : null;
|
|
50
67
|
}
|
|
51
68
|
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
if (!hookInput || hookInput.hook_event_name !== 'UserPromptSubmit') return false;
|
|
55
|
-
if (!hookInput.session_id) return false;
|
|
56
|
-
const text = String(hookInput.prompt ?? hookInput.user_prompt ?? hookInput.input ?? '');
|
|
69
|
+
const promptOf = (hookInput) => String(hookInput?.prompt ?? hookInput?.user_prompt ?? hookInput?.input ?? '');
|
|
70
|
+
const eligible = (hookInput) => Boolean(hookInput && hookInput.hook_event_name === 'UserPromptSubmit' && hookInput.session_id
|
|
57
71
|
// H2: a background task notification, slash-command scaffold, or other harness-authored message
|
|
58
72
|
// arrives on UserPromptSubmit exactly like real user text — arming the Stop-time grounding gate off
|
|
59
73
|
// one of these (because it happens to mention a rUv term) would demand a search_ruvnet call to
|
|
60
74
|
// close out a "turn" nobody had a hand in.
|
|
61
|
-
|
|
62
|
-
|
|
75
|
+
&& !isHarnessGenerated(promptOf(hookInput)));
|
|
76
|
+
|
|
77
|
+
/** Exported for the unit test: pure decision, no I/O. Gate 1 (the rUv-stack search requirement). */
|
|
78
|
+
export function shouldMark(hookInput) {
|
|
79
|
+
return eligible(hookInput) && ruvnetGate1Matches(promptOf(hookInput));
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/** Both arms for one prompt, or null when neither fires. Pure apart from the vocabulary it is given. */
|
|
83
|
+
export function armFor(hookInput, vocab = []) {
|
|
84
|
+
if (!eligible(hookInput)) return null;
|
|
85
|
+
const gate1 = ruvnetGate1Matches(promptOf(hookInput));
|
|
86
|
+
const c = classifyPrompt(promptOf(hookInput), vocab);
|
|
87
|
+
if (!gate1 && !c.assert) return null;
|
|
88
|
+
return { gate1, assert: c.assert, architecture: c.architecture, subjects: c.subjects };
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
export const STALE_MS = 2 * 3600_000;
|
|
92
|
+
/** A marker's JSON, or null. Old markers (no `gate1` field) were only ever written for Gate 1. */
|
|
93
|
+
export function readMarker(file) {
|
|
94
|
+
try {
|
|
95
|
+
const m = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
96
|
+
return m && typeof m === 'object' ? { gate1: m.gate1 !== false, assert: !!m.assert, architecture: !!m.architecture,
|
|
97
|
+
subjects: Array.isArray(m.subjects) ? m.subjects.map(String) : [], at: m.at } : null;
|
|
98
|
+
} catch { return null; }
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/** Merge an arm into an unconsumed marker WITHOUT moving its mtime (the turn boundary). */
|
|
102
|
+
export function writeArm(file, arm, meta = {}, now = Date.now()) {
|
|
103
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
104
|
+
let st = null;
|
|
105
|
+
try { st = fs.statSync(file); } catch { /* none yet */ }
|
|
106
|
+
const prev = st && now - st.mtimeMs < STALE_MS ? readMarker(file) : null;
|
|
107
|
+
const next = prev ? { ...meta, at: prev.at, gate1: prev.gate1 || arm.gate1, assert: prev.assert || arm.assert,
|
|
108
|
+
architecture: prev.architecture || arm.architecture, subjects: [...new Set([...prev.subjects, ...arm.subjects])].slice(0, 32) }
|
|
109
|
+
: { ...meta, at: new Date(now).toISOString(), ...arm };
|
|
110
|
+
fs.writeFileSync(file, JSON.stringify(next) + '\n');
|
|
111
|
+
if (prev) fs.utimesSync(file, st.atime, st.mtime);
|
|
112
|
+
return next;
|
|
63
113
|
}
|
|
64
114
|
|
|
65
115
|
async function main() {
|
|
@@ -69,17 +119,14 @@ async function main() {
|
|
|
69
119
|
hookInput = JSON.parse(raw || '{}');
|
|
70
120
|
} catch { process.exit(0); }
|
|
71
121
|
|
|
72
|
-
|
|
122
|
+
let arm = null;
|
|
123
|
+
try { arm = armFor(hookInput, loadVocabulary()); } catch { arm = shouldMark(hookInput) ? { gate1: true, assert: false, architecture: false, subjects: [] } : null; }
|
|
124
|
+
if (!arm) process.exit(0);
|
|
73
125
|
|
|
74
126
|
const file = markerPathFor(hookInput.session_id);
|
|
75
127
|
if (!file) process.exit(0);
|
|
76
128
|
try {
|
|
77
|
-
|
|
78
|
-
fs.writeFileSync(file, JSON.stringify({
|
|
79
|
-
at: new Date().toISOString(),
|
|
80
|
-
sessionId: hookInput.session_id,
|
|
81
|
-
turnId: hookInput.turn_id || hookInput.prompt_id || null,
|
|
82
|
-
}) + '\n');
|
|
129
|
+
writeArm(file, arm, { sessionId: hookInput.session_id, turnId: hookInput.turn_id || hookInput.prompt_id || null });
|
|
83
130
|
} catch { /* fail-open: no marker means the Stop gate stays silent, never a false block */ }
|
|
84
131
|
process.exit(0);
|
|
85
132
|
}
|
|
@@ -96,6 +96,21 @@ export function readStdinBounded({ maxBytes = 65536, idleMs = 50, emptyMs = 250
|
|
|
96
96
|
});
|
|
97
97
|
}
|
|
98
98
|
|
|
99
|
+
/**
|
|
100
|
+
* The Stop-hook payload with its provenance, shared by every Stop gate (was a verbatim copy in
|
|
101
|
+
* continuation-gate and grounding-turn-gate). Three sources, treated differently by callers:
|
|
102
|
+
* 'tty' (run bare, never force), 'unreadable' (read/parse failed — fs.readFileSync(0) throws EAGAIN
|
|
103
|
+
* intermittently on macOS; laundering that into {} would read as a fresh stop), 'stdin' (parsed — the
|
|
104
|
+
* only source allowed to force). ADR-043.
|
|
105
|
+
*/
|
|
106
|
+
export async function readStopHookInput() {
|
|
107
|
+
if (process.stdin.isTTY) return { __source: 'tty' };
|
|
108
|
+
try {
|
|
109
|
+
const raw = (await readStdinBounded()).toString('utf8');
|
|
110
|
+
return { ...JSON.parse(raw || '{}'), __source: 'stdin' };
|
|
111
|
+
} catch { return { __source: 'unreadable' }; }
|
|
112
|
+
}
|
|
113
|
+
|
|
99
114
|
/** The tool being invoked ("Bash", "Write", …), or "" if absent. */
|
|
100
115
|
export function toolName(ev) {
|
|
101
116
|
return ev && typeof ev.tool_name === 'string' ? ev.tool_name : '';
|
|
@@ -120,7 +120,9 @@ const TABLE = {
|
|
|
120
120
|
// way — it stopped trusting hook-shim.mjs as a blind generic spawner, not their presence here.
|
|
121
121
|
'learn-capture': { file: 'learn-capture.sh', interpreter: 'bash', mode: 'advisory', offBehavior: 'silence' },
|
|
122
122
|
'learn-flush': { file: 'learn-flush.mjs', interpreter: 'node', mode: 'advisory', offBehavior: 'silence' },
|
|
123
|
-
|
|
123
|
+
// 1 MiB, not 64 KiB: the Stop payload now carries `last_assistant_message`, and a long closing
|
|
124
|
+
// message truncated mid-JSON would parse as `{}` and silently drop the whole capture.
|
|
125
|
+
'session-snapshot': { file: 'session-snapshot-hook.mjs', interpreter: 'node', mode: 'advisory', offBehavior: 'run', stdinBytes: 1048576 },
|
|
124
126
|
'md-stamp': { file: 'md-stamp.mjs', interpreter: 'node', mode: 'advisory', offBehavior: 'silence' },
|
|
125
127
|
// THE EXTERNAL-SIGNAL WATCH PLANE, W1 OBSERVED (ADR-058 §D3; DDD-0013 Context 2). PostToolUse,
|
|
126
128
|
// matcher ^Bash$ (anchored — an unanchored matcher is F3/F4). Classifies gh/vercel/netlify/npm
|