ruvnet-brain 4.3.37 → 4.3.39
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/bin/install.mjs +32 -9
- package/kb/forge-update.mjs +1884 -0
- package/package.json +3 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/hooks/codex-hooks.json +4 -4
- package/plugin/hooks/hook-contracts.json +7 -7
- package/plugin/hooks/hooks.json +2 -2
- package/plugin/scripts/capability-claim-evidence.mjs +11 -2
- package/plugin/scripts/capacity-aware-parallel-work.mjs +71 -15
- package/plugin/scripts/codex-hook-wrapper.mjs +3 -0
- package/plugin/scripts/completion-claim-evidence.mjs +262 -0
- package/plugin/scripts/continuation-gate.mjs +105 -21
- package/plugin/scripts/continuation-objective.mjs +15 -0
- package/plugin/scripts/decision-gate.mjs +22 -2
- package/plugin/scripts/duplicate-gate.mjs +503 -0
- package/plugin/scripts/grounding-turn-evidence.mjs +339 -0
- package/plugin/scripts/grounding-turn-gate.mjs +77 -25
- package/plugin/scripts/grounding-turn-mark.mjs +61 -14
- package/plugin/scripts/hook-input.mjs +15 -0
- package/plugin/scripts/host-update.mjs +45 -0
- package/plugin/scripts/nightly-scheduler.mjs +34 -0
- package/plugin/scripts/session-start-budget.mjs +1 -0
- package/plugin/scripts/session-start-core.mjs +15 -2
- package/plugin/scripts/session-start-health.mjs +101 -1
- package/plugin/scripts/session-start-update-plane.mjs +71 -1
- package/scripts/approved-runtime.mjs +1 -1
- package/scripts/completion-claim-replay.mjs +114 -0
- package/scripts/corpus-canary.mjs +396 -0
- package/scripts/corpus-dispatch-decision.mjs +2 -2
- package/scripts/corpus-promotion.mjs +49 -0
- package/scripts/corpus-reconcile.mjs +27 -3
- package/scripts/corpus-watchdog.mjs +45 -6
- package/scripts/duplicate-gate-replay.mjs +98 -0
- package/scripts/grounding-turn-replay.mjs +131 -0
- package/scripts/nightly-watchdog.mjs +3 -3
- package/scripts/protected-release-invocation.mjs +1 -1
- package/scripts/release.mjs +180 -77
- package/scripts/single-source-check.mjs +12 -6
- package/scripts/wired-check.mjs +2 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ruvnet-brain",
|
|
3
|
-
"version": "4.3.
|
|
3
|
+
"version": "4.3.39",
|
|
4
4
|
"description": "One-command installer for RuvNet Brain \u2014 a portable, source-grounded brain over rUv's RuvNet building blocks, delivered as a Claude Code plugin so Claude uses the stack instead of fighting it.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -68,6 +68,7 @@
|
|
|
68
68
|
"cap:collect": "node scripts/rerank-cap-eval.mjs --collect",
|
|
69
69
|
"cap:report": "node scripts/rerank-cap-eval.mjs --report",
|
|
70
70
|
"cap:ab": "node scripts/rerank-cap-warm-ab.mjs",
|
|
71
|
+
"completion-claim:replay": "node scripts/completion-claim-replay.mjs",
|
|
71
72
|
"hooks:check": "node scripts/hook-retirement-check.mjs",
|
|
72
73
|
"release:qualify": "node scripts/release-qualification.mjs",
|
|
73
74
|
"rehearse:corpus": "node scripts/rehearse-corpus-pipeline.mjs",
|
|
@@ -96,6 +97,7 @@
|
|
|
96
97
|
"scripts/",
|
|
97
98
|
"kb/verify-citation.mjs",
|
|
98
99
|
"kb/retrieval-result.mjs",
|
|
100
|
+
"kb/forge-update.mjs",
|
|
99
101
|
"kb/corpus-release-identity.mjs",
|
|
100
102
|
"kb/capability-only.mjs",
|
|
101
103
|
"kb/capability-summaries/",
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ruvnet-brain",
|
|
3
3
|
"description": "RuvNet brain transplant for Claude Code — grounds every RuvNet decision in real source across 77 rUv repositories, prefers Ruflo / RuVector-RVF / AgentDB over training-prior defaults (pgvector, Pinecone, hand-rolled cosine), and can pull in any RuvNet repo on demand. Ships a UserPromptSubmit retrieve-and-inject grounding hook and a PreToolUse write gate that refuses ungrounded rUv-product code until search_ruvnet has been consulted (ADR-0012 / ADR-067).",
|
|
4
|
-
"version": "4.3.
|
|
4
|
+
"version": "4.3.39",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Stuart Kerr"
|
|
7
7
|
},
|
|
@@ -65,8 +65,8 @@
|
|
|
65
65
|
"hooks": [
|
|
66
66
|
{
|
|
67
67
|
"type": "command",
|
|
68
|
-
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\"
|
|
69
|
-
"timeout":
|
|
68
|
+
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\" 9000 unprompted-speech UserPromptSubmit",
|
|
69
|
+
"timeout": 10
|
|
70
70
|
},
|
|
71
71
|
{
|
|
72
72
|
"type": "command",
|
|
@@ -75,8 +75,8 @@
|
|
|
75
75
|
},
|
|
76
76
|
{
|
|
77
77
|
"type": "command",
|
|
78
|
-
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\"
|
|
79
|
-
"timeout":
|
|
78
|
+
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\" 5500 capacity-aware-parallel-work",
|
|
79
|
+
"timeout": 6
|
|
80
80
|
},
|
|
81
81
|
{
|
|
82
82
|
"type": "command",
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"_note": "The legacy automatic gate collection remains retired. What is permitted is the continuity plane below and nothing else: SessionStart restores the canonical project checkpoint; UserPromptSubmit runs the single unprompted-speech chokepoint; Stop may nudge one explicitly authorized project-scoped objective AND capture a project snapshot; PreCompact and SessionEnd capture a project snapshot. Version 3 of this file permitted exactly two handlers, which was not a safety property but a contradiction: it demanded a restore while forbidding any event that could WRITE the journal the restore reads, so the canonical store held zero progression rows. Adding to this list is still a deliberate act that must pass `npm run hooks:check`; what changed is that the shape can now express the plane that actually works. Amended 2026-09-11 (Stuart): the continuity-only charter had retired the ONLY enforcement of ADR-0012 — never write rUv-product code the brain has not seen — and the failure it exists to prevent recurred the day it was measured absent. The plane now also carries three grounding registrations: ground-ruvnet (UserPromptSubmit, grounding injection — a second owner of that event, scoped by ADR-040 §Amendment 2026-09-11 to directives rather than speech), decision-gate’s write route (PreToolUse, the one refuser, ADR-067), and grounding-stamp (PostToolUse on a successful search_ruvnet, the receipt that opens the write gate). Amended 2026-09-12: the 2026-09-11 claim that Codex PreToolUse/PostToolUse delivery had never been observed was measured with a prompt that never invoked a tool (`codex exec \"reply OK\"`), so it was an untested path, not a failing one. Re-measured with prompts that actually call a tool (a real apply_patch write, a real MCP search_ruvnet call against this repo's own server): both events FIRED on codex-cli 0.154.0 with real payloads. decision-gate's write route and grounding-stamp are now dual-host (see _codexCapture and contracts below); the bash route (exec_command) remains Claude-only — today's measurement did not exercise it. Also amended 2026-09-12: added grounding-turn-mark (UserPromptSubmit) and grounding-turn-gate (Stop), the \"answered without searching\" pair — ground-ruvnet's Gate 1 directive is advisory, so nothing previously checked whether the model complied before the turn ended. grounding-turn-mark records that Gate 1 fired for a turn; grounding-turn-gate forces continuation at Stop if grounding-stamp's own evidence shows no search_ruvnet call happened since. Both dual-host from the start (the Stop-block contract is already proven on Codex via continuation-gate). Added capacity-aware-parallel-work at UserPromptSubmit on both measured hosts: it is context-only, resource-bounded, and requires the coordinator to check actual tool/runtime slots; it does not spawn or claim workers.",
|
|
2
|
+
"_note": "The legacy automatic gate collection remains retired. What is permitted is the continuity plane below and nothing else: SessionStart restores the canonical project checkpoint; UserPromptSubmit runs the single unprompted-speech chokepoint; Stop may nudge one explicitly authorized project-scoped objective AND capture a project snapshot; PreCompact and SessionEnd capture a project snapshot. Version 3 of this file permitted exactly two handlers, which was not a safety property but a contradiction: it demanded a restore while forbidding any event that could WRITE the journal the restore reads, so the canonical store held zero progression rows. Adding to this list is still a deliberate act that must pass `npm run hooks:check`; what changed is that the shape can now express the plane that actually works. Amended 2026-09-11 (Stuart): the continuity-only charter had retired the ONLY enforcement of ADR-0012 — never write rUv-product code the brain has not seen — and the failure it exists to prevent recurred the day it was measured absent. The plane now also carries three grounding registrations: ground-ruvnet (UserPromptSubmit, grounding injection — a second owner of that event, scoped by ADR-040 §Amendment 2026-09-11 to directives rather than speech), decision-gate’s write route (PreToolUse, the one refuser, ADR-067), and grounding-stamp (PostToolUse on a successful search_ruvnet, the receipt that opens the write gate). Amended 2026-09-12: the 2026-09-11 claim that Codex PreToolUse/PostToolUse delivery had never been observed was measured with a prompt that never invoked a tool (`codex exec \"reply OK\"`), so it was an untested path, not a failing one. Re-measured with prompts that actually call a tool (a real apply_patch write, a real MCP search_ruvnet call against this repo's own server): both events FIRED on codex-cli 0.154.0 with real payloads. decision-gate's write route and grounding-stamp are now dual-host (see _codexCapture and contracts below); the bash route (exec_command) remains Claude-only — today's measurement did not exercise it. Also amended 2026-09-12: added grounding-turn-mark (UserPromptSubmit) and grounding-turn-gate (Stop), the \"answered without searching\" pair — ground-ruvnet's Gate 1 directive is advisory, so nothing previously checked whether the model complied before the turn ended. grounding-turn-mark records that Gate 1 fired for a turn; grounding-turn-gate forces continuation at Stop if grounding-stamp's own evidence shows no search_ruvnet call happened since. Both dual-host from the start (the Stop-block contract is already proven on Codex via continuation-gate). Added capacity-aware-parallel-work at UserPromptSubmit on both measured hosts: it is context-only, resource-bounded, and requires the coordinator to check actual tool/runtime slots; it does not spawn or claim workers. Amended 2026-09-30 (ADR-0030 decision point #1, no new registration): grounding-turn-mark also arms a turn when the prompt asks a capability/feasibility/architecture question about any tool or platform, and grounding-turn-gate then requires every capability claim in the final answer to be bound to a relevant STRONG source read this turn after the last weak one (a WebFetch body is a small model's summary: weak); on Claude it reads the transcript instead of stamp mtimes, which removed the measured 'no search recorded' false alarms. ADR-0030 #2/#3 run in shadow only.",
|
|
3
3
|
"_version": 8,
|
|
4
4
|
"_eventOwners": [
|
|
5
5
|
{
|
|
@@ -30,7 +30,7 @@
|
|
|
30
30
|
"claude",
|
|
31
31
|
"codex"
|
|
32
32
|
],
|
|
33
|
-
"responsibility": "May request one continuation for one explicitly authorized, project-scoped objective read from the user's own work ledger (ADR-043). Never a refusal, never a second task store."
|
|
33
|
+
"responsibility": "May request one continuation for one explicitly authorized, project-scoped objective read from the user's own work ledger (ADR-043). Never a refusal, never a second task store. Amended 2026-09-30 (ADR-074 class `completion`, no new registration): the same Stop body also requests ONE correction when the final answer claims work is done without a verification command run after the last state change this turn (Claude transcript), a named check, and a NOT-verified disclosure — on Codex, whose rollout format is not parsed, only the answer-side half is enforced and the transcript half is reported UNKNOWN. It also records first-person promises from the final answer into the SAME per-project ledger (Claude only; capped; closed only by a later evidence-passing completion claim, never by --done) and requests continuation on them under the same loop guards and cooldown."
|
|
34
34
|
},
|
|
35
35
|
{
|
|
36
36
|
"event": "Stop",
|
|
@@ -89,7 +89,7 @@
|
|
|
89
89
|
"claude",
|
|
90
90
|
"codex"
|
|
91
91
|
],
|
|
92
|
-
"responsibility": "The ONE process that may refuse a Write/Edit/MultiEdit/NotebookEdit/apply_patch (ADR-067): composes protect-state, hijack-ruvnet, ground-before-write (ADR-0012)
|
|
92
|
+
"responsibility": "The ONE process that may refuse a Write/Edit/MultiEdit/NotebookEdit/apply_patch (ADR-067): composes protect-state, hijack-ruvnet, ground-before-write (ADR-0012), adr-currency and duplicate-code (2026-09-30: a new code file or large new export duplicating existing code in this checkout, refused at most once per path per session) into one decision with one reason. Fails open on its own errors. apply_patch is Codex's raw write-tool name, added to the shared matcher 2026-09-12 after a real apply_patch call was measured to fire PreToolUse live (codex-cli 0.154.0); codex-hook-adapter.mjs normalizes it to the same Edit shape decision-gate.mjs's policies already read on Claude."
|
|
93
93
|
},
|
|
94
94
|
{
|
|
95
95
|
"event": "PostToolUse",
|
|
@@ -109,7 +109,7 @@
|
|
|
109
109
|
"claude",
|
|
110
110
|
"codex"
|
|
111
111
|
],
|
|
112
|
-
"responsibility": "Records that ground-ruvnet's Gate 1 (the same regex, proven byte-identical by test) matched this turn's prompt, so grounding-turn-gate has something to check at Stop — Stop's own payload carries no prompt text. Writes only a marker file under ~/.cache/ruvnet-brain/grounding-turn/; never blocks, never speaks to the model."
|
|
112
|
+
"responsibility": "Records that ground-ruvnet's Gate 1 (the same regex, proven byte-identical by test) matched this turn's prompt, so grounding-turn-gate has something to check at Stop — Stop's own payload carries no prompt text. Amended 2026-09-30: also arms when the prompt asks a capability/feasibility/architecture question about any tool or platform, recording the subjects it named; a marker not yet consumed is merged without moving its mtime (a queued mid-turn prompt re-dating it was a measured false-alarm cause). Writes only a marker file under ~/.cache/ruvnet-brain/grounding-turn/; never blocks, never speaks to the model."
|
|
113
113
|
},
|
|
114
114
|
{
|
|
115
115
|
"event": "Stop",
|
|
@@ -119,7 +119,7 @@
|
|
|
119
119
|
"claude",
|
|
120
120
|
"codex"
|
|
121
121
|
],
|
|
122
|
-
"responsibility": "The 'answered without searching' gate (2026-09-12): if grounding-turn-mark's marker shows Gate 1 fired this turn and grounding-stamp's own stamp evidence shows no search_ruvnet call happened since, forces continuation via the same hookSpecificOutput.additionalContext contract continuation-gate already uses on both hosts. A separate registration from continuation-gate — see grounding-turn-gate.mjs's header for why folding it into that file's ledger/cooldown semantics would corrupt them rather than extend them. Consumes its marker unconditionally so a stale one can never pressure an unrelated later turn."
|
|
122
|
+
"responsibility": "The 'answered without searching' gate (2026-09-12): if grounding-turn-mark's marker shows Gate 1 fired this turn and grounding-stamp's own stamp evidence shows no search_ruvnet call happened since, forces continuation via the same hookSpecificOutput.additionalContext contract continuation-gate already uses on both hosts. A separate registration from continuation-gate — see grounding-turn-gate.mjs's header for why folding it into that file's ledger/cooldown semantics would corrupt them rather than extend them. Consumes its marker unconditionally so a stale one can never pressure an unrelated later turn. Amended 2026-09-30 (ADR-0030 #1): on Claude, 'was search_ruvnet called' is read from the transcript, and on an armed turn each capability claim in the final answer must be bound to a relevant strong source read this turn after the last weak one, else ONE correction naming the claim and what was read; on Codex (rollout not parsed) only rUv-term claims are judged, from stamps, and the rest are UNKNOWN. Gates #2/#3 are shadow-logged, never delivered."
|
|
123
123
|
}
|
|
124
124
|
],
|
|
125
125
|
"_notInThisPlane": [
|
|
@@ -179,7 +179,7 @@
|
|
|
179
179
|
"mode": "blocking",
|
|
180
180
|
"offBehavior": "silence",
|
|
181
181
|
"matcher": "*",
|
|
182
|
-
"timeout":
|
|
182
|
+
"timeout": 9
|
|
183
183
|
},
|
|
184
184
|
{
|
|
185
185
|
"id": "continuation-gate",
|
|
@@ -250,7 +250,7 @@
|
|
|
250
250
|
"mode": "advisory",
|
|
251
251
|
"offBehavior": "run",
|
|
252
252
|
"matcher": "*",
|
|
253
|
-
"timeout":
|
|
253
|
+
"timeout": 6
|
|
254
254
|
},
|
|
255
255
|
{
|
|
256
256
|
"id": "decision-gate",
|
package/plugin/hooks/hooks.json
CHANGED
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
{
|
|
21
21
|
"type": "command",
|
|
22
22
|
"command": "node \"${CLAUDE_PLUGIN_ROOT}/scripts/hook-shim.mjs\" unprompted-speech UserPromptSubmit",
|
|
23
|
-
"timeout":
|
|
23
|
+
"timeout": 9
|
|
24
24
|
},
|
|
25
25
|
{
|
|
26
26
|
"type": "command",
|
|
@@ -30,7 +30,7 @@
|
|
|
30
30
|
{
|
|
31
31
|
"type": "command",
|
|
32
32
|
"command": "node \"${CLAUDE_PLUGIN_ROOT}/scripts/hook-shim.mjs\" capacity-aware-parallel-work || true",
|
|
33
|
-
"timeout":
|
|
33
|
+
"timeout": 6
|
|
34
34
|
},
|
|
35
35
|
{
|
|
36
36
|
"type": "command",
|
|
@@ -10,7 +10,7 @@ const HOSTS = Object.freeze(['claude', 'codex']);
|
|
|
10
10
|
const OS_LANES = Object.freeze(['linux', 'macos', 'windows']);
|
|
11
11
|
const CLAIM_CLASSES = Object.freeze(['installation', 'behavior', 'currentVersion', 'latestVersion', 'health']);
|
|
12
12
|
const VERSION = /\bv?(\d+\.\d+\.\d+(?:[-+][0-9A-Za-z.-]+)?)\b/;
|
|
13
|
-
const
|
|
13
|
+
const RUVNET_TOOL_DEFAULT = '(?:Ruflo|Claude Flow|Agentic Flow|Agentic QE|RuVector|Agent Browser|Ruv Swarm|AgentDB|RuLake|RuvNet Brain)';
|
|
14
14
|
const MAX_AGE_MS = 10 * 60_000;
|
|
15
15
|
const MAX_LEDGER_BYTES = 1024 * 1024;
|
|
16
16
|
const STOPWORDS = new Set(['a', 'an', 'and', 'are', 'can', 'could', 'does', 'for', 'has', 'have',
|
|
@@ -227,7 +227,16 @@ export function readLiveSurfaceReceipts({ file = null, env = process.env, limit
|
|
|
227
227
|
} catch { return []; }
|
|
228
228
|
}
|
|
229
229
|
|
|
230
|
-
|
|
230
|
+
/**
|
|
231
|
+
* Claim extraction. `tools` widens the subject vocabulary beyond RUVNET_TOOL (grounding-turn-gate.mjs
|
|
232
|
+
* passes repo aliases, installed stores and the armed prompt's subjects); the default is unchanged,
|
|
233
|
+
* so this file's own Stop audit keeps exactly its original scope.
|
|
234
|
+
*/
|
|
235
|
+
const escapeRe = (value) => String(value).replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
236
|
+
export function extractClaims(message, { tools = null } = {}) {
|
|
237
|
+
const RUVNET_TOOL = tools?.length
|
|
238
|
+
? `(?:${[...new Set(tools.map(String).filter(Boolean))].sort((a, b) => b.length - a.length).map(escapeRe).join('|')})`
|
|
239
|
+
: RUVNET_TOOL_DEFAULT;
|
|
231
240
|
const claims = [];
|
|
232
241
|
for (const sentence of safeMessage(message).split(/(?<=[.!?])\s+|\n+/).map((value) => value.trim()).filter(Boolean)) {
|
|
233
242
|
let match = new RegExp(`\\b(${RUVNET_TOOL})\\b[^.!?]{0,48}\\b(?:current|installed)\\s+version\\s+(?:is\\s+)?v?(\\d+\\.\\d+\\.\\d+(?:[-+][0-9A-Za-z.-]+)?)`, 'i').exec(sentence);
|
|
@@ -74,12 +74,35 @@ export function parseMacPressureOutput(text, { totalMemoryBytes, normalizedLoad
|
|
|
74
74
|
return { freePct, swapUsedBytes, compressorBytes, totalMemoryBytes, normalizedLoad };
|
|
75
75
|
}
|
|
76
76
|
|
|
77
|
-
/**
|
|
77
|
+
/**
|
|
78
|
+
* The total-agent ceiling when no host cap is reported, scaled to the machine. Owner rule (2026-09-30):
|
|
79
|
+
* a big machine must be used — 16 cores / 128 GB is ~10 agents when idle; small or unmeasured machines
|
|
80
|
+
* keep the conservative 4.
|
|
81
|
+
*/
|
|
82
|
+
export function hardwareAgentCeiling(cores) {
|
|
83
|
+
if (!Number.isInteger(cores) || cores < 1) return UNKNOWN_RUNTIME_TOTAL_AGENT_CEILING;
|
|
84
|
+
if (cores >= 12) return 10;
|
|
85
|
+
if (cores >= 8) return 6;
|
|
86
|
+
return UNKNOWN_RUNTIME_TOTAL_AGENT_CEILING;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Classify measured pressure. Memory probes can time out on a loaded machine, so a sample carrying only
|
|
91
|
+
* the load average is classified by load alone (an unmeasurable memory state must not silently turn a
|
|
92
|
+
* 16-core machine into a one-agent machine). Nothing measurable at all recommends serial work.
|
|
93
|
+
*/
|
|
78
94
|
export function pressureRecommendation(sample) {
|
|
79
|
-
|
|
80
|
-
sample.freePct, sample.swapUsedBytes, sample.compressorBytes,
|
|
81
|
-
|
|
82
|
-
|
|
95
|
+
const memoryKnown = !!sample && [
|
|
96
|
+
sample.freePct, sample.swapUsedBytes, sample.compressorBytes, sample.totalMemoryBytes,
|
|
97
|
+
].every(Number.isFinite) && sample.totalMemoryBytes > 0;
|
|
98
|
+
if (sample && !memoryKnown && Number.isFinite(sample.normalizedLoad) && sample.normalizedLoad >= 0) {
|
|
99
|
+
const load = sample.normalizedLoad;
|
|
100
|
+
if (load >= 1) return { tier: 'constrained', totalAgents: 1, reason: 'CPU oversubscribed (load-only measurement)' };
|
|
101
|
+
if (load >= 0.75) return { tier: 'high-cpu', totalAgents: 3, reason: 'CPU pressure is high (load-only measurement)' };
|
|
102
|
+
if (load >= 0.55) return { tier: 'moderate', totalAgents: 5, reason: 'CPU headroom is partial (load-only measurement)' };
|
|
103
|
+
return { tier: 'available', totalAgents: null, reason: 'CPU headroom is available (load-only measurement)' };
|
|
104
|
+
}
|
|
105
|
+
if (!sample || !memoryKnown || !Number.isFinite(sample.normalizedLoad)) {
|
|
83
106
|
return { tier: 'unknown', totalAgents: 1, reason: 'capacity signals unavailable' };
|
|
84
107
|
}
|
|
85
108
|
const compressedRatio = sample.compressorBytes / sample.totalMemoryBytes;
|
|
@@ -93,7 +116,7 @@ export function pressureRecommendation(sample) {
|
|
|
93
116
|
return { tier: 'high-cpu', totalAgents: 3, reason: 'CPU pressure is high' };
|
|
94
117
|
}
|
|
95
118
|
if (sample.freePct < 75 || compressedRatio >= 0.2 || sample.normalizedLoad >= 0.55) {
|
|
96
|
-
return { tier: 'moderate', totalAgents:
|
|
119
|
+
return { tier: 'moderate', totalAgents: 5, reason: 'resource headroom is partial' };
|
|
97
120
|
}
|
|
98
121
|
return { tier: 'available', totalAgents: null, reason: 'measured headroom is available' };
|
|
99
122
|
}
|
|
@@ -106,10 +129,15 @@ export function effectiveAgentRecommendation(sample, {
|
|
|
106
129
|
configuredMaxChildren = null,
|
|
107
130
|
runtimeTotalAgentCap = null,
|
|
108
131
|
workUnitCount = null,
|
|
132
|
+
cores = null,
|
|
109
133
|
} = {}) {
|
|
110
134
|
const pressure = pressureRecommendation(sample);
|
|
111
135
|
const runtimeCapKnown = Number.isInteger(runtimeTotalAgentCap) && runtimeTotalAgentCap >= 1;
|
|
112
|
-
const
|
|
136
|
+
const ceiling = hardwareAgentCeiling(cores);
|
|
137
|
+
// A host-reported cap is authoritative over the hardware ceiling; pressure tiers never exceed it.
|
|
138
|
+
const limits = [pressure.totalAgents === null
|
|
139
|
+
? (runtimeCapKnown ? runtimeTotalAgentCap : ceiling)
|
|
140
|
+
: Math.min(pressure.totalAgents, ceiling)];
|
|
113
141
|
if (Number.isInteger(configuredMaxChildren) && configuredMaxChildren >= 0) {
|
|
114
142
|
limits.push(configuredMaxChildren + 1); // configured workers plus the coordinating agent
|
|
115
143
|
}
|
|
@@ -139,8 +167,19 @@ function readInput() {
|
|
|
139
167
|
return Buffer.concat(chunks).toString('utf8');
|
|
140
168
|
}
|
|
141
169
|
|
|
170
|
+
// The load average alone, from the OS. Used when the macOS memory probe times out (it does, under
|
|
171
|
+
// exactly the heavy load that matters) and on other platforms. Windows reports no load average, so it
|
|
172
|
+
// stays unmeasured there rather than reading as an idle machine.
|
|
173
|
+
function collectLoadOnly() {
|
|
174
|
+
if (process.platform === 'win32') return null;
|
|
175
|
+
const cpuCount = os.cpus().length;
|
|
176
|
+
const load = os.loadavg()[0];
|
|
177
|
+
if (!cpuCount || !Number.isFinite(load) || load < 0) return null;
|
|
178
|
+
return { normalizedLoad: load / cpuCount, loadAvg1: load };
|
|
179
|
+
}
|
|
180
|
+
|
|
142
181
|
function collectMacPressure() {
|
|
143
|
-
if (process.platform !== 'darwin') return
|
|
182
|
+
if (process.platform !== 'darwin') return collectLoadOnly();
|
|
144
183
|
try {
|
|
145
184
|
// One bounded subprocess gathers pressure, swap, and compressor bytes. Never use raw free RAM
|
|
146
185
|
// as the capacity signal; memory_pressure, actual swap, compression, and normalized load drive
|
|
@@ -149,28 +188,42 @@ function collectMacPressure() {
|
|
|
149
188
|
encoding: 'utf8', timeout: PROBE_TIMEOUT_MS, maxBuffer: 24 * 1024,
|
|
150
189
|
env: { PATH: '/usr/bin:/bin:/usr/sbin:/sbin' },
|
|
151
190
|
});
|
|
152
|
-
if (probe.error || probe.status !== 0) return
|
|
191
|
+
if (probe.error || probe.status !== 0) return collectLoadOnly();
|
|
153
192
|
const cpuCount = os.cpus().length;
|
|
154
193
|
const load = os.loadavg()[0];
|
|
155
194
|
if (!cpuCount || !Number.isFinite(load) || load < 0) return null;
|
|
156
|
-
|
|
195
|
+
const parsed = parseMacPressureOutput(probe.stdout, {
|
|
157
196
|
totalMemoryBytes: os.totalmem(), normalizedLoad: load / cpuCount,
|
|
158
197
|
});
|
|
198
|
+
return parsed ? { ...parsed, loadAvg1: load } : collectLoadOnly();
|
|
159
199
|
} catch {
|
|
160
|
-
return
|
|
200
|
+
return collectLoadOnly();
|
|
161
201
|
}
|
|
162
202
|
}
|
|
163
203
|
|
|
164
|
-
|
|
204
|
+
/** The numbers behind the recommendation, so the owner can check them (owner 2026-09-30). */
|
|
205
|
+
export function describeMeasurement(sample, cores) {
|
|
206
|
+
if (!sample || !Number.isFinite(sample.normalizedLoad)) return 'capacity measurement unavailable';
|
|
207
|
+
const parts = [];
|
|
208
|
+
if (Number.isInteger(cores) && cores > 0) parts.push(`${cores} cores`);
|
|
209
|
+
if (Number.isFinite(sample.totalMemoryBytes) && sample.totalMemoryBytes > 0) parts.push(`${Math.round(sample.totalMemoryBytes / GIB)} GB RAM`);
|
|
210
|
+
parts.push(Number.isFinite(sample.loadAvg1) ? `load ${sample.loadAvg1.toFixed(1)} (${sample.normalizedLoad.toFixed(2)}/core)` : `load ${sample.normalizedLoad.toFixed(2)}/core`);
|
|
211
|
+
if (Number.isFinite(sample.freePct)) parts.push(`memory ${Math.round(sample.freePct)}% free`);
|
|
212
|
+
if (Number.isFinite(sample.swapUsedBytes)) parts.push(sample.swapUsedBytes > 0 ? 'swap in use' : 'no swap');
|
|
213
|
+
return parts.join(', ');
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
export function formatAdvisory(recommendation, measured = null) {
|
|
165
217
|
const n = recommendation.totalAgents;
|
|
166
218
|
const runtime = recommendation.runtimeCapKnown
|
|
167
219
|
? 'Do not exceed the host-reported runtime cap; configured concurrency is only a ceiling.'
|
|
168
|
-
: `This hook cannot see the live runtime/tool cap; ${
|
|
220
|
+
: `This hook cannot see the live runtime/tool cap; ${n} total is this machine's measured ceiling until the coordinator checks it. Configured concurrency is not proof of available slots.`;
|
|
169
221
|
const workerPlan = recommendation.workers > 0
|
|
170
222
|
? `At most ${recommendation.workers} worker agent${recommendation.workers === 1 ? '' : 's'}`
|
|
171
223
|
: 'No additional worker agents';
|
|
172
224
|
return [
|
|
173
225
|
'Capacity-aware parallel-work advisory (context only; no workers were started).',
|
|
226
|
+
...(measured ? [`Measured just now: ${measured}. State these numbers to the owner when you size a fan-out, and never claim parallel workers that are not running.`] : []),
|
|
174
227
|
`This prompt appears to contain independent work. Resource tier: ${recommendation.tier}; recommend no more than ${n} total agent${n === 1 ? '' : 's'} including the coordinator (${workerPlan}).`,
|
|
175
228
|
`${runtime} Treat configured concurrency as a ceiling only; never infer that configured slots are available.`,
|
|
176
229
|
'If a real agent-spawn/task tool is available and slots remain, launch actual workers now with non-overlapping deliverables and collect their results. If tools or slots are unavailable, continue serially and do not claim parallel workers exist.',
|
|
@@ -185,11 +238,14 @@ export function runCapacityHook(rawInput, sample) {
|
|
|
185
238
|
// "independent work" prose but nobody wrote it — never advise a parallel-work fan-out off of one.
|
|
186
239
|
if (isHarnessGenerated(prompt)) return '';
|
|
187
240
|
if (!isSubstantialParallelWork(prompt)) return '';
|
|
188
|
-
|
|
241
|
+
const measuredSample = sample === undefined ? collectMacPressure() : sample;
|
|
242
|
+
const cores = os.cpus().length;
|
|
243
|
+
return formatAdvisory(effectiveAgentRecommendation(measuredSample, {
|
|
189
244
|
configuredMaxChildren: input?.configured_max_children,
|
|
190
245
|
runtimeTotalAgentCap: input?.runtime_total_agent_cap,
|
|
191
246
|
workUnitCount: input?.independent_workstream_count ?? estimateWorkUnitCount(prompt),
|
|
192
|
-
|
|
247
|
+
cores,
|
|
248
|
+
}), describeMeasurement(measuredSample, cores));
|
|
193
249
|
}
|
|
194
250
|
|
|
195
251
|
function main() {
|
|
@@ -69,6 +69,9 @@ function timeoutFor(hookId) {
|
|
|
69
69
|
// tinguishable from a crash; 6000ms covers the gate's cap plus this chain's spawn overhead
|
|
70
70
|
// (measured 773–1145ms end-to-end warm, so ~150–400ms of that is the wrapper/adapter/shim).
|
|
71
71
|
if (hookId === 'decision-gate') return 6_000;
|
|
72
|
+
// Advisory capacity hook: 4 of 8 runs were killed at the old 2s host timeout on a loaded machine (measured
|
|
73
|
+
// 2026-09-30). Keep the wrapper's budget below the registration's own 5.5s so the wrapper, not the host, ends it.
|
|
74
|
+
if (hookId === 'capacity-aware-parallel-work') return 5_000;
|
|
72
75
|
if (hookId === 'ground-ruvnet' || hookId === 'unprompted-speech' || hookId === 'continuation-gate') {
|
|
73
76
|
return 8_500;
|
|
74
77
|
}
|
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* completion-claim-evidence.mjs — ADR-074's `completion` claim class, and the promise capture that
|
|
3
|
+
* rides the same boundary. Pure functions only: continuation-gate.mjs (the ONE Stop chokepoint) owns
|
|
4
|
+
* delivery, loop safety, cooldown and the ledger write. No hook registration of its own.
|
|
5
|
+
*
|
|
6
|
+
* WHY (owner, 2026-09-30): "You can never ever tell me something is done and implemented without
|
|
7
|
+
* having tested it end to end... That's how I ended up with a broken corpus that hasn't been
|
|
8
|
+
* successfully updated in 40 days because you never told me it was broken."
|
|
9
|
+
*
|
|
10
|
+
* THE RULE, deterministic, read from the host transcript of THIS turn (everything after the last
|
|
11
|
+
* genuine user message):
|
|
12
|
+
* A positive completion assertion ("fixed", "done", "shipped", "it will now…") PASSES only when
|
|
13
|
+
* (1) at least one verification command EXECUTED after the last state-changing action, with its
|
|
14
|
+
* result present and not an error, AND
|
|
15
|
+
* (2) the answer names the check (a "Verified:"-style line, or the executed command's own name), AND
|
|
16
|
+
* (3) the answer discloses what is NOT verified.
|
|
17
|
+
* Anything else is one correction request. No claim recognised → no verdict about the prose.
|
|
18
|
+
*
|
|
19
|
+
* HOSTS. Claude Code transcripts (`transcript_path`, JSONL) are parsed. Codex's Stop payload also
|
|
20
|
+
* carries `transcript_path`, but its rollout format is NOT parsed anywhere in this repository
|
|
21
|
+
* (project-progression-sources.mjs and turn-outcome-capture.mjs both decline it), so on Codex the
|
|
22
|
+
* transcript half is UNKNOWN and only the answer-side half (2)+(3) is enforced — disclosed in the
|
|
23
|
+
* correction text, never silently treated as proven.
|
|
24
|
+
*/
|
|
25
|
+
import { readSettledTranscript } from './turn-outcome-capture.mjs';
|
|
26
|
+
|
|
27
|
+
const MAX_SENTENCE = 400;
|
|
28
|
+
export const PROMISE_KIND = 'assistant-commitment';
|
|
29
|
+
export const PROMISE_CAP_PER_TURN = 3;
|
|
30
|
+
export const PROMISE_CAP_OPEN = 8;
|
|
31
|
+
|
|
32
|
+
// ── answer-side parsing ─────────────────────────────────────────────────────────────────────────
|
|
33
|
+
/** Code, inline code, quoted strings and block quotes are someone else's words, never a claim. */
|
|
34
|
+
export function strippedProse(message) {
|
|
35
|
+
return String(message || '')
|
|
36
|
+
.replace(/```[\s\S]*?```/g, '\n')
|
|
37
|
+
.replace(/`[^`\n]*`/g, ' ')
|
|
38
|
+
.replace(/"[^"\n]{0,300}"|“[^”\n]{0,300}”/g, ' ')
|
|
39
|
+
.replace(/^\s*>.*$/gm, ' ')
|
|
40
|
+
.replace(/^\s*\|.*\|\s*$/gm, ' '); // table rows are data cells, not sentences
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function sentences(message) {
|
|
44
|
+
return strippedProse(message)
|
|
45
|
+
.split(/(?<=[.!?])\s+|\n+/)
|
|
46
|
+
.map((s) => s.trim())
|
|
47
|
+
.filter((s) => s && s.length <= MAX_SENTENCE);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// A sentence that is itself a disclosure, a negation, a hedge or a condition is not a claim.
|
|
51
|
+
const NOT_A_CLAIM = /\b(?:not|never|no|nothing|none|cannot|can't|won't|isn't|aren't|wasn't|weren't|hasn't|haven't|didn't|doesn't|don't|without|unverified|untested|unproven|unknown|yet|if|unless|until|once|when|whether|should|would|could|might|may|maybe|probably|likely|expected|supposed|hopefully|assuming|pending|remaining|remains|todo|tbd|before|after|previously|earlier|already|yesterday|ago|last\s+(?:week|night|time|session|release)|claimed|claims|said|says|reported|told|neither|nor|as\s+soon\s+as|tell\s+me|I(?:'ll|’ll)|we(?:'ll|’ll)|will\s+(?!now\b)|(?:was|were)\s+working|damage\s+is\s+done)\b|\?/i;
|
|
52
|
+
const CLAIM_PATTERNS = [
|
|
53
|
+
// "X is fixed", "the gate is now live", "tests are passing", "it's done"
|
|
54
|
+
/\b(?:is|are|was|were|has\s+been|have\s+been|(?:it|that|this|what|everything|there)(?:'s|’s)|now)\s+(?:all\s+|fully\s+|now\s+|finally\s+|successfully\s+|actually\s+)*(?:fixed|done|complete|completed|resolved|shipped|deployed|live(?!\s+(?:in|inside|under|with|alongside))|implemented|landed|published|merged|released|verified|green|passing|operational|in\s+place|wired(?:\s+up)?|working(?!\s+(?:on|with|through|in|as|tree|copy|dir)))\b/i,
|
|
55
|
+
// "I fixed", "I've shipped", "we deployed"
|
|
56
|
+
/\b(?:I|we)(?:'ve|’ve|\s+have)?\s+(?:now\s+|just\s+|successfully\s+|also\s+|fully\s+)*(?:fixed|shipped|deployed|resolved|completed|implemented|landed|published|released|verified|finished|wired\s+up)\b/i,
|
|
57
|
+
// "Done." "Fixed:" "Shipped —" "✅ Deployed" at the start of a line
|
|
58
|
+
/^(?:[-*•]\s*|#+\s*|\*\*|✅\s*)*(?:done|fixed|shipped|deployed|resolved|complete|completed|implemented|all\s+done|all\s+set|all\s+green)\b\s*(?:[.!:—–-]|\*\*|$)/i,
|
|
59
|
+
// "now works", "it will now block …", "works end to end"
|
|
60
|
+
/\b(?:now\s+works|works\s+now|works\s+end[- ]to[- ]end|will\s+now\s+(?:work|fire|block|catch|run|stay|refuse|correct|speak|pass|succeed))\b/i,
|
|
61
|
+
];
|
|
62
|
+
|
|
63
|
+
/** Positive completion assertions, in message order. Deterministic; no model call. */
|
|
64
|
+
export function extractCompletionClaims(message) {
|
|
65
|
+
const out = [];
|
|
66
|
+
for (const sentence of sentences(message)) {
|
|
67
|
+
if (NOT_A_CLAIM.test(sentence)) continue;
|
|
68
|
+
if (CLAIM_PATTERNS.some((re) => re.test(sentence))) out.push({ class: 'completion', text: sentence });
|
|
69
|
+
}
|
|
70
|
+
return out;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const NAMED_CHECK = /(?:^|\n)\s*(?:[-*•]\s*|#+\s*|\*\*)*(?:verified|verification|evidence|proof|tested|checks?(?:\s+run)?|test\s+results?|measured)\b[^\n]{0,4}?(?:\*\*)?\s*[:—–]/i;
|
|
74
|
+
const DISCLOSES_GAPS = /\b(?:not\s+(?:yet\s+)?(?:verified|tested|checked|exercised|measured|proven|covered|run)|unverified|untested|unproven|what\s+i\s+did\s+not\s+(?:test|verify|check)|did\s+not\s+(?:test|verify|check|run|exercise)|didn't\s+(?:test|verify|check|run|exercise)|remains?\s+unknown|UNKNOWN)\b/i;
|
|
75
|
+
|
|
76
|
+
// ── transcript-side parsing (Claude Code JSONL) ─────────────────────────────────────────────────
|
|
77
|
+
const textOf = (content) => (typeof content === 'string' ? content
|
|
78
|
+
: Array.isArray(content) ? content.filter((c) => c?.type === 'text' && typeof c.text === 'string').map((c) => c.text).join('\n') : '');
|
|
79
|
+
|
|
80
|
+
const TMP_PATH = /^(?:\/tmp\/|\/private\/tmp\/|\/var\/folders\/|\$\{?TMPDIR|\/dev\/null)/;
|
|
81
|
+
const MUTATING_SEGMENT = [
|
|
82
|
+
/\bgit\s+(?:commit|push|merge|rebase|reset|revert|cherry-pick|tag|am|apply|restore|stash\s+(?:pop|apply)|mv|rm|switch|checkout)\b/,
|
|
83
|
+
/\bgh\s+(?:release\s+(?:create|edit|upload|delete)|pr\s+(?:create|merge|close|edit|ready)|workflow\s+run|run\s+rerun|issue\s+(?:create|close|edit|comment)|repo\s+edit|secret\s+set|variable\s+set|api\b[^|;&]*-X\s*(?:POST|PUT|PATCH|DELETE))\b/,
|
|
84
|
+
/\b(?:npm|pnpm|yarn)\s+(?:publish|version|install|i|add|uninstall|remove|ci|link|unpublish|dist-tag)\b/,
|
|
85
|
+
/\bnpm\s+run\s+[\w:.-]*(?:write|set|fix|deploy|publish|release|sync|install|update|apply|seed|stamp|migrate|format)[\w:.-]*/,
|
|
86
|
+
/\b(?:launchctl\s+(?:load|unload|bootstrap|bootout|enable|disable|kickstart)|crontab\s+(?!-l\b)\S|schtasks\s+\/(?:create|delete|change))/i,
|
|
87
|
+
/\b(?:vercel\s+(?:deploy|--prod|promote|alias)|netlify\s+deploy|docker\s+(?:push|build)|kubectl\s+(?:apply|delete|rollout)|terraform\s+apply|wrangler\s+(?:deploy|publish))\b/,
|
|
88
|
+
/\s--(?:write|apply|fix|install|enable-nightly|publish|commit|seed|set|update)\b/,
|
|
89
|
+
/\b(?:sed|perl)\s+(?:-[a-zA-Z]*i|-i)\b/,
|
|
90
|
+
];
|
|
91
|
+
const FILE_MUTATORS = /^(?:sudo\s+)?(?:rm|mv|cp|mkdir|touch|chmod|chown|ln|truncate|rsync|unzip|tee|install)\b/;
|
|
92
|
+
const TRIVIAL = /^(?:sudo\s+)?(?:ls|cat|head|tail|less|more|grep|egrep|rg|find|fd|wc|echo|printf|pwd|cd|which|type|command|sort|uniq|cut|tr|stat|file|date|sleep|true|false|export|set|source|\.|test|\[|jq|awk|sed|basename|dirname|realpath|env|git\s+(?:status|log|diff|show|branch|rev-parse|remote|config|worktree\s+list|stash\s+list))\b/;
|
|
93
|
+
|
|
94
|
+
function segments(command) {
|
|
95
|
+
return String(command || '').split(/&&|\|\||;|\n|\|/).map((s) => s.trim()).filter(Boolean);
|
|
96
|
+
}
|
|
97
|
+
function segmentMutates(segment) {
|
|
98
|
+
if (MUTATING_SEGMENT.some((re) => re.test(segment))) return true;
|
|
99
|
+
if (FILE_MUTATORS.test(segment)) {
|
|
100
|
+
const args = segment.split(/\s+/).slice(1).filter((a) => !a.startsWith('-'));
|
|
101
|
+
return !args.length || !args.every((a) => TMP_PATH.test(a.replace(/^['"]|['"]$/g, '')));
|
|
102
|
+
}
|
|
103
|
+
return /(?:^|[^0-9&>])>{1,2}\s*(?!&|\/dev\/null)(['"]?)([^\s'"]+)/.test(segment)
|
|
104
|
+
&& !TMP_PATH.test((/>{1,2}\s*['"]?([^\s'"]+)/.exec(segment) || [])[1] || '');
|
|
105
|
+
}
|
|
106
|
+
const segmentChecks = (segment) => !TRIVIAL.test(segment) && !segmentMutates(segment);
|
|
107
|
+
function checkName(segment) {
|
|
108
|
+
const words = segment.replace(/^(?:[A-Z_][A-Z0-9_]*=\S+\s+)+/, '').split(/\s+/);
|
|
109
|
+
const run = words.findIndex((w) => w === 'run');
|
|
110
|
+
if (/^(?:npm|pnpm|yarn)$/.test(words[0]) && run >= 0 && words[run + 1]) return words[run + 1];
|
|
111
|
+
if (words[0] === 'npx' && words[1]) return words[1];
|
|
112
|
+
if (words[0] === 'node' && words[1]) return words[1].split('/').pop().replace(/\.[cm]?js$/, '');
|
|
113
|
+
return words.slice(0, 2).join(' ');
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
const READ_ONLY_AGENTS = /^(?:explore|plan|claude-code-guide|statusline-setup)$/i;
|
|
117
|
+
const MCP_MUTATING = /__(?:create|update|delete|remove|publish|deploy|push|write|send|set|merge|upload|patch|put|post|add|rename|move|approve|promote|rollback|cancel|buy|store|edit|import|reset|stop|terminate|spawn|execute)[a-z_-]*$/i;
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* The current turn of a Claude JSONL transcript: every main-thread record after the last genuine
|
|
121
|
+
* user message, plus that message's text. Shared by claudeTurnEvents() and grounding-turn-evidence.mjs
|
|
122
|
+
* so both gates agree on where a turn starts.
|
|
123
|
+
*/
|
|
124
|
+
export function currentTurnRecords(lines) {
|
|
125
|
+
const recs = [];
|
|
126
|
+
for (const l of lines || []) { try { const o = JSON.parse(l); if (!o?.isSidechain) recs.push(o); } catch { /* torn line */ } }
|
|
127
|
+
let start = -1;
|
|
128
|
+
recs.forEach((o, i) => {
|
|
129
|
+
const c = o?.message?.content;
|
|
130
|
+
const isToolResult = Array.isArray(c) && c.some((x) => x?.type === 'tool_result');
|
|
131
|
+
if (o?.type === 'user' && (o.message?.role || 'user') === 'user' && !isToolResult && !o.isMeta && textOf(c).trim()) start = i;
|
|
132
|
+
});
|
|
133
|
+
return { boundaryFound: start >= 0, prompt: start >= 0 ? textOf(recs[start].message?.content) : '', recs: recs.slice(start + 1) };
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/** Ordered events for the current turn of a Claude JSONL transcript. */
|
|
137
|
+
export function claudeTurnEvents(lines) {
|
|
138
|
+
const { boundaryFound, recs: turnRecs } = currentTurnRecords(lines);
|
|
139
|
+
const results = new Map();
|
|
140
|
+
for (const o of turnRecs) {
|
|
141
|
+
const c = o?.message?.content;
|
|
142
|
+
if (!Array.isArray(c)) continue;
|
|
143
|
+
for (const r of c) {
|
|
144
|
+
if (r?.type === 'tool_result' && r.tool_use_id) {
|
|
145
|
+
results.set(r.tool_use_id, { error: r.is_error === true || o.toolUseResult?.interrupted === true,
|
|
146
|
+
present: textOf(r.content).trim().length > 0 || String(o.toolUseResult?.stdout || '').trim().length > 0 });
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
const events = [];
|
|
151
|
+
for (const o of turnRecs) {
|
|
152
|
+
const c = o?.message?.content;
|
|
153
|
+
if (o?.type !== 'assistant' || !Array.isArray(c)) continue;
|
|
154
|
+
for (const u of c) {
|
|
155
|
+
if (u?.type !== 'tool_use') continue;
|
|
156
|
+
const input = u.input || {};
|
|
157
|
+
const result = results.get(u.id) || { present: false, error: false };
|
|
158
|
+
if (['Edit', 'Write', 'MultiEdit', 'NotebookEdit'].includes(u.name)) {
|
|
159
|
+
events.push({ kind: 'change', what: `${u.name} ${input.file_path || input.notebook_path || ''}`.trim() });
|
|
160
|
+
} else if (u.name === 'Bash') {
|
|
161
|
+
for (const seg of segments(input.command)) {
|
|
162
|
+
if (segmentMutates(seg)) events.push({ kind: 'change', what: seg.slice(0, 120) });
|
|
163
|
+
else if (segmentChecks(seg)) events.push({ kind: 'check', what: seg.slice(0, 120), name: checkName(seg), ...result });
|
|
164
|
+
}
|
|
165
|
+
} else if (u.name === 'Agent' || u.name === 'Task') {
|
|
166
|
+
if (!READ_ONLY_AGENTS.test(String(input.subagent_type || ''))) events.push({ kind: 'change', what: `${u.name} ${input.subagent_type || 'agent'}` });
|
|
167
|
+
} else if (String(u.name || '').startsWith('mcp__')) {
|
|
168
|
+
events.push(MCP_MUTATING.test(u.name) ? { kind: 'change', what: u.name }
|
|
169
|
+
: { kind: 'check', what: u.name, name: u.name.split('__').pop(), ...result });
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
return { boundaryFound, events };
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/** Verification that ran AFTER the last state change this turn, with a present, non-error result. */
|
|
177
|
+
export function postChangeVerification(turn) {
|
|
178
|
+
const events = turn?.events || [];
|
|
179
|
+
let lastChange = -1;
|
|
180
|
+
events.forEach((e, i) => { if (e.kind === 'change') lastChange = i; });
|
|
181
|
+
const checks = events.slice(lastChange + 1).filter((e) => e.kind === 'check' && e.present && !e.error);
|
|
182
|
+
return { lastChange: lastChange >= 0 ? events[lastChange].what : null, checks,
|
|
183
|
+
staleChecks: events.slice(0, Math.max(0, lastChange)).filter((e) => e.kind === 'check').length };
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
export function readClaudeTurn(transcriptPath, { read = readSettledTranscript } = {}) {
|
|
187
|
+
if (typeof transcriptPath !== 'string' || !/\.jsonl$/i.test(transcriptPath)) return null;
|
|
188
|
+
try { return claudeTurnEvents(read(transcriptPath, { maxMs: 0 })); } catch { return null; }
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* The completion audit. `turn` is claudeTurnEvents() output, or null when the host transcript is
|
|
193
|
+
* unavailable/unparsed (then the transcript half is UNKNOWN and says so).
|
|
194
|
+
*/
|
|
195
|
+
export function auditCompletionClaims(message, { turn = null, host = 'claude' } = {}) {
|
|
196
|
+
const claims = extractCompletionClaims(message);
|
|
197
|
+
if (!claims.length) return { verdict: 'NONE', claims };
|
|
198
|
+
const text = String(message || '');
|
|
199
|
+
const verification = turn ? postChangeVerification(turn) : null;
|
|
200
|
+
const names = verification?.checks.map((c) => String(c.name || '').toLowerCase()).filter((n) => n.length >= 3) || [];
|
|
201
|
+
const namesCheck = NAMED_CHECK.test(text) || names.some((n) => text.toLowerCase().includes(n));
|
|
202
|
+
const disclosesGaps = DISCLOSES_GAPS.test(strippedProse(text).replace(/```[\s\S]*?```/g, ''))
|
|
203
|
+
|| DISCLOSES_GAPS.test(text);
|
|
204
|
+
const problems = [];
|
|
205
|
+
if (!turn) problems.push(`this host's (${host}) transcript is unavailable or not parsed, so whether a check ran after the change is UNKNOWN`);
|
|
206
|
+
else if (!verification.checks.length) {
|
|
207
|
+
problems.push(verification.lastChange
|
|
208
|
+
? `no end-to-end check ran this turn after the last change (${verification.lastChange})`
|
|
209
|
+
: 'no end-to-end check ran this turn');
|
|
210
|
+
}
|
|
211
|
+
if (!namesCheck) problems.push('the answer does not name the check that proves it');
|
|
212
|
+
if (!disclosesGaps) problems.push('the answer does not disclose what is NOT verified');
|
|
213
|
+
const transcriptOk = turn ? verification.checks.length > 0 : false;
|
|
214
|
+
const verdict = transcriptOk && namesCheck && disclosesGaps ? 'PASS' : !turn && namesCheck && disclosesGaps ? 'UNKNOWN' : 'FAIL';
|
|
215
|
+
return { verdict, claims, verification, namesCheck, disclosesGaps, problems };
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
// ── promises (first-person commitments) ─────────────────────────────────────────────────────────
|
|
219
|
+
const PROMISE_LEAD = /\b(?:I(?:'ll|’ll|\s+will)|I(?:'m|’m|\s+am)\s+going\s+to|next,?\s+I(?:'ll|’ll|\s+will)|I\s+commit\s+to)\s+(?:now\s+|next\s+|then\s+|also\s+|immediately\s+|first\s+)*([a-z][^.!?\n]{6,158})/i;
|
|
220
|
+
const PROMISE_REJECT = /\?|:\*\*|\b(?:if|unless|once|when|whenever|as\s+soon\s+as|after|the\s+moment|until|whether|shortly|regularly|periodically|every\s+\d+|check\s+(?:back|again|on\s+it)|keep\s+(?:watching|an\s+eye)|maybe|might|could|would|probably|perhaps|possibly|option|alternatively|either|or\s+I\s+(?:can|could)|want\s+me|should\s+I|shall\s+I|let\s+you\s+know|wait|await|stand\s+by|hold\s+off|leave\s+(?:it|that|this)\s+to\s+you|happy\s+to|glad\s+to|be\s+honest|note\s+that|keep\s+(?:that|this)\s+in\s+mind|try\s+to|need\s+your|your\s+(?:approval|go-ahead|call|decision|confirmation))\b/i;
|
|
221
|
+
const TRIVIAL_PROMISE = /^(?:not|never|be|say|mention|note|admit|point|flag|stop|stay|leave|keep|summari[sz]e|explain|answer|respond|reply|report|remember|give|tell|send|paste|bring|hand|show|walk|notify|post|update\s+you|come\s+back|get\s+back|handle|only|stand|work)\b/i;
|
|
222
|
+
|
|
223
|
+
export const normalizePromise = (value) => String(value || '').toLowerCase().replace(/[`*_"“”'’]/g, '')
|
|
224
|
+
.replace(/[^a-z0-9./:+-]+/g, ' ').trim().slice(0, 160);
|
|
225
|
+
|
|
226
|
+
/** Non-hedged, present-session, first-person commitments in the FINAL answer. */
|
|
227
|
+
export function extractCommitments(message) {
|
|
228
|
+
const out = [];
|
|
229
|
+
const seen = new Set();
|
|
230
|
+
const lines = strippedProse(message).split('\n');
|
|
231
|
+
let inPlan = false;
|
|
232
|
+
for (const raw of lines) {
|
|
233
|
+
const line = raw.trim();
|
|
234
|
+
if (/^(?:[-*•]\s*|#+\s*|\*\*)*my\s+plan\b.{0,20}[:—–]\s*$/i.test(line)) { inPlan = true; continue; }
|
|
235
|
+
const item = inPlan ? /^(?:\d+[.)]|[-*•])\s+(.{6,160})$/.exec(line) : null;
|
|
236
|
+
if (inPlan && !item && line) inPlan = false;
|
|
237
|
+
const candidates = item ? [item[1]] : sentences(line).map((s) => PROMISE_LEAD.exec(s)?.[1]).filter(Boolean);
|
|
238
|
+
const whole = item ? item[1] : line;
|
|
239
|
+
for (const action of candidates) {
|
|
240
|
+
if (PROMISE_REJECT.test(whole) || TRIVIAL_PROMISE.test(action.trim())) continue;
|
|
241
|
+
const text = action.replace(/\s+/g, ' ').trim().replace(/[,;:]$/, '');
|
|
242
|
+
const key = normalizePromise(text);
|
|
243
|
+
if (key.length < 6 || seen.has(key)) continue;
|
|
244
|
+
seen.add(key);
|
|
245
|
+
out.push({ text, key });
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
return out.slice(0, PROMISE_CAP_PER_TURN);
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
const STOP = new Set(['the', 'a', 'an', 'and', 'or', 'to', 'of', 'in', 'on', 'for', 'with', 'it', 'is', 'are',
|
|
252
|
+
'this', 'that', 'now', 'next', 'then', 'will', 'be', 'i', 'we', 'all', 'so', 'as', 'at', 'by', 'from']);
|
|
253
|
+
const tokens = (value) => normalizePromise(value).split(' ').filter((t) => t.length > 2 && !STOP.has(t));
|
|
254
|
+
|
|
255
|
+
/** A passing completion claim closes a promise only when it names the same work. */
|
|
256
|
+
export function claimClosesPromise(claimText, promiseText) {
|
|
257
|
+
const want = tokens(promiseText);
|
|
258
|
+
if (!want.length) return false;
|
|
259
|
+
const have = new Set(tokens(claimText));
|
|
260
|
+
const hit = want.filter((t) => have.has(t)).length;
|
|
261
|
+
return hit >= 2 && hit / want.length >= 0.4;
|
|
262
|
+
}
|