claude-code-session-manager 0.80.0 → 0.81.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/dist/assets/{AgentLibrary-zS3jw_1e.js → AgentLibrary-psZYVM2w.js} +1 -1
  2. package/dist/assets/{DataModel-Cy_vxTpi.js → DataModel-BMied5pg.js} +1 -1
  3. package/dist/assets/{History-C6JRuqfT.js → History-BFC0oaKc.js} +1 -1
  4. package/dist/assets/{Hooks-BafPy9mB.js → Hooks-CTLfO9G8.js} +1 -1
  5. package/dist/assets/{HostBilko-BZwhQOFt.js → HostBilko-D4I0Cpwn.js} +1 -1
  6. package/dist/assets/{Library-C8JDDliz.js → Library-BTzS8KsS.js} +1 -1
  7. package/dist/assets/{ListDetail-CqiOdwLc.js → ListDetail-CuRmT008.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-CyLyP67L.js → MarkdownEditor-BgQGtvWo.js} +1 -1
  9. package/dist/assets/{McpServers-BzMv-_84.js → McpServers-DiVQUK57.js} +1 -1
  10. package/dist/assets/{Memory-DSBYQdJR.js → Memory-DSu15JmL.js} +1 -1
  11. package/dist/assets/{Panel-CLUhkNNA.js → Panel-CD5wxGSR.js} +1 -1
  12. package/dist/assets/{Permissions-BfC2-HN4.js → Permissions-C_tAE6yJ.js} +1 -1
  13. package/dist/assets/{Plugins-BKi40jT5.js → Plugins-kwI7W-eK.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-BzFw4KhD.js → ProvenanceBadge-CQceOgsH.js} +1 -1
  15. package/dist/assets/{SaveBar-avk2p9jv.js → SaveBar-BDk5e3Pp.js} +1 -1
  16. package/dist/assets/{Scheduler-Bf_6MdJo.js → Scheduler-DRnvEYzv.js} +1 -1
  17. package/dist/assets/{ScopeSwitcher-C-RwYUVZ.js → ScopeSwitcher-CeifaOlq.js} +1 -1
  18. package/dist/assets/{Settings-Djd8OoBA.js → Settings-BWQ1Utop.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-DuogY6s7.js → SkillReferenceGraph-CW6e1SW1.js} +1 -1
  20. package/dist/assets/{Skills-D_qAqxZ_.js → Skills-BLwFB0E4.js} +1 -1
  21. package/dist/assets/{SystemPrompt-DbHFLQV3.js → SystemPrompt-DC4ZTArJ.js} +1 -1
  22. package/dist/assets/{TagLibrary-C2y91BT0.js → TagLibrary-pMeGfjUb.js} +1 -1
  23. package/dist/assets/{TiptapBody-D9iz4xQx.js → TiptapBody-BHFid2pZ.js} +1 -1
  24. package/dist/assets/{Toggle-BGnFL2E5.js → Toggle-D9eYoZh4.js} +1 -1
  25. package/dist/assets/{index-_2ARyFDj.js → index-CuyM9vAP.js} +4 -4
  26. package/dist/assets/{settingsSchema-JK15eJU8.js → settingsSchema-DrxC67uZ.js} +1 -1
  27. package/dist/index.html +1 -1
  28. package/package.json +1 -1
  29. package/plugins/session-manager-dev/skills/develop/standards.md +1 -0
  30. package/scripts/project-pages-logic/dist/logic.cjs +12 -12
  31. package/scripts/render-project-pages/dist/renderer.cjs +22 -22
  32. package/src/main/__tests__/rcaReport.test.cjs +24 -0
  33. package/src/main/__tests__/runVerify-blocked-by-foreign-wip.test.cjs +58 -0
  34. package/src/main/__tests__/runVerify-policy-denial.test.cjs +89 -0
  35. package/src/main/__tests__/scheduler-already-satisfied-on-main.test.cjs +105 -0
  36. package/src/main/__tests__/scheduler-blocked-by-foreign-wip.test.cjs +107 -0
  37. package/src/main/__tests__/scheduler-finalize-dispatch-guards.test.cjs +229 -0
  38. package/src/main/__tests__/scheduler-looks-done.test.cjs +141 -1
  39. package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +81 -0
  40. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +205 -2
  41. package/src/main/lib/__tests__/branchSweep.test.cjs +164 -0
  42. package/src/main/lib/__tests__/fixtures/204-mercury-steam-horse.log.txt +13 -0
  43. package/src/main/lib/__tests__/landedSinceRun.test.cjs +61 -1
  44. package/src/main/lib/__tests__/rateLimitWindow.test.cjs +88 -0
  45. package/src/main/lib/__tests__/reaperHelpers.test.cjs +120 -1
  46. package/src/main/lib/branchSweep.cjs +127 -0
  47. package/src/main/lib/gitWorktree.cjs +20 -0
  48. package/src/main/lib/landedSinceRun.cjs +41 -1
  49. package/src/main/lib/rateLimitWindow.cjs +62 -0
  50. package/src/main/lib/rcaReport.cjs +18 -3
  51. package/src/main/lib/reaperHelpers.cjs +169 -3
  52. package/src/main/lib/scheduleJobTransitions.cjs +16 -3
  53. package/src/main/runVerify.cjs +71 -3
  54. package/src/main/scheduler.cjs +741 -31
@@ -52,6 +52,7 @@ const VERDICT_LABELS = {
52
52
  pass_no_commit_already_shipped: 'PASS with no commit — deliverables already shipped',
53
53
  pass_no_commit_prior_run_verified: 'PASS with no commit — prior run of this slug already landed the work',
54
54
  silent_no_op: 'no commit, clean tree — no evidence of work',
55
+ blocked_by_foreign_wip_streak: 'blocked by a sibling job\'s foreign WIP 3x in a row',
55
56
  };
56
57
 
57
58
  function humanVerdict(verdict) {
@@ -69,14 +70,18 @@ const FAILURE_CLASSES = {
69
70
  UNCOMMITTED: 'uncommitted-changes',
70
71
  TRANSCRIPT_ERRORS: 'transcript-errors',
71
72
  ABANDONED_BACKGROUND_TASK: 'abandoned-background-task',
73
+ BLOCKED_BY_FOREIGN_WIP: 'blocked-by-foreign-wip',
72
74
  UNKNOWN: 'unknown',
73
75
  };
74
76
 
75
77
  // ─── Recovery actions — one per failure class, machine-readable ────────────
76
78
  // Closed set the scheduler routes on: 'archive' (stale re-run, do not
77
- // re-queue), 'resume-and-commit' (PRD 1111's --resume dispatch owns this),
78
- // 'verify-and-close' (the work likely landed, just missing its sentinel),
79
- // 'investigate' (the only class that still buys a fix-plan investigation).
79
+ // re-queue — also the closest fit for BLOCKED_BY_FOREIGN_WIP: there is
80
+ // nothing to investigate, and re-queuing is already owned by
81
+ // reconcile()'s requeueForeignWipBlockedJobs, not this action), 'resume-and-commit'
82
+ // (PRD 1111's --resume dispatch owns this), 'verify-and-close' (the work
83
+ // likely landed, just missing its sentinel), 'investigate' (the only class
84
+ // that still buys a fix-plan investigation).
80
85
  const RECOVERY_ACTIONS = {
81
86
  [FAILURE_CLASSES.ALREADY_SHIPPED]: 'archive',
82
87
  [FAILURE_CLASSES.SELF_QUEUE]: 'investigate',
@@ -96,6 +101,7 @@ const RECOVERY_ACTIONS = {
96
101
  // carries the "check for salvaged work, commit before re-implementing"
97
102
  // guidance into that cold-read investigation.
98
103
  [FAILURE_CLASSES.ABANDONED_BACKGROUND_TASK]: 'investigate',
104
+ [FAILURE_CLASSES.BLOCKED_BY_FOREIGN_WIP]: 'archive',
99
105
  [FAILURE_CLASSES.UNKNOWN]: 'investigate',
100
106
  };
101
107
 
@@ -121,6 +127,8 @@ const PREVENTION_HINTS = {
121
127
  'Recover or annotate every error within ~10 lines (e.g. `# expected/handled: <why>`) instead of leaving a bare Traceback near the end of the transcript.',
122
128
  [FAILURE_CLASSES.ABANDONED_BACKGROUND_TASK]:
123
129
  'A headless run cannot receive a background-task completion notification — never let a long Bash command auto-background past its foreground timeout and then wait for it. Run long commands with an explicit bound (`timeout <N> <cmd>`) so they finish in the foreground, or poll their output file directly instead of waiting on Monitor/notification.',
130
+ [FAILURE_CLASSES.BLOCKED_BY_FOREIGN_WIP]:
131
+ 'This job correctly reported `SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP` against a sibling job\'s in-flight, uncommitted files three times in a row — auto-requeue is exhausted (reconcile() only clears this once the blocked paths go clean). This is not a defect in this PRD\'s own work: check on the sibling job/human that owns the still-dirty path(s) named on the job row, and either wait for them to commit or manually reset this job to `pending` once the tree is clear. Do not author a fix-plan PRD against it.',
124
132
  [FAILURE_CLASSES.UNKNOWN]:
125
133
  'Re-run the acceptance criteria gate locally against the failure log to pin down the specific break before re-queuing.',
126
134
  };
@@ -159,6 +167,13 @@ const POST_AC_OVERRUN_MIN_TAIL_FRACTION = 0.3;
159
167
  function classifyFailure({ verdict, logTail }) {
160
168
  const lines = (logTail || '').split('\n');
161
169
 
170
+ // Checked before every other rule: an unambiguous, materially-checked
171
+ // verdict (the scheduler itself validated the executor's claimed paths
172
+ // against the job's disclosed foreign-WIP manifest before ever landing
173
+ // this verdict — see validateForeignWipBlockClaim in scheduler.cjs) needs
174
+ // no log-tail heuristics to classify.
175
+ if (verdict === 'blocked_by_foreign_wip_streak') return FAILURE_CLASSES.BLOCKED_BY_FOREIGN_WIP;
176
+
162
177
  // Checked first, before SELF_QUEUE/STUCK_LOOP: a correct executor that finds
163
178
  // its acceptance criteria already satisfied by a prior commit makes no
164
179
  // change and truthfully prints a PASS sentinel, so the run lands
@@ -8,6 +8,7 @@
8
8
  */
9
9
 
10
10
  const fs = require('node:fs');
11
+ const path = require('node:path');
11
12
  const { readTail } = require('./fileTail.cjs');
12
13
  const { detectRateLimitInLog } = require('./rateLimitDetect.cjs');
13
14
 
@@ -34,6 +35,90 @@ function claudePidAlive(pid) {
34
35
  }
35
36
  }
36
37
 
38
+ /**
39
+ * findLiveProcessForJob(job, { worktreeDir }) → pid | null
40
+ *
41
+ * Positive liveness scan for a 'running' row whose `runtime.pid` is missing —
42
+ * the case a missing pid must NOT be read as "the process is gone" (2026-09-06
43
+ * incident: 234-uranus-eight-tails-ox marked failed/never_ran while PID
44
+ * 2174739 was a live `claude -p` still writing to that job's own worktree).
45
+ *
46
+ * Linux-`/proc` only. Scans every numeric `/proc/<pid>` entry and matches
47
+ * either: `/proc/<pid>/cwd` resolves to `worktreeDir` (or a path nested under
48
+ * it), or `/proc/<pid>/cmdline` contains both `claude` and the job's slug
49
+ * (fallback for a worktree-disabled/in-place run, where there is no dedicated
50
+ * worktreeDir to match against). Returns the first matching pid, or null if
51
+ * none is found.
52
+ *
53
+ * Safe fallback by construction: on any platform without `/proc` (macOS,
54
+ * Windows) `fs.readdirSync('/proc')` throws and this returns null immediately
55
+ * — i.e. exactly today's behaviour (fail toward terminalizing), never a hang
56
+ * or a thrown error propagating to the caller.
57
+ */
58
+ function findLiveProcessForJob(job, { worktreeDir } = {}) {
59
+ const slug = job?.slug;
60
+ let entries;
61
+ try {
62
+ entries = fs.readdirSync('/proc');
63
+ } catch {
64
+ return null;
65
+ }
66
+ for (const name of entries) {
67
+ if (!/^\d+$/.test(name)) continue;
68
+ const pid = Number(name);
69
+ if (!pid || pid <= 1) continue;
70
+ if (worktreeDir) {
71
+ try {
72
+ const cwdLink = fs.readlinkSync(`/proc/${pid}/cwd`);
73
+ if (cwdLink === worktreeDir || cwdLink.startsWith(worktreeDir + path.sep)) return pid;
74
+ } catch {
75
+ // Process exited mid-scan, or permission denied — try the argv
76
+ // fallback below before giving up on this pid.
77
+ }
78
+ }
79
+ if (slug) {
80
+ try {
81
+ const cmd = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8').replace(/\0/g, ' ');
82
+ if (/\bclaude\b/.test(cmd) && cmd.includes(slug)) return pid;
83
+ } catch {
84
+ // Same as above — process gone or unreadable, keep scanning.
85
+ }
86
+ }
87
+ }
88
+ return null;
89
+ }
90
+
91
+ /**
92
+ * True when `logPath` exists and has non-zero content — the literal "did the
93
+ * run dir produce any log output" check that gates whether a pidless reap may
94
+ * assert `gateOutcome: 'never_ran'`. A run whose log has real bytes in it DID
95
+ * run, regardless of whether classifyRunOutcome found a clean result event in
96
+ * it — asserting never_ran in that case would be a false claim (see
97
+ * resolvePidlessGateOutcome below).
98
+ */
99
+ function logHasOutput(logPath) {
100
+ if (!logPath) return false;
101
+ try {
102
+ return fs.statSync(logPath).size > 0;
103
+ } catch {
104
+ return false;
105
+ }
106
+ }
107
+
108
+ /**
109
+ * Gate-outcome for a pidless reap, once no live process was found for it.
110
+ * `never_ran` is asserted ONLY when the run dir produced no log output at
111
+ * all — mapOutcomeToGateOutcome's own 'no_result' → 'never_ran' mapping is
112
+ * otherwise too broad here: a log with real content but no clean result event
113
+ * (e.g. killed mid-turn) proves the job DID run, so that case is reported as
114
+ * 'failed' instead of the false 'never_ran'.
115
+ */
116
+ function resolvePidlessGateOutcome(outcome, hasOutput) {
117
+ if (!hasOutput) return 'never_ran';
118
+ const mapped = mapOutcomeToGateOutcome(outcome);
119
+ return mapped === 'never_ran' ? 'failed' : mapped;
120
+ }
121
+
37
122
  /**
38
123
  * Classify the terminal outcome of a completed run by reading the last 64 KB
39
124
  * of its log file and scanning for the LAST `{"type":"result"}` JSONL event.
@@ -129,10 +214,21 @@ const ORPHAN_REQUEUE_CAP = 5;
129
214
  * A pidless row whose `startedAt` is missing or unparseable is neither
130
215
  * reaped nor skipped silently — age can't be proven, so it is surfaced in
131
216
  * `warnings` instead (the caller logs it) and left alone.
217
+ *
218
+ * `findLiveProcess` (optional, `(job) → pid | null`) is consulted for a
219
+ * pidless row ONLY once its age clears `grace` — i.e. right before it would
220
+ * otherwise be terminalized. A pid it finds means the process is actually
221
+ * alive despite the missing runtime.pid record: the row is diverted into
222
+ * `recovered` (never `reapable`) so the caller can re-stamp the pid and leave
223
+ * the row `running`, instead of terminalizing a job that is still doing real
224
+ * work (2026-09-06 incident — see findLiveProcessForJob's header). Omitting
225
+ * `findLiveProcess` (existing callers/tests) preserves prior behaviour
226
+ * exactly: every pidless row past grace reaps, none are ever recovered.
132
227
  */
133
- function selectReapableJobs(jobs, now, { pidAlive, grace } = {}) {
228
+ function selectReapableJobs(jobs, now, { pidAlive, grace, findLiveProcess } = {}) {
134
229
  const reapable = [];
135
230
  const warnings = [];
231
+ const recovered = [];
136
232
  for (const j of jobs ?? []) {
137
233
  if (j.status !== 'running') continue;
138
234
  const pid = j.runtime?.pid;
@@ -148,6 +244,11 @@ function selectReapableJobs(jobs, now, { pidAlive, grace } = {}) {
148
244
  }
149
245
  const ageMs = now - startedAt;
150
246
  if (ageMs < grace) continue; // spawn may still be mid-flight
247
+ const livePid = typeof findLiveProcess === 'function' ? findLiveProcess(j) : null;
248
+ if (livePid) {
249
+ recovered.push({ slug: j.slug, pid: livePid });
250
+ continue;
251
+ }
151
252
  reapable.push({
152
253
  slug: j.slug,
153
254
  pid: null,
@@ -155,7 +256,72 @@ function selectReapableJobs(jobs, now, { pidAlive, grace } = {}) {
155
256
  reason: `reaped: no runtime.pid recorded after ${Math.round(grace / 60_000)}m — spawn never completed`,
156
257
  });
157
258
  }
158
- return { reapable, warnings };
259
+ return { reapable, warnings, recovered };
260
+ }
261
+
262
+ /**
263
+ * isAlreadySatisfiedOnMain(commits) → { sha, verdict, reason } | null
264
+ *
265
+ * Pure decision layer for the finish-protocol commit-guard's second,
266
+ * independently-evidenced route to 'completed' (PRD 1136). The commit-guard
267
+ * in scheduler.cjs parks a clean exit / no-commit / clean-tree run as
268
+ * `needs_review` ("finish protocol incomplete") because that shape is
269
+ * normally the strongest signal that nothing happened — but it is also
270
+ * exactly the shape a run produces when its PRD's work was ALREADY merged to
271
+ * main before the run dispatched (2026-09-06: 1133-reaper-must-verify-
272
+ * integration-before-completed and 1134-land-stranded-sm-job-branches, both
273
+ * false-negatived this way and then auto-fix-minted a redundant `-fix-`
274
+ * child against a codebase where the change was already present).
275
+ *
276
+ * Takes the already-git-queried list of commit SHAs (newest first, caller's
277
+ * job — see scheduler.cjs's findSatisfyingCommitOnMain) that are reachable
278
+ * from `main`, newer than the job's `queuedAt`, and touch the PRD's own
279
+ * declared paths. Returns the winning verdict naming the satisfying sha, or
280
+ * `null` when there is no such commit — the caller must then leave the
281
+ * existing 'finish protocol incomplete' → needs_review verdict untouched.
282
+ * This function never widens what counts as evidence; it only decides what
283
+ * to do once the caller's git query has already proven a satisfying commit
284
+ * exists, so it can never turn a genuine no-op into a false 'completed'.
285
+ */
286
+ function isAlreadySatisfiedOnMain(commits) {
287
+ if (!Array.isArray(commits) || commits.length === 0) return null;
288
+ const sha = commits[0];
289
+ return {
290
+ sha,
291
+ verdict: 'already_satisfied_on_main',
292
+ reason: `already satisfied by ${sha} — a commit on main newer than this run's queuedAt already touches this PRD's declared paths`,
293
+ };
294
+ }
295
+
296
+ /**
297
+ * resolveCommitGuardOutcome(guardVerdict, satisfyingCommits) → verdict | null
298
+ *
299
+ * The full finalize-time decision this PRD adds: takes commitGuardVerdict's
300
+ * own output (scheduler.cjs) plus the already-git-queried satisfying-commit
301
+ * list (scheduler.cjs's findSatisfyingCommitOnMain) and decides which verdict
302
+ * actually wins. Only ever touches the 'silent_no_op' shape — a guardVerdict
303
+ * of null (no violation) or 'uncommitted_changes' (real dirt left behind)
304
+ * passes through completely untouched, so this can never weaken the
305
+ * uncommitted-changes guarantee PRD 1133 introduced. Pure — no I/O, no git —
306
+ * so the whole finalize decision is directly unit-testable without spawning
307
+ * a real job.
308
+ */
309
+ function resolveCommitGuardOutcome(guardVerdict, satisfyingCommits) {
310
+ if (!guardVerdict || guardVerdict.verdict !== 'silent_no_op') return guardVerdict ?? null;
311
+ const satisfied = isAlreadySatisfiedOnMain(satisfyingCommits);
312
+ if (!satisfied) return guardVerdict;
313
+ return { verdict: satisfied.verdict, reason: satisfied.reason, satisfyingSha: satisfied.sha };
159
314
  }
160
315
 
161
- module.exports = { claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs };
316
+ module.exports = {
317
+ claudePidAlive,
318
+ classifyRunOutcome,
319
+ mapOutcomeToGateOutcome,
320
+ ORPHAN_REQUEUE_CAP,
321
+ selectReapableJobs,
322
+ findLiveProcessForJob,
323
+ logHasOutput,
324
+ resolvePidlessGateOutcome,
325
+ isAlreadySatisfiedOnMain,
326
+ resolveCommitGuardOutcome,
327
+ };
@@ -41,13 +41,26 @@ const STATUS_HISTORY_CAP = 20;
41
41
  * Explicit from->to edges. Every real assignment site in scheduler.cjs maps
42
42
  * onto one of these (verified against the 16 sites this module replaces):
43
43
  * - pending->running (dispatch), pending->completed (archived-PRD skip,
44
- * manual archive of an already-shipped PRD), pending->failed (admin
45
- * cancelJob on a not-yet-started job)
44
+ * manual archive of an already-shipped PRD; also source
45
+ * 'spawnJob:dispatch-sidecar-reconcile' — a pending row about to
46
+ * dispatch whose newest run-sidecar already shows a completed-equivalent
47
+ * outcome finished at/after this row's own last pending transition is
48
+ * finalized from that sidecar instead of spawning a redundant re-run;
49
+ * 2026-09-06 incident: a silently-dropped finalize left a slug 'pending'
50
+ * forever and the dispatcher re-fired it three times), pending->failed
51
+ * (admin cancelJob on a not-yet-started job)
46
52
  * - running->completed|failed|needs_review (normal run outcomes, reaper),
47
53
  * running->skipped (spawnJob:skip-archived's 'prd-missing' case — no
48
54
  * executor ever ran; kept distinct from 'completed' so unrun work can't
49
55
  * read as shipped), running->pending (halt/rate-limit reset,
50
- * transient-failure retry)
56
+ * transient-failure retry). Also reached with verdict
57
+ * 'reaped_without_integration' (source 'reapDeadRunningJobs', PRD 1133): a
58
+ * reaped job whose result event looked successful but whose work cannot
59
+ * be shown to have landed — a worktree branch still holding unmerged
60
+ * commits, or an in-place run whose HEAD never advanced during the run
61
+ * window — is parked here instead of 'completed', naming the branch (or
62
+ * the lack of any commit) so the stranded work is never silently treated
63
+ * as shipped.
51
64
  * - investigating->failed|needs_review (restore prior status once the
52
65
  * investigation probe exits), investigating->completed (defensive: the
53
66
  * restored prior status could in principle be 'completed' if a caller
@@ -70,6 +70,19 @@ function isHarnessToolError(content) {
70
70
  || /\bNo such tool available\b/.test(content);
71
71
  }
72
72
 
73
+ /**
74
+ * A PreToolUse hook denial (`guard-destructive-git.cjs`, `guard-prd-writes.cjs`,
75
+ * `guard-inline-implementation.cjs`, or the harness's own `Blocked: sleep N
76
+ * followed by: ...` form) never executed the tool it names — it says nothing
77
+ * about whether the task succeeded, and the model is expected to adapt and
78
+ * retry. (Incident: PRD 1106, 2026-09-02 — four guard-destructive-git denials
79
+ * in a fully-implemented, correctly-isolated run.)
80
+ */
81
+ function isPolicyDenial(content) {
82
+ if (typeof content !== 'string' || !content) return false;
83
+ return /^(?:Error:\s*)?Blocked:/.test(content);
84
+ }
85
+
73
86
  function detectPattern(content) {
74
87
  if (typeof content !== 'string' || !content) return null;
75
88
 
@@ -506,16 +519,17 @@ function checkDeps(queueEntry, allJobs, prdBody) {
506
519
  // ─── sentinel scanner ─────────────────────────────────────────────────────────
507
520
 
508
521
  /**
509
- * Scan for a `SCHEDULER_VERDICT: PASS|FAIL` sentinel line in the run output.
522
+ * Scan for a `SCHEDULER_VERDICT: PASS|FAIL|BLOCKED_BY_FOREIGN_WIP` sentinel
523
+ * line in the run output.
510
524
  *
511
525
  * Checks `resultEvent.resultText` first (the agent's final message), then the
512
526
  * last tool_result content. Anchored to line-start so prose mentioning the
513
527
  * string in mid-sentence does not match.
514
528
  *
515
- * Returns 'pass', 'fail', or null.
529
+ * Returns 'pass', 'fail', 'blocked_by_foreign_wip', or null.
516
530
  */
517
531
  function scanSentinel(resultEvent, events) {
518
- const RE = /^SCHEDULER_VERDICT:\s*(PASS|FAIL)\b/m;
532
+ const RE = /^SCHEDULER_VERDICT:\s*(PASS|FAIL|BLOCKED_BY_FOREIGN_WIP)\b/m;
519
533
 
520
534
  if (resultEvent) {
521
535
  const m = resultEvent.resultText.match(RE);
@@ -534,6 +548,44 @@ function scanSentinel(resultEvent, events) {
534
548
  return null;
535
549
  }
536
550
 
551
+ /**
552
+ * Scan for a `FOREIGN_WIP_PATHS: <comma-separated paths>` evidence line —
553
+ * the mandatory companion to a `SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP`
554
+ * sentinel (see FINISH_PROTOCOL in scheduler.cjs). Same two-source scan order
555
+ * as scanSentinel (resultEvent.resultText, then the last tool_result) so the
556
+ * two scanners always agree on which "final say" they're reading from.
557
+ *
558
+ * Returns a deduplicated array of trimmed path strings (possibly empty — an
559
+ * executor that emits the verdict with no paths line, or an empty one, gives
560
+ * the caller nothing to validate, which the caller must treat as an invalid
561
+ * claim, not an empty-but-valid one).
562
+ */
563
+ function scanForeignWipPathsClaim(resultEvent, events) {
564
+ const RE = /^FOREIGN_WIP_PATHS:\s*(.+)$/m;
565
+
566
+ const parse = (text) => {
567
+ const m = typeof text === 'string' ? text.match(RE) : null;
568
+ if (!m) return null;
569
+ return [...new Set(m[1].split(',').map((p) => p.trim()).filter(Boolean))];
570
+ };
571
+
572
+ if (resultEvent) {
573
+ const paths = parse(resultEvent.resultText);
574
+ if (paths) return paths;
575
+ }
576
+
577
+ let lastToolResult = null;
578
+ for (const ev of events) {
579
+ if (ev.kind === 'tool_result') lastToolResult = ev;
580
+ }
581
+ if (lastToolResult) {
582
+ const paths = parse(lastToolResult.content);
583
+ if (paths) return paths;
584
+ }
585
+
586
+ return [];
587
+ }
588
+
537
589
  // ─── merge-main postcondition exemption ──────────────────────────────────────
538
590
 
539
591
  /**
@@ -853,6 +905,20 @@ async function verifyRun({ runDir, prdPath, queueEntry, allJobs = [], committedD
853
905
  // seen in 58-web-remote-correctness-batch, 2026-06-10).
854
906
  if (isHarnessToolError(ev.content)) continue;
855
907
 
908
+ // A PreToolUse policy denial (a guard hook blocked the call before it
909
+ // ran) is exempt from both the is_error scan and the content-pattern
910
+ // scan for the same reason as a harness tool error above — the denied
911
+ // call never executed. Unlike a harness tool error, record it as an
912
+ // annotation so the denial stays visible in the verdict record instead
913
+ // of vanishing silently.
914
+ if (isPolicyDenial(ev.content)) {
915
+ annotations.push({
916
+ verdict: 'policy_denial',
917
+ reason: `PreToolUse policy denial at event ${i}: ${ev.content.slice(0, 200)}`,
918
+ });
919
+ continue;
920
+ }
921
+
856
922
  // A tool_result carrying a non-null parent_tool_use_id happened INSIDE a Task
857
923
  // subagent's own execution, not in the main agent's. Subagents do ordinary
858
924
  // exploratory work (greps that exit 1 on no-match, ls on a path that may not
@@ -1147,6 +1213,7 @@ module.exports = {
1147
1213
  // Exposed for unit tests.
1148
1214
  detectPattern,
1149
1215
  isHarnessToolError,
1216
+ isPolicyDenial,
1150
1217
  isSelfRecovered,
1151
1218
  normalizeDescForRecovery,
1152
1219
  toolUseName,
@@ -1155,6 +1222,7 @@ module.exports = {
1155
1222
  checkDeps,
1156
1223
  parseLog,
1157
1224
  scanSentinel,
1225
+ scanForeignWipPathsClaim,
1158
1226
  isMergeMainSlug,
1159
1227
  extractMergeMainPrNumber,
1160
1228
  checkMergeablePr,