claude-code-session-manager 0.80.0 → 0.81.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-zS3jw_1e.js → AgentLibrary-psZYVM2w.js} +1 -1
- package/dist/assets/{DataModel-Cy_vxTpi.js → DataModel-BMied5pg.js} +1 -1
- package/dist/assets/{History-C6JRuqfT.js → History-BFC0oaKc.js} +1 -1
- package/dist/assets/{Hooks-BafPy9mB.js → Hooks-CTLfO9G8.js} +1 -1
- package/dist/assets/{HostBilko-BZwhQOFt.js → HostBilko-D4I0Cpwn.js} +1 -1
- package/dist/assets/{Library-C8JDDliz.js → Library-BTzS8KsS.js} +1 -1
- package/dist/assets/{ListDetail-CqiOdwLc.js → ListDetail-CuRmT008.js} +1 -1
- package/dist/assets/{MarkdownEditor-CyLyP67L.js → MarkdownEditor-BgQGtvWo.js} +1 -1
- package/dist/assets/{McpServers-BzMv-_84.js → McpServers-DiVQUK57.js} +1 -1
- package/dist/assets/{Memory-DSBYQdJR.js → Memory-DSu15JmL.js} +1 -1
- package/dist/assets/{Panel-CLUhkNNA.js → Panel-CD5wxGSR.js} +1 -1
- package/dist/assets/{Permissions-BfC2-HN4.js → Permissions-C_tAE6yJ.js} +1 -1
- package/dist/assets/{Plugins-BKi40jT5.js → Plugins-kwI7W-eK.js} +2 -2
- package/dist/assets/{ProvenanceBadge-BzFw4KhD.js → ProvenanceBadge-CQceOgsH.js} +1 -1
- package/dist/assets/{SaveBar-avk2p9jv.js → SaveBar-BDk5e3Pp.js} +1 -1
- package/dist/assets/{Scheduler-Bf_6MdJo.js → Scheduler-DRnvEYzv.js} +1 -1
- package/dist/assets/{ScopeSwitcher-C-RwYUVZ.js → ScopeSwitcher-CeifaOlq.js} +1 -1
- package/dist/assets/{Settings-Djd8OoBA.js → Settings-BWQ1Utop.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-DuogY6s7.js → SkillReferenceGraph-CW6e1SW1.js} +1 -1
- package/dist/assets/{Skills-D_qAqxZ_.js → Skills-BLwFB0E4.js} +1 -1
- package/dist/assets/{SystemPrompt-DbHFLQV3.js → SystemPrompt-DC4ZTArJ.js} +1 -1
- package/dist/assets/{TagLibrary-C2y91BT0.js → TagLibrary-pMeGfjUb.js} +1 -1
- package/dist/assets/{TiptapBody-D9iz4xQx.js → TiptapBody-BHFid2pZ.js} +1 -1
- package/dist/assets/{Toggle-BGnFL2E5.js → Toggle-D9eYoZh4.js} +1 -1
- package/dist/assets/{index-_2ARyFDj.js → index-CuyM9vAP.js} +4 -4
- package/dist/assets/{settingsSchema-JK15eJU8.js → settingsSchema-DrxC67uZ.js} +1 -1
- package/dist/index.html +1 -1
- package/package.json +1 -1
- package/plugins/session-manager-dev/skills/develop/standards.md +1 -0
- package/scripts/project-pages-logic/dist/logic.cjs +12 -12
- package/scripts/render-project-pages/dist/renderer.cjs +22 -22
- package/src/main/__tests__/rcaReport.test.cjs +24 -0
- package/src/main/__tests__/runVerify-blocked-by-foreign-wip.test.cjs +58 -0
- package/src/main/__tests__/runVerify-policy-denial.test.cjs +89 -0
- package/src/main/__tests__/scheduler-already-satisfied-on-main.test.cjs +105 -0
- package/src/main/__tests__/scheduler-blocked-by-foreign-wip.test.cjs +107 -0
- package/src/main/__tests__/scheduler-finalize-dispatch-guards.test.cjs +229 -0
- package/src/main/__tests__/scheduler-looks-done.test.cjs +141 -1
- package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +81 -0
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +205 -2
- package/src/main/lib/__tests__/branchSweep.test.cjs +164 -0
- package/src/main/lib/__tests__/fixtures/204-mercury-steam-horse.log.txt +13 -0
- package/src/main/lib/__tests__/landedSinceRun.test.cjs +61 -1
- package/src/main/lib/__tests__/rateLimitWindow.test.cjs +88 -0
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +120 -1
- package/src/main/lib/branchSweep.cjs +127 -0
- package/src/main/lib/gitWorktree.cjs +20 -0
- package/src/main/lib/landedSinceRun.cjs +41 -1
- package/src/main/lib/rateLimitWindow.cjs +62 -0
- package/src/main/lib/rcaReport.cjs +18 -3
- package/src/main/lib/reaperHelpers.cjs +169 -3
- package/src/main/lib/scheduleJobTransitions.cjs +16 -3
- package/src/main/runVerify.cjs +71 -3
- package/src/main/scheduler.cjs +741 -31
|
@@ -52,6 +52,7 @@ const VERDICT_LABELS = {
|
|
|
52
52
|
pass_no_commit_already_shipped: 'PASS with no commit — deliverables already shipped',
|
|
53
53
|
pass_no_commit_prior_run_verified: 'PASS with no commit — prior run of this slug already landed the work',
|
|
54
54
|
silent_no_op: 'no commit, clean tree — no evidence of work',
|
|
55
|
+
blocked_by_foreign_wip_streak: 'blocked by a sibling job\'s foreign WIP 3x in a row',
|
|
55
56
|
};
|
|
56
57
|
|
|
57
58
|
function humanVerdict(verdict) {
|
|
@@ -69,14 +70,18 @@ const FAILURE_CLASSES = {
|
|
|
69
70
|
UNCOMMITTED: 'uncommitted-changes',
|
|
70
71
|
TRANSCRIPT_ERRORS: 'transcript-errors',
|
|
71
72
|
ABANDONED_BACKGROUND_TASK: 'abandoned-background-task',
|
|
73
|
+
BLOCKED_BY_FOREIGN_WIP: 'blocked-by-foreign-wip',
|
|
72
74
|
UNKNOWN: 'unknown',
|
|
73
75
|
};
|
|
74
76
|
|
|
75
77
|
// ─── Recovery actions — one per failure class, machine-readable ────────────
|
|
76
78
|
// Closed set the scheduler routes on: 'archive' (stale re-run, do not
|
|
77
|
-
// re-queue
|
|
78
|
-
//
|
|
79
|
-
// '
|
|
79
|
+
// re-queue — also the closest fit for BLOCKED_BY_FOREIGN_WIP: there is
|
|
80
|
+
// nothing to investigate, and re-queuing is already owned by
|
|
81
|
+
// reconcile()'s requeueForeignWipBlockedJobs, not this action), 'resume-and-commit'
|
|
82
|
+
// (PRD 1111's --resume dispatch owns this), 'verify-and-close' (the work
|
|
83
|
+
// likely landed, just missing its sentinel), 'investigate' (the only class
|
|
84
|
+
// that still buys a fix-plan investigation).
|
|
80
85
|
const RECOVERY_ACTIONS = {
|
|
81
86
|
[FAILURE_CLASSES.ALREADY_SHIPPED]: 'archive',
|
|
82
87
|
[FAILURE_CLASSES.SELF_QUEUE]: 'investigate',
|
|
@@ -96,6 +101,7 @@ const RECOVERY_ACTIONS = {
|
|
|
96
101
|
// carries the "check for salvaged work, commit before re-implementing"
|
|
97
102
|
// guidance into that cold-read investigation.
|
|
98
103
|
[FAILURE_CLASSES.ABANDONED_BACKGROUND_TASK]: 'investigate',
|
|
104
|
+
[FAILURE_CLASSES.BLOCKED_BY_FOREIGN_WIP]: 'archive',
|
|
99
105
|
[FAILURE_CLASSES.UNKNOWN]: 'investigate',
|
|
100
106
|
};
|
|
101
107
|
|
|
@@ -121,6 +127,8 @@ const PREVENTION_HINTS = {
|
|
|
121
127
|
'Recover or annotate every error within ~10 lines (e.g. `# expected/handled: <why>`) instead of leaving a bare Traceback near the end of the transcript.',
|
|
122
128
|
[FAILURE_CLASSES.ABANDONED_BACKGROUND_TASK]:
|
|
123
129
|
'A headless run cannot receive a background-task completion notification — never let a long Bash command auto-background past its foreground timeout and then wait for it. Run long commands with an explicit bound (`timeout <N> <cmd>`) so they finish in the foreground, or poll their output file directly instead of waiting on Monitor/notification.',
|
|
130
|
+
[FAILURE_CLASSES.BLOCKED_BY_FOREIGN_WIP]:
|
|
131
|
+
'This job correctly reported `SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP` against a sibling job\'s in-flight, uncommitted files three times in a row — auto-requeue is exhausted (reconcile() only clears this once the blocked paths go clean). This is not a defect in this PRD\'s own work: check on the sibling job/human that owns the still-dirty path(s) named on the job row, and either wait for them to commit or manually reset this job to `pending` once the tree is clear. Do not author a fix-plan PRD against it.',
|
|
124
132
|
[FAILURE_CLASSES.UNKNOWN]:
|
|
125
133
|
'Re-run the acceptance criteria gate locally against the failure log to pin down the specific break before re-queuing.',
|
|
126
134
|
};
|
|
@@ -159,6 +167,13 @@ const POST_AC_OVERRUN_MIN_TAIL_FRACTION = 0.3;
|
|
|
159
167
|
function classifyFailure({ verdict, logTail }) {
|
|
160
168
|
const lines = (logTail || '').split('\n');
|
|
161
169
|
|
|
170
|
+
// Checked before every other rule: an unambiguous, materially-checked
|
|
171
|
+
// verdict (the scheduler itself validated the executor's claimed paths
|
|
172
|
+
// against the job's disclosed foreign-WIP manifest before ever landing
|
|
173
|
+
// this verdict — see validateForeignWipBlockClaim in scheduler.cjs) needs
|
|
174
|
+
// no log-tail heuristics to classify.
|
|
175
|
+
if (verdict === 'blocked_by_foreign_wip_streak') return FAILURE_CLASSES.BLOCKED_BY_FOREIGN_WIP;
|
|
176
|
+
|
|
162
177
|
// Checked first, before SELF_QUEUE/STUCK_LOOP: a correct executor that finds
|
|
163
178
|
// its acceptance criteria already satisfied by a prior commit makes no
|
|
164
179
|
// change and truthfully prints a PASS sentinel, so the run lands
|
|
@@ -8,6 +8,7 @@
|
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
const fs = require('node:fs');
|
|
11
|
+
const path = require('node:path');
|
|
11
12
|
const { readTail } = require('./fileTail.cjs');
|
|
12
13
|
const { detectRateLimitInLog } = require('./rateLimitDetect.cjs');
|
|
13
14
|
|
|
@@ -34,6 +35,90 @@ function claudePidAlive(pid) {
|
|
|
34
35
|
}
|
|
35
36
|
}
|
|
36
37
|
|
|
38
|
+
/**
|
|
39
|
+
* findLiveProcessForJob(job, { worktreeDir }) → pid | null
|
|
40
|
+
*
|
|
41
|
+
* Positive liveness scan for a 'running' row whose `runtime.pid` is missing —
|
|
42
|
+
* the case a missing pid must NOT be read as "the process is gone" (2026-09-06
|
|
43
|
+
* incident: 234-uranus-eight-tails-ox marked failed/never_ran while PID
|
|
44
|
+
* 2174739 was a live `claude -p` still writing to that job's own worktree).
|
|
45
|
+
*
|
|
46
|
+
* Linux-`/proc` only. Scans every numeric `/proc/<pid>` entry and matches
|
|
47
|
+
* either: `/proc/<pid>/cwd` resolves to `worktreeDir` (or a path nested under
|
|
48
|
+
* it), or `/proc/<pid>/cmdline` contains both `claude` and the job's slug
|
|
49
|
+
* (fallback for a worktree-disabled/in-place run, where there is no dedicated
|
|
50
|
+
* worktreeDir to match against). Returns the first matching pid, or null if
|
|
51
|
+
* none is found.
|
|
52
|
+
*
|
|
53
|
+
* Safe fallback by construction: on any platform without `/proc` (macOS,
|
|
54
|
+
* Windows) `fs.readdirSync('/proc')` throws and this returns null immediately
|
|
55
|
+
* — i.e. exactly today's behaviour (fail toward terminalizing), never a hang
|
|
56
|
+
* or a thrown error propagating to the caller.
|
|
57
|
+
*/
|
|
58
|
+
function findLiveProcessForJob(job, { worktreeDir } = {}) {
|
|
59
|
+
const slug = job?.slug;
|
|
60
|
+
let entries;
|
|
61
|
+
try {
|
|
62
|
+
entries = fs.readdirSync('/proc');
|
|
63
|
+
} catch {
|
|
64
|
+
return null;
|
|
65
|
+
}
|
|
66
|
+
for (const name of entries) {
|
|
67
|
+
if (!/^\d+$/.test(name)) continue;
|
|
68
|
+
const pid = Number(name);
|
|
69
|
+
if (!pid || pid <= 1) continue;
|
|
70
|
+
if (worktreeDir) {
|
|
71
|
+
try {
|
|
72
|
+
const cwdLink = fs.readlinkSync(`/proc/${pid}/cwd`);
|
|
73
|
+
if (cwdLink === worktreeDir || cwdLink.startsWith(worktreeDir + path.sep)) return pid;
|
|
74
|
+
} catch {
|
|
75
|
+
// Process exited mid-scan, or permission denied — try the argv
|
|
76
|
+
// fallback below before giving up on this pid.
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
if (slug) {
|
|
80
|
+
try {
|
|
81
|
+
const cmd = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8').replace(/\0/g, ' ');
|
|
82
|
+
if (/\bclaude\b/.test(cmd) && cmd.includes(slug)) return pid;
|
|
83
|
+
} catch {
|
|
84
|
+
// Same as above — process gone or unreadable, keep scanning.
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
return null;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* True when `logPath` exists and has non-zero content — the literal "did the
|
|
93
|
+
* run dir produce any log output" check that gates whether a pidless reap may
|
|
94
|
+
* assert `gateOutcome: 'never_ran'`. A run whose log has real bytes in it DID
|
|
95
|
+
* run, regardless of whether classifyRunOutcome found a clean result event in
|
|
96
|
+
* it — asserting never_ran in that case would be a false claim (see
|
|
97
|
+
* resolvePidlessGateOutcome below).
|
|
98
|
+
*/
|
|
99
|
+
function logHasOutput(logPath) {
|
|
100
|
+
if (!logPath) return false;
|
|
101
|
+
try {
|
|
102
|
+
return fs.statSync(logPath).size > 0;
|
|
103
|
+
} catch {
|
|
104
|
+
return false;
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Gate-outcome for a pidless reap, once no live process was found for it.
|
|
110
|
+
* `never_ran` is asserted ONLY when the run dir produced no log output at
|
|
111
|
+
* all — mapOutcomeToGateOutcome's own 'no_result' → 'never_ran' mapping is
|
|
112
|
+
* otherwise too broad here: a log with real content but no clean result event
|
|
113
|
+
* (e.g. killed mid-turn) proves the job DID run, so that case is reported as
|
|
114
|
+
* 'failed' instead of the false 'never_ran'.
|
|
115
|
+
*/
|
|
116
|
+
function resolvePidlessGateOutcome(outcome, hasOutput) {
|
|
117
|
+
if (!hasOutput) return 'never_ran';
|
|
118
|
+
const mapped = mapOutcomeToGateOutcome(outcome);
|
|
119
|
+
return mapped === 'never_ran' ? 'failed' : mapped;
|
|
120
|
+
}
|
|
121
|
+
|
|
37
122
|
/**
|
|
38
123
|
* Classify the terminal outcome of a completed run by reading the last 64 KB
|
|
39
124
|
* of its log file and scanning for the LAST `{"type":"result"}` JSONL event.
|
|
@@ -129,10 +214,21 @@ const ORPHAN_REQUEUE_CAP = 5;
|
|
|
129
214
|
* A pidless row whose `startedAt` is missing or unparseable is neither
|
|
130
215
|
* reaped nor skipped silently — age can't be proven, so it is surfaced in
|
|
131
216
|
* `warnings` instead (the caller logs it) and left alone.
|
|
217
|
+
*
|
|
218
|
+
* `findLiveProcess` (optional, `(job) → pid | null`) is consulted for a
|
|
219
|
+
* pidless row ONLY once its age clears `grace` — i.e. right before it would
|
|
220
|
+
* otherwise be terminalized. A pid it finds means the process is actually
|
|
221
|
+
* alive despite the missing runtime.pid record: the row is diverted into
|
|
222
|
+
* `recovered` (never `reapable`) so the caller can re-stamp the pid and leave
|
|
223
|
+
* the row `running`, instead of terminalizing a job that is still doing real
|
|
224
|
+
* work (2026-09-06 incident — see findLiveProcessForJob's header). Omitting
|
|
225
|
+
* `findLiveProcess` (existing callers/tests) preserves prior behaviour
|
|
226
|
+
* exactly: every pidless row past grace reaps, none are ever recovered.
|
|
132
227
|
*/
|
|
133
|
-
function selectReapableJobs(jobs, now, { pidAlive, grace } = {}) {
|
|
228
|
+
function selectReapableJobs(jobs, now, { pidAlive, grace, findLiveProcess } = {}) {
|
|
134
229
|
const reapable = [];
|
|
135
230
|
const warnings = [];
|
|
231
|
+
const recovered = [];
|
|
136
232
|
for (const j of jobs ?? []) {
|
|
137
233
|
if (j.status !== 'running') continue;
|
|
138
234
|
const pid = j.runtime?.pid;
|
|
@@ -148,6 +244,11 @@ function selectReapableJobs(jobs, now, { pidAlive, grace } = {}) {
|
|
|
148
244
|
}
|
|
149
245
|
const ageMs = now - startedAt;
|
|
150
246
|
if (ageMs < grace) continue; // spawn may still be mid-flight
|
|
247
|
+
const livePid = typeof findLiveProcess === 'function' ? findLiveProcess(j) : null;
|
|
248
|
+
if (livePid) {
|
|
249
|
+
recovered.push({ slug: j.slug, pid: livePid });
|
|
250
|
+
continue;
|
|
251
|
+
}
|
|
151
252
|
reapable.push({
|
|
152
253
|
slug: j.slug,
|
|
153
254
|
pid: null,
|
|
@@ -155,7 +256,72 @@ function selectReapableJobs(jobs, now, { pidAlive, grace } = {}) {
|
|
|
155
256
|
reason: `reaped: no runtime.pid recorded after ${Math.round(grace / 60_000)}m — spawn never completed`,
|
|
156
257
|
});
|
|
157
258
|
}
|
|
158
|
-
return { reapable, warnings };
|
|
259
|
+
return { reapable, warnings, recovered };
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/**
|
|
263
|
+
* isAlreadySatisfiedOnMain(commits) → { sha, verdict, reason } | null
|
|
264
|
+
*
|
|
265
|
+
* Pure decision layer for the finish-protocol commit-guard's second,
|
|
266
|
+
* independently-evidenced route to 'completed' (PRD 1136). The commit-guard
|
|
267
|
+
* in scheduler.cjs parks a clean exit / no-commit / clean-tree run as
|
|
268
|
+
* `needs_review` ("finish protocol incomplete") because that shape is
|
|
269
|
+
* normally the strongest signal that nothing happened — but it is also
|
|
270
|
+
* exactly the shape a run produces when its PRD's work was ALREADY merged to
|
|
271
|
+
* main before the run dispatched (2026-09-06: 1133-reaper-must-verify-
|
|
272
|
+
* integration-before-completed and 1134-land-stranded-sm-job-branches, both
|
|
273
|
+
* false-negatived this way and then auto-fix-minted a redundant `-fix-`
|
|
274
|
+
* child against a codebase where the change was already present).
|
|
275
|
+
*
|
|
276
|
+
* Takes the already-git-queried list of commit SHAs (newest first, caller's
|
|
277
|
+
* job — see scheduler.cjs's findSatisfyingCommitOnMain) that are reachable
|
|
278
|
+
* from `main`, newer than the job's `queuedAt`, and touch the PRD's own
|
|
279
|
+
* declared paths. Returns the winning verdict naming the satisfying sha, or
|
|
280
|
+
* `null` when there is no such commit — the caller must then leave the
|
|
281
|
+
* existing 'finish protocol incomplete' → needs_review verdict untouched.
|
|
282
|
+
* This function never widens what counts as evidence; it only decides what
|
|
283
|
+
* to do once the caller's git query has already proven a satisfying commit
|
|
284
|
+
* exists, so it can never turn a genuine no-op into a false 'completed'.
|
|
285
|
+
*/
|
|
286
|
+
function isAlreadySatisfiedOnMain(commits) {
|
|
287
|
+
if (!Array.isArray(commits) || commits.length === 0) return null;
|
|
288
|
+
const sha = commits[0];
|
|
289
|
+
return {
|
|
290
|
+
sha,
|
|
291
|
+
verdict: 'already_satisfied_on_main',
|
|
292
|
+
reason: `already satisfied by ${sha} — a commit on main newer than this run's queuedAt already touches this PRD's declared paths`,
|
|
293
|
+
};
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
/**
|
|
297
|
+
* resolveCommitGuardOutcome(guardVerdict, satisfyingCommits) → verdict | null
|
|
298
|
+
*
|
|
299
|
+
* The full finalize-time decision this PRD adds: takes commitGuardVerdict's
|
|
300
|
+
* own output (scheduler.cjs) plus the already-git-queried satisfying-commit
|
|
301
|
+
* list (scheduler.cjs's findSatisfyingCommitOnMain) and decides which verdict
|
|
302
|
+
* actually wins. Only ever touches the 'silent_no_op' shape — a guardVerdict
|
|
303
|
+
* of null (no violation) or 'uncommitted_changes' (real dirt left behind)
|
|
304
|
+
* passes through completely untouched, so this can never weaken the
|
|
305
|
+
* uncommitted-changes guarantee PRD 1133 introduced. Pure — no I/O, no git —
|
|
306
|
+
* so the whole finalize decision is directly unit-testable without spawning
|
|
307
|
+
* a real job.
|
|
308
|
+
*/
|
|
309
|
+
function resolveCommitGuardOutcome(guardVerdict, satisfyingCommits) {
|
|
310
|
+
if (!guardVerdict || guardVerdict.verdict !== 'silent_no_op') return guardVerdict ?? null;
|
|
311
|
+
const satisfied = isAlreadySatisfiedOnMain(satisfyingCommits);
|
|
312
|
+
if (!satisfied) return guardVerdict;
|
|
313
|
+
return { verdict: satisfied.verdict, reason: satisfied.reason, satisfyingSha: satisfied.sha };
|
|
159
314
|
}
|
|
160
315
|
|
|
161
|
-
module.exports = {
|
|
316
|
+
module.exports = {
|
|
317
|
+
claudePidAlive,
|
|
318
|
+
classifyRunOutcome,
|
|
319
|
+
mapOutcomeToGateOutcome,
|
|
320
|
+
ORPHAN_REQUEUE_CAP,
|
|
321
|
+
selectReapableJobs,
|
|
322
|
+
findLiveProcessForJob,
|
|
323
|
+
logHasOutput,
|
|
324
|
+
resolvePidlessGateOutcome,
|
|
325
|
+
isAlreadySatisfiedOnMain,
|
|
326
|
+
resolveCommitGuardOutcome,
|
|
327
|
+
};
|
|
@@ -41,13 +41,26 @@ const STATUS_HISTORY_CAP = 20;
|
|
|
41
41
|
* Explicit from->to edges. Every real assignment site in scheduler.cjs maps
|
|
42
42
|
* onto one of these (verified against the 16 sites this module replaces):
|
|
43
43
|
* - pending->running (dispatch), pending->completed (archived-PRD skip,
|
|
44
|
-
* manual archive of an already-shipped PRD
|
|
45
|
-
*
|
|
44
|
+
* manual archive of an already-shipped PRD; also source
|
|
45
|
+
* 'spawnJob:dispatch-sidecar-reconcile' — a pending row about to
|
|
46
|
+
* dispatch whose newest run-sidecar already shows a completed-equivalent
|
|
47
|
+
* outcome finished at/after this row's own last pending transition is
|
|
48
|
+
* finalized from that sidecar instead of spawning a redundant re-run;
|
|
49
|
+
* 2026-09-06 incident: a silently-dropped finalize left a slug 'pending'
|
|
50
|
+
* forever and the dispatcher re-fired it three times), pending->failed
|
|
51
|
+
* (admin cancelJob on a not-yet-started job)
|
|
46
52
|
* - running->completed|failed|needs_review (normal run outcomes, reaper),
|
|
47
53
|
* running->skipped (spawnJob:skip-archived's 'prd-missing' case — no
|
|
48
54
|
* executor ever ran; kept distinct from 'completed' so unrun work can't
|
|
49
55
|
* read as shipped), running->pending (halt/rate-limit reset,
|
|
50
|
-
* transient-failure retry)
|
|
56
|
+
* transient-failure retry). Also reached with verdict
|
|
57
|
+
* 'reaped_without_integration' (source 'reapDeadRunningJobs', PRD 1133): a
|
|
58
|
+
* reaped job whose result event looked successful but whose work cannot
|
|
59
|
+
* be shown to have landed — a worktree branch still holding unmerged
|
|
60
|
+
* commits, or an in-place run whose HEAD never advanced during the run
|
|
61
|
+
* window — is parked here instead of 'completed', naming the branch (or
|
|
62
|
+
* the lack of any commit) so the stranded work is never silently treated
|
|
63
|
+
* as shipped.
|
|
51
64
|
* - investigating->failed|needs_review (restore prior status once the
|
|
52
65
|
* investigation probe exits), investigating->completed (defensive: the
|
|
53
66
|
* restored prior status could in principle be 'completed' if a caller
|
package/src/main/runVerify.cjs
CHANGED
|
@@ -70,6 +70,19 @@ function isHarnessToolError(content) {
|
|
|
70
70
|
|| /\bNo such tool available\b/.test(content);
|
|
71
71
|
}
|
|
72
72
|
|
|
73
|
+
/**
|
|
74
|
+
* A PreToolUse hook denial (`guard-destructive-git.cjs`, `guard-prd-writes.cjs`,
|
|
75
|
+
* `guard-inline-implementation.cjs`, or the harness's own `Blocked: sleep N
|
|
76
|
+
* followed by: ...` form) never executed the tool it names — it says nothing
|
|
77
|
+
* about whether the task succeeded, and the model is expected to adapt and
|
|
78
|
+
* retry. (Incident: PRD 1106, 2026-09-02 — four guard-destructive-git denials
|
|
79
|
+
* in a fully-implemented, correctly-isolated run.)
|
|
80
|
+
*/
|
|
81
|
+
function isPolicyDenial(content) {
|
|
82
|
+
if (typeof content !== 'string' || !content) return false;
|
|
83
|
+
return /^(?:Error:\s*)?Blocked:/.test(content);
|
|
84
|
+
}
|
|
85
|
+
|
|
73
86
|
function detectPattern(content) {
|
|
74
87
|
if (typeof content !== 'string' || !content) return null;
|
|
75
88
|
|
|
@@ -506,16 +519,17 @@ function checkDeps(queueEntry, allJobs, prdBody) {
|
|
|
506
519
|
// ─── sentinel scanner ─────────────────────────────────────────────────────────
|
|
507
520
|
|
|
508
521
|
/**
|
|
509
|
-
* Scan for a `SCHEDULER_VERDICT: PASS|FAIL` sentinel
|
|
522
|
+
* Scan for a `SCHEDULER_VERDICT: PASS|FAIL|BLOCKED_BY_FOREIGN_WIP` sentinel
|
|
523
|
+
* line in the run output.
|
|
510
524
|
*
|
|
511
525
|
* Checks `resultEvent.resultText` first (the agent's final message), then the
|
|
512
526
|
* last tool_result content. Anchored to line-start so prose mentioning the
|
|
513
527
|
* string in mid-sentence does not match.
|
|
514
528
|
*
|
|
515
|
-
* Returns 'pass', 'fail', or null.
|
|
529
|
+
* Returns 'pass', 'fail', 'blocked_by_foreign_wip', or null.
|
|
516
530
|
*/
|
|
517
531
|
function scanSentinel(resultEvent, events) {
|
|
518
|
-
const RE = /^SCHEDULER_VERDICT:\s*(PASS|FAIL)\b/m;
|
|
532
|
+
const RE = /^SCHEDULER_VERDICT:\s*(PASS|FAIL|BLOCKED_BY_FOREIGN_WIP)\b/m;
|
|
519
533
|
|
|
520
534
|
if (resultEvent) {
|
|
521
535
|
const m = resultEvent.resultText.match(RE);
|
|
@@ -534,6 +548,44 @@ function scanSentinel(resultEvent, events) {
|
|
|
534
548
|
return null;
|
|
535
549
|
}
|
|
536
550
|
|
|
551
|
+
/**
|
|
552
|
+
* Scan for a `FOREIGN_WIP_PATHS: <comma-separated paths>` evidence line —
|
|
553
|
+
* the mandatory companion to a `SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP`
|
|
554
|
+
* sentinel (see FINISH_PROTOCOL in scheduler.cjs). Same two-source scan order
|
|
555
|
+
* as scanSentinel (resultEvent.resultText, then the last tool_result) so the
|
|
556
|
+
* two scanners always agree on which "final say" they're reading from.
|
|
557
|
+
*
|
|
558
|
+
* Returns a deduplicated array of trimmed path strings (possibly empty — an
|
|
559
|
+
* executor that emits the verdict with no paths line, or an empty one, gives
|
|
560
|
+
* the caller nothing to validate, which the caller must treat as an invalid
|
|
561
|
+
* claim, not an empty-but-valid one).
|
|
562
|
+
*/
|
|
563
|
+
function scanForeignWipPathsClaim(resultEvent, events) {
|
|
564
|
+
const RE = /^FOREIGN_WIP_PATHS:\s*(.+)$/m;
|
|
565
|
+
|
|
566
|
+
const parse = (text) => {
|
|
567
|
+
const m = typeof text === 'string' ? text.match(RE) : null;
|
|
568
|
+
if (!m) return null;
|
|
569
|
+
return [...new Set(m[1].split(',').map((p) => p.trim()).filter(Boolean))];
|
|
570
|
+
};
|
|
571
|
+
|
|
572
|
+
if (resultEvent) {
|
|
573
|
+
const paths = parse(resultEvent.resultText);
|
|
574
|
+
if (paths) return paths;
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
let lastToolResult = null;
|
|
578
|
+
for (const ev of events) {
|
|
579
|
+
if (ev.kind === 'tool_result') lastToolResult = ev;
|
|
580
|
+
}
|
|
581
|
+
if (lastToolResult) {
|
|
582
|
+
const paths = parse(lastToolResult.content);
|
|
583
|
+
if (paths) return paths;
|
|
584
|
+
}
|
|
585
|
+
|
|
586
|
+
return [];
|
|
587
|
+
}
|
|
588
|
+
|
|
537
589
|
// ─── merge-main postcondition exemption ──────────────────────────────────────
|
|
538
590
|
|
|
539
591
|
/**
|
|
@@ -853,6 +905,20 @@ async function verifyRun({ runDir, prdPath, queueEntry, allJobs = [], committedD
|
|
|
853
905
|
// seen in 58-web-remote-correctness-batch, 2026-06-10).
|
|
854
906
|
if (isHarnessToolError(ev.content)) continue;
|
|
855
907
|
|
|
908
|
+
// A PreToolUse policy denial (a guard hook blocked the call before it
|
|
909
|
+
// ran) is exempt from both the is_error scan and the content-pattern
|
|
910
|
+
// scan for the same reason as a harness tool error above — the denied
|
|
911
|
+
// call never executed. Unlike a harness tool error, record it as an
|
|
912
|
+
// annotation so the denial stays visible in the verdict record instead
|
|
913
|
+
// of vanishing silently.
|
|
914
|
+
if (isPolicyDenial(ev.content)) {
|
|
915
|
+
annotations.push({
|
|
916
|
+
verdict: 'policy_denial',
|
|
917
|
+
reason: `PreToolUse policy denial at event ${i}: ${ev.content.slice(0, 200)}`,
|
|
918
|
+
});
|
|
919
|
+
continue;
|
|
920
|
+
}
|
|
921
|
+
|
|
856
922
|
// A tool_result carrying a non-null parent_tool_use_id happened INSIDE a Task
|
|
857
923
|
// subagent's own execution, not in the main agent's. Subagents do ordinary
|
|
858
924
|
// exploratory work (greps that exit 1 on no-match, ls on a path that may not
|
|
@@ -1147,6 +1213,7 @@ module.exports = {
|
|
|
1147
1213
|
// Exposed for unit tests.
|
|
1148
1214
|
detectPattern,
|
|
1149
1215
|
isHarnessToolError,
|
|
1216
|
+
isPolicyDenial,
|
|
1150
1217
|
isSelfRecovered,
|
|
1151
1218
|
normalizeDescForRecovery,
|
|
1152
1219
|
toolUseName,
|
|
@@ -1155,6 +1222,7 @@ module.exports = {
|
|
|
1155
1222
|
checkDeps,
|
|
1156
1223
|
parseLog,
|
|
1157
1224
|
scanSentinel,
|
|
1225
|
+
scanForeignWipPathsClaim,
|
|
1158
1226
|
isMergeMainSlug,
|
|
1159
1227
|
extractMergeMainPrNumber,
|
|
1160
1228
|
checkMergeablePr,
|