claude-code-session-manager 0.76.0 → 0.77.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (123) hide show
  1. package/dist/assets/{AgentLibrary-CBx9l4zN.js → AgentLibrary-B2ie8bbw.js} +2 -2
  2. package/dist/assets/{DataModel-Bf0EIE_t.js → DataModel-BIJPYw32.js} +1 -1
  3. package/dist/assets/{History-CpdtWhC8.js → History-CeY6dk9S.js} +2 -2
  4. package/dist/assets/{Hooks-DyUbMDmg.js → Hooks-BFH2ocKg.js} +2 -2
  5. package/dist/assets/{HostBilko-By-wIpry.js → HostBilko-36gj9wLz.js} +1 -1
  6. package/dist/assets/{Library-CQmo4QVC.js → Library-C-hBct39.js} +1 -1
  7. package/dist/assets/{ListDetail-BQMd6NOm.js → ListDetail-CNq64VWV.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-DEp43FXX.js → MarkdownEditor-Bh3qt5-1.js} +1 -1
  9. package/dist/assets/{McpServers-CLarzwqA.js → McpServers-DpGN0oyz.js} +1 -1
  10. package/dist/assets/{Memory-B0sCdIy1.js → Memory-D59hUjC4.js} +6 -6
  11. package/dist/assets/{Panel-BhWPVOCD.js → Panel-DCgbaoci.js} +1 -1
  12. package/dist/assets/{Permissions-Ddlq8T_O.js → Permissions-DAmQ0DYV.js} +2 -2
  13. package/dist/assets/{Plugins-D2oA_2Jl.js → Plugins-Dyfgn6Is.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-DgAgavUM.js → ProvenanceBadge-BiYhPO1U.js} +1 -1
  15. package/dist/assets/SaveBar-RV7B6sOh.js +1 -0
  16. package/dist/assets/Scheduler-BPaNqx1b.js +14 -0
  17. package/dist/assets/{ScopeSwitcher-C_zWEtIl.js → ScopeSwitcher-P4mdLGNU.js} +1 -1
  18. package/dist/assets/{Settings-2Vx3X5SI.js → Settings-BL4vf5aX.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-BDEUjlTQ.js → SkillReferenceGraph-BRBDyi1_.js} +1 -1
  20. package/dist/assets/{Skills-Cmrz_LeN.js → Skills-BV08gDUH.js} +2 -2
  21. package/dist/assets/{SystemPrompt-DVA1eYDP.js → SystemPrompt-CLftSsDw.js} +1 -1
  22. package/dist/assets/TagLibrary-Bp8jGsd5.js +1 -0
  23. package/dist/assets/{TiptapBody-DmPc3amD.js → TiptapBody-jCpuB6E5.js} +1 -1
  24. package/dist/assets/{Toggle-zfd5LJkK.js → Toggle-D2paA1xf.js} +1 -1
  25. package/dist/assets/{index-B_4PNh9T.js → index-BDRSqBl3.js} +175 -175
  26. package/dist/assets/{index-DIjnPkRN.css → index-CYhdtisq.css} +1 -1
  27. package/dist/assets/{settingsSchema-B9es6fdA.js → settingsSchema-6IOLjZZN.js} +1 -1
  28. package/dist/index.html +2 -2
  29. package/package.json +8 -2
  30. package/plugins/session-manager-dev/skills/develop/standards.md +1 -1
  31. package/scripts/project-pages-logic/dist/logic.cjs +4709 -0
  32. package/scripts/render-project-pages/dist/renderer.cjs +18900 -0
  33. package/scripts/render-project-pages.cjs +70 -0
  34. package/scripts/scheduler-mcp-server.cjs +115 -1
  35. package/scripts/validate-project-pages-summary.cjs +62 -0
  36. package/src/main/__tests__/agentModelResolve.test.cjs +66 -0
  37. package/src/main/__tests__/health-delegation-chain.test.cjs +2 -1
  38. package/src/main/__tests__/prdAgentType.test.cjs +103 -0
  39. package/src/main/__tests__/prdCreate.test.cjs +138 -0
  40. package/src/main/__tests__/prdFrontmatterAgentType.test.cjs +117 -0
  41. package/src/main/__tests__/prdFrontmatterQuietMachine.test.cjs +108 -0
  42. package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +485 -0
  43. package/src/main/__tests__/projectPages.test.cjs +73 -1
  44. package/src/main/__tests__/rcaReport.test.cjs +54 -0
  45. package/src/main/__tests__/runVerify.test.cjs +94 -0
  46. package/src/main/__tests__/scheduler-autofix-select.test.cjs +43 -0
  47. package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +103 -0
  48. package/src/main/__tests__/scheduler-effective-concurrency.test.cjs +10 -0
  49. package/src/main/__tests__/scheduler-foreign-wip-manifest.test.cjs +78 -0
  50. package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +242 -0
  51. package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +31 -0
  52. package/src/main/__tests__/scheduler-launch-failure.test.cjs +201 -0
  53. package/src/main/__tests__/scheduler-leftover-fields.test.cjs +52 -0
  54. package/src/main/__tests__/scheduler-looks-done.test.cjs +241 -0
  55. package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +135 -0
  56. package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +222 -0
  57. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +147 -0
  58. package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +212 -0
  59. package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +194 -0
  60. package/src/main/__tests__/seedAgentPersonas.test.cjs +75 -14
  61. package/src/main/config.cjs +4 -1
  62. package/src/main/index.cjs +8 -1
  63. package/src/main/ipcSchemas.cjs +51 -0
  64. package/src/main/lib/__tests__/childWithLog.test.cjs +78 -0
  65. package/src/main/lib/__tests__/delegationReadiness.test.cjs +152 -2
  66. package/src/main/lib/__tests__/epicWorktreeMint.test.cjs +4 -2
  67. package/src/main/lib/__tests__/fixChainDepth.test.cjs +40 -0
  68. package/src/main/lib/__tests__/gitWorktree.test.cjs +277 -4
  69. package/src/main/lib/__tests__/gitWorktreeSalvageDelta.test.cjs +153 -0
  70. package/src/main/lib/__tests__/jobWorktree.test.cjs +5 -3
  71. package/src/main/lib/__tests__/landedSinceRun.test.cjs +73 -0
  72. package/src/main/lib/__tests__/launchFailure.test.cjs +220 -0
  73. package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +1 -0
  74. package/src/main/lib/__tests__/opsOwnership.test.cjs +7 -0
  75. package/src/main/lib/__tests__/prdDeclaredPaths.test.cjs +82 -0
  76. package/src/main/lib/__tests__/queueHealth.test.cjs +58 -0
  77. package/src/main/lib/__tests__/quietMachineLease.test.cjs +39 -0
  78. package/src/main/lib/__tests__/reaperHelpers.test.cjs +22 -1
  79. package/src/main/lib/__tests__/schedulerBatchLaunchHold.test.cjs +125 -0
  80. package/src/main/lib/__tests__/schedulerBatchQuietMachine.test.cjs +109 -0
  81. package/src/main/lib/__tests__/schedulerMcpServerHeadlessRefusal.test.cjs +71 -0
  82. package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +350 -0
  83. package/src/main/lib/agentModelResolve.cjs +58 -0
  84. package/src/main/lib/childWithLog.cjs +40 -5
  85. package/src/main/lib/claudeBin.cjs +54 -1
  86. package/src/main/lib/definitionOfDone.cjs +3 -2
  87. package/src/main/lib/delegationReadiness.cjs +115 -9
  88. package/src/main/lib/epicWorktreeMint.cjs +5 -2
  89. package/src/main/lib/fixChainDepth.cjs +45 -0
  90. package/src/main/lib/gitWorktree.cjs +464 -19
  91. package/src/main/lib/jobWorktree.cjs +1 -0
  92. package/src/main/lib/landedSinceRun.cjs +55 -0
  93. package/src/main/lib/launchFailure.cjs +357 -0
  94. package/src/main/lib/mcpToolCatalog.cjs +87 -2
  95. package/src/main/lib/opsOwnership.cjs +12 -0
  96. package/src/main/lib/prdAgentType.cjs +84 -0
  97. package/src/main/lib/prdCreate.cjs +57 -1
  98. package/src/main/lib/prdDeclaredPaths.cjs +70 -0
  99. package/src/main/lib/prdFrontmatter.cjs +17 -3
  100. package/src/main/lib/projectHomeAdminRoutes.cjs +402 -0
  101. package/src/main/lib/projectPageSummarySchema.cjs +181 -0
  102. package/src/main/lib/queueHealth.cjs +38 -0
  103. package/src/main/lib/queueStore.cjs +9 -2
  104. package/src/main/lib/quietMachineLease.cjs +48 -0
  105. package/src/main/lib/rcaReport.cjs +53 -3
  106. package/src/main/lib/reaperHelpers.cjs +18 -1
  107. package/src/main/lib/scheduleJobSchema.cjs +31 -0
  108. package/src/main/lib/scheduleJobTransitions.cjs +6 -2
  109. package/src/main/lib/schedulerBatch.cjs +133 -29
  110. package/src/main/lib/schedulerConfig.cjs +19 -0
  111. package/src/main/projectPages.cjs +160 -2
  112. package/src/main/runVerify.cjs +50 -9
  113. package/src/main/scheduler/prdParser.cjs +18 -1
  114. package/src/main/scheduler.cjs +1371 -97
  115. package/src/main/seedAgentPersonas.cjs +62 -21
  116. package/src/main/templates/project-pages-catalog.json +741 -0
  117. package/src/main/templates/project-pages-pipeline.md +417 -0
  118. package/src/preload/api.d.ts +118 -2
  119. package/src/preload/index.cjs +7 -0
  120. package/src/seed/agents/project-home-builder.md +59 -0
  121. package/dist/assets/SaveBar-Qvc4Ek-H.js +0 -1
  122. package/dist/assets/Scheduler-BmYJvNzK.js +0 -14
  123. package/dist/assets/TagLibrary-DYJGAKZu.js +0 -1
@@ -53,9 +53,12 @@ const { ipcMain } = require('electron');
53
53
  const billing = require('./usage.cjs');
54
54
  const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
55
55
  const supervisor = require('./supervisor.cjs');
56
- const { resolveClaudeBin } = require('./lib/claudeBin.cjs');
56
+ const { resolveClaudeBin, probeClaudeVersion } = require('./lib/claudeBin.cjs');
57
+ const launchFailure = require('./lib/launchFailure.cjs');
58
+ const { appendError } = require('./lib/opsErrorLog.cjs');
57
59
  const { readTail } = require('./lib/fileTail.cjs');
58
- const { claudePidAlive, classifyRunOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
60
+ const { claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
61
+ const { computeQueueHealth } = require('./lib/queueHealth.cjs');
59
62
  const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
60
63
  const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
61
64
  const { sendIfAlive } = require('./lib/sendToRenderer.cjs');
@@ -68,6 +71,8 @@ const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs')
68
71
  const promptSessionTranscript = require('./promptSessionTranscript.cjs');
69
72
  const { verifyRun } = require('./runVerify.cjs');
70
73
  const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
74
+ const { landedSinceRun } = require('./lib/landedSinceRun.cjs');
75
+ const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
71
76
  const logs = require('./logs.cjs');
72
77
  const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
73
78
  const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
@@ -104,6 +109,7 @@ const queueOps = require('./queueOps.cjs');
104
109
  // home-dir layout.
105
110
  const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
106
111
  const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
112
+ const agentModelResolve = require('./lib/agentModelResolve.cjs');
107
113
  const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
108
114
  const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
109
115
  const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
@@ -127,6 +133,7 @@ function resolveOriginSessionId(cwd, epicId) {
127
133
  return session && typeof session.claudeSessionId === 'string' ? session.claudeSessionId : null;
128
134
  }
129
135
  const sessionSlots = require('./lib/sessionSlots.cjs');
136
+ const quietMachineLease = require('./lib/quietMachineLease.cjs');
130
137
  const jobWorktree = require('./lib/jobWorktree.cjs');
131
138
  const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
132
139
  const queueStore = require('./lib/queueStore.cjs');
@@ -184,6 +191,21 @@ const RESULT_TEXT_TAIL_BYTES = 64 * 1024;
184
191
  const IDLE_OUTPUT_KILL_MS = 20 * 60_000;
185
192
  const IDLE_CHECK_INTERVAL_MS = 60_000;
186
193
 
194
+ // Foreground Bash budget for every spawned `claude -p` job (executor +
195
+ // investigation). The Claude Code harness auto-backgrounds any foreground
196
+ // Bash command past its own default (120s) or max (600s) timeout and returns
197
+ // a tool result promising a later notification — but a headless single-shot
198
+ // run has no later turn, so that notification can never arrive and the run
199
+ // dead-ends mid-verification with no commit and no verdict. Raising these
200
+ // via the child's env moves that trap out of reach of normal gate commands
201
+ // (test suites, builds). BASH_MAX_TIMEOUT_MS MUST stay strictly below
202
+ // IDLE_OUTPUT_KILL_MS with real margin: a long foreground Bash emits no
203
+ // stream-json events while it runs, so the log mtime stalls and the
204
+ // idle-tail watchdog above would SIGTERM the job mid-gate if the two ever
205
+ // crossed — trading one silent failure for another.
206
+ const BASH_DEFAULT_TIMEOUT_MS = 600_000; // 10 min
207
+ const BASH_MAX_TIMEOUT_MS = 900_000; // 15 min — must stay below IDLE_OUTPUT_KILL_MS
208
+
187
209
  // Boot reconciliation: a job left 'running' by an app restart/crash whose log
188
210
  // shows neither success nor a real failure result was merely interrupted — the
189
211
  // host died, the PRD didn't. Re-queue it up to this many times before giving up
@@ -216,10 +238,12 @@ for it to return. Never start a verification command as a background task
216
238
  (no background Bash) and then call Monitor, TaskOutput, or ScheduleWakeup to
217
239
  pick up its result later — a headless \`claude -p\` run has no later turn, so
218
240
  nothing ever delivers that notification and the run dies mid-verification with
219
- no commit and no verdict. For a long-running command, bound it yourself with
220
- the shell (e.g. \`timeout 300 npm test\`) and a matching foreground tool
221
- timeout; if it still cannot finish inside budget, stop and emit
222
- SCHEDULER_VERDICT: FAIL with the reason instead of deferring it.
241
+ no commit and no verdict. Your foreground Bash budget for this run is
242
+ ${BASH_DEFAULT_TIMEOUT_MS / 1000}s by default, up to ${BASH_MAX_TIMEOUT_MS / 1000}s max
243
+ — size your own \`timeout <n>\` wrapper (e.g. \`timeout ${Math.floor(BASH_MAX_TIMEOUT_MS / 1000)} npm test\`)
244
+ to fit inside that ceiling; if a gate command still cannot finish inside
245
+ budget, stop and emit SCHEDULER_VERDICT: FAIL with the reason instead of
246
+ deferring it.
223
247
 
224
248
  1. CODE REVIEW — run \`/code-review --fix\` on your changes and apply the fixes it
225
249
  surfaces (correctness first). For any finding you judge a false positive, say
@@ -303,6 +327,156 @@ function gitHead(cwd) {
303
327
  });
304
328
  }
305
329
 
330
+ // Return the current `git stash list` entries in cwd as raw lines
331
+ // "<hash> <ref> <subject>" (hash is stable even as ref indices shift when a
332
+ // new entry is pushed on top), or null when the guard does not apply (cwd is
333
+ // not a git work tree, git is missing, or the call errors). Never throws.
334
+ function stashList(cwd) {
335
+ return new Promise((resolve) => {
336
+ if (!cwd) { resolve(null); return; }
337
+ execFile(
338
+ 'git',
339
+ ['-C', cwd, 'stash', 'list', '--format=%H %gd %gs'],
340
+ { timeout: 10_000, windowsHide: true },
341
+ (err, stdout) => {
342
+ if (err) { resolve(null); return; }
343
+ resolve(String(stdout || '').split('\n').filter(Boolean));
344
+ },
345
+ );
346
+ });
347
+ }
348
+
349
+ // Parse one `stashList()` line into { hash, ref, subject }. Pure, exported
350
+ // for unit testing. Returns null for a malformed line.
351
+ function parseStashLine(line) {
352
+ const m = /^(\S+)\s+(\S+)\s+(.*)$/.exec(String(line || ''));
353
+ return m ? { hash: m[1], ref: m[2], subject: m[3] } : null;
354
+ }
355
+
356
+ // Paths touched by any commit landed in cwd strictly between headBefore and
357
+ // headAfter. Returns [] when no commit landed (headBefore === headAfter, or
358
+ // either is missing) — used by the shared-tree guard below to tell a path
359
+ // the job legitimately committed apart from a path that just silently went
360
+ // quiet with nothing to explain it. Never throws.
361
+ function pathsChangedSince(cwd, headBefore, headAfter) {
362
+ return new Promise((resolve) => {
363
+ if (!cwd || !headBefore || !headAfter || headBefore === headAfter) { resolve([]); return; }
364
+ execFile(
365
+ 'git',
366
+ ['-C', cwd, 'diff', '--name-only', `${headBefore}..${headAfter}`],
367
+ { timeout: 10_000, windowsHide: true },
368
+ (err, stdout) => { resolve(err ? [] : String(stdout || '').split('\n').filter(Boolean)); },
369
+ );
370
+ });
371
+ }
372
+
373
+ // Restore ONE specific stash ref (never a blanket pop of "whatever is on
374
+ // top") into cwd: apply, then drop only on a clean apply. On conflict the
375
+ // entry is left in place — never dropped, never forced — so the operator's
376
+ // own `git stash pop`/`apply` still works afterward. Never throws.
377
+ function restoreSpecificStash(cwd, ref) {
378
+ return new Promise((resolve) => {
379
+ execFile('git', ['-C', cwd, 'stash', 'apply', ref], { timeout: 10_000, windowsHide: true }, (applyErr, _stdout, applyStderr) => {
380
+ if (applyErr) {
381
+ resolve({ ok: false, error: String(applyStderr || applyErr.message || applyErr).trim().split('\n')[0] });
382
+ return;
383
+ }
384
+ execFile('git', ['-C', cwd, 'stash', 'drop', ref], { timeout: 10_000, windowsHide: true }, () => {
385
+ resolve({ ok: true });
386
+ });
387
+ });
388
+ });
389
+ }
390
+
391
+ // Diff a before/after `stashList()` pair plus a before/after dirty-path pair
392
+ // to find what an in-place job silently discarded from a tree it shares with
393
+ // something else (Incident: social-signals-trader 2026-09-01, a blanket
394
+ // `git stash` reverted a live operator config edit with no error anywhere).
395
+ // Two independent signals, either of which means the job discarded state it
396
+ // did not create:
397
+ // - newStashes: a stash entry now present that wasn't in the baseline —
398
+ // the job ran `git stash` itself.
399
+ // - reverted: a path that was dirty in the baseline, is clean now, and was
400
+ // not touched by any commit landed during the run — the job reset/
401
+ // checked-out over pre-existing uncommitted work without stashing it.
402
+ // Pure/no I/O — the guard's git calls happen at the call site
403
+ // (checkSharedTreeGuard). Exported for unit testing.
404
+ function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAfter, pathsCommittedDuringRun }) {
405
+ const beforeHashes = new Set((stashBefore || []).map((l) => parseStashLine(l)?.hash).filter(Boolean));
406
+ const newStashes = (stashAfter || [])
407
+ .map(parseStashLine)
408
+ .filter((e) => e && !beforeHashes.has(e.hash));
409
+ const dirtyAfterSet = new Set(dirtyAfter || []);
410
+ const committedSet = new Set(pathsCommittedDuringRun || []);
411
+ const reverted = (dirtyBefore || []).filter((p) => !dirtyAfterSet.has(p) && !committedSet.has(p));
412
+ return { newStashes, reverted };
413
+ }
414
+
415
+ // Post-run shared-tree guard for an IN-PLACE job (worktree.ok === false —
416
+ // callers must gate on that; a worktree-isolated run's git state can never
417
+ // leak into guardCwd, so there is nothing here to check). Best-effort: never
418
+ // throws, never changes the job's exit code. Restores exactly one
419
+ // executor-created stash (never guesses when there are 2+); reports anything
420
+ // it can't safely resolve on the returned object so the caller can surface it
421
+ // on the job row instead of finishing silently green.
422
+ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug }) {
423
+ try {
424
+ const [stashAfter, headAfter] = await Promise.all([
425
+ module.exports.stashList(cwd),
426
+ module.exports.gitHead(cwd),
427
+ ]);
428
+ const pathsCommittedDuringRun = await module.exports.pathsChangedSince(cwd, headBefore, headAfter);
429
+ // First pass: which stashes are new. Decided before charging anything
430
+ // against dirtyBaseline — a path this run's own stash covers must not be
431
+ // judged "reverted" using dirty state captured before the restore below
432
+ // has had a chance to bring it back.
433
+ const { newStashes } = module.exports.evaluateSharedTreeGuard({
434
+ stashBefore: stashBaseline,
435
+ stashAfter,
436
+ dirtyBefore: [],
437
+ dirtyAfter: [],
438
+ pathsCommittedDuringRun,
439
+ });
440
+
441
+ const result = {};
442
+ if (newStashes.length === 1) {
443
+ const [entry] = newStashes;
444
+ const restore = await module.exports.restoreSpecificStash(cwd, entry.ref);
445
+ if (restore.ok) {
446
+ result.restoredStash = entry.ref;
447
+ console.log(`[scheduler] ${slug}: restored a stash the job created in the shared tree (${entry.ref})`);
448
+ } else {
449
+ result.restoreFailed = `${entry.ref}: ${restore.error || 'apply failed'}`;
450
+ console.error(`[scheduler] ${slug}: shared-tree guard could not restore ${entry.ref}: ${restore.error}`);
451
+ }
452
+ } else if (newStashes.length > 1) {
453
+ result.ambiguousStashes = newStashes.map((e) => e.ref);
454
+ console.error(`[scheduler] ${slug}: shared-tree guard found ${newStashes.length} stashes the job created — ambiguous, not auto-restoring (${result.ambiguousStashes.join(', ')})`);
455
+ }
456
+
457
+ // Second pass: recompute "reverted" against the tree's dirty state AFTER
458
+ // any restore attempt above, so a path that came back via a successfully
459
+ // restored stash is not ALSO reported as an unexplained revert (it was
460
+ // explained — by the stash this guard just restored).
461
+ const dirtyAfter = await module.exports.uncommittedChanges(cwd);
462
+ const { reverted } = module.exports.evaluateSharedTreeGuard({
463
+ stashBefore: stashBaseline,
464
+ stashAfter,
465
+ dirtyBefore: dirtyBaseline,
466
+ dirtyAfter,
467
+ pathsCommittedDuringRun,
468
+ });
469
+ if (reverted.length) {
470
+ result.reverted = reverted;
471
+ console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
472
+ }
473
+ return (result.restoredStash || result.restoreFailed || result.ambiguousStashes || result.reverted) ? result : null;
474
+ } catch (e) {
475
+ console.error(`[scheduler] ${slug}: shared-tree guard error`, e);
476
+ return null;
477
+ }
478
+ }
479
+
306
480
  // True when cwd is inside a git repository. Used to keep a non-git cwd (e.g.
307
481
  // a scratch dir like /tmp) from ever being handed to an investigation's
308
482
  // fix-plan as its cwd — the commit guard, worktree isolation, and
@@ -1563,9 +1737,11 @@ async function reconcile(state) {
1563
1737
  // membership, so moving the file between Epic dirs must re-point the row.
1564
1738
  epicId: p.epicId ?? job.epicId ?? null,
1565
1739
  dependsOn: p.dependsOn,
1740
+ quietMachine: p.quietMachine === true,
1566
1741
  originSessionId: job.originSessionId
1567
1742
  ?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
1568
1743
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1744
+ agentType: p.agentType ?? job.agentType ?? null,
1569
1745
  };
1570
1746
  // Adopt path: a row parked 'quarantined' (no createdVia provenance when
1571
1747
  // discovered) whose PRD file now carries a stamp — written via the
@@ -1674,8 +1850,10 @@ async function reconcile(state) {
1674
1850
  sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
1675
1851
  epicId: p.epicId ?? inv.row?.epicId ?? null,
1676
1852
  dependsOn: p.dependsOn,
1853
+ quietMachine: p.quietMachine === true,
1677
1854
  originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
1678
1855
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1856
+ agentType: p.agentType ?? inv.row?.agentType ?? null,
1679
1857
  };
1680
1858
  const reason = `reconcile: repaired invalid status ${JSON.stringify(oldStatus)}`;
1681
1859
  // A repair is not a lifecycle transition — the corrupted `status` was
@@ -1794,8 +1972,10 @@ async function reconcile(state) {
1794
1972
  sourceTabId: p.sourceTabId,
1795
1973
  epicId: p.epicId ?? null,
1796
1974
  dependsOn: p.dependsOn,
1975
+ quietMachine: p.quietMachine === true,
1797
1976
  originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
1798
1977
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1978
+ agentType: p.agentType ?? null,
1799
1979
  status: 'pending',
1800
1980
  // Enqueue time (PRD 1086/1087): the cross-project fairness tiebreak and
1801
1981
  // the starvation escalation both need a provable age for a pending row;
@@ -2051,6 +2231,10 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
2051
2231
  lastRunAt: state.lastRunAt,
2052
2232
  nextReset: getNextResetCached(),
2053
2233
  paused: state.paused,
2234
+ // Launch circuit breaker (issue #11): which personas cannot launch right
2235
+ // now and why, plus any degraded-mode env in force. Empty objects when healthy.
2236
+ launchBlocks: state.launchBlocks ?? {},
2237
+ launchMitigations: state.launchMitigations ?? {},
2054
2238
  utilization: cachedUtilization,
2055
2239
  pollHealth: {
2056
2240
  lastPollAt,
@@ -2194,7 +2378,17 @@ async function setPaused(reason, resumeAtIso) {
2194
2378
 
2195
2379
  async function clearPause(source) {
2196
2380
  if (resumeTimer) { clearTimeout(resumeTimer); resumeTimer = null; }
2381
+ const humanOverride = source === 'manual' || source === 'run-now';
2197
2382
  const wasPaused = await mutate((s) => {
2383
+ // A human Resume / Run now also re-closes every launch circuit breaker:
2384
+ // the operator is asserting the environment is fixed (CLI updated,
2385
+ // re-logged-in). The next dispatch of each persona is its probe; if the
2386
+ // environment is still broken the breaker simply re-arms.
2387
+ if (humanOverride && s.launchBlocks && Object.keys(s.launchBlocks).length) {
2388
+ console.log(`[scheduler] clearPause (${source}): clearing launch blocks [${Object.keys(s.launchBlocks).join(', ')}]`);
2389
+ for (const j of s.jobs) if (j.status === 'pending' && j.heldReason && /^launch blocked/.test(j.heldReason)) delete j.heldReason;
2390
+ s.launchBlocks = {};
2391
+ }
2198
2392
  if (!s.paused) return false;
2199
2393
  console.log(`[scheduler] clearPause (${source || 'manual'})`);
2200
2394
  s.paused = null;
@@ -2242,6 +2436,29 @@ function resetJobFields(job, errorMsg, opts = {}) {
2242
2436
  job.error = errorMsg ?? null;
2243
2437
  delete job.runtime;
2244
2438
  delete job.verifierVerdict;
2439
+ delete job.uncommittedPaths;
2440
+ delete job.resumeRecoveryAttempted;
2441
+ // Same "this run's outcome, not durable across a reset" category as the
2442
+ // fields above — a stale 'archive' recoveryAction from a prior life of this
2443
+ // slug must never survive a reset and silently exclude a genuinely-new
2444
+ // needs_review episode from selectAutoFixTargets (applyRcaClassification
2445
+ // only overwrites these on a successful RCA write, so without this they
2446
+ // can otherwise linger forever when RCA is disabled or errors).
2447
+ delete job.rcaFailureClass;
2448
+ delete job.rcaRecoveryAction;
2449
+ // Like exitCode: this run's outcome, not durable across a reset — a stale
2450
+ // leak badge from a prior attempt must not linger once the job re-fires.
2451
+ delete job.leakedDescendants;
2452
+ // A pending row is about to re-run fresh — a stale leftover badge or a
2453
+ // stale pre-run baseline from the attempt that just ended must not linger
2454
+ // and be mistaken for THIS (not-yet-run) attempt's own output. spawnJob
2455
+ // persists a brand-new guardBaseline at the next dispatch.
2456
+ delete job.guardBaseline;
2457
+ delete job.guardHeadBefore;
2458
+ delete job.leftoverPaths;
2459
+ delete job.leftoverCount;
2460
+ delete job.leftoverPathsTruncated;
2461
+ delete job.preRunDirtyPaths;
2245
2462
  // Deliberately NOT deleting job.landedCommit: it must outlive a reset so a
2246
2463
  // re-fired run of this same slug can pass it to verifyRun as
2247
2464
  // priorLandedCommit (pass_no_commit_prior_run_verified exemption).
@@ -2792,7 +3009,190 @@ function commitGuardVerdict({ newlyDirty, siblingRunning, ranInWorktree, jobSelf
2792
3009
  reason: `finish protocol incomplete: ${dirty.length} uncommitted file(s) left in working tree (e.g. ${sample})${salvageNote}`,
2793
3010
  downgradeTo: 'needs_review',
2794
3011
  annotations: carried.length ? carried : undefined,
3012
+ // The exact dirty-path list, persisted on the job row (see the
3013
+ // commit-guard call site) so a later resume-recovery attempt
3014
+ // (selectResumeRecoveryTarget) can name these paths without re-running
3015
+ // `git status` against a tree that may have moved on since.
3016
+ dirtyPaths: dirty,
3017
+ };
3018
+ }
3019
+
3020
+ // Every path list this job leaves attributed on the row is capped here so a
3021
+ // pathological run (thousands of newly-dirty files) never bloats queue.json
3022
+ // or history.jsonl — the count is still recorded in full via leftoverCount,
3023
+ // only the displayed sample is capped.
3024
+ const LEFTOVER_PATHS_CAP = 50;
3025
+
3026
+ /**
3027
+ * Pure: turn a newly-dirty path list (or null, meaning "couldn't tell" —
3028
+ * never "left nothing") into the `leftoverPaths`/`leftoverCount`/
3029
+ * `leftoverPathsTruncated` triple stamped on a terminal job row, or null when
3030
+ * there is nothing to attribute (empty list, or the list itself is
3031
+ * unavailable). One shape for both the worktree-leftover path and the
3032
+ * in-place baseline-delta path — see this function's callers in spawnJob and
3033
+ * reapDeadRunningJobs, both of which diff against a persisted pre-run
3034
+ * baseline so a human's or a sibling's pre-existing WIP is never
3035
+ * misattributed to this job.
3036
+ */
3037
+ function leftoverFieldsFrom(paths) {
3038
+ if (!Array.isArray(paths) || paths.length === 0) return null;
3039
+ const fields = {
3040
+ leftoverPaths: paths.slice(0, LEFTOVER_PATHS_CAP),
3041
+ leftoverCount: paths.length,
2795
3042
  };
3043
+ if (paths.length > LEFTOVER_PATHS_CAP) fields.leftoverPathsTruncated = true;
3044
+ return fields;
3045
+ }
3046
+
3047
+ /** Stamps (or clears) the leftover-attribution fields on a job row in place. */
3048
+ function applyLeftoverFields(row, paths) {
3049
+ delete row.leftoverPaths;
3050
+ delete row.leftoverCount;
3051
+ delete row.leftoverPathsTruncated;
3052
+ const fields = leftoverFieldsFrom(paths);
3053
+ if (fields) Object.assign(row, fields);
3054
+ }
3055
+
3056
+ // Same bloat concern as LEFTOVER_PATHS_CAP, applied to the PRE-run dirty
3057
+ // snapshot (foreign WIP the job did not create) instead of the post-run
3058
+ // leftover delta.
3059
+ const PRE_RUN_DIRTY_PATHS_CAP = 200;
3060
+
3061
+ /**
3062
+ * Pure: cap a dirty-path list at PRE_RUN_DIRTY_PATHS_CAP, appending a
3063
+ * `+N more` marker entry when truncated, so queue.json/history.jsonl never
3064
+ * take on an unbounded row for a pathologically dirty shared tree. Returns
3065
+ * [] for null/empty input (never null) — callers gate storage/prompt
3066
+ * injection on `.length` the same way carriedPaths already does.
3067
+ */
3068
+ function capDirtyPaths(paths, cap = PRE_RUN_DIRTY_PATHS_CAP) {
3069
+ if (!Array.isArray(paths) || paths.length === 0) return [];
3070
+ if (paths.length <= cap) return paths.slice();
3071
+ return [...paths.slice(0, cap), `+${paths.length - cap} more`];
3072
+ }
3073
+
3074
+ // Stable, machine-greppable delimiter — a downstream PRD (verifier scoring
3075
+ // foreign-WIP test failures separately) greps the executor log for this
3076
+ // exact marker, so its text must never be reworded casually.
3077
+ const FOREIGN_WIP_DELIMITER = '--- FOREIGN WORKING-TREE STATE (not your work) ---';
3078
+ const FOREIGN_WIP_END_DELIMITER = '--- END FOREIGN WORKING-TREE STATE ---';
3079
+
3080
+ /**
3081
+ * Pure: build the executor-prompt section warning about pre-existing dirty
3082
+ * paths this job does not own — either base WIP carried into an isolated
3083
+ * worktree (PRD 1094's carriedPaths, checked first since it's the more
3084
+ * specific/authoritative case) or the raw pre-run dirty snapshot of a shared
3085
+ * (non-isolated) tree. Returns '' when both lists are empty so a clean spawn
3086
+ * produces a byte-identical prompt to before this section existed.
3087
+ */
3088
+ function buildForeignWipSection({ preRunDirtyPaths, carriedPaths } = {}) {
3089
+ const carried = Array.isArray(carriedPaths) ? carriedPaths.filter(Boolean) : [];
3090
+ if (carried.length) {
3091
+ return [
3092
+ FOREIGN_WIP_DELIMITER,
3093
+ 'This job is running in an isolated git worktree, but the following paths carry uncommitted base-tree work-in-progress that was carried into this checkout so the tree is self-consistent. The authoritative copy of these files lives in the MAIN tree, not this worktree.',
3094
+ 'These files were already modified before this job started. They are NOT this job\'s work:',
3095
+ ...carried.map((p) => ` ${p}`),
3096
+ 'Do not stage, commit, revert, or stash these paths. A test failure confined to these paths is not this job\'s regression.',
3097
+ FOREIGN_WIP_END_DELIMITER,
3098
+ ].join('\n');
3099
+ }
3100
+ const dirty = Array.isArray(preRunDirtyPaths) ? preRunDirtyPaths.filter(Boolean) : [];
3101
+ if (dirty.length) {
3102
+ return [
3103
+ FOREIGN_WIP_DELIMITER,
3104
+ 'This job is running in a SHARED working tree (not isolated in its own worktree). The following paths were already modified when this job started:',
3105
+ ...dirty.map((p) => ` ${p}`),
3106
+ 'These files are NOT this job\'s work. Do not stage, commit, revert, or stash them. A test failure confined to these paths is not this job\'s regression.',
3107
+ FOREIGN_WIP_END_DELIMITER,
3108
+ ].join('\n');
3109
+ }
3110
+ return '';
3111
+ }
3112
+
3113
+ /**
3114
+ * Resume-first recovery (PRD 1111). A job parked in needs_review with verdict
3115
+ * 'uncommitted_changes' has a live claude session (job.sessionId, minted by
3116
+ * spawnJob's `--session-id`) that already has full context of the work it
3117
+ * left uncommitted — resuming it via `claude -p --resume <sessionId>` lets it
3118
+ * finish its own finish-protocol COMMIT step, instead of spawnInvestigation
3119
+ * cold-reading the log to author a fix-plan PRD that a FRESH session then has
3120
+ * to re-derive that same context for. Pure/no I/O so the eligibility rule can
3121
+ * be unit-tested directly, matching classifyFailureOutcome/commitGuardVerdict.
3122
+ *
3123
+ * Bounded to exactly one attempt via job.resumeRecoveryAttempted, stamped
3124
+ * atomically with the 'running' transition inside spawnJob's own dispatch
3125
+ * mutate (see spawnJob) — never here — so a crash between this function
3126
+ * returning a target and the resume child actually spawning cannot leave the
3127
+ * job re-eligible.
3128
+ *
3129
+ * Kill-switch: SM_RESUME_RECOVERY_DISABLE=1 restores today's behaviour
3130
+ * exactly (always returns null), mirroring SM_RCA_DISABLE/SM_DOD_DISABLE.
3131
+ */
3132
+ function selectResumeRecoveryTarget(job) {
3133
+ if (process.env.SM_RESUME_RECOVERY_DISABLE === '1') return null;
3134
+ if (!job || job.status !== 'needs_review') return null;
3135
+ if (job.verifierVerdict !== 'uncommitted_changes') return null;
3136
+ if (typeof job.sessionId !== 'string' || job.sessionId.length === 0) return null;
3137
+ if (job.resumeRecoveryAttempted === true) return null;
3138
+ const dirtyPaths = Array.isArray(job.uncommittedPaths)
3139
+ ? job.uncommittedPaths.filter((p) => typeof p === 'string' && p.length > 0)
3140
+ : [];
3141
+ if (!dirtyPaths.length) return null;
3142
+ return { slug: job.slug, sessionId: job.sessionId, dirtyPaths, salvagePatch: job.salvagePatch || null };
3143
+ }
3144
+
3145
+ /**
3146
+ * Short deterministic preamble for a resume-recovery dispatch — NEVER the
3147
+ * original PRD body (the resumed session already has that in its own
3148
+ * conversation history; re-embedding it would just waste context and risk
3149
+ * contradicting whatever state the session actually left behind). Names the
3150
+ * exact paths recorded on the parked job row so the resumed run can verify
3151
+ * them on disk before trusting them, rather than re-deriving them itself.
3152
+ */
3153
+ function buildResumeRecoveryPreamble({ dirtyPaths, salvagePatch }) {
3154
+ const pathList = dirtyPaths.map((p) => `- ${p}`).join('\n');
3155
+ const salvageLine = salvagePatch
3156
+ ? `\nA salvage patch of this work was also captured at: ${salvagePatch} — apply it if any of the paths above are missing from the working tree.\n`
3157
+ : '';
3158
+ return `RESUME RECOVERY: your previous run in this same session left uncommitted work on disk and exited before the finish protocol's COMMIT step ran. This is a continuation of that same session, not a new task — do not restart from scratch.
3159
+
3160
+ The following path(s) were recorded as uncommitted when this job was parked for review:
3161
+ ${pathList}
3162
+ ${salvageLine}
3163
+ Do the following now:
3164
+ 1. Run \`git status\` and verify each path above is present on disk and reflects your intended work. If a path is missing, investigate before recreating it — don't blindly redo work that may already be committed or salvaged elsewhere.
3165
+ 2. Run the project's verification gate (typecheck/lint/tests) in the FOREGROUND — wait for it to finish and read its real exit code before proceeding. Do not background it.
3166
+ 3. If the gate is green, stage exactly the paths you created or modified for this work and commit them: \`git add <path> [<path>...] && git commit -m "<type>(<scope>): <summary>"\`.
3167
+ 4. If the gate is red, fix it, then commit.
3168
+
3169
+ As the LAST LINE of your final result text, emit exactly one of:
3170
+ SCHEDULER_VERDICT: PASS
3171
+ SCHEDULER_VERDICT: FAIL <one-line reason>
3172
+ Print PASS only once the commit above has actually landed.`;
3173
+ }
3174
+
3175
+ /**
3176
+ * Pure argv builder for a `claude -p` child spawn, shared so the
3177
+ * resume-vs-fresh-session choice is made in exactly one place. `resume`
3178
+ * selects `--resume <sessionId>` (reconnect) INSTEAD of `--session-id
3179
+ * <sessionId>` (mint) — the two flags are mutually exclusive, never both.
3180
+ * `--model` is always explicit (never left to the CLI's drifting default —
3181
+ * see conventions.md). `systemPrompt`, when given (the PRD's `agentType`
3182
+ * persona body, resolved by agentModelResolve.cjs's resolvePrdPersonaForSpawn),
3183
+ * is passed as `--append-system-prompt` so the executor IS that persona at
3184
+ * launch rather than being asked in prose to adopt one.
3185
+ */
3186
+ function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }) {
3187
+ return [
3188
+ '-p', prompt,
3189
+ '--model', model,
3190
+ ...(systemPrompt ? ['--append-system-prompt', systemPrompt] : []),
3191
+ '--dangerously-skip-permissions',
3192
+ '--output-format', 'stream-json',
3193
+ '--verbose',
3194
+ ...(resume ? ['--resume', sessionId] : ['--session-id', sessionId]),
3195
+ ];
2796
3196
  }
2797
3197
 
2798
3198
  // ---------- execution ----------
@@ -2813,7 +3213,7 @@ function pickRunDir() {
2813
3213
  * Watchdogs are declared as an array; the result-tailer's exit-code mapping
2814
3214
  * (success+killedBySignal → 0) is scheduler-specific and lives in onExit.
2815
3215
  */
2816
- async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
3216
+ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget = null, foreignWip = null, launchEnv = null) {
2817
3217
  const logPath = path.join(runDir, `${job.slug}.log`);
2818
3218
  const metaPath = path.join(runDir, `${job.slug}.meta.json`);
2819
3219
  // `cwd` stays the MAIN tree throughout — PRD lookup (findPrdDir/prdPathForJob)
@@ -2823,7 +3223,10 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2823
3223
  const cwd = job.cwd || defaultCwd;
2824
3224
  const spawnCwd = execCwd || cwd;
2825
3225
  const startedAt = Date.now();
2826
- const sessionId = randomUUID();
3226
+ // Resume mode (PRD 1111) reconnects to the SAME session that left the
3227
+ // uncommitted work — reusing its id via `--resume` instead of minting a
3228
+ // fresh one via `--session-id` is the entire point of the recovery.
3229
+ const sessionId = resumeTarget ? resumeTarget.sessionId : randomUUID();
2827
3230
 
2828
3231
  // Phase 1: open log fd so we can emit pre-spawn diagnostics (early-exit
2829
3232
  // error paths) before the child is created. withChildAndLog takes ownership
@@ -2848,14 +3251,23 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2848
3251
  return { exitCode: -1, durationMs: 0, error: errMsg, sessionId };
2849
3252
  }
2850
3253
 
3254
+ let prompt;
3255
+ let prdPath = null;
3256
+ if (resumeTarget) {
3257
+ // Resume mode (PRD 1111): a short deterministic preamble naming the
3258
+ // recorded dirty paths, NEVER the original PRD body — the resumed
3259
+ // session already has that in its own conversation history via
3260
+ // --resume, and re-embedding it here would just contradict whatever
3261
+ // state the session actually left on disk.
3262
+ prompt = buildResumeRecoveryPreamble({ dirtyPaths: resumeTarget.dirtyPaths, salvagePatch: resumeTarget.salvagePatch });
3263
+ } else {
2851
3264
  // Read full PRD body fresh from disk (queue stored only the preview).
2852
3265
  // Resolve through findPrdDir's full candidate search (legacy flat dir +
2853
3266
  // every project's Epic-scoped dirs) first, so the common case — a live
2854
3267
  // Epic-scoped PRD — is a first-try hit instead of probing the retired flat
2855
3268
  // dir and only then falling back.
2856
- let prompt;
2857
3269
  const resolvedDir = await findPrdDir(job.slug);
2858
- let prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
3270
+ prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
2859
3271
  try {
2860
3272
  const parsed = await parsePrd(prdPath);
2861
3273
  // The review → security-review → verify → commit finish sequence is
@@ -2911,14 +3323,17 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2911
3323
  return { exitCode: -1, durationMs: 0, error: e?.message };
2912
3324
  }
2913
3325
  }
3326
+ } // end resumeTarget ? preamble : normal-PRD-read
2914
3327
 
3328
+ let contextDigestApplied = false;
3329
+ let originSessionId = null;
3330
+ if (!resumeTarget) {
2915
3331
  // Prepend the Epic's own session digest (PRD 950/958) when this job traces
2916
3332
  // back to a known Epic — additive only, never mutates the PRD body itself.
2917
3333
  // A missing/unresolved epicId or a digest build failure is a silent no-op:
2918
3334
  // the PRD's own body must remain sufficient to complete the job on its own.
2919
3335
  const digestEpicId = job.epicId ?? job.sourcePromptId ?? null;
2920
- const originSessionId = resolveOriginSessionId(cwd, digestEpicId);
2921
- let contextDigestApplied = false;
3336
+ originSessionId = resolveOriginSessionId(cwd, digestEpicId);
2922
3337
  let digestText = '';
2923
3338
  if (originSessionId) {
2924
3339
  try {
@@ -2929,12 +3344,39 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2929
3344
  digestText = '';
2930
3345
  }
2931
3346
  }
3347
+ // Quiet-machine degraded dispatch (PRD 1107): this job opted into
3348
+ // `quietMachine: true` but waited past quietMachineWaitMs() without the
3349
+ // machine ever going quiet, so pickNextBatch dispatched it anyway rather
3350
+ // than wedge the queue forever. Told to the executor as a plain prompt
3351
+ // line — its own wall-clock/timing acceptance criteria were measured (or
3352
+ // will be measured) under CPU contention from sibling jobs, not on a
3353
+ // quiet machine, so it should not report a timing result as trustworthy
3354
+ // without saying so.
3355
+ if (job.quietLeaseDegraded === true) {
3356
+ prompt = `NOTE: this job requested \`quietMachine: true\` but the machine never went idle within the `
3357
+ + `configured wait window, so it was dispatched anyway (degraded). Any timing/frame-rate/performance `
3358
+ + `measurement in this run may be affected by CPU contention from other concurrent jobs — say so explicitly `
3359
+ + `in your result rather than reporting it as a clean measurement.\n\n${prompt}`;
3360
+ }
2932
3361
  // Always route through composeExecutorPrompt (even with an empty digest)
2933
3362
  // so the finish protocol is appended in the prompt's tail exactly once,
2934
3363
  // after any digest fence rather than concatenated ahead of it.
2935
3364
  prompt = composeExecutorPrompt({ prdBody: prompt, digestText, finishProtocol: FINISH_PROTOCOL });
2936
3365
 
2937
- const promptCheck = validatePromptForSpawn(prompt, prdPath);
3366
+ // Foreign-WIP manifest (starry-night-ships PRD 148 postmortem): the
3367
+ // scheduler already knows, at spawn time, which dirty paths this job did
3368
+ // not create — either a shared tree's pre-existing dirty set or worktree
3369
+ // WIP carried in from the base tree (PRD 1094). Telling the executor
3370
+ // explicitly here means it never has to bisect by content to prove a test
3371
+ // failure isn't its own regression. '' (clean spawn) leaves prompt
3372
+ // byte-identical to before this section existed.
3373
+ const foreignWipSection = buildForeignWipSection(foreignWip || {});
3374
+ if (foreignWipSection) {
3375
+ prompt = `${prompt}\n\n${foreignWipSection}`;
3376
+ }
3377
+ } // end !resumeTarget digest/finish-protocol composition
3378
+
3379
+ const promptCheck = validatePromptForSpawn(prompt, resumeTarget ? `<resume recovery preamble for ${job.slug}>` : prdPath);
2938
3380
  if (!promptCheck.ok) {
2939
3381
  safeLog(`[scheduler] ${promptCheck.error}\n`);
2940
3382
  closeFd();
@@ -2942,6 +3384,15 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2942
3384
  return { exitCode: -1, durationMs: 0, error: promptCheck.error, sessionId };
2943
3385
  }
2944
3386
 
3387
+ // PRD agentType → persona + model (PRD 1115): resolved for both fresh and
3388
+ // resume dispatches, keyed off job.agentType (persisted on the queue row
3389
+ // by reconcile()) rather than re-reading the PRD file — a resumed session
3390
+ // must keep launching as the SAME persona it started as. Never throws;
3391
+ // a dangling/absent agentType falls back to no persona + FALLBACK_MODEL
3392
+ // and is logged once by resolvePrdPersonaForSpawn itself.
3393
+ const personaResolution = await agentModelResolve.resolvePrdPersonaForSpawn({ cwd, agentType: job.agentType });
3394
+ safeLog(`[scheduler] agentType=${job.agentType || '(none)'} persona=${personaResolution.personaPath || '(fallback — no persona applied)'} model=${personaResolution.model}\n`);
3395
+
2945
3396
  return await new Promise((resolve) => {
2946
3397
  const claudeBin = resolveClaudeBin();
2947
3398
  // Strip Claude Code env and secrets that leak in when session-manager is
@@ -2954,7 +3405,26 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2954
3405
  // originProjectRoot so a job running inside its own worktree can still
2955
3406
  // resolve the real project for create-prd/open-session/readiness. See
2956
3407
  // projectRootResolve.cjs.
2957
- const childEnv = cleanChildEnv({ PATH: pathWithUserBins(), SM_PROJECT_ROOT: cwd });
3408
+ // SM_SCHEDULER_JOB_SLUG marks the child (and the MCP servers it
3409
+ // inherits its env to) as a headless scheduled executor, so
3410
+ // scheduler-mcp-server.cjs can refuse scheduler_create_prd from inside a
3411
+ // run (issue #11 list C1 — the PRD 460 self-queue incident). Only a
3412
+ // persona whose whole job is decomposition may still queue.
3413
+ // `launchEnv` is the launch circuit breaker's degraded-mode env (e.g.
3414
+ // MAX_THINKING_TOKENS=0 while an outdated CLI's thinking parameter is
3415
+ // being rejected — lib/launchFailure.cjs); applied last so it wins.
3416
+ const childEnv = cleanChildEnv({
3417
+ PATH: pathWithUserBins(),
3418
+ SM_PROJECT_ROOT: cwd,
3419
+ SM_SCHEDULER_JOB_SLUG: job.slug,
3420
+ SM_SCHEDULER_JOB_MAY_QUEUE: job.agentType === 'architect' ? '1' : '0',
3421
+ BASH_DEFAULT_TIMEOUT_MS: String(BASH_DEFAULT_TIMEOUT_MS),
3422
+ BASH_MAX_TIMEOUT_MS: String(BASH_MAX_TIMEOUT_MS),
3423
+ ...(launchEnv && typeof launchEnv === 'object' ? launchEnv : {}),
3424
+ });
3425
+ if (launchEnv && Object.keys(launchEnv).length) {
3426
+ safeLog(`[scheduler] launch mitigation env applied: ${Object.entries(launchEnv).map(([k, v]) => `${k}=${v}`).join(' ')}\n`);
3427
+ }
2958
3428
 
2959
3429
  // Track whether the agent has emitted a `result` event in its JSONL stream.
2960
3430
  // null until seen; then one of "success" | "error_max_turns" | … per the
@@ -3061,14 +3531,16 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
3061
3531
  closeFd,
3062
3532
  spawn: {
3063
3533
  command: claudeBin,
3064
- args: [
3065
- '-p', prompt,
3066
- '--model', 'sonnet',
3067
- '--dangerously-skip-permissions',
3068
- '--output-format', 'stream-json',
3069
- '--verbose',
3070
- '--session-id', sessionId,
3071
- ],
3534
+ // Resume mode passes `--resume <sessionId>` (reconnect to the SAME
3535
+ // session) INSTEAD of `--session-id <sessionId>` (mint a new one) —
3536
+ // never both, see buildClaudeSpawnArgs.
3537
+ args: buildClaudeSpawnArgs({
3538
+ prompt,
3539
+ model: personaResolution.model,
3540
+ sessionId,
3541
+ resume: !!resumeTarget,
3542
+ systemPrompt: personaResolution.systemPrompt,
3543
+ }),
3072
3544
  options: {
3073
3545
  cwd: spawnCwd,
3074
3546
  env: childEnv,
@@ -3081,8 +3553,13 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
3081
3553
  },
3082
3554
  },
3083
3555
  watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog],
3084
- onExit({ exitCode, signal, killedByWatchdog: _kbw, error, spawnFailed, safeLog: sl }) {
3556
+ onExit({ exitCode, signal, killedByWatchdog: _kbw, error, spawnFailed, leakedDescendants, safeLog: sl }) {
3085
3557
  const durationMs = Date.now() - startedAt;
3558
+ const leaked = leakedDescendants ?? [];
3559
+ if (leaked.length > 0) {
3560
+ sl(`\n[scheduler] leaked ${leaked.length} descendant(s) swept from job process group: ` +
3561
+ `${leaked.map((p) => `pid=${p.pid} comm=${p.comm} pcpu=${p.pcpu} etimes=${p.etimes}s`).join(', ')}\n`);
3562
+ }
3086
3563
 
3087
3564
  if (error) {
3088
3565
  // Covers both synchronous spawn failure and child 'error' events.
@@ -3092,8 +3569,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
3092
3569
  sl(`\n[scheduler] ${errMsg}\n`);
3093
3570
  // Sync write: inside a Promise executor callback; must flush meta
3094
3571
  // before resolve() so the spawnJob mutate() that follows sees it.
3095
- config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA, originSessionId, contextDigestApplied });
3096
- resolve({ exitCode: -1, durationMs, error: errMsg, sessionId });
3572
+ config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA, originSessionId, contextDigestApplied });
3573
+ resolve({ exitCode: -1, durationMs, error: errMsg, leakedDescendants: leaked, sessionId });
3097
3574
  return;
3098
3575
  }
3099
3576
 
@@ -3116,16 +3593,33 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
3116
3593
  `duration=${Math.round(durationMs / 1000)}s\n`);
3117
3594
  const rateLimited = effectiveCode !== 0 && detectRateLimitInLog(logPath);
3118
3595
  const networkError = effectiveCode !== 0 && !rateLimited && detectNetworkErrorInLog(logPath);
3596
+ // Non-run detection (issue #11 lists A1–A3): the harness's `result`
3597
+ // event tells us whether the model ever got a turn. A first-request
3598
+ // API rejection (num_turns ≤ 1, output_tokens 0, `API Error:` text)
3599
+ // is a broken ENVIRONMENT, not a failed PRD — spawnJob routes it to
3600
+ // the launch circuit breaker instead of failed/investigation.
3601
+ const resultStats = launchFailure.readResultEvent(logPath);
3602
+ const launchFailed = (effectiveCode !== 0 && !rateLimited && !networkError)
3603
+ ? launchFailure.classifyLaunchFailure(resultStats)
3604
+ : null;
3605
+ if (launchFailed) {
3606
+ sl(`\n[scheduler] LAUNCH FAILURE (${launchFailed.kind}${launchFailed.httpStatus ? ` HTTP ${launchFailed.httpStatus}` : ''}): ` +
3607
+ `${launchFailed.message} — no turn was taken; this is not a PRD failure\n`);
3608
+ }
3119
3609
  // Sync write: child 'exit' handler must flush meta before resolve()
3120
3610
  // so the spawnJob mutate() that follows sees the persisted exit code.
3121
3611
  config.writeJsonSync(metaPath, {
3122
3612
  slug: job.slug, cwd, sessionId, exitCode: effectiveCode, rateLimited, networkError,
3123
- startedAt, finishedAt: Date.now(), durationMs,
3613
+ launchFailure: launchFailed,
3614
+ numTurns: resultStats?.numTurns ?? null, outputTokens: resultStats?.outputTokens ?? null,
3615
+ totalCostUsd: resultStats?.totalCostUsd ?? null, terminalReasonFromHarness: resultStats?.terminalReason ?? null,
3616
+ launchEnvApplied: launchEnv && Object.keys(launchEnv).length ? Object.keys(launchEnv) : [],
3617
+ startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
3124
3618
  agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
3125
3619
  schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
3126
3620
  originSessionId, contextDigestApplied,
3127
3621
  });
3128
- resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, sessionId });
3622
+ resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, launchFailure: launchFailed, resultStats, leakedDescendants: leaked, sessionId });
3129
3623
  },
3130
3624
  });
3131
3625
 
@@ -3190,7 +3684,27 @@ function healTargetForFix(fixSlug, jobs) {
3190
3684
  * spawnInvestigation computes.
3191
3685
  */
3192
3686
  function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group }) {
3193
- return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.
3687
+ const abandonedBackgroundTaskNote = failedJob.verifierVerdict === 'abandoned_background_task' ? `
3688
+
3689
+ # Known failure class: abandoned background task
3690
+ This job's verifier verdict is \`abandoned_background_task\`: the transcript shows a Bash command
3691
+ auto-backgrounded past its foreground timeout, and the run ended waiting for a "you will be
3692
+ notified when it completes" callback a headless run structurally cannot receive. This is NOT
3693
+ evidence the work failed — it is evidence the run stopped short of its finish protocol. The work is
3694
+ usually already written and correct; only the commit is missing.
3695
+
3696
+ By the time this investigation runs, the failed job's isolated worktree (if it ran in one) has
3697
+ already been cleaned up — \`${cwd}\` is the BASE repo, not that worktree, so a plain \`git status\`/
3698
+ \`git diff\` there will usually show nothing even though real work was produced. The scheduler
3699
+ salvages any uncommitted diff from a killed job's worktree BEFORE deleting it${
3700
+ failedJob.salvagePatch ? `, and this job's salvage patch was captured at:\n\n ${failedJob.salvagePatch}` : ', to a `.uncommitted.patch` file next to the run log — check the run dir for one'
3701
+ }.
3702
+
3703
+ The fix-plan PRD you write for this MUST instruct its executor to, in order:
3704
+ 1. Check for a salvage patch (named \`<slug>.uncommitted.patch\` in the run directory${failedJob.salvagePatch ? `, e.g. \`${failedJob.salvagePatch}\`` : ''}) and, if found, apply it to the working tree BEFORE inspecting \`git status\`/\`git diff\` in ${cwd} for uncommitted changes matching the original PRD's acceptance criteria.
3705
+ 2. If the work is present (via the applied patch or already in the tree) and satisfies the acceptance criteria, run the project's verify commands and COMMIT it — do not re-implement or re-plan the PRD from scratch.
3706
+ 3. Only fall back to re-implementing whatever acceptance criteria are genuinely missing after applying any salvage patch, not the whole PRD.` : '';
3707
+ return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${abandonedBackgroundTaskNote}
3194
3708
 
3195
3709
  # Failed job
3196
3710
  - Slug: ${failedJob.slug}
@@ -3308,7 +3822,37 @@ function readRunOutcomeSidecars(runDir, slug) {
3308
3822
  * or if the run being investigated actually verified clean (nothing to fix —
3309
3823
  * see shouldSkipInvestigationForCleanRun).
3310
3824
  */
3825
+ const INVESTIGATION_LAUNCH_KEY = 'investigation';
3826
+
3311
3827
  async function spawnInvestigation(failedJob, runDir) {
3828
+ // The probe launches with the same CLI as the job it diagnoses. While
3829
+ // that CLI cannot launch at all (launch circuit breaker, issue #11 list
3830
+ // B1: probes e4f82da2/d374e6bf died on the same HTTP 400 as the runs
3831
+ // they were investigating) there is nothing to diagnose — skip, loudly.
3832
+ {
3833
+ const state = await readQueue().catch(() => null);
3834
+ const block = state?.launchBlocks?.[INVESTIGATION_LAUNCH_KEY];
3835
+ const jobBlock = state?.launchBlocks?.[launchFailure.launchBlockKeyFor(failedJob)];
3836
+ const gate = launchFailure.evaluateLaunchGate(block || jobBlock, { now: Date.now(), claudeVersion: await probeClaudeVersion() });
3837
+ if (gate.state === 'blocked') {
3838
+ console.log(`[scheduler] skip investigation: ${failedJob.slug} — ${gate.reason}`);
3839
+ await mutate((s) => {
3840
+ const j = s.jobs.find((x) => x.slug === failedJob.slug);
3841
+ if (j) { j.autoFixOutcome = 'launch-blocked'; j.autoFixNote = gate.reason; }
3842
+ }).catch(() => {});
3843
+ return { deferred: false };
3844
+ }
3845
+ }
3846
+ // Resume-first recovery (PRD 1111) always gets first refusal — a job
3847
+ // eligible for a bounded `--resume` dispatch must never also get a
3848
+ // cold-read fix-plan PRD authored in the same pass. selectResumeRecoveryTarget
3849
+ // returns null for every job shape spawnInvestigation is normally called
3850
+ // with (e.g. plain 'failed' jobs never carry verifierVerdict
3851
+ // 'uncommitted_changes'), so this is a no-op for the common case.
3852
+ if (selectResumeRecoveryTarget(failedJob)) {
3853
+ console.log(`[scheduler] skip investigation: ${failedJob.slug} is resume-recovery eligible`);
3854
+ return { deferred: false };
3855
+ }
3312
3856
  if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth)) {
3313
3857
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
3314
3858
  return { deferred: false };
@@ -3418,7 +3962,11 @@ async function spawnInvestigation(failedJob, runDir) {
3418
3962
  await broadcast({ flush: true });
3419
3963
 
3420
3964
  const claudeBin = resolveClaudeBin();
3421
- const childEnv = cleanChildEnv({ PATH: pathWithUserBins() }); // Homebrew/user bins for macOS
3965
+ const childEnv = cleanChildEnv({
3966
+ PATH: pathWithUserBins(), // Homebrew/user bins for macOS
3967
+ BASH_DEFAULT_TIMEOUT_MS: String(BASH_DEFAULT_TIMEOUT_MS),
3968
+ BASH_MAX_TIMEOUT_MS: String(BASH_MAX_TIMEOUT_MS),
3969
+ });
3422
3970
 
3423
3971
  // Investigation needs only a deadman watchdog — no idle-tail or result-tail
3424
3972
  // since investigations are short-running Opus probes with a hard ceiling.
@@ -3480,6 +4028,23 @@ async function spawnInvestigation(failedJob, runDir) {
3480
4028
  return;
3481
4029
  }
3482
4030
  sl(`\n[scheduler] investigation exit code=${exitCode}\n`);
4031
+ if (exitCode !== 0) {
4032
+ const probeResult = launchFailure.readResultEvent(investigationLogPath);
4033
+ const probeLaunchFailure = launchFailure.classifyLaunchFailure(probeResult);
4034
+ if (probeLaunchFailure) {
4035
+ sl(`\n[scheduler] investigation LAUNCH FAILURE (${probeLaunchFailure.kind}): ${probeLaunchFailure.message} — arming '${INVESTIGATION_LAUNCH_KEY}' launch block\n`);
4036
+ probeClaudeVersion().then((claudeVersion) => mutate((s) => {
4037
+ s.launchBlocks = s.launchBlocks || {};
4038
+ s.launchBlocks[INVESTIGATION_LAUNCH_KEY] = launchFailure.armLaunchBlock(s.launchBlocks[INVESTIGATION_LAUNCH_KEY] || null, {
4039
+ kind: probeLaunchFailure.kind, httpStatus: probeLaunchFailure.httpStatus, message: probeLaunchFailure.message,
4040
+ now: Date.now(), claudeVersion, slug: failedJob.slug, runId: failedJob.runId ?? null,
4041
+ });
4042
+ const j = s.jobs.find((x) => x.slug === failedJob.slug);
4043
+ if (j) { j.autoFixOutcome = 'launch-blocked'; j.autoFixNote = `investigation probe never ran: ${probeLaunchFailure.message}`; }
4044
+ })).catch(() => {});
4045
+ return;
4046
+ }
4047
+ }
3483
4048
  // Fold the investigation's <RCA> summary into the root-cause report already
3484
4049
  // written for this job (needs_review jobs only — writeRcaReport no-ops when
3485
4050
  // failedJob has no verifierVerdict, e.g. plain 'failed' jobs never got one).
@@ -3547,7 +4112,122 @@ async function spawnInvestigation(failedJob, runDir) {
3547
4112
  }
3548
4113
  }
3549
4114
 
3550
- async function spawnJob(job, runId, runDir, defaultCwd) {
4115
+ /**
4116
+ * computeLaunchHolds(state) → Map<slug, reason>
4117
+ *
4118
+ * The launch circuit breaker's per-tick view (lib/launchFailure.cjs, issue
4119
+ * #11): every pending row whose persona is blocked is held with its reason;
4120
+ * when a persona's backoff has elapsed exactly ONE of its pending rows is
4121
+ * left pickable (the half-open probe) and the rest are held behind it. A
4122
+ * CLI version change drops the block outright — that is the incident's real
4123
+ * fix (`claude update`) and the queue must resume on the next tick.
4124
+ * Mutates nothing; spawnJob makes the durable decision at dispatch.
4125
+ */
4126
+ async function computeLaunchHolds(state, { now = Date.now(), claudeVersion } = {}) {
4127
+ const held = new Map();
4128
+ const blocks = state?.launchBlocks;
4129
+ if (!blocks || typeof blocks !== 'object' || !Object.keys(blocks).length) return held;
4130
+ const version = claudeVersion === undefined ? await probeClaudeVersion() : claudeVersion;
4131
+ const probeAllowed = new Set();
4132
+ for (const j of state.jobs || []) {
4133
+ if (j.status !== 'pending') continue;
4134
+ const key = launchFailure.launchBlockKeyFor(j);
4135
+ const block = blocks[key];
4136
+ if (!block) continue;
4137
+ const gate = launchFailure.evaluateLaunchGate(block, { now, claudeVersion: version });
4138
+ if (gate.state === 'open') continue;
4139
+ if (gate.state === 'probe' && !probeAllowed.has(key)) {
4140
+ probeAllowed.add(key);
4141
+ continue;
4142
+ }
4143
+ held.set(j.slug, gate.state === 'probe'
4144
+ ? `launch blocked (${block.kind}) — waiting for this tick's probe of '${key}'`
4145
+ : gate.reason);
4146
+ }
4147
+ return held;
4148
+ }
4149
+
4150
+ /**
4151
+ * A run that never got a turn (res.launchFailure — see executeJob's onExit)
4152
+ * is routed here instead of the failed/investigation path (issue #11 lists
4153
+ * A1–A3, B1): the row goes back to `pending` carrying the API's own message
4154
+ * as its error, no retry budget is consumed, no auto-fix probe is spawned
4155
+ * (it would die the same way), and the persona's launch circuit breaker is
4156
+ * armed so the queue stops re-dispatching identical doomed launches while
4157
+ * still self-healing on backoff / CLI update / human Retry.
4158
+ */
4159
+ /**
4160
+ * Pure state mutation behind handleLaunchFailure (exported for tests): arms
4161
+ * the persona's breaker and returns the job's `running` row to `pending`
4162
+ * carrying the API message. Returns the armed block.
4163
+ */
4164
+ function applyLaunchFailure(s, { job, lf, runId, launchKey, mitigationApplied, claudeVersion, now = Date.now() }) {
4165
+ s.launchBlocks = s.launchBlocks || {};
4166
+ s.launchMitigations = s.launchMitigations || {};
4167
+ const prev = s.launchBlocks[launchKey] || null;
4168
+ const armed = launchFailure.armLaunchBlock(prev, {
4169
+ kind: lf.kind, httpStatus: lf.httpStatus, message: lf.message, now, claudeVersion,
4170
+ slug: job.slug, runId, mitigationApplied,
4171
+ });
4172
+ s.launchBlocks[launchKey] = armed;
4173
+ // A mitigation that was in force and still failed is no longer proven —
4174
+ // drop it so the hint and the next probe are honest.
4175
+ if (mitigationApplied && s.launchMitigations[launchKey]) delete s.launchMitigations[launchKey];
4176
+ const i = (s.jobs || []).findIndex((x) => x.slug === job.slug);
4177
+ if (i >= 0 && s.jobs[i].status === 'running') {
4178
+ const prevCount = s.jobs[i].launchFailure?.count ?? 0;
4179
+ const msg = `launch failure (${lf.kind}${lf.httpStatus ? ` HTTP ${lf.httpStatus}` : ''}): ${lf.message}`;
4180
+ resetJobFields(s.jobs[i], msg, { source: 'spawnJob:launch-failure' });
4181
+ s.jobs[i].launchFailure = {
4182
+ kind: lf.kind, httpStatus: lf.httpStatus ?? null, message: lf.message,
4183
+ at: new Date(now).toISOString(), runId, count: prevCount + 1, mitigationApplied,
4184
+ };
4185
+ s.jobs[i].terminalReason = `launch_failure:${lf.kind}`;
4186
+ s.jobs[i].heldReason = armed.exhausted
4187
+ ? `launch blocked (${lf.kind}) after ${armed.attempts} failed probe(s) — ${armed.hint}`
4188
+ : `launch blocked (${lf.kind}) — re-probe at ${armed.until}. ${armed.hint}`;
4189
+ }
4190
+ return armed;
4191
+ }
4192
+
4193
+ async function handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion }) {
4194
+ const lf = res.launchFailure;
4195
+ const now = Date.now();
4196
+ const mitigationApplied = !!(launchEnv && Object.keys(launchEnv).length);
4197
+ let armed = null;
4198
+ await mutate((s) => {
4199
+ armed = applyLaunchFailure(s, { job, lf, runId, launchKey, mitigationApplied, claudeVersion, now });
4200
+ });
4201
+ launchFailure.writeOutcomeSidecar(runDir, job.slug, {
4202
+ runId,
4203
+ exitCode: res.exitCode,
4204
+ durationMs: res.durationMs ?? null,
4205
+ numTurns: res.resultStats?.numTurns ?? null,
4206
+ outputTokens: res.resultStats?.outputTokens ?? null,
4207
+ totalCostUsd: res.resultStats?.totalCostUsd ?? null,
4208
+ verdict: null,
4209
+ status: 'pending',
4210
+ terminalReason: `launch_failure:${lf.kind}`,
4211
+ launchFailure: { kind: lf.kind, httpStatus: lf.httpStatus ?? null, message: lf.message },
4212
+ launchEnvApplied: launchEnv ? Object.keys(launchEnv) : [],
4213
+ filesChanged: 0,
4214
+ landedCommit: null,
4215
+ });
4216
+ try {
4217
+ appendError({
4218
+ cwd: job.cwd || DEFAULT_PROJECT_CWD,
4219
+ scope: 'scheduler',
4220
+ level: 'error',
4221
+ message: `launch failure (${lf.kind}) for ${job.slug}: ${lf.message} — persona '${launchKey}' blocked, attempt ${armed?.attempts}${armed?.exhausted ? ' (exhausted; needs CLI update or Retry)' : ''}`,
4222
+ meta: { slug: job.slug, runId, kind: lf.kind, httpStatus: lf.httpStatus ?? null, claudeVersion: claudeVersion ?? null, mitigationApplied, hint: armed?.hint },
4223
+ });
4224
+ } catch { /* durable logging must never break the queue */ }
4225
+ console.error(`[scheduler] ${job.slug}: LAUNCH FAILURE (${lf.kind}${lf.httpStatus ? ` HTTP ${lf.httpStatus}` : ''}) — ${lf.message}. ` +
4226
+ `Persona '${launchKey}' blocked (attempt ${armed?.attempts}${armed?.until ? `, re-probe at ${armed.until}` : ', exhausted'}). ${armed?.hint}`);
4227
+ await broadcast({ flush: true });
4228
+ }
4229
+
4230
+ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
3551
4231
  // Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
3552
4232
  // — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
3553
4233
  // leaves the job pending; the next tick retries when a slot frees up.
@@ -3557,13 +4237,108 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3557
4237
  return;
3558
4238
  }
3559
4239
  runningSet.add(job.slug);
4240
+ // Exclusive quiet-machine lease (PRD 1107) — acquired here, in the same
4241
+ // slot-acquire/dispatch step as sessionSlots, and released in this
4242
+ // function's own finally below alongside sessionSlots.release, so every
4243
+ // exit path (normal exit, timeout, SIGTERM, crash) that already frees the
4244
+ // session slot also frees the lease. pickNextBatch only ever hands this
4245
+ // function a quietMachine job when the lease was free at pick time, so
4246
+ // acquire() here should never fail in practice — but check anyway rather
4247
+ // than assume, since a lease held by a stale slug would otherwise wedge
4248
+ // silently.
4249
+ const quietLeaseAcquired = job.quietMachine === true && quietMachineLease.acquire(job.slug);
3560
4250
  try {
4251
+ // Worktree isolation cap check (PRD 1112) — probed BEFORE the job is
4252
+ // marked 'running', so a job that can't get isolation right now is a
4253
+ // DEFERRAL, not a fallback: it stays 'pending' and is retried on the
4254
+ // next dispatch pass, exactly like the sessionSlots miss above, instead
4255
+ // of degrading into an in-place run in a tree a sibling job may be
4256
+ // actively writing to (the shared-tree collision this cap exists to
4257
+ // prevent). Every OTHER worktree.ok===false reason (not a git repo,
4258
+ // disabled, carry-over failure) keeps the existing in-place fallback —
4259
+ // only the cap-reached reason is a deferral, checked here via
4260
+ // createJobWorktree's own reason string so the two paths never
4261
+ // silently drift out of sync with gitWorktree.cjs's actual wording.
4262
+ const preflightWorktree = resumeTarget
4263
+ ? { ok: false, reason: 'resume-recovery: running in place to reuse the session\'s prior working tree' }
4264
+ : await jobWorktree.createJobWorktree({ cwd: job.cwd || defaultCwd, slug: job.slug });
4265
+ if (!preflightWorktree.ok && /^worktree cap reached\b/.test(preflightWorktree.reason || '')) {
4266
+ console.log(`[scheduler] ${job.slug}: deferring — ${preflightWorktree.reason}`);
4267
+ await mutate((s) => {
4268
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4269
+ if (idx >= 0) s.jobs[idx].heldReason = preflightWorktree.reason;
4270
+ });
4271
+ await broadcast({ flush: true });
4272
+ return;
4273
+ }
4274
+ // Launch circuit breaker (lib/launchFailure.cjs, issue #11). Re-evaluated
4275
+ // here, not just in tickQueue, because the block can change between the
4276
+ // pick and this dispatch (another job's probe just failed). 'blocked' →
4277
+ // hold the row; 'probe' → this job is the single half-open probe and is
4278
+ // stamped as such so no sibling probes the same broken persona at once.
4279
+ const launchKey = launchFailure.launchBlockKeyFor(job);
4280
+ const claudeVersionNow = await probeClaudeVersion();
4281
+ let launchEnv = null;
4282
+ let launchProbe = false;
4283
+ const launchGate = await mutate((s) => {
4284
+ s.launchBlocks = s.launchBlocks || {};
4285
+ s.launchMitigations = s.launchMitigations || {};
4286
+ const mitigation = s.launchMitigations[launchKey];
4287
+ if (mitigation && claudeVersionNow && mitigation.claudeVersion && mitigation.claudeVersion !== claudeVersionNow) {
4288
+ console.log(`[scheduler] launch gate: CLI version changed (${mitigation.claudeVersion} → ${claudeVersionNow}) — dropping ${launchKey} mitigation to retry a clean launch`);
4289
+ delete s.launchMitigations[launchKey];
4290
+ }
4291
+ const block = s.launchBlocks[launchKey];
4292
+ const gate = launchFailure.evaluateLaunchGate(block, { now: Date.now(), claudeVersion: claudeVersionNow });
4293
+ if (gate.state === 'open' && block) {
4294
+ console.log(`[scheduler] launch gate: clearing ${launchKey} block (${gate.reason})`);
4295
+ delete s.launchBlocks[launchKey];
4296
+ }
4297
+ if (gate.state === 'blocked') {
4298
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4299
+ if (idx >= 0) s.jobs[idx].heldReason = gate.reason;
4300
+ return gate;
4301
+ }
4302
+ if (gate.state === 'probe') {
4303
+ block.probing = { slug: job.slug, at: new Date().toISOString() };
4304
+ launchProbe = true;
4305
+ launchEnv = block.mitigationEnv || null;
4306
+ } else if (s.launchMitigations[launchKey]?.env) {
4307
+ launchEnv = { ...s.launchMitigations[launchKey].env };
4308
+ }
4309
+ return gate;
4310
+ });
4311
+ if (launchGate.state === 'blocked') {
4312
+ console.log(`[scheduler] ${job.slug}: deferring — ${launchGate.reason}`);
4313
+ await broadcast({ flush: true });
4314
+ return;
4315
+ }
4316
+ if (launchProbe) {
4317
+ console.log(`[scheduler] ${job.slug}: dispatching as launch probe for '${launchKey}'${launchEnv ? ` with mitigation ${JSON.stringify(launchEnv)}` : ''}`);
4318
+ }
4319
+
3561
4320
  await mutate((s) => {
3562
4321
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
3563
4322
  if (idx >= 0) {
3564
- transitionJob(s.jobs[idx], 'running', { reason: 'dispatched for execution', source: 'spawnJob:dispatch' });
4323
+ transitionJob(s.jobs[idx], 'running', {
4324
+ reason: resumeTarget ? 'dispatched for resume-recovery' : 'dispatched for execution',
4325
+ source: 'spawnJob:dispatch',
4326
+ });
4327
+ delete s.jobs[idx].heldReason;
3565
4328
  s.jobs[idx].runId = runId;
3566
4329
  s.jobs[idx].startedAt = new Date().toISOString();
4330
+ if (job.quietMachine === true) {
4331
+ s.jobs[idx].quietMachine = true;
4332
+ s.jobs[idx].quietLeaseDegraded = job.quietLeaseDegraded === true;
4333
+ }
4334
+ // Stamp the bounded one-attempt marker BEFORE the resume spawn, in
4335
+ // the SAME mutate as the 'running' transition, so an app crash
4336
+ // between here and the child actually spawning still leaves this
4337
+ // job un-retriable (selectResumeRecoveryTarget returns null once
4338
+ // this is true) rather than silently re-firing forever.
4339
+ if (resumeTarget) {
4340
+ s.jobs[idx].resumeRecoveryAttempted = true;
4341
+ }
3567
4342
  }
3568
4343
  });
3569
4344
  await broadcast({ flush: true });
@@ -3573,15 +4348,50 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3573
4348
  const guardCwd = job.cwd || defaultCwd;
3574
4349
  const guardBaseline = await uncommittedChanges(guardCwd);
3575
4350
  const guardHeadBefore = await gitHead(guardCwd);
4351
+ // Shared-tree stash guard baseline (incident 2026-09-01): captured
4352
+ // unconditionally, before worktree isolation is even attempted, so an
4353
+ // in-place run always has a true pre-run snapshot to diff against. See
4354
+ // checkSharedTreeGuard below, gated to in-place runs only.
4355
+ const stashBaseline = await stashList(guardCwd);
4356
+
4357
+ // Persist the pre-run baseline onto the row itself (not just the local
4358
+ // variable) so a finalizer that never reaches the rest of THIS function
4359
+ // — namely reapDeadRunningJobs, when the process vanishes mid-run — can
4360
+ // still compute a truthful newly-dirty delta instead of having no
4361
+ // baseline at all. `runtime` (unlike this) is deleted on finalize; this
4362
+ // survives until the finalize mutate below explicitly clears it.
4363
+ //
4364
+ // preRunDirtyPaths is the SAME snapshot, capped and reworked into the
4365
+ // executor-facing manifest (buildForeignWipSection) telling the job which
4366
+ // paths it does not own — unlike guardBaseline/guardHeadBefore, it is
4367
+ // deliberately left on the row through to history.jsonl (not deleted at
4368
+ // finalize) so a post-hoc reader can tell whether a completed job ran
4369
+ // against foreign WIP.
4370
+ const preRunDirtyPaths = capDirtyPaths(guardBaseline);
4371
+ await mutate((s) => {
4372
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4373
+ if (idx >= 0) {
4374
+ s.jobs[idx].guardBaseline = guardBaseline || [];
4375
+ s.jobs[idx].guardHeadBefore = guardHeadBefore || null;
4376
+ if (preRunDirtyPaths.length) s.jobs[idx].preRunDirtyPaths = preRunDirtyPaths;
4377
+ else delete s.jobs[idx].preRunDirtyPaths;
4378
+ }
4379
+ });
3576
4380
 
3577
4381
  // Worktree isolation (PRD 994): give this job its own linked `git worktree`
3578
4382
  // checkout so its edits/tests/commit never collide with a sibling job or
3579
4383
  // an interactive session in the SAME repo. `worktree.ok` is false (with a
3580
- // logged reason) for a non-git cwd, a dirty base tree, the cap being hit,
3581
- // or SM_JOB_WORKTREE_DISABLE=1 — every case falls back to running in place,
3582
- // never a hard failure. See jobWorktree.cjs's header comment for why
3583
- // job.cwd (guardCwd) itself is NEVER repointed at the worktree dir.
3584
- const worktree = await jobWorktree.createJobWorktree({ cwd: guardCwd, slug: job.slug });
4384
+ // logged reason) for a non-git cwd, a dirty base tree, or
4385
+ // SM_JOB_WORKTREE_DISABLE=1 — every case falls back to running in place,
4386
+ // never a hard failure. (The cap-reached reason was already handled above
4387
+ // as a pre-dispatch DEFERRAL — a job never reaches this point with that
4388
+ // reason.) See jobWorktree.cjs's header comment for why job.cwd
4389
+ // (guardCwd) itself is NEVER repointed at the worktree dir. Reuses
4390
+ // `preflightWorktree` computed above the 'running' transition — it
4391
+ // already IS this job's worktree attempt (or already-created checkout),
4392
+ // so calling createJobWorktree a second time here would double-create
4393
+ // (or double-count the cap) for the exact same job.
4394
+ const worktree = preflightWorktree;
3585
4395
  if (worktree.ok) {
3586
4396
  console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
3587
4397
  } else {
@@ -3597,6 +4407,16 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3597
4407
  });
3598
4408
  }
3599
4409
  }
4410
+ // Base-tree WIP carried into the worktree (createWorktree, PRD 1094) —
4411
+ // recorded on the job row so integration can exclude these paths from
4412
+ // the branch diff below, and so it's queryable from the queue.
4413
+ const carriedPaths = (worktree.ok && Array.isArray(worktree.carriedPaths)) ? worktree.carriedPaths : [];
4414
+ if (carriedPaths.length) {
4415
+ await mutate((s) => {
4416
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4417
+ if (idx >= 0) s.jobs[idx].carriedPaths = carriedPaths;
4418
+ });
4419
+ }
3600
4420
 
3601
4421
  // Integrate the job's branch back into guardCwd's own HEAD, THEN tear the
3602
4422
  // worktree checkout down — both must happen BEFORE any git read below
@@ -3614,7 +4434,17 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3614
4434
  let res;
3615
4435
  let worktreeLeftoverDirty = [];
3616
4436
  let worktreeIntegrationFailure = null;
3617
- let worktreeSalvagePatch = null;
4437
+ // A job's uncommitted-work patch, whichever isolation mode produced it —
4438
+ // set by EITHER branch below, never both (worktree.ok picks exactly one
4439
+ // shape for the whole run). Named generically (not "worktree...") because
4440
+ // an in-place run salvages one too (PRD 1098).
4441
+ let salvagePatch = null;
4442
+ // Which foreign-WIP shape applies to THIS run: an isolated worktree only
4443
+ // ever needs to disclose carriedPaths (its checkout starts clean apart
4444
+ // from those carried paths); an in-place/shared-tree run discloses the
4445
+ // raw pre-run dirty snapshot instead. Never both — see
4446
+ // buildForeignWipSection.
4447
+ const foreignWip = worktree.ok ? { carriedPaths } : { preRunDirtyPaths };
3618
4448
  try {
3619
4449
  res = await executeJob(job, runDir, defaultCwd, async (pid, sessionId, cwd) => {
3620
4450
  await mutate((s) => {
@@ -3625,7 +4455,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3625
4455
  }
3626
4456
  });
3627
4457
  await broadcast({ flush: true });
3628
- }, worktree.ok ? worktree.dir : undefined);
4458
+ }, worktree.ok ? worktree.dir : undefined, resumeTarget, foreignWip, launchEnv);
3629
4459
  } finally {
3630
4460
  if (worktree.ok) {
3631
4461
  worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
@@ -3638,11 +4468,14 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3638
4468
  const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
3639
4469
  const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
3640
4470
  if (salvage && salvage.ok) {
3641
- worktreeSalvagePatch = salvagePath;
4471
+ salvagePatch = salvagePath;
3642
4472
  console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
3643
4473
  }
3644
4474
  }
3645
- const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug });
4475
+ const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
4476
+ if (integration.ok && integration.reason === 'carried-wip-only') {
4477
+ console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
4478
+ }
3646
4479
  if (!integration.ok) {
3647
4480
  worktreeIntegrationFailure = integration.reason;
3648
4481
  console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
@@ -3655,9 +4488,86 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3655
4488
  branch: worktree.branch,
3656
4489
  keepBranch: !integration.ok,
3657
4490
  });
4491
+ } else {
4492
+ // In-place run (non-git cwd, cap reached, env-disabled, or a carry-over
4493
+ // failure) — there is no throwaway checkout to diff, so salvage only
4494
+ // the DELTA this job itself dirtied: paths in guardBaseline are a
4495
+ // human's or a sibling job's pre-existing WIP and must never appear in
4496
+ // this job's patch. Runs for every exit code (finally always fires
4497
+ // once `res` resolves, success or not) including signal deaths and the
4498
+ // rate-limited/halt path — a killed in-place run is exactly the case
4499
+ // this exists to cover. Never mutates guardCwd's index or stashes:
4500
+ // salvageDirtyDelta is read-only (git status + git diff only).
4501
+ try {
4502
+ const after = await uncommittedChanges(guardCwd);
4503
+ if (after) {
4504
+ const baseSet = new Set(guardBaseline || []);
4505
+ const deltaPaths = after.filter((p) => !baseSet.has(p));
4506
+ if (deltaPaths.length) {
4507
+ const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
4508
+ const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: guardCwd, paths: deltaPaths, outFile: salvagePath });
4509
+ if (salvage && salvage.ok) {
4510
+ salvagePatch = salvagePath;
4511
+ console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted in-place diff (${deltaPaths.length} path(s)) to ${salvagePath}`);
4512
+ }
4513
+ }
4514
+ }
4515
+ } catch (e) {
4516
+ console.error(`[scheduler] ${job.slug}: in-place salvage failed`, e);
4517
+ }
3658
4518
  }
3659
4519
  }
3660
4520
 
4521
+ // Newly-dirty leftover computation — hoisted OUT of the exit===0 branch
4522
+ // (below) so it runs for every terminal outcome: exit 0, any non-zero
4523
+ // exit including 137/143, and the rate-limited/halt path alike. This is
4524
+ // the exact same shape the exit=0 commit-guard and the transient-failure
4525
+ // classifier each used to compute independently (guardCwd's own
4526
+ // baseline-delta UNION worktreeLeftoverDirty, which is already
4527
+ // inherently-new since it came from a fresh worktree checkout with no
4528
+ // baseline to diff against) — computed once here and reused by both
4529
+ // below, plus by the terminal-finalize mutate for leftoverPaths/
4530
+ // leftoverCount. null only when git-status itself is unavailable
4531
+ // (non-git cwd / git errored) — NEVER treated as "left nothing", exactly
4532
+ // like every other best-effort git-state check in this function.
4533
+ const afterGuardCwd = await uncommittedChanges(guardCwd);
4534
+ const newlyDirtyAll = afterGuardCwd === null
4535
+ ? null
4536
+ : [...new Set([
4537
+ ...afterGuardCwd.filter((p) => !new Set(guardBaseline || []).has(p)),
4538
+ ...worktreeLeftoverDirty,
4539
+ ])];
4540
+
4541
+ if (res.launchFailure) {
4542
+ await handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion: claudeVersionNow });
4543
+ return;
4544
+ }
4545
+ if (launchFailure.resultShowsRealTurn(res.resultStats)) {
4546
+ // The launch worked (whatever happens to the run next) — close the
4547
+ // breaker for this persona. If the probe only got through thanks to a
4548
+ // mitigation env, keep applying that env to every later launch of the
4549
+ // persona until the CLI version changes; otherwise the very next job
4550
+ // would fail the same way and re-arm the block (a flap per job).
4551
+ await mutate((s) => {
4552
+ const block = s.launchBlocks?.[launchKey];
4553
+ if (!block) return;
4554
+ delete s.launchBlocks[launchKey];
4555
+ if (launchEnv && Object.keys(launchEnv).length) {
4556
+ s.launchMitigations = s.launchMitigations || {};
4557
+ s.launchMitigations[launchKey] = {
4558
+ kind: block.kind,
4559
+ env: { ...launchEnv },
4560
+ since: new Date().toISOString(),
4561
+ claudeVersion: claudeVersionNow ?? block.claudeVersion ?? null,
4562
+ hint: launchFailure.launchFailureHint(block.kind, { claudeVersion: claudeVersionNow ?? block.claudeVersion }),
4563
+ };
4564
+ console.log(`[scheduler] launch gate: ${launchKey} recovered via mitigation ${JSON.stringify(launchEnv)} — kept in force until the CLI version changes`);
4565
+ } else {
4566
+ console.log(`[scheduler] launch gate: ${launchKey} recovered — block cleared after ${block.attempts} failed probe(s)`);
4567
+ }
4568
+ });
4569
+ }
4570
+
3661
4571
  if (res.rateLimited) {
3662
4572
  const resetIso = await refreshNextReset().catch(() => cachedNextReset);
3663
4573
  await setPaused('rate_limit', resetIso);
@@ -3699,6 +4609,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3699
4609
  // pass it back into verifyRun as priorLandedCommit (see the
3700
4610
  // pass_no_commit_prior_run_verified exemption in runVerify.cjs).
3701
4611
  let jobLandedCommitThisRun = null;
4612
+ let sharedTreeGuard = null;
3702
4613
  if (res.exitCode === 0 && !res.rateLimited) {
3703
4614
  // Detect whether the job self-committed by comparing HEAD before/after.
3704
4615
  // Used by the sentinel override: SCHEDULER_VERDICT: PASS + a landed
@@ -3771,20 +4682,12 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3771
4682
  const guardWillRefire = verifyResult && verifyResult.downgradeTo === 'pending';
3772
4683
  const guardIsLegitimateNoOp = verifyResult && COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict);
3773
4684
  if (res.exitCode === 0 && !res.rateLimited && !guardWillRefire && !guardIsLegitimateNoOp) {
3774
- const after = await uncommittedChanges(guardCwd);
3775
- // after === null means non-git cwd (or git errored) — best-effort skip,
3776
- // same as always; only a git-status result (even an empty one) counts
3777
- // as evidence for the zero-edit path.
3778
- if (after !== null) {
3779
- const baseSet = new Set(guardBaseline || []);
3780
- // worktreeLeftoverDirty was captured from a FRESH checkout (no baseline
3781
- // to diff against — every path in it is inherently new) right before
3782
- // the worktree was torn down, so it must be counted here or a job's
3783
- // uncommitted leftovers silently vanish with the worktree.
3784
- const newlyDirty = [...new Set([
3785
- ...after.filter((p) => !baseSet.has(p)),
3786
- ...worktreeLeftoverDirty,
3787
- ])];
4685
+ // afterGuardCwd === null means non-git cwd (or git errored) —
4686
+ // best-effort skip, same as always; only a git-status result (even an
4687
+ // empty one) counts as evidence for the zero-edit path. newlyDirtyAll
4688
+ // was computed once, above, right after the try/finally.
4689
+ if (afterGuardCwd !== null) {
4690
+ const newlyDirty = newlyDirtyAll;
3788
4691
  const guardState = await readQueue().catch(() => ({ jobs: [] }));
3789
4692
  const siblingRunning = (guardState.jobs || []).some(
3790
4693
  (j) => j.slug !== job.slug && j.status === 'running' && (j.cwd || defaultCwd) === guardCwd,
@@ -3799,7 +4702,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3799
4702
  legitimateNoOp: guardIsLegitimateNoOp,
3800
4703
  isFixPlanJob: isFixPlanSlug(job.slug),
3801
4704
  verifyResult,
3802
- salvagePatch: worktreeSalvagePatch,
4705
+ salvagePatch,
3803
4706
  });
3804
4707
  if (guardVerdict) {
3805
4708
  verifyResult = guardVerdict;
@@ -3823,6 +4726,36 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3823
4726
  };
3824
4727
  }
3825
4728
 
4729
+ // Shared-tree stash guard (incident 2026-09-01): only meaningful for an
4730
+ // IN-PLACE run — worktree.ok isolates the job's git state into its own
4731
+ // checkout, so nothing there can leak into guardCwd. Best-effort and run
4732
+ // regardless of exit code: a job can discard shared state on its way to
4733
+ // a non-zero exit just as easily as on a clean one.
4734
+ if (!worktree.ok) {
4735
+ sharedTreeGuard = await module.exports.checkSharedTreeGuard({
4736
+ cwd: guardCwd,
4737
+ stashBaseline,
4738
+ dirtyBaseline: guardBaseline,
4739
+ headBefore: guardHeadBefore,
4740
+ slug: job.slug,
4741
+ });
4742
+ // A restored stash alone isn't silence — it's logged loudly above and
4743
+ // surfaced on the job row below — but a path that's still missing
4744
+ // (restore failed, or two-plus stashes we refused to guess between, or
4745
+ // a revert with no stash to restore at all) must not finish green.
4746
+ if (sharedTreeGuard && (sharedTreeGuard.restoreFailed || sharedTreeGuard.ambiguousStashes || sharedTreeGuard.reverted)) {
4747
+ verifyResult = {
4748
+ verdict: 'shared_tree_reverted',
4749
+ reason: sharedTreeGuard.reverted
4750
+ ? `job discarded pre-existing state in the shared tree: ${sharedTreeGuard.reverted.length} path(s) reverted with no commit to explain it (${sharedTreeGuard.reverted.slice(0, 3).join(', ')})`
4751
+ : sharedTreeGuard.restoreFailed
4752
+ ? `job stashed the shared tree and the stash could not be auto-restored: ${sharedTreeGuard.restoreFailed}`
4753
+ : `job created ${sharedTreeGuard.ambiguousStashes.length} stashes in the shared tree — ambiguous, not auto-restored (${sharedTreeGuard.ambiguousStashes.join(', ')})`,
4754
+ downgradeTo: 'needs_review',
4755
+ };
4756
+ }
4757
+ }
4758
+
3826
4759
  // SIGTERM commit check: reuse the same commit-window scan the exit=0
3827
4760
  // guard uses above (one commit-detection path, not two) to see whether a
3828
4761
  // 143 (SIGTERM) run still landed a deliverable before it died. Scoped
@@ -3845,6 +4778,8 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3845
4778
  let needsInvestigationNow = false;
3846
4779
  let investigationJobSnapshot = null;
3847
4780
  let needsReviewRcaSnapshot = null;
4781
+ let resumeRecoveryJob = null;
4782
+ let resumeRecoveryTarget = null;
3848
4783
  let terminalNotifySnapshot = null;
3849
4784
  const newlyCompletedPrds = [];
3850
4785
  await mutate((s) => {
@@ -3891,10 +4826,11 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3891
4826
  transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
3892
4827
  s.jobs[i2].finishedAt = new Date().toISOString();
3893
4828
  s.jobs[i2].exitCode = res.exitCode;
3894
- if (worktreeSalvagePatch) {
3895
- s.jobs[i2].worktreeSalvagePatch = worktreeSalvagePatch;
4829
+ s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
4830
+ if (salvagePatch) {
4831
+ s.jobs[i2].salvagePatch = salvagePatch;
3896
4832
  } else {
3897
- delete s.jobs[i2].worktreeSalvagePatch;
4833
+ delete s.jobs[i2].salvagePatch;
3898
4834
  }
3899
4835
  s.jobs[i2].error = effectiveStatus === 'needs_review'
3900
4836
  ? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
@@ -3918,6 +4854,26 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3918
4854
  } else {
3919
4855
  delete s.jobs[i2].verifierVerdict;
3920
4856
  }
4857
+ // Closed-set outcome taxonomy (issue #11 list A2) so a queue row
4858
+ // says WHY it ended without anyone opening the transcript.
4859
+ s.jobs[i2].terminalReason = launchFailure.deriveTerminalReason({
4860
+ effectiveStatus, exitCode: res.exitCode, verifyResult, sigtermOverride, worktreeIntegrationFailure,
4861
+ });
4862
+ delete s.jobs[i2].launchFailure;
4863
+ delete s.jobs[i2].heldReason;
4864
+ // Persist the commit-guard's exact dirty-path list (verdict
4865
+ // 'uncommitted_changes' only) so a later resume-recovery attempt
4866
+ // (selectResumeRecoveryTarget) can name these paths without
4867
+ // re-running `git status` against a tree that may have moved on.
4868
+ if (verifyResult?.verdict === 'uncommitted_changes' && Array.isArray(verifyResult.dirtyPaths)) {
4869
+ // Capped the same way preRunDirtyPaths/leftoverPaths are — an
4870
+ // uncapped list here would let a pathologically dirty tree bloat
4871
+ // queue.json/history.jsonl and the resume-recovery prompt built
4872
+ // from it (buildResumeRecoveryPreamble/selectResumeRecoveryTarget).
4873
+ s.jobs[i2].uncommittedPaths = capDirtyPaths(verifyResult.dirtyPaths);
4874
+ } else {
4875
+ delete s.jobs[i2].uncommittedPaths;
4876
+ }
3921
4877
  // Non-blocking notes (e.g. a recovered missing-dependency probe, or a
3922
4878
  // pattern hit demoted because a materially-checkable verdict outranked
3923
4879
  // it) — surfaced even on completed jobs so the signal isn't lost.
@@ -3928,7 +4884,29 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3928
4884
  } else {
3929
4885
  delete s.jobs[i2].verifierAnnotations;
3930
4886
  }
4887
+ // Shared-tree guard outcome (restored stash / unresolved revert /
4888
+ // ambiguous stashes) — visible on the row even when a restored
4889
+ // stash left the run otherwise green, so it's never silent.
4890
+ if (sharedTreeGuard) {
4891
+ s.jobs[i2].sharedTreeGuard = sharedTreeGuard;
4892
+ } else {
4893
+ delete s.jobs[i2].sharedTreeGuard;
4894
+ }
3931
4895
  delete s.jobs[i2].runtime;
4896
+ // Pre-run baseline no longer needed once this run has finalized —
4897
+ // its whole purpose (letting THIS finalize compute a truthful
4898
+ // delta) is done; a fresh one is captured at the next dispatch.
4899
+ delete s.jobs[i2].guardBaseline;
4900
+ delete s.jobs[i2].guardHeadBefore;
4901
+ // Leftover-attribution fields (PRD: capture+surface uncommitted
4902
+ // work on every terminal path, not just exit=0) — set for EVERY
4903
+ // terminal outcome above (completed/failed/needs_review alike),
4904
+ // not just the exit=0 commit-guard branch, so a bare `failed` row
4905
+ // is visually distinguishable from one that quietly left work
4906
+ // behind. newlyDirtyAll is null when git-status was unavailable
4907
+ // (non-git cwd) — applyLeftoverFields treats null like "nothing to
4908
+ // attribute" via its Array.isArray guard, same as an empty array.
4909
+ applyLeftoverFields(s.jobs[i2], newlyDirtyAll);
3932
4910
 
3933
4911
  if (isNotifiableTerminalStatus(effectiveStatus)) {
3934
4912
  terminalNotifySnapshot = { ...s.jobs[i2] };
@@ -3946,6 +4924,19 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3946
4924
  // takes the treatAsPending branch above and never reaches here).
3947
4925
  needsReviewRcaSnapshot = { ...s.jobs[i2] };
3948
4926
 
4927
+ // Resume-first recovery (PRD 1111): evaluated BEFORE the auto-fix
4928
+ // eligibility check below — a job whose verdict is
4929
+ // 'uncommitted_changes' with a live sessionId gets one bounded
4930
+ // `--resume` dispatch instead of a cold-read fix-plan
4931
+ // investigation. Snapshot only (no I/O inside mutate()); the
4932
+ // actual dispatch happens outside mutate(), below. Never sets
4933
+ // needsInvestigationNow — the two are mutually exclusive for the
4934
+ // same tick, mirroring the `else if` used outside mutate().
4935
+ const target = selectResumeRecoveryTarget(s.jobs[i2]);
4936
+ if (target) {
4937
+ resumeRecoveryJob = { ...s.jobs[i2] };
4938
+ resumeRecoveryTarget = target;
4939
+ } else {
3949
4940
  // Same-tick auto-fix (feedback 2026-07-12): rather than waiting up to
3950
4941
  // 10 min for reverifyNeedsReview()'s periodic pass, check right here
3951
4942
  // whether this job qualifies for auto-fix (same eligibility rule
@@ -3969,6 +4960,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3969
4960
  needsInvestigationNow = true;
3970
4961
  investigationJobSnapshot = { ...s.jobs[i2] };
3971
4962
  }
4963
+ }
3972
4964
  }
3973
4965
  // Auto-promote: when a fix-* PRD completes successfully, the original
3974
4966
  // failed PRD's work is logically done. Flip its status to 'completed'
@@ -3984,6 +4976,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3984
4976
  orig.exitCode = 0;
3985
4977
  orig.error = null;
3986
4978
  orig.completedBy = job.slug;
4979
+ delete orig.looksDone;
3987
4980
  if (priorStatus === 'needs_review') {
3988
4981
  delete orig.verifierVerdict;
3989
4982
  }
@@ -3998,6 +4991,24 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3998
4991
  }
3999
4992
  await broadcast({ flush: true });
4000
4993
 
4994
+ // Per-run outcome sidecar (issue #11 list B5): turns/tokens/verdict in one
4995
+ // small JSON next to the log so fleet health never needs a transcript parse.
4996
+ launchFailure.writeOutcomeSidecar(runDir, job.slug, {
4997
+ runId,
4998
+ exitCode: res.exitCode,
4999
+ durationMs: res.durationMs ?? null,
5000
+ numTurns: res.resultStats?.numTurns ?? null,
5001
+ outputTokens: res.resultStats?.outputTokens ?? null,
5002
+ totalCostUsd: res.resultStats?.totalCostUsd ?? null,
5003
+ verdict: verifyResult?.verdict ?? (res.exitCode === 0 ? 'clean' : null),
5004
+ status: terminalNotifySnapshot?.status ?? failedJobSnapshot?.status ?? null,
5005
+ terminalReason: terminalNotifySnapshot?.terminalReason ?? failedJobSnapshot?.terminalReason ?? null,
5006
+ launchFailure: null,
5007
+ launchEnvApplied: launchEnv ? Object.keys(launchEnv) : [],
5008
+ filesChanged: Array.isArray(newlyDirtyAll) ? newlyDirtyAll.length : null,
5009
+ landedCommit: jobLandedCommitThisRun ?? null,
5010
+ });
5011
+
4001
5012
  if (terminalNotifySnapshot) {
4002
5013
  notifyOriginatingTab(terminalNotifySnapshot).catch((e) => {
4003
5014
  console.error('[scheduler] notifyOriginatingTab error', job.slug, e);
@@ -4013,12 +5024,28 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
4013
5024
  verdict: needsReviewRcaSnapshot.verifierVerdict,
4014
5025
  annotations: needsReviewRcaSnapshot.verifierAnnotations,
4015
5026
  })
4016
- .then((report) => notifyNeedsReview(needsReviewRcaSnapshot, report))
5027
+ .then(async (report) => {
5028
+ // Persist the classification onto the parked job row so the scheduler
5029
+ // can route on it (e.g. selectAutoFixTargets excluding 'archive')
5030
+ // without re-parsing the RCA markdown on every pass.
5031
+ await mutate((s) => {
5032
+ const j = s.jobs.find((x) => x.slug === needsReviewRcaSnapshot.slug);
5033
+ applyRcaClassification(j, report);
5034
+ }).catch(() => {});
5035
+ return notifyNeedsReview(needsReviewRcaSnapshot, report);
5036
+ })
4017
5037
  .catch((e) => {
4018
5038
  console.error('[scheduler] writeRcaReport error', job.slug, e);
4019
5039
  });
4020
5040
  }
4021
5041
 
5042
+ if (resumeRecoveryJob && resumeRecoveryTarget) {
5043
+ console.log(`[scheduler] needs_review ${job.slug} → resume-recovery (session ${resumeRecoveryTarget.sessionId}, ${resumeRecoveryTarget.dirtyPaths.length} dirty path(s))`);
5044
+ spawnResumeRecovery(resumeRecoveryJob, resumeRecoveryTarget).catch((e) => {
5045
+ console.error('[scheduler] spawnResumeRecovery error', job.slug, e);
5046
+ });
5047
+ }
5048
+
4022
5049
  if (actuallyFailed && failedJobSnapshot) {
4023
5050
  // Transient-failure detector. A 143/137 exit is ALWAYS a signal kill — the
4024
5051
  // agent never self-exits with those — so the only question is WHO killed it.
@@ -4036,25 +5063,11 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
4036
5063
  // the threshold and still fall through to investigation.
4037
5064
  const ec = failedJobSnapshot.exitCode;
4038
5065
  const retries = failedJobSnapshot.transientRetries ?? 0;
4039
- // Only pay for the extra git status call when the failure is plausibly
4040
- // transient — a real code failure never needs the dirty-tree check.
4041
5066
  const maybeTransient = (ec === 143 || ec === 137) || res.networkError === true;
4042
- let newlyDirtyCount = 0;
4043
- let dirtySample = '';
4044
- if (maybeTransient) {
4045
- const afterFailure = await uncommittedChanges(guardCwd);
4046
- const baseSet = new Set(guardBaseline || []);
4047
- // See the commit-guard block above: worktreeLeftoverDirty was captured
4048
- // (and the checkout already torn down) before this point, so it must
4049
- // be folded in here too or a transiently-killed job's leftover WIP
4050
- // silently disappears with its worktree.
4051
- const newlyDirty = [...new Set([
4052
- ...(afterFailure || []).filter((p) => !baseSet.has(p)),
4053
- ...worktreeLeftoverDirty,
4054
- ])];
4055
- newlyDirtyCount = newlyDirty.length;
4056
- dirtySample = newlyDirty.slice(0, 3).join(', ');
4057
- }
5067
+ // newlyDirtyAll was computed once, above, right after the try/finally —
5068
+ // reused here rather than re-querying git status a third time.
5069
+ const newlyDirtyCount = maybeTransient ? (newlyDirtyAll || []).length : 0;
5070
+ const dirtySample = maybeTransient ? (newlyDirtyAll || []).slice(0, 3).join(', ') : '';
4058
5071
  const decision = classifyFailureOutcome({
4059
5072
  exitCode: ec,
4060
5073
  networkError: res.networkError,
@@ -4074,7 +5087,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
4074
5087
  });
4075
5088
  await broadcast({ flush: true });
4076
5089
  } else if (decision.action === 'fail-dirty') {
4077
- const salvageNote = worktreeSalvagePatch ? ` — recoverable from salvage patch ${worktreeSalvagePatch}` : '';
5090
+ const salvageNote = salvagePatch ? ` — recoverable from salvage patch ${salvagePatch}` : '';
4078
5091
  console.log(`[scheduler] transient failure (${decision.transientKind}) for ${job.slug} left ${newlyDirtyCount} uncommitted file(s) (e.g. ${dirtySample})${salvageNote} — not auto-requeuing`);
4079
5092
  await mutate((s) => {
4080
5093
  const i = s.jobs.findIndex((x) => x.slug === job.slug);
@@ -4115,11 +5128,32 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
4115
5128
  runningSet.delete(job.slug);
4116
5129
  // Slot release notifies subscribed pumps (chat lane) machine-wide.
4117
5130
  sessionSlots.release(slotToken);
5131
+ // Release the exclusive quiet-machine lease on EVERY exit path this
5132
+ // finally covers (normal exit, timeout, SIGTERM, crash) — see the
5133
+ // acquire-site comment above. Bounded: a lease this function never
5134
+ // acquired is simply a no-op release.
5135
+ if (quietLeaseAcquired) quietMachineLease.release(job.slug);
4118
5136
  // Each job completion is a signal to advance the queue.
4119
5137
  tickQueue().catch(() => {});
4120
5138
  }
4121
5139
  }
4122
5140
 
5141
+ /**
5142
+ * Dispatch a resume-recovery attempt (PRD 1111) for a job already found
5143
+ * eligible by selectResumeRecoveryTarget. Thin wrapper around spawnJob —
5144
+ * reuses its entire slot-acquire/worktree/verify/commit-guard/finalize
5145
+ * machinery unchanged, so a resume run that itself parks or fails falls
5146
+ * through to the SAME spawnInvestigation fallback any other run would, with
5147
+ * zero special-casing. `job` and `resumeTarget` must be snapshots taken
5148
+ * BEFORE this call (this function does no eligibility re-check — spawnJob's
5149
+ * own dispatch mutate is what stamps resumeRecoveryAttempted, atomically
5150
+ * with the 'running' transition).
5151
+ */
5152
+ async function spawnResumeRecovery(job, resumeTarget) {
5153
+ const { runId, dir: runDir } = pickRunDir();
5154
+ await spawnJob(job, runId, runDir, job.cwd || DEFAULT_PROJECT_CWD, resumeTarget);
5155
+ }
5156
+
4123
5157
  // Serialized ticker: prevents two concurrent tickQueue() calls from racing
4124
5158
  // on the same pending jobs. A simple promise tail suffices since pickNextBatch
4125
5159
  // is synchronous and spawnJob is fire-and-forget.
@@ -4153,7 +5187,13 @@ function tickQueue({ bypassLoadGate = false } = {}) {
4153
5187
  // cap that sessionSlots.cjs was written to replace — which silently
4154
5188
  // ceilinged the queue at 3 while the pool the user configured said 5.
4155
5189
  const freeSlots = sessionSlots.available();
4156
- const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots);
5190
+ const heldSlugs = await computeLaunchHolds(state);
5191
+ const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
5192
+ leaseHeld: quietMachineLease.isHeld(),
5193
+ machineInUse: sessionSlots.inUse(),
5194
+ now: Date.now(),
5195
+ heldSlugs,
5196
+ });
4157
5197
  if (batch.length === 0 && freeSlots === 0) {
4158
5198
  const snap = sessionSlots.snapshot();
4159
5199
  const pendingCount = state.jobs.filter((j) => j.status === 'pending').length;
@@ -4328,6 +5368,30 @@ async function maybeLaunchWhenAvailable(state) {
4328
5368
 
4329
5369
  // ---------- dead-process reaper ----------
4330
5370
 
5371
+ // Queue-health sweep cadence: hangs off reapDeadRunningJobs's own cycle
5372
+ // counter (it already runs once per poll tick) rather than a second timer,
5373
+ // so its cadence can never drift from the poll cadence or double-fire
5374
+ // across a backoff reset.
5375
+ let queueHealthSweepCycle = 0;
5376
+ const QUEUE_HEALTH_SWEEP_EVERY_N_CYCLES = 20;
5377
+
5378
+ /**
5379
+ * runQueueHealthSweep(jobs) — read-only reporting pass over the queue
5380
+ * snapshot reapDeadRunningJobs already read this cycle. Never transitions a
5381
+ * job, never archives a PRD, never spawns anything; only logs and appends
5382
+ * an audit event for any project with drift worth a human glance.
5383
+ */
5384
+ function runQueueHealthSweep(jobs) {
5385
+ try {
5386
+ for (const { cwd, neverRan, looksDone, stuck } of computeQueueHealth(jobs)) {
5387
+ console.log(`[scheduler] queue-health ${cwd}: ${neverRan} never_ran, ${looksDone} looks-done, ${stuck} stuck`);
5388
+ appendAuditEvent('scheduler_queue_health', { cwd, neverRan, looksDone, stuck });
5389
+ }
5390
+ } catch (e) {
5391
+ console.warn('[scheduler] queue-health sweep error', e?.message);
5392
+ }
5393
+ }
5394
+
4331
5395
  /**
4332
5396
  * Scan running jobs, identify those whose claude process is provably dead OR
4333
5397
  * whose spawn never got far enough to record a runtime.pid in the first
@@ -4362,22 +5426,79 @@ async function reapDeadRunningJobs() {
4362
5426
  // Absent/empty run dir → classifyRunOutcome finds no result event →
4363
5427
  // 'no_result' → non-success below → filed as failed, never completed.
4364
5428
  const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
4365
- dead.push({ slug, pid, outcome, pidless, reason });
5429
+ // A pidless reap means the spawn never got far enough to record a
5430
+ // pid — the gate could not possibly have run, regardless of what
5431
+ // classifyRunOutcome makes of an absent/empty log.
5432
+ const gateOutcome = pidless ? 'never_ran' : mapOutcomeToGateOutcome(outcome);
5433
+ dead.push({ slug, pid, outcome, gateOutcome, pidless, reason });
5434
+ }
5435
+
5436
+ queueHealthSweepCycle += 1;
5437
+ if (queueHealthSweepCycle % QUEUE_HEALTH_SWEEP_EVERY_N_CYCLES === 0) {
5438
+ runQueueHealthSweep(state.jobs);
4366
5439
  }
5440
+
4367
5441
  if (dead.length === 0) return;
4368
5442
 
4369
- await mutate((s) => {
4370
- for (const { slug, pid, outcome, pidless, reason } of dead) {
5443
+ await mutate(async (s) => {
5444
+ for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
4371
5445
  const idx = s.jobs.findIndex((x) => x.slug === slug);
4372
5446
  if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
4373
5447
  const success = outcome === 'success';
4374
- const transitionReason = pidless ? reason : `reaped: process gone (outcome=${outcome})`;
5448
+
5449
+ // Best-effort in-place leftover computation: a job whose owning
5450
+ // process vanished without spawnJob()'s own finally block ever
5451
+ // running (the exact case this reaper exists for) never got that
5452
+ // block's salvage OR leftover-attribution pass either. Only
5453
+ // attempted when the row carries a persisted pre-run baseline
5454
+ // (guardBaseline, persisted by spawnJob at dispatch — see there).
5455
+ // With no baseline there is no safe way to tell this job's own dirt
5456
+ // from a human's or a sibling's pre-existing WIP, so this skips
5457
+ // rather than ever dumping/attributing the whole tree.
5458
+ let deltaPaths = null;
5459
+ if (Array.isArray(s.jobs[idx].guardBaseline) && s.jobs[idx].runId) {
5460
+ try {
5461
+ const rowCwd = s.jobs[idx].cwd || s.config?.defaultCwd || DEFAULT_PROJECT_CWD;
5462
+ const after = await uncommittedChanges(rowCwd);
5463
+ if (after) {
5464
+ const baseSet = new Set(s.jobs[idx].guardBaseline);
5465
+ deltaPaths = after.filter((p) => !baseSet.has(p));
5466
+ if (deltaPaths.length) {
5467
+ const salvagePath = path.join(RUNS_DIR, s.jobs[idx].runId, `${slug}.uncommitted.patch`);
5468
+ const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: rowCwd, paths: deltaPaths, outFile: salvagePath });
5469
+ if (salvage && salvage.ok) {
5470
+ s.jobs[idx].salvagePatch = salvagePath;
5471
+ console.log(`[scheduler] reapDeadRunningJobs: salvaged ${salvage.bytes} byte(s) of uncommitted in-place diff for ${slug} to ${salvagePath}`);
5472
+ }
5473
+ }
5474
+ }
5475
+ } catch (e) {
5476
+ console.error(`[scheduler] reapDeadRunningJobs: in-place salvage failed for ${slug}`, e);
5477
+ }
5478
+ }
5479
+ const leftoverSuffix = deltaPaths && deltaPaths.length
5480
+ ? ` — left ${deltaPaths.length} files uncommitted`
5481
+ : '';
5482
+ const transitionReason = (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
5483
+
4375
5484
  transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
4376
5485
  s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
4377
5486
  s.jobs[idx].finishedAt = new Date().toISOString();
4378
5487
  s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
5488
+ s.jobs[idx].gateOutcome = gateOutcome;
4379
5489
  delete s.jobs[idx].runtime;
5490
+ delete s.jobs[idx].guardBaseline;
5491
+ delete s.jobs[idx].guardHeadBefore;
5492
+ applyLeftoverFields(s.jobs[idx], deltaPaths);
4380
5493
  runningSet.delete(slug);
5494
+ // A dead job reaped here never reached spawnJob's own finally block
5495
+ // (that's this reaper's whole reason to exist — see its header
5496
+ // comment) — so if it held the quiet-machine lease, spawnJob never
5497
+ // got the chance to release it. Release it here too, or a
5498
+ // quietMachine job whose process silently vanished (OOM, a crash
5499
+ // with no exit event) wedges the lease held forever and stalls
5500
+ // dispatch for every project until the app restarts.
5501
+ if (s.jobs[idx].quietMachine === true) quietMachineLease.release(slug);
4381
5502
  if (pidless) {
4382
5503
  console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
4383
5504
  appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
@@ -4570,7 +5691,7 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
4570
5691
  // investigation jobs correctly found "nothing to fix" but were flagged
4571
5692
  // anyway). For non-fix-plan jobs the exemption never applies, so rescanning
4572
5693
  // their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
4573
- const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'pass_no_commit', 'pass_no_commit_already_shipped']);
5694
+ const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'abandoned_background_task', 'pass_no_commit', 'pass_no_commit_already_shipped']);
4574
5695
 
4575
5696
  // Bounds fix-plan recursion: cap N permits at most N+1 fix jobs per original
4576
5697
  // slug (depth 1 = the original job, depth 2 = its `-fix`, depth 3+ is
@@ -4707,22 +5828,48 @@ function isPlanUnqueued(job, queuedSlugs) {
4707
5828
  * Bias to needs_review: a false yellow costs a human glance, a false green
4708
5829
  * costs a silently-unfixed bug — which is exactly what happened.
4709
5830
  */
5831
+ // abandoned_background_task shares no_verdict_sentinel's exact rescan path
5832
+ // (same "sentinel === null && !commitEvidence" gate in runVerify, same
5833
+ // committedDuringRun repo-wide-not-per-job attribution problem) — the PRD 983
5834
+ // incident mechanism above applies identically, so it gets the same guard
5835
+ // rather than a carve-out that would silently reopen the same false-heal hole.
5836
+ const NO_ATTRIBUTABLE_COMMIT_VERDICTS = new Set(['no_verdict_sentinel', 'abandoned_background_task']);
5837
+
4710
5838
  function healRefusalReason(job, verdict, committedDuringRun) {
4711
5839
  if (!job || !verdict) return null;
4712
5840
  if (!COMPLETED_EQUIVALENT_VERDICTS.has(verdict.verdict)) return null;
4713
- if (job.verifierVerdict !== 'no_verdict_sentinel') return null;
5841
+ if (!NO_ATTRIBUTABLE_COMMIT_VERDICTS.has(job.verifierVerdict)) return null;
4714
5842
  // A commit this job actually recorded as its own is real evidence; the
4715
5843
  // repo-wide window scan is not.
4716
5844
  if (job.landedCommit) return null;
4717
- return 'no_verdict_sentinel with no job-attributable commit — refusing to heal'
5845
+ return `${job.verifierVerdict} with no job-attributable commit — refusing to heal`
4718
5846
  + ` (committedInWindow=${committedDuringRun === true} is repo-wide, not proof this job delivered)`;
4719
5847
  }
4720
5848
 
5849
+ /**
5850
+ * True when a `failed` job's failure is unverified-shaped — no result event
5851
+ * was ever recorded for its run (classifyRunOutcome === 'no_result'), so no
5852
+ * SCHEDULER_VERDICT sentinel could have been parsed either, OR it already
5853
+ * carries a RESCANNABLE_VERDICTS verifierVerdict. A row that failed with a
5854
+ * real result event (classifyRunOutcome === 'failed', i.e. a genuine red
5855
+ * gate or a real non-zero-exit error) is excluded — that failure is
5856
+ * evidence, not silence, and must never become a heal candidate (PRD 1102).
5857
+ */
5858
+ function isFailedUnverifiedShaped(job) {
5859
+ if (!job || job.status !== 'failed') return false;
5860
+ if (job.verifierVerdict && RESCANNABLE_VERDICTS.has(job.verifierVerdict)) return true;
5861
+ const runId = job.runId || resolveRunId(job);
5862
+ if (!runId) return false;
5863
+ const logPath = path.join(RUNS_DIR, runId, `${job.slug}.log`);
5864
+ return classifyRunOutcome(logPath) === 'no_result';
5865
+ }
5866
+
4721
5867
  function isRescanCandidate(job) {
4722
- return !!job
4723
- && job.status === 'needs_review'
4724
- && !!(job.runId || resolveRunId(job))
4725
- && RESCANNABLE_VERDICTS.has(job.verifierVerdict);
5868
+ if (!job) return false;
5869
+ if (!(job.runId || resolveRunId(job))) return false;
5870
+ if (job.status === 'needs_review') return RESCANNABLE_VERDICTS.has(job.verifierVerdict);
5871
+ if (job.status === 'failed') return isFailedUnverifiedShaped(job);
5872
+ return false;
4726
5873
  }
4727
5874
 
4728
5875
  /**
@@ -4756,10 +5903,36 @@ function isRescanCandidate(job) {
4756
5903
  * exhausted retry is excluded
4757
5904
  * - no fix sibling on disk (fixSlugExists) or already in the queue
4758
5905
  */
5906
+ /**
5907
+ * Persist a writeRcaReport() result onto its job row — job.rcaFailureClass /
5908
+ * job.rcaRecoveryAction — so selectAutoFixTargets and future routing can read
5909
+ * the classification straight off the queue row instead of re-parsing the RCA
5910
+ * markdown. Pure mutation of the passed-in job object; no I/O. A no-op when
5911
+ * the job is missing, has moved off needs_review (e.g. resumed and completed
5912
+ * before this async write landed), or the report was never filed (disabled,
5913
+ * error, etc). Returns whether it applied, for callers/tests that want to
5914
+ * assert on it.
5915
+ */
5916
+ function applyRcaClassification(job, report) {
5917
+ if (!job || job.status !== 'needs_review' || !report?.filed) return false;
5918
+ job.rcaFailureClass = report.failureClass;
5919
+ job.rcaRecoveryAction = report.recoveryAction;
5920
+ return true;
5921
+ }
5922
+
4759
5923
  function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRunId }) {
4760
5924
  const slugsInQueue = new Set(jobs.map((j) => j.slug));
4761
5925
  return jobs.filter((job) => {
4762
5926
  if (job.status !== 'needs_review') return false;
5927
+ // A stale re-run whose work already shipped (rcaReport's 'already-shipped'
5928
+ // class) must never buy a fix-plan PRD — there is nothing to fix, and the
5929
+ // correct recovery (archiving the PRD) is a human/reconcile action, not
5930
+ // an investigation.
5931
+ if (job.rcaRecoveryAction === 'archive') return false;
5932
+ // Resume-first recovery (PRD 1111): a job still eligible for its one
5933
+ // bounded `--resume` attempt must never also become a fix-plan target
5934
+ // in the same pass — see spawnInvestigation's own identical guard.
5935
+ if (selectResumeRecoveryTarget(job)) return false;
4763
5936
  const runId = job.runId || resolveJobRunId(job);
4764
5937
  if (!runId) return false;
4765
5938
  if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth)) return false;
@@ -4795,12 +5968,52 @@ function isEligibleForImmediateAutoFix(job, allJobs, fixSlugExists) {
4795
5968
  return targets.some((t) => t.slug === job.slug);
4796
5969
  }
4797
5970
 
5971
+ /**
5972
+ * Widened evidence check (PRD 1102): does at least one commit land AFTER
5973
+ * this job's run window that touches a path the PRD itself declares? Scoped
5974
+ * to the PRD's own declared paths (never the whole repo) so a sibling job's
5975
+ * unrelated commit is not credited to this one — see healRefusalReason's own
5976
+ * rationale for why unscoped, repo-wide evidence is not attribution.
5977
+ *
5978
+ * Returns null (no annotation, never fabricated) when the PRD names no
5979
+ * paths — the caller then has only the existing, already-computed
5980
+ * committedInWindow signal to go on, same as before this PRD.
5981
+ *
5982
+ * @returns {Promise<{commits: string[], paths: string[], detectedAt: string} | null>}
5983
+ */
5984
+ async function computeLooksDone(job) {
5985
+ const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
5986
+ const paths = declaredPathsForPrd(prdPath);
5987
+ if (!paths.length) return null;
5988
+ await fetchAllRefs(job.cwd);
5989
+ const commits = await landedSinceRun(job.cwd, job.startedAt, paths);
5990
+ if (!commits.length) return null;
5991
+ return { commits, paths, detectedAt: new Date().toISOString() };
5992
+ }
5993
+
4798
5994
  async function reverifyNeedsReview() {
4799
5995
  const snap = await readQueue();
4800
5996
  const candidates = snap.jobs.filter(isRescanCandidate);
4801
5997
  const healed = [];
4802
5998
  const leftForReview = [];
5999
+ const looksDoneUpdates = [];
4803
6000
  for (const job of candidates) {
6001
+ if (job.status === 'failed') {
6002
+ // A failed row never runs the transcript-verifier rescan below — that
6003
+ // machinery (verifyRun/COMPLETED_EQUIVALENT_VERDICTS) exists to
6004
+ // auto-COMPLETE a stale needs_review row, and a failed row must never
6005
+ // auto-complete through this pass (see the AC's conservative-in-the-
6006
+ // completing-direction constraint). The only thing a failed candidate
6007
+ // can gain here is a looksDone annotation + a failed → needs_review
6008
+ // transition, for a human to confirm.
6009
+ const looksDone = await computeLooksDone(job);
6010
+ if (looksDone) {
6011
+ looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: true });
6012
+ } else {
6013
+ leftForReview.push({ slug: job.slug, reason: 'failed, unverified-shaped run — no post-window evidence on declared paths' });
6014
+ }
6015
+ continue;
6016
+ }
4804
6017
  const runDir = path.join(RUNS_DIR, job.runId || resolveRunId(job));
4805
6018
  const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
4806
6019
  // Derive committedDuringRun from the recorded run window. The live
@@ -4827,13 +6040,45 @@ async function reverifyNeedsReview() {
4827
6040
  });
4828
6041
  } catch { leftForReview.push({ slug: job.slug, reason: 'verifyRun threw' }); continue; }
4829
6042
  const refusal = healRefusalReason(job, v, committedDuringRun);
6043
+ let stillOpen = true;
4830
6044
  if (refusal) {
4831
6045
  leftForReview.push({ slug: job.slug, reason: refusal });
4832
6046
  } else if (v && COMPLETED_EQUIVALENT_VERDICTS.has(v.verdict)) {
4833
6047
  healed.push(job.slug);
6048
+ stillOpen = false;
4834
6049
  } else {
4835
6050
  leftForReview.push({ slug: job.slug, reason: v ? `${v.verdict}: ${v.reason}` : 'null verdict' });
4836
6051
  }
6052
+ // Still needs_review after the existing heal pass — widen the evidence
6053
+ // window before giving up on it entirely (unchanged heal semantics for
6054
+ // rows that already qualified above; this only adds an annotation).
6055
+ if (stillOpen) {
6056
+ const looksDone = await computeLooksDone(job);
6057
+ if (looksDone) {
6058
+ looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
6059
+ }
6060
+ }
6061
+ }
6062
+ if (looksDoneUpdates.length) {
6063
+ const bySlug = new Map(looksDoneUpdates.map((u) => [u.slug, u]));
6064
+ await mutate((s) => {
6065
+ for (const j of s.jobs) {
6066
+ const u = bySlug.get(j.slug);
6067
+ if (!u) continue;
6068
+ if (u.fromFailed) {
6069
+ transitionJob(j, 'needs_review', {
6070
+ reason: 'looks done — commit(s) since this run touch this PRD\'s declared paths; confirm before archiving',
6071
+ source: 'reverifyNeedsReview:looksDone',
6072
+ });
6073
+ }
6074
+ if (j.status !== 'needs_review') continue;
6075
+ j.looksDone = u.looksDone;
6076
+ const shaList = u.looksDone.commits.slice(0, 5).map((c) => c.slice(0, 7)).join(', ');
6077
+ j.error = `looks done — ${u.looksDone.commits.length} commit(s) since this run touch this PRD's paths (${shaList}); confirm before archiving`;
6078
+ }
6079
+ });
6080
+ console.log(`[scheduler] boot reverify: looksDone annotated for ${looksDoneUpdates.length} row(s): ${looksDoneUpdates.map((u) => u.slug).join(', ')}`);
6081
+ await broadcast();
4837
6082
  }
4838
6083
  if (healed.length) {
4839
6084
  const healSet = new Set(healed);
@@ -4844,6 +6089,7 @@ async function reverifyNeedsReview() {
4844
6089
  transitionJob(j, 'completed', { reason: 'boot reverify: stale needs_review healed', source: 'reverifyNeedsReview:heal' });
4845
6090
  j.error = null;
4846
6091
  delete j.verifierVerdict;
6092
+ delete j.looksDone;
4847
6093
  healedPrds.push({ slug: j.slug, cwd: j.cwd });
4848
6094
  }
4849
6095
  }
@@ -4883,6 +6129,7 @@ async function reverifyNeedsReview() {
4883
6129
  orig.exitCode = 0;
4884
6130
  orig.error = null;
4885
6131
  orig.completedBy = job.slug;
6132
+ delete orig.looksDone;
4886
6133
  if (priorStatus === 'needs_review') delete orig.verifierVerdict;
4887
6134
  promoted.push(`${orig.slug} (was ${priorStatus}, via ${job.slug})`);
4888
6135
  promotedPrds.push({ slug: orig.slug, cwd: orig.cwd });
@@ -4949,14 +6196,40 @@ async function reverifyNeedsReview() {
4949
6196
  await broadcast();
4950
6197
  }
4951
6198
 
6199
+ // The annotate mutate above only runs conditionally — when it didn't fire,
6200
+ // afterHealForAnnotate is still the current on-disk state, so reuse it
6201
+ // instead of re-reading queue.json twice more back-to-back for the
6202
+ // resume-recovery and auto-fix passes below (neither of which mutates
6203
+ // synchronously: spawnResumeRecovery/spawnJob's own writes land later).
6204
+ const queueForResumeAndAutofix = (unresolvable.length || exhaustedAutoFix.length || planUnqueued.length)
6205
+ ? await readQueue()
6206
+ : afterHealForAnnotate;
6207
+
6208
+ // Resume-first recovery (PRD 1111): before any fix-plan investigation is
6209
+ // authored below, offer the bounded one-attempt `--resume` dispatch to any
6210
+ // needs_review job this periodic pass finds still eligible — e.g. one the
6211
+ // same-tick check in spawnJob missed because the app restarted between
6212
+ // that job parking and this pass running. selectAutoFixTargets below
6213
+ // already excludes every job this loop dispatches, so a resumable job
6214
+ // never also gets a fix-plan PRD authored in the same pass.
6215
+ {
6216
+ for (const job of queueForResumeAndAutofix.jobs) {
6217
+ const target = selectResumeRecoveryTarget(job);
6218
+ if (!target) continue;
6219
+ console.log(`[scheduler] resume-recovery: needs_review ${job.slug} → resuming session ${target.sessionId}`);
6220
+ spawnResumeRecovery(job, target).catch((e) => {
6221
+ console.error('[scheduler] spawnResumeRecovery error', job.slug, e);
6222
+ });
6223
+ }
6224
+ }
6225
+
4952
6226
  // Auto-fix: spawn a fix-plan investigation for each job still in
4953
6227
  // needs_review after the heal pass (kill-switch: SM_AUTOFIX_DISABLE=1).
4954
6228
  // spawnInvestigation early-returns once investigationsInFlight reaches
4955
6229
  // MAX_CONCURRENT_INVESTIGATIONS (queues the rest for retry), so this loop
4956
6230
  // cannot fan out past the cap regardless of how many targets are selected.
4957
6231
  if (process.env.SM_AUTOFIX_DISABLE !== '1') {
4958
- const afterHeal = await readQueue();
4959
- const targets = selectAutoFixTargets(afterHeal.jobs, {
6232
+ const targets = selectAutoFixTargets(queueForResumeAndAutofix.jobs, {
4960
6233
  fixSlugExists: (s) => candidatePrdsDirs().some((dir) => fs.existsSync(path.join(dir, `${s}.md`))),
4961
6234
  });
4962
6235
  for (const job of targets) {
@@ -4985,7 +6258,7 @@ async function reverifyNeedsReview() {
4985
6258
  }
4986
6259
  }
4987
6260
 
4988
- return { rescanned: candidates.length, healed, leftForReview };
6261
+ return { rescanned: candidates.length, healed, leftForReview, looksDone: looksDoneUpdates.map((u) => u.slug) };
4989
6262
  }
4990
6263
 
4991
6264
  /**
@@ -5695,8 +6968,8 @@ async function init() {
5695
6968
  // there" (parallelGroup/estimateMinutes/sourcePromptId/epicId/
5696
6969
  // archivedStatus); `fields=full` restores them.
5697
6970
  function toCompactPrdEntry(entry) {
5698
- const { slug, title, cwd, mtimeMs, archived, status } = entry;
5699
- return { slug, title, cwd, mtimeMs, archived, status };
6971
+ const { slug, title, cwd, mtimeMs, archived, status, agentType } = entry;
6972
+ return { slug, title, cwd, mtimeMs, archived, status, agentType };
5700
6973
  }
5701
6974
 
5702
6975
  /**
@@ -5743,6 +7016,7 @@ async function listPrdsInternal() {
5743
7016
  estimateMinutes: parsed.estimateMinutes,
5744
7017
  sourcePromptId: parsed.sourcePromptId,
5745
7018
  epicId: parsed.epicId ?? null,
7019
+ agentType: parsed.agentType ?? null,
5746
7020
  mtimeMs: stat.mtimeMs,
5747
7021
  archived,
5748
7022
  };
@@ -5952,7 +7226,7 @@ const remote = {
5952
7226
 
5953
7227
  async listJobs() {
5954
7228
  const state = await readQueue();
5955
- return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd }));
7229
+ return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd, agentType: j.agentType ?? null }));
5956
7230
  },
5957
7231
 
5958
7232
  // Single queue row lookup, used by cancelJob/updatePrd's status guards and
@@ -6191,4 +7465,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
6191
7465
  });
6192
7466
  }
6193
7467
 
6194
- module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS };
7468
+ module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure };