claude-code-session-manager 0.75.3 → 0.77.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (164) hide show
  1. package/dist/assets/{AgentLibrary-CzQqcObq.js → AgentLibrary-B2ie8bbw.js} +2 -2
  2. package/dist/assets/{DataModel-Bj_WlLz8.js → DataModel-BIJPYw32.js} +1 -1
  3. package/dist/assets/{History-DnSi_OHm.js → History-CeY6dk9S.js} +2 -2
  4. package/dist/assets/{Hooks-0BB0dp3S.js → Hooks-BFH2ocKg.js} +2 -2
  5. package/dist/assets/{HostBilko-DHpwwsLQ.js → HostBilko-36gj9wLz.js} +1 -1
  6. package/dist/assets/{Library-CaJVqVvi.js → Library-C-hBct39.js} +1 -1
  7. package/dist/assets/{ListDetail-C1W2HmC2.js → ListDetail-CNq64VWV.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-5Ob9FW3z.js → MarkdownEditor-Bh3qt5-1.js} +1 -1
  9. package/dist/assets/{McpServers-JxCSfm1S.js → McpServers-DpGN0oyz.js} +1 -1
  10. package/dist/assets/{Memory-BDeqlqwH.js → Memory-D59hUjC4.js} +6 -6
  11. package/dist/assets/{Panel-Dh9ZHuEj.js → Panel-DCgbaoci.js} +1 -1
  12. package/dist/assets/{Permissions-DXy-CbEY.js → Permissions-DAmQ0DYV.js} +2 -2
  13. package/dist/assets/{Plugins-_n1Iuc8T.js → Plugins-Dyfgn6Is.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-BP_evfxE.js → ProvenanceBadge-BiYhPO1U.js} +1 -1
  15. package/dist/assets/SaveBar-RV7B6sOh.js +1 -0
  16. package/dist/assets/Scheduler-BPaNqx1b.js +14 -0
  17. package/dist/assets/{ScopeSwitcher-CAWzM6RI.js → ScopeSwitcher-P4mdLGNU.js} +1 -1
  18. package/dist/assets/{Settings-DRRozLyT.js → Settings-BL4vf5aX.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-DGHDWlz4.js → SkillReferenceGraph-BRBDyi1_.js} +1 -1
  20. package/dist/assets/{Skills-D8L66eiX.js → Skills-BV08gDUH.js} +2 -2
  21. package/dist/assets/{SystemPrompt-CYtUsonD.js → SystemPrompt-CLftSsDw.js} +1 -1
  22. package/dist/assets/TagLibrary-Bp8jGsd5.js +1 -0
  23. package/dist/assets/{TiptapBody-B2hRgbPE.js → TiptapBody-jCpuB6E5.js} +1 -1
  24. package/dist/assets/{Toggle-BTwsbxam.js → Toggle-D2paA1xf.js} +1 -1
  25. package/dist/assets/{index-DijufvkJ.js → index-BDRSqBl3.js} +704 -704
  26. package/dist/assets/{index-CMLnzdZC.css → index-CYhdtisq.css} +1 -1
  27. package/dist/assets/{settingsSchema-D6wzxAi6.js → settingsSchema-6IOLjZZN.js} +1 -1
  28. package/dist/index.html +2 -2
  29. package/package.json +8 -2
  30. package/plugins/session-manager-dev/skills/develop/standards.md +1 -1
  31. package/scripts/lib/activeSessions.cjs +116 -6
  32. package/scripts/project-pages-logic/dist/logic.cjs +4709 -0
  33. package/scripts/render-project-pages/dist/renderer.cjs +18900 -0
  34. package/scripts/render-project-pages.cjs +70 -0
  35. package/scripts/scheduler-mcp-server.cjs +269 -96
  36. package/scripts/validate-project-pages-summary.cjs +62 -0
  37. package/src/main/__tests__/agentModelResolve.test.cjs +66 -0
  38. package/src/main/__tests__/epicStatusMirror.test.cjs +110 -0
  39. package/src/main/__tests__/health-delegation-chain.test.cjs +106 -0
  40. package/src/main/__tests__/prdAdminRoutes.test.cjs +295 -0
  41. package/src/main/__tests__/prdAgentType.test.cjs +103 -0
  42. package/src/main/__tests__/prdCreate.test.cjs +247 -0
  43. package/src/main/__tests__/prdFrontmatterAgentType.test.cjs +117 -0
  44. package/src/main/__tests__/prdFrontmatterQuietMachine.test.cjs +108 -0
  45. package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +485 -0
  46. package/src/main/__tests__/projectPages.test.cjs +73 -1
  47. package/src/main/__tests__/rcaReport.test.cjs +54 -0
  48. package/src/main/__tests__/runVerify.test.cjs +94 -0
  49. package/src/main/__tests__/scheduler-autofix-select.test.cjs +58 -3
  50. package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +103 -0
  51. package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +41 -0
  52. package/src/main/__tests__/scheduler-effective-concurrency.test.cjs +10 -0
  53. package/src/main/__tests__/scheduler-foreign-wip-manifest.test.cjs +78 -0
  54. package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +242 -0
  55. package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +31 -0
  56. package/src/main/__tests__/scheduler-launch-failure.test.cjs +201 -0
  57. package/src/main/__tests__/scheduler-leftover-fields.test.cjs +52 -0
  58. package/src/main/__tests__/scheduler-looks-done.test.cjs +241 -0
  59. package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +135 -0
  60. package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +222 -0
  61. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +207 -1
  62. package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +212 -0
  63. package/src/main/__tests__/scheduler-stranded-investigation.test.cjs +185 -0
  64. package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +194 -0
  65. package/src/main/__tests__/seedAgentPersonas.test.cjs +75 -14
  66. package/src/main/__tests__/seedSchedulerMcp.test.cjs +66 -0
  67. package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -5
  68. package/src/main/bilkoHost.cjs +4 -3
  69. package/src/main/chatRunner.cjs +6 -1
  70. package/src/main/config.cjs +25 -33
  71. package/src/main/health.cjs +153 -2
  72. package/src/main/index.cjs +64 -5
  73. package/src/main/ipcSchemas.cjs +69 -1
  74. package/src/main/lib/__tests__/activeIndexRebuild.test.cjs +179 -0
  75. package/src/main/lib/__tests__/childWithLog.test.cjs +141 -0
  76. package/src/main/lib/__tests__/delegationReadiness.test.cjs +391 -42
  77. package/src/main/lib/__tests__/ephemeralCwd.test.cjs +91 -0
  78. package/src/main/lib/__tests__/epicWorktreeMint.test.cjs +5 -3
  79. package/src/main/lib/__tests__/fixChainDepth.test.cjs +40 -0
  80. package/src/main/lib/__tests__/gitWorktree.test.cjs +290 -5
  81. package/src/main/lib/__tests__/gitWorktreeSalvage.test.cjs +107 -0
  82. package/src/main/lib/__tests__/gitWorktreeSalvageDelta.test.cjs +153 -0
  83. package/src/main/lib/__tests__/jobWorktree.test.cjs +6 -4
  84. package/src/main/lib/__tests__/landedSinceRun.test.cjs +73 -0
  85. package/src/main/lib/__tests__/launchFailure.test.cjs +220 -0
  86. package/src/main/lib/__tests__/loadGate.test.cjs +159 -0
  87. package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +102 -0
  88. package/src/main/lib/__tests__/opsOwnership.test.cjs +7 -0
  89. package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +151 -0
  90. package/src/main/lib/__tests__/opsRootResolve.test.cjs +149 -0
  91. package/src/main/lib/__tests__/prdDeclaredPaths.test.cjs +82 -0
  92. package/src/main/lib/__tests__/projectRootResolve.test.cjs +148 -0
  93. package/src/main/lib/__tests__/queueHealth.test.cjs +58 -0
  94. package/src/main/lib/__tests__/quietMachineLease.test.cjs +39 -0
  95. package/src/main/lib/__tests__/reaperHelpers.test.cjs +133 -0
  96. package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +19 -9
  97. package/src/main/lib/__tests__/schedulerBatchFairness.test.cjs +213 -0
  98. package/src/main/lib/__tests__/schedulerBatchLaunchHold.test.cjs +125 -0
  99. package/src/main/lib/__tests__/schedulerBatchProjectCap.test.cjs +127 -0
  100. package/src/main/lib/__tests__/schedulerBatchQuietMachine.test.cjs +109 -0
  101. package/src/main/lib/__tests__/schedulerMcpServerHeadlessRefusal.test.cjs +71 -0
  102. package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +217 -0
  103. package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +350 -0
  104. package/src/main/lib/activeIndexMerge.cjs +15 -0
  105. package/src/main/lib/activeIndexRebuild.cjs +133 -0
  106. package/src/main/lib/agentModelResolve.cjs +58 -0
  107. package/src/main/lib/buildTarget.cjs +3 -2
  108. package/src/main/lib/childWithLog.cjs +69 -2
  109. package/src/main/lib/claudeBin.cjs +54 -1
  110. package/src/main/lib/crossProjectFeedback.cjs +8 -1
  111. package/src/main/lib/definitionOfDone.cjs +3 -2
  112. package/src/main/lib/delegationReadiness.cjs +514 -26
  113. package/src/main/lib/ephemeralCwd.cjs +78 -0
  114. package/src/main/lib/epicDelegationStats.cjs +2 -1
  115. package/src/main/lib/epicMint.cjs +17 -1
  116. package/src/main/lib/epicStatusMirror.cjs +95 -0
  117. package/src/main/lib/epicValidationHook.cjs +2 -1
  118. package/src/main/lib/epicWorktreeMint.cjs +5 -2
  119. package/src/main/lib/fixChainDepth.cjs +45 -0
  120. package/src/main/lib/gitWorktree.cjs +520 -21
  121. package/src/main/lib/jobWorktree.cjs +2 -0
  122. package/src/main/lib/landedSinceRun.cjs +55 -0
  123. package/src/main/lib/launchFailure.cjs +357 -0
  124. package/src/main/lib/loadGate.cjs +134 -0
  125. package/src/main/lib/mcpToolCatalog.cjs +370 -0
  126. package/src/main/lib/opsErrorLog.cjs +12 -1
  127. package/src/main/lib/opsOwnership.cjs +106 -0
  128. package/src/main/lib/prdAdminRoutes.cjs +43 -3
  129. package/src/main/lib/prdAgentType.cjs +84 -0
  130. package/src/main/lib/prdCreate.cjs +103 -15
  131. package/src/main/lib/prdDeclaredPaths.cjs +70 -0
  132. package/src/main/lib/prdFrontmatter.cjs +17 -3
  133. package/src/main/lib/prdLocations.cjs +13 -6
  134. package/src/main/lib/projectHomeAdminRoutes.cjs +402 -0
  135. package/src/main/lib/projectPageSummarySchema.cjs +181 -0
  136. package/src/main/lib/projectRootResolve.cjs +134 -0
  137. package/src/main/lib/promptSessionSchema.cjs +7 -0
  138. package/src/main/lib/queueHealth.cjs +38 -0
  139. package/src/main/lib/queueStore.cjs +40 -7
  140. package/src/main/lib/quietMachineLease.cjs +48 -0
  141. package/src/main/lib/rcaReport.cjs +54 -4
  142. package/src/main/lib/reaperHelpers.cjs +64 -1
  143. package/src/main/lib/scheduleJobSchema.cjs +31 -0
  144. package/src/main/lib/scheduleJobTransitions.cjs +6 -2
  145. package/src/main/lib/schedulerBatch.cjs +301 -55
  146. package/src/main/lib/schedulerConfig.cjs +99 -0
  147. package/src/main/projectBrief.cjs +3 -2
  148. package/src/main/projectPages.cjs +162 -3
  149. package/src/main/promptSessionTranscript.cjs +0 -0
  150. package/src/main/pty.cjs +5 -0
  151. package/src/main/queueOps.cjs +15 -8
  152. package/src/main/runVerify.cjs +50 -9
  153. package/src/main/scheduler/prdParser.cjs +18 -1
  154. package/src/main/scheduler.cjs +1701 -130
  155. package/src/main/seedAgentPersonas.cjs +62 -21
  156. package/src/main/seedSchedulerMcp.cjs +58 -4
  157. package/src/main/templates/project-pages-catalog.json +741 -0
  158. package/src/main/templates/project-pages-pipeline.md +417 -0
  159. package/src/preload/api.d.ts +187 -3
  160. package/src/preload/index.cjs +9 -0
  161. package/src/seed/agents/project-home-builder.md +59 -0
  162. package/dist/assets/SaveBar-D-gCUx4n.js +0 -1
  163. package/dist/assets/Scheduler-Bpd4OGju.js +0 -14
  164. package/dist/assets/TagLibrary-E5CLeuVk.js +0 -1
@@ -53,9 +53,13 @@ const { ipcMain } = require('electron');
53
53
  const billing = require('./usage.cjs');
54
54
  const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
55
55
  const supervisor = require('./supervisor.cjs');
56
- const { resolveClaudeBin } = require('./lib/claudeBin.cjs');
56
+ const { resolveClaudeBin, probeClaudeVersion } = require('./lib/claudeBin.cjs');
57
+ const launchFailure = require('./lib/launchFailure.cjs');
58
+ const { appendError } = require('./lib/opsErrorLog.cjs');
57
59
  const { readTail } = require('./lib/fileTail.cjs');
58
- const { claudePidAlive, classifyRunOutcome, ORPHAN_REQUEUE_CAP } = require('./lib/reaperHelpers.cjs');
60
+ const { claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
61
+ const { computeQueueHealth } = require('./lib/queueHealth.cjs');
62
+ const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
59
63
  const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
60
64
  const { sendIfAlive } = require('./lib/sendToRenderer.cjs');
61
65
  const { createBroadcastCoalescer } = require('./lib/broadcastCoalescer.cjs');
@@ -67,8 +71,10 @@ const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs')
67
71
  const promptSessionTranscript = require('./promptSessionTranscript.cjs');
68
72
  const { verifyRun } = require('./runVerify.cjs');
69
73
  const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
74
+ const { landedSinceRun } = require('./lib/landedSinceRun.cjs');
75
+ const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
70
76
  const logs = require('./logs.cjs');
71
- const { schemas, validated } = require('./ipcSchemas.cjs');
77
+ const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
72
78
  const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
73
79
  const {
74
80
  POLL_INTERVAL_MS,
@@ -78,6 +84,9 @@ const {
78
84
  QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
79
85
  JOB_OVERRUN_FACTOR: JOB_OVERRUN_FACTOR_DEFAULT,
80
86
  JOB_OVERRUN_FLOOR_MS: JOB_OVERRUN_FLOOR_MS_DEFAULT,
87
+ PIDLESS_SPAWN_GRACE_MS,
88
+ INVESTIGATION_MAX_MS,
89
+ STARVATION_ESCALATE_MS,
81
90
  } = require('./lib/schedulerConfig.cjs');
82
91
  const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
83
92
  ? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
@@ -88,7 +97,7 @@ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
88
97
  const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
89
98
  ? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
90
99
  : JOB_OVERRUN_FLOOR_MS_DEFAULT;
91
- const { pickForProject, pickNextBatch, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
100
+ const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
92
101
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
93
102
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
94
103
  const queueHistory = require('./lib/queueHistory.cjs');
@@ -100,7 +109,8 @@ const queueOps = require('./queueOps.cjs');
100
109
  // home-dir layout.
101
110
  const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
102
111
  const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
103
- const { transitionJob, STATUS_HISTORY_CAP } = require('./lib/scheduleJobTransitions.cjs');
112
+ const agentModelResolve = require('./lib/agentModelResolve.cjs');
113
+ const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
104
114
  const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
105
115
  const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
106
116
  const { appendAuditEvent } = require('./lib/auditLog.cjs');
@@ -123,6 +133,7 @@ function resolveOriginSessionId(cwd, epicId) {
123
133
  return session && typeof session.claudeSessionId === 'string' ? session.claudeSessionId : null;
124
134
  }
125
135
  const sessionSlots = require('./lib/sessionSlots.cjs');
136
+ const quietMachineLease = require('./lib/quietMachineLease.cjs');
126
137
  const jobWorktree = require('./lib/jobWorktree.cjs');
127
138
  const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
128
139
  const queueStore = require('./lib/queueStore.cjs');
@@ -180,6 +191,21 @@ const RESULT_TEXT_TAIL_BYTES = 64 * 1024;
180
191
  const IDLE_OUTPUT_KILL_MS = 20 * 60_000;
181
192
  const IDLE_CHECK_INTERVAL_MS = 60_000;
182
193
 
194
+ // Foreground Bash budget for every spawned `claude -p` job (executor +
195
+ // investigation). The Claude Code harness auto-backgrounds any foreground
196
+ // Bash command past its own default (120s) or max (600s) timeout and returns
197
+ // a tool result promising a later notification — but a headless single-shot
198
+ // run has no later turn, so that notification can never arrive and the run
199
+ // dead-ends mid-verification with no commit and no verdict. Raising these
200
+ // via the child's env moves that trap out of reach of normal gate commands
201
+ // (test suites, builds). BASH_MAX_TIMEOUT_MS MUST stay strictly below
202
+ // IDLE_OUTPUT_KILL_MS with real margin: a long foreground Bash emits no
203
+ // stream-json events while it runs, so the log mtime stalls and the
204
+ // idle-tail watchdog above would SIGTERM the job mid-gate if the two ever
205
+ // crossed — trading one silent failure for another.
206
+ const BASH_DEFAULT_TIMEOUT_MS = 600_000; // 10 min
207
+ const BASH_MAX_TIMEOUT_MS = 900_000; // 15 min — must stay below IDLE_OUTPUT_KILL_MS
208
+
183
209
  // Boot reconciliation: a job left 'running' by an app restart/crash whose log
184
210
  // shows neither success nor a real failure result was merely interrupted — the
185
211
  // host died, the PRD didn't. Re-queue it up to this many times before giving up
@@ -205,6 +231,20 @@ const FINISH_PROTOCOL = `
205
231
  Once every acceptance-criteria line above is satisfied, finish in this EXACT
206
232
  sequence. Do not stop before the commit lands; committing is part of the job.
207
233
 
234
+ RUN VERIFICATION IN THE FOREGROUND — this applies to the whole run, not just
235
+ step 3 below: every test/typecheck/lint/build command you run, whether while
236
+ implementing the AC or during VERIFY, must run SYNCHRONOUSLY and you must wait
237
+ for it to return. Never start a verification command as a background task
238
+ (no background Bash) and then call Monitor, TaskOutput, or ScheduleWakeup to
239
+ pick up its result later — a headless \`claude -p\` run has no later turn, so
240
+ nothing ever delivers that notification and the run dies mid-verification with
241
+ no commit and no verdict. Your foreground Bash budget for this run is
242
+ ${BASH_DEFAULT_TIMEOUT_MS / 1000}s by default, up to ${BASH_MAX_TIMEOUT_MS / 1000}s max
243
+ — size your own \`timeout <n>\` wrapper (e.g. \`timeout ${Math.floor(BASH_MAX_TIMEOUT_MS / 1000)} npm test\`)
244
+ to fit inside that ceiling; if a gate command still cannot finish inside
245
+ budget, stop and emit SCHEDULER_VERDICT: FAIL with the reason instead of
246
+ deferring it.
247
+
208
248
  1. CODE REVIEW — run \`/code-review --fix\` on your changes and apply the fixes it
209
249
  surfaces (correctness first). For any finding you judge a false positive, say
210
250
  why in your result; do not silently skip it. If \`/code-review\` is not
@@ -287,6 +327,156 @@ function gitHead(cwd) {
287
327
  });
288
328
  }
289
329
 
330
+ // Return the current `git stash list` entries in cwd as raw lines
331
+ // "<hash> <ref> <subject>" (hash is stable even as ref indices shift when a
332
+ // new entry is pushed on top), or null when the guard does not apply (cwd is
333
+ // not a git work tree, git is missing, or the call errors). Never throws.
334
+ function stashList(cwd) {
335
+ return new Promise((resolve) => {
336
+ if (!cwd) { resolve(null); return; }
337
+ execFile(
338
+ 'git',
339
+ ['-C', cwd, 'stash', 'list', '--format=%H %gd %gs'],
340
+ { timeout: 10_000, windowsHide: true },
341
+ (err, stdout) => {
342
+ if (err) { resolve(null); return; }
343
+ resolve(String(stdout || '').split('\n').filter(Boolean));
344
+ },
345
+ );
346
+ });
347
+ }
348
+
349
+ // Parse one `stashList()` line into { hash, ref, subject }. Pure, exported
350
+ // for unit testing. Returns null for a malformed line.
351
+ function parseStashLine(line) {
352
+ const m = /^(\S+)\s+(\S+)\s+(.*)$/.exec(String(line || ''));
353
+ return m ? { hash: m[1], ref: m[2], subject: m[3] } : null;
354
+ }
355
+
356
+ // Paths touched by any commit landed in cwd strictly between headBefore and
357
+ // headAfter. Returns [] when no commit landed (headBefore === headAfter, or
358
+ // either is missing) — used by the shared-tree guard below to tell a path
359
+ // the job legitimately committed apart from a path that just silently went
360
+ // quiet with nothing to explain it. Never throws.
361
+ function pathsChangedSince(cwd, headBefore, headAfter) {
362
+ return new Promise((resolve) => {
363
+ if (!cwd || !headBefore || !headAfter || headBefore === headAfter) { resolve([]); return; }
364
+ execFile(
365
+ 'git',
366
+ ['-C', cwd, 'diff', '--name-only', `${headBefore}..${headAfter}`],
367
+ { timeout: 10_000, windowsHide: true },
368
+ (err, stdout) => { resolve(err ? [] : String(stdout || '').split('\n').filter(Boolean)); },
369
+ );
370
+ });
371
+ }
372
+
373
+ // Restore ONE specific stash ref (never a blanket pop of "whatever is on
374
+ // top") into cwd: apply, then drop only on a clean apply. On conflict the
375
+ // entry is left in place — never dropped, never forced — so the operator's
376
+ // own `git stash pop`/`apply` still works afterward. Never throws.
377
+ function restoreSpecificStash(cwd, ref) {
378
+ return new Promise((resolve) => {
379
+ execFile('git', ['-C', cwd, 'stash', 'apply', ref], { timeout: 10_000, windowsHide: true }, (applyErr, _stdout, applyStderr) => {
380
+ if (applyErr) {
381
+ resolve({ ok: false, error: String(applyStderr || applyErr.message || applyErr).trim().split('\n')[0] });
382
+ return;
383
+ }
384
+ execFile('git', ['-C', cwd, 'stash', 'drop', ref], { timeout: 10_000, windowsHide: true }, () => {
385
+ resolve({ ok: true });
386
+ });
387
+ });
388
+ });
389
+ }
390
+
391
+ // Diff a before/after `stashList()` pair plus a before/after dirty-path pair
392
+ // to find what an in-place job silently discarded from a tree it shares with
393
+ // something else (Incident: social-signals-trader 2026-09-01, a blanket
394
+ // `git stash` reverted a live operator config edit with no error anywhere).
395
+ // Two independent signals, either of which means the job discarded state it
396
+ // did not create:
397
+ // - newStashes: a stash entry now present that wasn't in the baseline —
398
+ // the job ran `git stash` itself.
399
+ // - reverted: a path that was dirty in the baseline, is clean now, and was
400
+ // not touched by any commit landed during the run — the job reset/
401
+ // checked-out over pre-existing uncommitted work without stashing it.
402
+ // Pure/no I/O — the guard's git calls happen at the call site
403
+ // (checkSharedTreeGuard). Exported for unit testing.
404
+ function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAfter, pathsCommittedDuringRun }) {
405
+ const beforeHashes = new Set((stashBefore || []).map((l) => parseStashLine(l)?.hash).filter(Boolean));
406
+ const newStashes = (stashAfter || [])
407
+ .map(parseStashLine)
408
+ .filter((e) => e && !beforeHashes.has(e.hash));
409
+ const dirtyAfterSet = new Set(dirtyAfter || []);
410
+ const committedSet = new Set(pathsCommittedDuringRun || []);
411
+ const reverted = (dirtyBefore || []).filter((p) => !dirtyAfterSet.has(p) && !committedSet.has(p));
412
+ return { newStashes, reverted };
413
+ }
414
+
415
+ // Post-run shared-tree guard for an IN-PLACE job (worktree.ok === false —
416
+ // callers must gate on that; a worktree-isolated run's git state can never
417
+ // leak into guardCwd, so there is nothing here to check). Best-effort: never
418
+ // throws, never changes the job's exit code. Restores exactly one
419
+ // executor-created stash (never guesses when there are 2+); reports anything
420
+ // it can't safely resolve on the returned object so the caller can surface it
421
+ // on the job row instead of finishing silently green.
422
+ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug }) {
423
+ try {
424
+ const [stashAfter, headAfter] = await Promise.all([
425
+ module.exports.stashList(cwd),
426
+ module.exports.gitHead(cwd),
427
+ ]);
428
+ const pathsCommittedDuringRun = await module.exports.pathsChangedSince(cwd, headBefore, headAfter);
429
+ // First pass: which stashes are new. Decided before charging anything
430
+ // against dirtyBaseline — a path this run's own stash covers must not be
431
+ // judged "reverted" using dirty state captured before the restore below
432
+ // has had a chance to bring it back.
433
+ const { newStashes } = module.exports.evaluateSharedTreeGuard({
434
+ stashBefore: stashBaseline,
435
+ stashAfter,
436
+ dirtyBefore: [],
437
+ dirtyAfter: [],
438
+ pathsCommittedDuringRun,
439
+ });
440
+
441
+ const result = {};
442
+ if (newStashes.length === 1) {
443
+ const [entry] = newStashes;
444
+ const restore = await module.exports.restoreSpecificStash(cwd, entry.ref);
445
+ if (restore.ok) {
446
+ result.restoredStash = entry.ref;
447
+ console.log(`[scheduler] ${slug}: restored a stash the job created in the shared tree (${entry.ref})`);
448
+ } else {
449
+ result.restoreFailed = `${entry.ref}: ${restore.error || 'apply failed'}`;
450
+ console.error(`[scheduler] ${slug}: shared-tree guard could not restore ${entry.ref}: ${restore.error}`);
451
+ }
452
+ } else if (newStashes.length > 1) {
453
+ result.ambiguousStashes = newStashes.map((e) => e.ref);
454
+ console.error(`[scheduler] ${slug}: shared-tree guard found ${newStashes.length} stashes the job created — ambiguous, not auto-restoring (${result.ambiguousStashes.join(', ')})`);
455
+ }
456
+
457
+ // Second pass: recompute "reverted" against the tree's dirty state AFTER
458
+ // any restore attempt above, so a path that came back via a successfully
459
+ // restored stash is not ALSO reported as an unexplained revert (it was
460
+ // explained — by the stash this guard just restored).
461
+ const dirtyAfter = await module.exports.uncommittedChanges(cwd);
462
+ const { reverted } = module.exports.evaluateSharedTreeGuard({
463
+ stashBefore: stashBaseline,
464
+ stashAfter,
465
+ dirtyBefore: dirtyBaseline,
466
+ dirtyAfter,
467
+ pathsCommittedDuringRun,
468
+ });
469
+ if (reverted.length) {
470
+ result.reverted = reverted;
471
+ console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
472
+ }
473
+ return (result.restoredStash || result.restoreFailed || result.ambiguousStashes || result.reverted) ? result : null;
474
+ } catch (e) {
475
+ console.error(`[scheduler] ${slug}: shared-tree guard error`, e);
476
+ return null;
477
+ }
478
+ }
479
+
290
480
  // True when cwd is inside a git repository. Used to keep a non-git cwd (e.g.
291
481
  // a scratch dir like /tmp) from ever being handed to an investigation's
292
482
  // fix-plan as its cwd — the commit guard, worktree isolation, and
@@ -690,6 +880,43 @@ async function safeSlugPath(slug) {
690
880
  return safeSlugPathIn(dir, slug);
691
881
  }
692
882
 
883
+ /**
884
+ * The two distinct failure modes safeSlugPath collapses into one nullable
885
+ * return (the defect this fixes — see the PRD that added this helper's
886
+ * Goal): a slug that fails SCHEDULE_SLUG_RE is a caller mistake ("invalid
887
+ * slug"), while a well-formed slug that exists in no candidate PRD dir is a
888
+ * lookup miss ("unknown slug") — an agent retrying the first as if it were
889
+ * the second (or vice versa) burns a turn on the wrong fix. Returns
890
+ * `{ ok: true, path }` or `{ ok: false, reason: 'invalid-slug' | 'not-found' }`.
891
+ * `cwd`, if given, narrows the search to that one project's own PRD dirs
892
+ * (prdDirForCwd + its Epic-scoped dirs — same pattern as getPrdParsed);
893
+ * omitted, it searches every candidate dir machine-wide via findPrdDir.
894
+ */
895
+ async function resolveSlugOrReason(slug, cwd) {
896
+ if (!SCHEDULE_SLUG_RE.test(slug)) return { ok: false, reason: 'invalid-slug' };
897
+ if (cwd) {
898
+ for (const dir of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
899
+ const p = safeSlugPathIn(dir, slug);
900
+ if (!p) continue;
901
+ try {
902
+ await fsp.access(p);
903
+ return { ok: true, path: p };
904
+ } catch { /* not in this dir — try the next candidate */ }
905
+ }
906
+ return { ok: false, reason: 'not-found' };
907
+ }
908
+ const dir = await findPrdDir(slug);
909
+ if (!dir) return { ok: false, reason: 'not-found' };
910
+ const p = safeSlugPathIn(dir, slug);
911
+ if (!p) return { ok: false, reason: 'not-found' };
912
+ return { ok: true, path: p };
913
+ }
914
+
915
+ /** Actionable message for `resolveSlugOrReason`'s 'not-found' reason. */
916
+ function unknownSlugMessage(slug) {
917
+ return `unknown slug "${slug}": no PRD file with that name in any known project — call scheduler_list_prds (optionally with cwd) to see what exists`;
918
+ }
919
+
693
920
  /**
694
921
  * Move a completed job's `<slug>.md` out of its PRD dir into that dir's
695
922
  * sibling `prds-archived/`, so a finished slug can't be re-fired by the
@@ -1098,6 +1325,73 @@ function findOverrunningJobs(jobs, now, { factor, floorMs } = {}) {
1098
1325
  return out;
1099
1326
  }
1100
1327
 
1328
+ /**
1329
+ * findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
1330
+ * → [{ slug, cwd, ageMs, restoreStatus }]
1331
+ *
1332
+ * Pure (besides the warn-log side effect on the two unprovable-age cases
1333
+ * below), no other IO. spawnInvestigation's own restore of a job's
1334
+ * pre-investigation status runs entirely inside the process that spawned the
1335
+ * probe (its withChildAndLog onExit handler, or the synchronous-throw catch
1336
+ * path) — so a job left 'investigating' when the app itself dies or restarts
1337
+ * has NOTHING left to restore it. The comment at spawnInvestigation's onExit
1338
+ * asserts "'investigating' must never be the job's resting state"; this is
1339
+ * the sweep that makes that true across a restart, not just within one.
1340
+ *
1341
+ * A row qualifies only when ALL of:
1342
+ * - status is 'investigating'
1343
+ * - its most recent transition INTO 'investigating' (statusHistory's last
1344
+ * `to === 'investigating'` entry — a job can be investigated more than
1345
+ * once across its life, e.g. a retried auto-fix) is older than `maxMs`
1346
+ * - it has no live probe process behind it (checked via runtime.pid, set by
1347
+ * spawnInvestigation once its child spawns and cleared on every restore
1348
+ * path, the same shape reapDeadRunningJobs already uses for 'running' rows)
1349
+ *
1350
+ * `restoreStatus` is that transition entry's `from` — the exact value
1351
+ * spawnInvestigation itself would have restored to (`failedJob.status ||
1352
+ * 'failed'`), which for a row that already finished and recorded
1353
+ * finishedAt+exitCode (the burrow-834 shape) is whatever terminal status was
1354
+ * computed for that outcome BEFORE the probe was spawned — this sweep never
1355
+ * re-derives it from exitCode, only replays the already-recorded decision.
1356
+ *
1357
+ * A row with no recoverable transition timestamp cannot have its age proven,
1358
+ * so it is warn-logged and left alone rather than guessed at — same posture
1359
+ * as findStaleQuarantinedJobs/findOverrunningJobs.
1360
+ *
1361
+ * `restoreStatus` is validated against LEGAL_TRANSITIONS['investigating']
1362
+ * before being returned — `statusHistory`'s `from` should only ever be
1363
+ * 'failed' or 'needs_review' (the only two states LEGAL_TRANSITIONS allows
1364
+ * into 'investigating'), but a corrupted/unexpected value must not be handed
1365
+ * straight to transitionJob: an illegal target is refused outright (row stays
1366
+ * stuck at 'investigating', re-detected as stranded every sweep with no path
1367
+ * out), so an out-of-set `from` falls back to 'failed' here instead.
1368
+ */
1369
+ const INVESTIGATING_RESTORE_TARGETS = new Set(LEGAL_TRANSITIONS.investigating);
1370
+ function findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive) {
1371
+ const out = [];
1372
+ for (const j of jobs ?? []) {
1373
+ if (j.status !== 'investigating') continue;
1374
+ const entries = (j.statusHistory || []).filter((h) => h.to === 'investigating');
1375
+ const entry = entries[entries.length - 1];
1376
+ if (!entry) {
1377
+ console.warn(`[scheduler] findStrandedInvestigations: ${j.slug} is 'investigating' with no statusHistory entry recording the transition — cannot prove age, leaving alone`);
1378
+ continue;
1379
+ }
1380
+ const since = Date.parse(entry.at ?? '');
1381
+ if (Number.isNaN(since)) {
1382
+ console.warn(`[scheduler] findStrandedInvestigations: ${j.slug} has an unparseable investigating-transition timestamp (${entry.at}) — cannot prove age, leaving alone`);
1383
+ continue;
1384
+ }
1385
+ const ageMs = now - since;
1386
+ if (ageMs < maxMs) continue; // a live probe must not be yanked out from under itself
1387
+ const pid = j.runtime?.pid;
1388
+ if (pid && isAlive(pid)) continue; // probe genuinely still running — not stranded
1389
+ const restoreStatus = INVESTIGATING_RESTORE_TARGETS.has(entry.from) ? entry.from : 'failed';
1390
+ out.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, restoreStatus });
1391
+ }
1392
+ return out;
1393
+ }
1394
+
1101
1395
  // An empty queue and an unreadable queue are NOT the same thing, and
1102
1396
  // conflating them is destructive: reconcile() treats every PRD .md with no
1103
1397
  // matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
@@ -1443,9 +1737,11 @@ async function reconcile(state) {
1443
1737
  // membership, so moving the file between Epic dirs must re-point the row.
1444
1738
  epicId: p.epicId ?? job.epicId ?? null,
1445
1739
  dependsOn: p.dependsOn,
1740
+ quietMachine: p.quietMachine === true,
1446
1741
  originSessionId: job.originSessionId
1447
1742
  ?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
1448
1743
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1744
+ agentType: p.agentType ?? job.agentType ?? null,
1449
1745
  };
1450
1746
  // Adopt path: a row parked 'quarantined' (no createdVia provenance when
1451
1747
  // discovered) whose PRD file now carries a stamp — written via the
@@ -1554,8 +1850,10 @@ async function reconcile(state) {
1554
1850
  sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
1555
1851
  epicId: p.epicId ?? inv.row?.epicId ?? null,
1556
1852
  dependsOn: p.dependsOn,
1853
+ quietMachine: p.quietMachine === true,
1557
1854
  originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
1558
1855
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1856
+ agentType: p.agentType ?? inv.row?.agentType ?? null,
1559
1857
  };
1560
1858
  const reason = `reconcile: repaired invalid status ${JSON.stringify(oldStatus)}`;
1561
1859
  // A repair is not a lifecycle transition — the corrupted `status` was
@@ -1674,9 +1972,15 @@ async function reconcile(state) {
1674
1972
  sourceTabId: p.sourceTabId,
1675
1973
  epicId: p.epicId ?? null,
1676
1974
  dependsOn: p.dependsOn,
1975
+ quietMachine: p.quietMachine === true,
1677
1976
  originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
1678
1977
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1978
+ agentType: p.agentType ?? null,
1679
1979
  status: 'pending',
1980
+ // Enqueue time (PRD 1086/1087): the cross-project fairness tiebreak and
1981
+ // the starvation escalation both need a provable age for a pending row;
1982
+ // before this stamp a freshly minted row carried no timestamp at all.
1983
+ queuedAt: new Date().toISOString(),
1680
1984
  runId: null,
1681
1985
  startedAt: null,
1682
1986
  finishedAt: null,
@@ -1864,6 +2168,9 @@ function drainDeferredInvestigation() {
1864
2168
  let cancelToken = { cancelled: false };
1865
2169
  // Last memory-gate observation; included in snapshot for renderer visibility.
1866
2170
  let lastMemGate = null;
2171
+ // CPU-load launch gate (PRD 1085, lib/loadGate.cjs) — innermost launch
2172
+ // predicate after pool → project cap → memory. Withholds launches only.
2173
+ const loadGate = createLoadGate();
1867
2174
 
1868
2175
  // Last tickQueue outcome, kept for the UI. tickQueue already computes a precise
1869
2176
  // reason for every way a batch can come back empty (dependency holds, slot
@@ -1924,6 +2231,10 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
1924
2231
  lastRunAt: state.lastRunAt,
1925
2232
  nextReset: getNextResetCached(),
1926
2233
  paused: state.paused,
2234
+ // Launch circuit breaker (issue #11): which personas cannot launch right
2235
+ // now and why, plus any degraded-mode env in force. Empty objects when healthy.
2236
+ launchBlocks: state.launchBlocks ?? {},
2237
+ launchMitigations: state.launchMitigations ?? {},
1927
2238
  utilization: cachedUtilization,
1928
2239
  pollHealth: {
1929
2240
  lastPollAt,
@@ -1932,6 +2243,8 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
1932
2243
  lastFailureKind,
1933
2244
  },
1934
2245
  memGate: lastMemGate,
2246
+ // Why nothing is launching when the box is CPU-saturated (PRD 1085).
2247
+ loadGate: loadGate.snapshot(),
1935
2248
  lastTick,
1936
2249
  // The machine-wide slot pool IS the concurrency limit — there is no
1937
2250
  // separate scheduler cap any more. `source` distinguishes the
@@ -2065,7 +2378,17 @@ async function setPaused(reason, resumeAtIso) {
2065
2378
 
2066
2379
  async function clearPause(source) {
2067
2380
  if (resumeTimer) { clearTimeout(resumeTimer); resumeTimer = null; }
2381
+ const humanOverride = source === 'manual' || source === 'run-now';
2068
2382
  const wasPaused = await mutate((s) => {
2383
+ // A human Resume / Run now also re-closes every launch circuit breaker:
2384
+ // the operator is asserting the environment is fixed (CLI updated,
2385
+ // re-logged-in). The next dispatch of each persona is its probe; if the
2386
+ // environment is still broken the breaker simply re-arms.
2387
+ if (humanOverride && s.launchBlocks && Object.keys(s.launchBlocks).length) {
2388
+ console.log(`[scheduler] clearPause (${source}): clearing launch blocks [${Object.keys(s.launchBlocks).join(', ')}]`);
2389
+ for (const j of s.jobs) if (j.status === 'pending' && j.heldReason && /^launch blocked/.test(j.heldReason)) delete j.heldReason;
2390
+ s.launchBlocks = {};
2391
+ }
2069
2392
  if (!s.paused) return false;
2070
2393
  console.log(`[scheduler] clearPause (${source || 'manual'})`);
2071
2394
  s.paused = null;
@@ -2113,6 +2436,29 @@ function resetJobFields(job, errorMsg, opts = {}) {
2113
2436
  job.error = errorMsg ?? null;
2114
2437
  delete job.runtime;
2115
2438
  delete job.verifierVerdict;
2439
+ delete job.uncommittedPaths;
2440
+ delete job.resumeRecoveryAttempted;
2441
+ // Same "this run's outcome, not durable across a reset" category as the
2442
+ // fields above — a stale 'archive' recoveryAction from a prior life of this
2443
+ // slug must never survive a reset and silently exclude a genuinely-new
2444
+ // needs_review episode from selectAutoFixTargets (applyRcaClassification
2445
+ // only overwrites these on a successful RCA write, so without this they
2446
+ // can otherwise linger forever when RCA is disabled or errors).
2447
+ delete job.rcaFailureClass;
2448
+ delete job.rcaRecoveryAction;
2449
+ // Like exitCode: this run's outcome, not durable across a reset — a stale
2450
+ // leak badge from a prior attempt must not linger once the job re-fires.
2451
+ delete job.leakedDescendants;
2452
+ // A pending row is about to re-run fresh — a stale leftover badge or a
2453
+ // stale pre-run baseline from the attempt that just ended must not linger
2454
+ // and be mistaken for THIS (not-yet-run) attempt's own output. spawnJob
2455
+ // persists a brand-new guardBaseline at the next dispatch.
2456
+ delete job.guardBaseline;
2457
+ delete job.guardHeadBefore;
2458
+ delete job.leftoverPaths;
2459
+ delete job.leftoverCount;
2460
+ delete job.leftoverPathsTruncated;
2461
+ delete job.preRunDirtyPaths;
2116
2462
  // Deliberately NOT deleting job.landedCommit: it must outlive a reset so a
2117
2463
  // re-fired run of this same slug can pass it to verifyRun as
2118
2464
  // priorLandedCommit (pass_no_commit_prior_run_verified exemption).
@@ -2609,9 +2955,15 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
2609
2955
  * tree was dirty — closed by widening that call site's condition, not by
2610
2956
  * changing this function's four defenses below, which still apply to both
2611
2957
  * shapes identically:
2612
- * - siblingRunning: a concurrent job in the same cwd makes working-tree
2613
- * evidence unreliable in both directions (extra dirt OR a clean tree
2614
- * that isn't this job's doing).
2958
+ * - siblingRunning: on a SHARED tree only — a concurrent job in the same
2959
+ * cwd makes working-tree evidence unreliable in both directions (extra
2960
+ * dirt OR a clean tree that isn't this job's doing). Suppressed by
2961
+ * ranInWorktree: when this job ran in its own git worktree, the
2962
+ * newly-dirty set and the integrated HEAD are attributable to this job
2963
+ * alone regardless of what siblings were doing concurrently in their own
2964
+ * worktrees, so the excuse does not apply (PRD 109 shipped 'completed'
2965
+ * with nothing committed specifically because this carve-out fired
2966
+ * unconditionally during a high-concurrency run).
2615
2967
  * - jobSelfCommitted: HEAD moved during the run, so the job's deliverable
2616
2968
  * landed even if dirt (from a concurrent actor) remains.
2617
2969
  * - legitimateNoOp (COMPLETED_EQUIVALENT_VERDICTS): runVerify.cjs's own
@@ -2631,8 +2983,8 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
2631
2983
  * still a genuine finish-protocol violation (incident:
2632
2984
  * 523-fix-bounded-fix-plan-retry, 2026-07-12).
2633
2985
  */
2634
- function commitGuardVerdict({ newlyDirty, siblingRunning, jobSelfCommitted, legitimateNoOp, isFixPlanJob, verifyResult }) {
2635
- if (siblingRunning || jobSelfCommitted || legitimateNoOp) return null;
2986
+ function commitGuardVerdict({ newlyDirty, siblingRunning, ranInWorktree, jobSelfCommitted, legitimateNoOp, isFixPlanJob, verifyResult, salvagePatch }) {
2987
+ if ((siblingRunning && !ranInWorktree) || jobSelfCommitted || legitimateNoOp) return null;
2636
2988
  const dirty = newlyDirty || [];
2637
2989
  if (dirty.length === 0 && isFixPlanJob) return null;
2638
2990
 
@@ -2651,12 +3003,196 @@ function commitGuardVerdict({ newlyDirty, siblingRunning, jobSelfCommitted, legi
2651
3003
  }
2652
3004
 
2653
3005
  const sample = dirty.slice(0, 3).join(', ');
3006
+ const salvageNote = salvagePatch ? ` — recoverable from salvage patch ${salvagePatch}` : '';
2654
3007
  return {
2655
3008
  verdict: 'uncommitted_changes',
2656
- reason: `finish protocol incomplete: ${dirty.length} uncommitted file(s) left in working tree (e.g. ${sample})`,
3009
+ reason: `finish protocol incomplete: ${dirty.length} uncommitted file(s) left in working tree (e.g. ${sample})${salvageNote}`,
2657
3010
  downgradeTo: 'needs_review',
2658
3011
  annotations: carried.length ? carried : undefined,
3012
+ // The exact dirty-path list, persisted on the job row (see the
3013
+ // commit-guard call site) so a later resume-recovery attempt
3014
+ // (selectResumeRecoveryTarget) can name these paths without re-running
3015
+ // `git status` against a tree that may have moved on since.
3016
+ dirtyPaths: dirty,
3017
+ };
3018
+ }
3019
+
3020
+ // Every path list this job leaves attributed on the row is capped here so a
3021
+ // pathological run (thousands of newly-dirty files) never bloats queue.json
3022
+ // or history.jsonl — the count is still recorded in full via leftoverCount,
3023
+ // only the displayed sample is capped.
3024
+ const LEFTOVER_PATHS_CAP = 50;
3025
+
3026
+ /**
3027
+ * Pure: turn a newly-dirty path list (or null, meaning "couldn't tell" —
3028
+ * never "left nothing") into the `leftoverPaths`/`leftoverCount`/
3029
+ * `leftoverPathsTruncated` triple stamped on a terminal job row, or null when
3030
+ * there is nothing to attribute (empty list, or the list itself is
3031
+ * unavailable). One shape for both the worktree-leftover path and the
3032
+ * in-place baseline-delta path — see this function's callers in spawnJob and
3033
+ * reapDeadRunningJobs, both of which diff against a persisted pre-run
3034
+ * baseline so a human's or a sibling's pre-existing WIP is never
3035
+ * misattributed to this job.
3036
+ */
3037
+ function leftoverFieldsFrom(paths) {
3038
+ if (!Array.isArray(paths) || paths.length === 0) return null;
3039
+ const fields = {
3040
+ leftoverPaths: paths.slice(0, LEFTOVER_PATHS_CAP),
3041
+ leftoverCount: paths.length,
2659
3042
  };
3043
+ if (paths.length > LEFTOVER_PATHS_CAP) fields.leftoverPathsTruncated = true;
3044
+ return fields;
3045
+ }
3046
+
3047
+ /** Stamps (or clears) the leftover-attribution fields on a job row in place. */
3048
+ function applyLeftoverFields(row, paths) {
3049
+ delete row.leftoverPaths;
3050
+ delete row.leftoverCount;
3051
+ delete row.leftoverPathsTruncated;
3052
+ const fields = leftoverFieldsFrom(paths);
3053
+ if (fields) Object.assign(row, fields);
3054
+ }
3055
+
3056
+ // Same bloat concern as LEFTOVER_PATHS_CAP, applied to the PRE-run dirty
3057
+ // snapshot (foreign WIP the job did not create) instead of the post-run
3058
+ // leftover delta.
3059
+ const PRE_RUN_DIRTY_PATHS_CAP = 200;
3060
+
3061
+ /**
3062
+ * Pure: cap a dirty-path list at PRE_RUN_DIRTY_PATHS_CAP, appending a
3063
+ * `+N more` marker entry when truncated, so queue.json/history.jsonl never
3064
+ * take on an unbounded row for a pathologically dirty shared tree. Returns
3065
+ * [] for null/empty input (never null) — callers gate storage/prompt
3066
+ * injection on `.length` the same way carriedPaths already does.
3067
+ */
3068
+ function capDirtyPaths(paths, cap = PRE_RUN_DIRTY_PATHS_CAP) {
3069
+ if (!Array.isArray(paths) || paths.length === 0) return [];
3070
+ if (paths.length <= cap) return paths.slice();
3071
+ return [...paths.slice(0, cap), `+${paths.length - cap} more`];
3072
+ }
3073
+
3074
+ // Stable, machine-greppable delimiter — a downstream PRD (verifier scoring
3075
+ // foreign-WIP test failures separately) greps the executor log for this
3076
+ // exact marker, so its text must never be reworded casually.
3077
+ const FOREIGN_WIP_DELIMITER = '--- FOREIGN WORKING-TREE STATE (not your work) ---';
3078
+ const FOREIGN_WIP_END_DELIMITER = '--- END FOREIGN WORKING-TREE STATE ---';
3079
+
3080
+ /**
3081
+ * Pure: build the executor-prompt section warning about pre-existing dirty
3082
+ * paths this job does not own — either base WIP carried into an isolated
3083
+ * worktree (PRD 1094's carriedPaths, checked first since it's the more
3084
+ * specific/authoritative case) or the raw pre-run dirty snapshot of a shared
3085
+ * (non-isolated) tree. Returns '' when both lists are empty so a clean spawn
3086
+ * produces a byte-identical prompt to before this section existed.
3087
+ */
3088
+ function buildForeignWipSection({ preRunDirtyPaths, carriedPaths } = {}) {
3089
+ const carried = Array.isArray(carriedPaths) ? carriedPaths.filter(Boolean) : [];
3090
+ if (carried.length) {
3091
+ return [
3092
+ FOREIGN_WIP_DELIMITER,
3093
+ 'This job is running in an isolated git worktree, but the following paths carry uncommitted base-tree work-in-progress that was carried into this checkout so the tree is self-consistent. The authoritative copy of these files lives in the MAIN tree, not this worktree.',
3094
+ 'These files were already modified before this job started. They are NOT this job\'s work:',
3095
+ ...carried.map((p) => ` ${p}`),
3096
+ 'Do not stage, commit, revert, or stash these paths. A test failure confined to these paths is not this job\'s regression.',
3097
+ FOREIGN_WIP_END_DELIMITER,
3098
+ ].join('\n');
3099
+ }
3100
+ const dirty = Array.isArray(preRunDirtyPaths) ? preRunDirtyPaths.filter(Boolean) : [];
3101
+ if (dirty.length) {
3102
+ return [
3103
+ FOREIGN_WIP_DELIMITER,
3104
+ 'This job is running in a SHARED working tree (not isolated in its own worktree). The following paths were already modified when this job started:',
3105
+ ...dirty.map((p) => ` ${p}`),
3106
+ 'These files are NOT this job\'s work. Do not stage, commit, revert, or stash them. A test failure confined to these paths is not this job\'s regression.',
3107
+ FOREIGN_WIP_END_DELIMITER,
3108
+ ].join('\n');
3109
+ }
3110
+ return '';
3111
+ }
3112
+
3113
+ /**
3114
+ * Resume-first recovery (PRD 1111). A job parked in needs_review with verdict
3115
+ * 'uncommitted_changes' has a live claude session (job.sessionId, minted by
3116
+ * spawnJob's `--session-id`) that already has full context of the work it
3117
+ * left uncommitted — resuming it via `claude -p --resume <sessionId>` lets it
3118
+ * finish its own finish-protocol COMMIT step, instead of spawnInvestigation
3119
+ * cold-reading the log to author a fix-plan PRD that a FRESH session then has
3120
+ * to re-derive that same context for. Pure/no I/O so the eligibility rule can
3121
+ * be unit-tested directly, matching classifyFailureOutcome/commitGuardVerdict.
3122
+ *
3123
+ * Bounded to exactly one attempt via job.resumeRecoveryAttempted, stamped
3124
+ * atomically with the 'running' transition inside spawnJob's own dispatch
3125
+ * mutate (see spawnJob) — never here — so a crash between this function
3126
+ * returning a target and the resume child actually spawning cannot leave the
3127
+ * job re-eligible.
3128
+ *
3129
+ * Kill-switch: SM_RESUME_RECOVERY_DISABLE=1 restores today's behaviour
3130
+ * exactly (always returns null), mirroring SM_RCA_DISABLE/SM_DOD_DISABLE.
3131
+ */
3132
+ function selectResumeRecoveryTarget(job) {
3133
+ if (process.env.SM_RESUME_RECOVERY_DISABLE === '1') return null;
3134
+ if (!job || job.status !== 'needs_review') return null;
3135
+ if (job.verifierVerdict !== 'uncommitted_changes') return null;
3136
+ if (typeof job.sessionId !== 'string' || job.sessionId.length === 0) return null;
3137
+ if (job.resumeRecoveryAttempted === true) return null;
3138
+ const dirtyPaths = Array.isArray(job.uncommittedPaths)
3139
+ ? job.uncommittedPaths.filter((p) => typeof p === 'string' && p.length > 0)
3140
+ : [];
3141
+ if (!dirtyPaths.length) return null;
3142
+ return { slug: job.slug, sessionId: job.sessionId, dirtyPaths, salvagePatch: job.salvagePatch || null };
3143
+ }
3144
+
3145
+ /**
3146
+ * Short deterministic preamble for a resume-recovery dispatch — NEVER the
3147
+ * original PRD body (the resumed session already has that in its own
3148
+ * conversation history; re-embedding it would just waste context and risk
3149
+ * contradicting whatever state the session actually left behind). Names the
3150
+ * exact paths recorded on the parked job row so the resumed run can verify
3151
+ * them on disk before trusting them, rather than re-deriving them itself.
3152
+ */
3153
+ function buildResumeRecoveryPreamble({ dirtyPaths, salvagePatch }) {
3154
+ const pathList = dirtyPaths.map((p) => `- ${p}`).join('\n');
3155
+ const salvageLine = salvagePatch
3156
+ ? `\nA salvage patch of this work was also captured at: ${salvagePatch} — apply it if any of the paths above are missing from the working tree.\n`
3157
+ : '';
3158
+ return `RESUME RECOVERY: your previous run in this same session left uncommitted work on disk and exited before the finish protocol's COMMIT step ran. This is a continuation of that same session, not a new task — do not restart from scratch.
3159
+
3160
+ The following path(s) were recorded as uncommitted when this job was parked for review:
3161
+ ${pathList}
3162
+ ${salvageLine}
3163
+ Do the following now:
3164
+ 1. Run \`git status\` and verify each path above is present on disk and reflects your intended work. If a path is missing, investigate before recreating it — don't blindly redo work that may already be committed or salvaged elsewhere.
3165
+ 2. Run the project's verification gate (typecheck/lint/tests) in the FOREGROUND — wait for it to finish and read its real exit code before proceeding. Do not background it.
3166
+ 3. If the gate is green, stage exactly the paths you created or modified for this work and commit them: \`git add <path> [<path>...] && git commit -m "<type>(<scope>): <summary>"\`.
3167
+ 4. If the gate is red, fix it, then commit.
3168
+
3169
+ As the LAST LINE of your final result text, emit exactly one of:
3170
+ SCHEDULER_VERDICT: PASS
3171
+ SCHEDULER_VERDICT: FAIL <one-line reason>
3172
+ Print PASS only once the commit above has actually landed.`;
3173
+ }
3174
+
3175
+ /**
3176
+ * Pure argv builder for a `claude -p` child spawn, shared so the
3177
+ * resume-vs-fresh-session choice is made in exactly one place. `resume`
3178
+ * selects `--resume <sessionId>` (reconnect) INSTEAD of `--session-id
3179
+ * <sessionId>` (mint) — the two flags are mutually exclusive, never both.
3180
+ * `--model` is always explicit (never left to the CLI's drifting default —
3181
+ * see conventions.md). `systemPrompt`, when given (the PRD's `agentType`
3182
+ * persona body, resolved by agentModelResolve.cjs's resolvePrdPersonaForSpawn),
3183
+ * is passed as `--append-system-prompt` so the executor IS that persona at
3184
+ * launch rather than being asked in prose to adopt one.
3185
+ */
3186
+ function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }) {
3187
+ return [
3188
+ '-p', prompt,
3189
+ '--model', model,
3190
+ ...(systemPrompt ? ['--append-system-prompt', systemPrompt] : []),
3191
+ '--dangerously-skip-permissions',
3192
+ '--output-format', 'stream-json',
3193
+ '--verbose',
3194
+ ...(resume ? ['--resume', sessionId] : ['--session-id', sessionId]),
3195
+ ];
2660
3196
  }
2661
3197
 
2662
3198
  // ---------- execution ----------
@@ -2677,7 +3213,7 @@ function pickRunDir() {
2677
3213
  * Watchdogs are declared as an array; the result-tailer's exit-code mapping
2678
3214
  * (success+killedBySignal → 0) is scheduler-specific and lives in onExit.
2679
3215
  */
2680
- async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
3216
+ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget = null, foreignWip = null, launchEnv = null) {
2681
3217
  const logPath = path.join(runDir, `${job.slug}.log`);
2682
3218
  const metaPath = path.join(runDir, `${job.slug}.meta.json`);
2683
3219
  // `cwd` stays the MAIN tree throughout — PRD lookup (findPrdDir/prdPathForJob)
@@ -2687,7 +3223,10 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2687
3223
  const cwd = job.cwd || defaultCwd;
2688
3224
  const spawnCwd = execCwd || cwd;
2689
3225
  const startedAt = Date.now();
2690
- const sessionId = randomUUID();
3226
+ // Resume mode (PRD 1111) reconnects to the SAME session that left the
3227
+ // uncommitted work — reusing its id via `--resume` instead of minting a
3228
+ // fresh one via `--session-id` is the entire point of the recovery.
3229
+ const sessionId = resumeTarget ? resumeTarget.sessionId : randomUUID();
2691
3230
 
2692
3231
  // Phase 1: open log fd so we can emit pre-spawn diagnostics (early-exit
2693
3232
  // error paths) before the child is created. withChildAndLog takes ownership
@@ -2712,14 +3251,23 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2712
3251
  return { exitCode: -1, durationMs: 0, error: errMsg, sessionId };
2713
3252
  }
2714
3253
 
3254
+ let prompt;
3255
+ let prdPath = null;
3256
+ if (resumeTarget) {
3257
+ // Resume mode (PRD 1111): a short deterministic preamble naming the
3258
+ // recorded dirty paths, NEVER the original PRD body — the resumed
3259
+ // session already has that in its own conversation history via
3260
+ // --resume, and re-embedding it here would just contradict whatever
3261
+ // state the session actually left on disk.
3262
+ prompt = buildResumeRecoveryPreamble({ dirtyPaths: resumeTarget.dirtyPaths, salvagePatch: resumeTarget.salvagePatch });
3263
+ } else {
2715
3264
  // Read full PRD body fresh from disk (queue stored only the preview).
2716
3265
  // Resolve through findPrdDir's full candidate search (legacy flat dir +
2717
3266
  // every project's Epic-scoped dirs) first, so the common case — a live
2718
3267
  // Epic-scoped PRD — is a first-try hit instead of probing the retired flat
2719
3268
  // dir and only then falling back.
2720
- let prompt;
2721
3269
  const resolvedDir = await findPrdDir(job.slug);
2722
- let prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
3270
+ prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
2723
3271
  try {
2724
3272
  const parsed = await parsePrd(prdPath);
2725
3273
  // The review → security-review → verify → commit finish sequence is
@@ -2775,14 +3323,17 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2775
3323
  return { exitCode: -1, durationMs: 0, error: e?.message };
2776
3324
  }
2777
3325
  }
3326
+ } // end resumeTarget ? preamble : normal-PRD-read
2778
3327
 
3328
+ let contextDigestApplied = false;
3329
+ let originSessionId = null;
3330
+ if (!resumeTarget) {
2779
3331
  // Prepend the Epic's own session digest (PRD 950/958) when this job traces
2780
3332
  // back to a known Epic — additive only, never mutates the PRD body itself.
2781
3333
  // A missing/unresolved epicId or a digest build failure is a silent no-op:
2782
3334
  // the PRD's own body must remain sufficient to complete the job on its own.
2783
3335
  const digestEpicId = job.epicId ?? job.sourcePromptId ?? null;
2784
- const originSessionId = resolveOriginSessionId(cwd, digestEpicId);
2785
- let contextDigestApplied = false;
3336
+ originSessionId = resolveOriginSessionId(cwd, digestEpicId);
2786
3337
  let digestText = '';
2787
3338
  if (originSessionId) {
2788
3339
  try {
@@ -2793,12 +3344,39 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2793
3344
  digestText = '';
2794
3345
  }
2795
3346
  }
3347
+ // Quiet-machine degraded dispatch (PRD 1107): this job opted into
3348
+ // `quietMachine: true` but waited past quietMachineWaitMs() without the
3349
+ // machine ever going quiet, so pickNextBatch dispatched it anyway rather
3350
+ // than wedge the queue forever. Told to the executor as a plain prompt
3351
+ // line — its own wall-clock/timing acceptance criteria were measured (or
3352
+ // will be measured) under CPU contention from sibling jobs, not on a
3353
+ // quiet machine, so it should not report a timing result as trustworthy
3354
+ // without saying so.
3355
+ if (job.quietLeaseDegraded === true) {
3356
+ prompt = `NOTE: this job requested \`quietMachine: true\` but the machine never went idle within the `
3357
+ + `configured wait window, so it was dispatched anyway (degraded). Any timing/frame-rate/performance `
3358
+ + `measurement in this run may be affected by CPU contention from other concurrent jobs — say so explicitly `
3359
+ + `in your result rather than reporting it as a clean measurement.\n\n${prompt}`;
3360
+ }
2796
3361
  // Always route through composeExecutorPrompt (even with an empty digest)
2797
3362
  // so the finish protocol is appended in the prompt's tail exactly once,
2798
3363
  // after any digest fence rather than concatenated ahead of it.
2799
3364
  prompt = composeExecutorPrompt({ prdBody: prompt, digestText, finishProtocol: FINISH_PROTOCOL });
2800
3365
 
2801
- const promptCheck = validatePromptForSpawn(prompt, prdPath);
3366
+ // Foreign-WIP manifest (starry-night-ships PRD 148 postmortem): the
3367
+ // scheduler already knows, at spawn time, which dirty paths this job did
3368
+ // not create — either a shared tree's pre-existing dirty set or worktree
3369
+ // WIP carried in from the base tree (PRD 1094). Telling the executor
3370
+ // explicitly here means it never has to bisect by content to prove a test
3371
+ // failure isn't its own regression. '' (clean spawn) leaves prompt
3372
+ // byte-identical to before this section existed.
3373
+ const foreignWipSection = buildForeignWipSection(foreignWip || {});
3374
+ if (foreignWipSection) {
3375
+ prompt = `${prompt}\n\n${foreignWipSection}`;
3376
+ }
3377
+ } // end !resumeTarget digest/finish-protocol composition
3378
+
3379
+ const promptCheck = validatePromptForSpawn(prompt, resumeTarget ? `<resume recovery preamble for ${job.slug}>` : prdPath);
2802
3380
  if (!promptCheck.ok) {
2803
3381
  safeLog(`[scheduler] ${promptCheck.error}\n`);
2804
3382
  closeFd();
@@ -2806,6 +3384,15 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2806
3384
  return { exitCode: -1, durationMs: 0, error: promptCheck.error, sessionId };
2807
3385
  }
2808
3386
 
3387
+ // PRD agentType → persona + model (PRD 1115): resolved for both fresh and
3388
+ // resume dispatches, keyed off job.agentType (persisted on the queue row
3389
+ // by reconcile()) rather than re-reading the PRD file — a resumed session
3390
+ // must keep launching as the SAME persona it started as. Never throws;
3391
+ // a dangling/absent agentType falls back to no persona + FALLBACK_MODEL
3392
+ // and is logged once by resolvePrdPersonaForSpawn itself.
3393
+ const personaResolution = await agentModelResolve.resolvePrdPersonaForSpawn({ cwd, agentType: job.agentType });
3394
+ safeLog(`[scheduler] agentType=${job.agentType || '(none)'} persona=${personaResolution.personaPath || '(fallback — no persona applied)'} model=${personaResolution.model}\n`);
3395
+
2809
3396
  return await new Promise((resolve) => {
2810
3397
  const claudeBin = resolveClaudeBin();
2811
3398
  // Strip Claude Code env and secrets that leak in when session-manager is
@@ -2813,7 +3400,31 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2813
3400
  // overrides `--model sonnet`, so scheduled jobs burn Opus credits silently.
2814
3401
  // PATH must include Homebrew/user bins or the job's node/git children ENOENT
2815
3402
  // when Electron was launched from Finder/Dock on macOS (stripped PATH).
2816
- const childEnv = cleanChildEnv({ PATH: pathWithUserBins() });
3403
+ // SM_PROJECT_ROOT is the main-tree cwd (never spawnCwd, which may be a
3404
+ // job/epic worktree) — forwarded by scheduler-mcp-server.cjs as
3405
+ // originProjectRoot so a job running inside its own worktree can still
3406
+ // resolve the real project for create-prd/open-session/readiness. See
3407
+ // projectRootResolve.cjs.
3408
+ // SM_SCHEDULER_JOB_SLUG marks the child (and the MCP servers it
3409
+ // inherits its env to) as a headless scheduled executor, so
3410
+ // scheduler-mcp-server.cjs can refuse scheduler_create_prd from inside a
3411
+ // run (issue #11 list C1 — the PRD 460 self-queue incident). Only a
3412
+ // persona whose whole job is decomposition may still queue.
3413
+ // `launchEnv` is the launch circuit breaker's degraded-mode env (e.g.
3414
+ // MAX_THINKING_TOKENS=0 while an outdated CLI's thinking parameter is
3415
+ // being rejected — lib/launchFailure.cjs); applied last so it wins.
3416
+ const childEnv = cleanChildEnv({
3417
+ PATH: pathWithUserBins(),
3418
+ SM_PROJECT_ROOT: cwd,
3419
+ SM_SCHEDULER_JOB_SLUG: job.slug,
3420
+ SM_SCHEDULER_JOB_MAY_QUEUE: job.agentType === 'architect' ? '1' : '0',
3421
+ BASH_DEFAULT_TIMEOUT_MS: String(BASH_DEFAULT_TIMEOUT_MS),
3422
+ BASH_MAX_TIMEOUT_MS: String(BASH_MAX_TIMEOUT_MS),
3423
+ ...(launchEnv && typeof launchEnv === 'object' ? launchEnv : {}),
3424
+ });
3425
+ if (launchEnv && Object.keys(launchEnv).length) {
3426
+ safeLog(`[scheduler] launch mitigation env applied: ${Object.entries(launchEnv).map(([k, v]) => `${k}=${v}`).join(' ')}\n`);
3427
+ }
2817
3428
 
2818
3429
  // Track whether the agent has emitted a `result` event in its JSONL stream.
2819
3430
  // null until seen; then one of "success" | "error_max_turns" | … per the
@@ -2920,14 +3531,16 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2920
3531
  closeFd,
2921
3532
  spawn: {
2922
3533
  command: claudeBin,
2923
- args: [
2924
- '-p', prompt,
2925
- '--model', 'sonnet',
2926
- '--dangerously-skip-permissions',
2927
- '--output-format', 'stream-json',
2928
- '--verbose',
2929
- '--session-id', sessionId,
2930
- ],
3534
+ // Resume mode passes `--resume <sessionId>` (reconnect to the SAME
3535
+ // session) INSTEAD of `--session-id <sessionId>` (mint a new one) —
3536
+ // never both, see buildClaudeSpawnArgs.
3537
+ args: buildClaudeSpawnArgs({
3538
+ prompt,
3539
+ model: personaResolution.model,
3540
+ sessionId,
3541
+ resume: !!resumeTarget,
3542
+ systemPrompt: personaResolution.systemPrompt,
3543
+ }),
2931
3544
  options: {
2932
3545
  cwd: spawnCwd,
2933
3546
  env: childEnv,
@@ -2940,8 +3553,13 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2940
3553
  },
2941
3554
  },
2942
3555
  watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog],
2943
- onExit({ exitCode, signal, killedByWatchdog: _kbw, error, spawnFailed, safeLog: sl }) {
3556
+ onExit({ exitCode, signal, killedByWatchdog: _kbw, error, spawnFailed, leakedDescendants, safeLog: sl }) {
2944
3557
  const durationMs = Date.now() - startedAt;
3558
+ const leaked = leakedDescendants ?? [];
3559
+ if (leaked.length > 0) {
3560
+ sl(`\n[scheduler] leaked ${leaked.length} descendant(s) swept from job process group: ` +
3561
+ `${leaked.map((p) => `pid=${p.pid} comm=${p.comm} pcpu=${p.pcpu} etimes=${p.etimes}s`).join(', ')}\n`);
3562
+ }
2945
3563
 
2946
3564
  if (error) {
2947
3565
  // Covers both synchronous spawn failure and child 'error' events.
@@ -2951,8 +3569,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2951
3569
  sl(`\n[scheduler] ${errMsg}\n`);
2952
3570
  // Sync write: inside a Promise executor callback; must flush meta
2953
3571
  // before resolve() so the spawnJob mutate() that follows sees it.
2954
- config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA, originSessionId, contextDigestApplied });
2955
- resolve({ exitCode: -1, durationMs, error: errMsg, sessionId });
3572
+ config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA, originSessionId, contextDigestApplied });
3573
+ resolve({ exitCode: -1, durationMs, error: errMsg, leakedDescendants: leaked, sessionId });
2956
3574
  return;
2957
3575
  }
2958
3576
 
@@ -2975,16 +3593,33 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2975
3593
  `duration=${Math.round(durationMs / 1000)}s\n`);
2976
3594
  const rateLimited = effectiveCode !== 0 && detectRateLimitInLog(logPath);
2977
3595
  const networkError = effectiveCode !== 0 && !rateLimited && detectNetworkErrorInLog(logPath);
3596
+ // Non-run detection (issue #11 lists A1–A3): the harness's `result`
3597
+ // event tells us whether the model ever got a turn. A first-request
3598
+ // API rejection (num_turns ≤ 1, output_tokens 0, `API Error:` text)
3599
+ // is a broken ENVIRONMENT, not a failed PRD — spawnJob routes it to
3600
+ // the launch circuit breaker instead of failed/investigation.
3601
+ const resultStats = launchFailure.readResultEvent(logPath);
3602
+ const launchFailed = (effectiveCode !== 0 && !rateLimited && !networkError)
3603
+ ? launchFailure.classifyLaunchFailure(resultStats)
3604
+ : null;
3605
+ if (launchFailed) {
3606
+ sl(`\n[scheduler] LAUNCH FAILURE (${launchFailed.kind}${launchFailed.httpStatus ? ` HTTP ${launchFailed.httpStatus}` : ''}): ` +
3607
+ `${launchFailed.message} — no turn was taken; this is not a PRD failure\n`);
3608
+ }
2978
3609
  // Sync write: child 'exit' handler must flush meta before resolve()
2979
3610
  // so the spawnJob mutate() that follows sees the persisted exit code.
2980
3611
  config.writeJsonSync(metaPath, {
2981
3612
  slug: job.slug, cwd, sessionId, exitCode: effectiveCode, rateLimited, networkError,
2982
- startedAt, finishedAt: Date.now(), durationMs,
3613
+ launchFailure: launchFailed,
3614
+ numTurns: resultStats?.numTurns ?? null, outputTokens: resultStats?.outputTokens ?? null,
3615
+ totalCostUsd: resultStats?.totalCostUsd ?? null, terminalReasonFromHarness: resultStats?.terminalReason ?? null,
3616
+ launchEnvApplied: launchEnv && Object.keys(launchEnv).length ? Object.keys(launchEnv) : [],
3617
+ startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
2983
3618
  agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
2984
3619
  schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
2985
3620
  originSessionId, contextDigestApplied,
2986
3621
  });
2987
- resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, sessionId });
3622
+ resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, launchFailure: launchFailed, resultStats, leakedDescendants: leaked, sessionId });
2988
3623
  },
2989
3624
  });
2990
3625
 
@@ -3049,7 +3684,27 @@ function healTargetForFix(fixSlug, jobs) {
3049
3684
  * spawnInvestigation computes.
3050
3685
  */
3051
3686
  function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group }) {
3052
- return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.
3687
+ const abandonedBackgroundTaskNote = failedJob.verifierVerdict === 'abandoned_background_task' ? `
3688
+
3689
+ # Known failure class: abandoned background task
3690
+ This job's verifier verdict is \`abandoned_background_task\`: the transcript shows a Bash command
3691
+ auto-backgrounded past its foreground timeout, and the run ended waiting for a "you will be
3692
+ notified when it completes" callback a headless run structurally cannot receive. This is NOT
3693
+ evidence the work failed — it is evidence the run stopped short of its finish protocol. The work is
3694
+ usually already written and correct; only the commit is missing.
3695
+
3696
+ By the time this investigation runs, the failed job's isolated worktree (if it ran in one) has
3697
+ already been cleaned up — \`${cwd}\` is the BASE repo, not that worktree, so a plain \`git status\`/
3698
+ \`git diff\` there will usually show nothing even though real work was produced. The scheduler
3699
+ salvages any uncommitted diff from a killed job's worktree BEFORE deleting it${
3700
+ failedJob.salvagePatch ? `, and this job's salvage patch was captured at:\n\n ${failedJob.salvagePatch}` : ', to a `.uncommitted.patch` file next to the run log — check the run dir for one'
3701
+ }.
3702
+
3703
+ The fix-plan PRD you write for this MUST instruct its executor to, in order:
3704
+ 1. Check for a salvage patch (named \`<slug>.uncommitted.patch\` in the run directory${failedJob.salvagePatch ? `, e.g. \`${failedJob.salvagePatch}\`` : ''}) and, if found, apply it to the working tree BEFORE inspecting \`git status\`/\`git diff\` in ${cwd} for uncommitted changes matching the original PRD's acceptance criteria.
3705
+ 2. If the work is present (via the applied patch or already in the tree) and satisfies the acceptance criteria, run the project's verify commands and COMMIT it — do not re-implement or re-plan the PRD from scratch.
3706
+ 3. Only fall back to re-implementing whatever acceptance criteria are genuinely missing after applying any salvage patch, not the whole PRD.` : '';
3707
+ return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${abandonedBackgroundTaskNote}
3053
3708
 
3054
3709
  # Failed job
3055
3710
  - Slug: ${failedJob.slug}
@@ -3167,7 +3822,37 @@ function readRunOutcomeSidecars(runDir, slug) {
3167
3822
  * or if the run being investigated actually verified clean (nothing to fix —
3168
3823
  * see shouldSkipInvestigationForCleanRun).
3169
3824
  */
3825
+ const INVESTIGATION_LAUNCH_KEY = 'investigation';
3826
+
3170
3827
  async function spawnInvestigation(failedJob, runDir) {
3828
+ // The probe launches with the same CLI as the job it diagnoses. While
3829
+ // that CLI cannot launch at all (launch circuit breaker, issue #11 list
3830
+ // B1: probes e4f82da2/d374e6bf died on the same HTTP 400 as the runs
3831
+ // they were investigating) there is nothing to diagnose — skip, loudly.
3832
+ {
3833
+ const state = await readQueue().catch(() => null);
3834
+ const block = state?.launchBlocks?.[INVESTIGATION_LAUNCH_KEY];
3835
+ const jobBlock = state?.launchBlocks?.[launchFailure.launchBlockKeyFor(failedJob)];
3836
+ const gate = launchFailure.evaluateLaunchGate(block || jobBlock, { now: Date.now(), claudeVersion: await probeClaudeVersion() });
3837
+ if (gate.state === 'blocked') {
3838
+ console.log(`[scheduler] skip investigation: ${failedJob.slug} — ${gate.reason}`);
3839
+ await mutate((s) => {
3840
+ const j = s.jobs.find((x) => x.slug === failedJob.slug);
3841
+ if (j) { j.autoFixOutcome = 'launch-blocked'; j.autoFixNote = gate.reason; }
3842
+ }).catch(() => {});
3843
+ return { deferred: false };
3844
+ }
3845
+ }
3846
+ // Resume-first recovery (PRD 1111) always gets first refusal — a job
3847
+ // eligible for a bounded `--resume` dispatch must never also get a
3848
+ // cold-read fix-plan PRD authored in the same pass. selectResumeRecoveryTarget
3849
+ // returns null for every job shape spawnInvestigation is normally called
3850
+ // with (e.g. plain 'failed' jobs never carry verifierVerdict
3851
+ // 'uncommitted_changes'), so this is a no-op for the common case.
3852
+ if (selectResumeRecoveryTarget(failedJob)) {
3853
+ console.log(`[scheduler] skip investigation: ${failedJob.slug} is resume-recovery eligible`);
3854
+ return { deferred: false };
3855
+ }
3171
3856
  if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth)) {
3172
3857
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
3173
3858
  return { deferred: false };
@@ -3277,7 +3962,11 @@ async function spawnInvestigation(failedJob, runDir) {
3277
3962
  await broadcast({ flush: true });
3278
3963
 
3279
3964
  const claudeBin = resolveClaudeBin();
3280
- const childEnv = cleanChildEnv({ PATH: pathWithUserBins() }); // Homebrew/user bins for macOS
3965
+ const childEnv = cleanChildEnv({
3966
+ PATH: pathWithUserBins(), // Homebrew/user bins for macOS
3967
+ BASH_DEFAULT_TIMEOUT_MS: String(BASH_DEFAULT_TIMEOUT_MS),
3968
+ BASH_MAX_TIMEOUT_MS: String(BASH_MAX_TIMEOUT_MS),
3969
+ });
3281
3970
 
3282
3971
  // Investigation needs only a deadman watchdog — no idle-tail or result-tail
3283
3972
  // since investigations are short-running Opus probes with a hard ceiling.
@@ -3317,7 +4006,10 @@ async function spawnInvestigation(failedJob, runDir) {
3317
4006
  // 'investigating' must never be the job's resting state.
3318
4007
  mutate((s) => {
3319
4008
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
3320
- if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation probe exited — restoring prior status', source: 'spawnInvestigation:onExit' });
4009
+ if (j && j.status === 'investigating') {
4010
+ transitionJob(j, failedJob.status || 'failed', { reason: 'investigation probe exited — restoring prior status', source: 'spawnInvestigation:onExit' });
4011
+ delete j.runtime;
4012
+ }
3321
4013
  })
3322
4014
  .then(() => broadcast({ flush: true }))
3323
4015
  .catch(() => {});
@@ -3336,6 +4028,23 @@ async function spawnInvestigation(failedJob, runDir) {
3336
4028
  return;
3337
4029
  }
3338
4030
  sl(`\n[scheduler] investigation exit code=${exitCode}\n`);
4031
+ if (exitCode !== 0) {
4032
+ const probeResult = launchFailure.readResultEvent(investigationLogPath);
4033
+ const probeLaunchFailure = launchFailure.classifyLaunchFailure(probeResult);
4034
+ if (probeLaunchFailure) {
4035
+ sl(`\n[scheduler] investigation LAUNCH FAILURE (${probeLaunchFailure.kind}): ${probeLaunchFailure.message} — arming '${INVESTIGATION_LAUNCH_KEY}' launch block\n`);
4036
+ probeClaudeVersion().then((claudeVersion) => mutate((s) => {
4037
+ s.launchBlocks = s.launchBlocks || {};
4038
+ s.launchBlocks[INVESTIGATION_LAUNCH_KEY] = launchFailure.armLaunchBlock(s.launchBlocks[INVESTIGATION_LAUNCH_KEY] || null, {
4039
+ kind: probeLaunchFailure.kind, httpStatus: probeLaunchFailure.httpStatus, message: probeLaunchFailure.message,
4040
+ now: Date.now(), claudeVersion, slug: failedJob.slug, runId: failedJob.runId ?? null,
4041
+ });
4042
+ const j = s.jobs.find((x) => x.slug === failedJob.slug);
4043
+ if (j) { j.autoFixOutcome = 'launch-blocked'; j.autoFixNote = `investigation probe never ran: ${probeLaunchFailure.message}`; }
4044
+ })).catch(() => {});
4045
+ return;
4046
+ }
4047
+ }
3339
4048
  // Fold the investigation's <RCA> summary into the root-cause report already
3340
4049
  // written for this job (needs_review jobs only — writeRcaReport no-ops when
3341
4050
  // failedJob has no verifierVerdict, e.g. plain 'failed' jobs never got one).
@@ -3374,6 +4083,15 @@ async function spawnInvestigation(failedJob, runDir) {
3374
4083
 
3375
4084
  if (child) {
3376
4085
  safeLog(`[scheduler] investigation pid=${child.pid}\n\n`);
4086
+ // Recorded so findStrandedInvestigations (a post-restart maintenance
4087
+ // sweep — the live process has no other way to know a probe is still
4088
+ // running) can tell a live probe apart from one whose owning process is
4089
+ // long gone, the same way reapDeadRunningJobs checks a running job's
4090
+ // runtime.pid.
4091
+ mutate((s) => {
4092
+ const j = s.jobs.find((x) => x.slug === failedJob.slug);
4093
+ if (j && j.status === 'investigating') j.runtime = { pid: child.pid };
4094
+ }).catch(() => {});
3377
4095
  }
3378
4096
  return { deferred: false };
3379
4097
  } catch (e) {
@@ -3383,7 +4101,10 @@ async function spawnInvestigation(failedJob, runDir) {
3383
4101
  releaseSlot();
3384
4102
  mutate((s) => {
3385
4103
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
3386
- if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation spawn threw before exiting — restoring prior status', source: 'spawnInvestigation:catch' });
4104
+ if (j && j.status === 'investigating') {
4105
+ transitionJob(j, failedJob.status || 'failed', { reason: 'investigation spawn threw before exiting — restoring prior status', source: 'spawnInvestigation:catch' });
4106
+ delete j.runtime;
4107
+ }
3387
4108
  })
3388
4109
  .then(() => broadcast({ flush: true }))
3389
4110
  .catch(() => {});
@@ -3391,7 +4112,122 @@ async function spawnInvestigation(failedJob, runDir) {
3391
4112
  }
3392
4113
  }
3393
4114
 
3394
- async function spawnJob(job, runId, runDir, defaultCwd) {
4115
+ /**
4116
+ * computeLaunchHolds(state) → Map<slug, reason>
4117
+ *
4118
+ * The launch circuit breaker's per-tick view (lib/launchFailure.cjs, issue
4119
+ * #11): every pending row whose persona is blocked is held with its reason;
4120
+ * when a persona's backoff has elapsed exactly ONE of its pending rows is
4121
+ * left pickable (the half-open probe) and the rest are held behind it. A
4122
+ * CLI version change drops the block outright — that is the incident's real
4123
+ * fix (`claude update`) and the queue must resume on the next tick.
4124
+ * Mutates nothing; spawnJob makes the durable decision at dispatch.
4125
+ */
4126
+ async function computeLaunchHolds(state, { now = Date.now(), claudeVersion } = {}) {
4127
+ const held = new Map();
4128
+ const blocks = state?.launchBlocks;
4129
+ if (!blocks || typeof blocks !== 'object' || !Object.keys(blocks).length) return held;
4130
+ const version = claudeVersion === undefined ? await probeClaudeVersion() : claudeVersion;
4131
+ const probeAllowed = new Set();
4132
+ for (const j of state.jobs || []) {
4133
+ if (j.status !== 'pending') continue;
4134
+ const key = launchFailure.launchBlockKeyFor(j);
4135
+ const block = blocks[key];
4136
+ if (!block) continue;
4137
+ const gate = launchFailure.evaluateLaunchGate(block, { now, claudeVersion: version });
4138
+ if (gate.state === 'open') continue;
4139
+ if (gate.state === 'probe' && !probeAllowed.has(key)) {
4140
+ probeAllowed.add(key);
4141
+ continue;
4142
+ }
4143
+ held.set(j.slug, gate.state === 'probe'
4144
+ ? `launch blocked (${block.kind}) — waiting for this tick's probe of '${key}'`
4145
+ : gate.reason);
4146
+ }
4147
+ return held;
4148
+ }
4149
+
4150
+ /**
4151
+ * A run that never got a turn (res.launchFailure — see executeJob's onExit)
4152
+ * is routed here instead of the failed/investigation path (issue #11 lists
4153
+ * A1–A3, B1): the row goes back to `pending` carrying the API's own message
4154
+ * as its error, no retry budget is consumed, no auto-fix probe is spawned
4155
+ * (it would die the same way), and the persona's launch circuit breaker is
4156
+ * armed so the queue stops re-dispatching identical doomed launches while
4157
+ * still self-healing on backoff / CLI update / human Retry.
4158
+ */
4159
+ /**
4160
+ * Pure state mutation behind handleLaunchFailure (exported for tests): arms
4161
+ * the persona's breaker and returns the job's `running` row to `pending`
4162
+ * carrying the API message. Returns the armed block.
4163
+ */
4164
+ function applyLaunchFailure(s, { job, lf, runId, launchKey, mitigationApplied, claudeVersion, now = Date.now() }) {
4165
+ s.launchBlocks = s.launchBlocks || {};
4166
+ s.launchMitigations = s.launchMitigations || {};
4167
+ const prev = s.launchBlocks[launchKey] || null;
4168
+ const armed = launchFailure.armLaunchBlock(prev, {
4169
+ kind: lf.kind, httpStatus: lf.httpStatus, message: lf.message, now, claudeVersion,
4170
+ slug: job.slug, runId, mitigationApplied,
4171
+ });
4172
+ s.launchBlocks[launchKey] = armed;
4173
+ // A mitigation that was in force and still failed is no longer proven —
4174
+ // drop it so the hint and the next probe are honest.
4175
+ if (mitigationApplied && s.launchMitigations[launchKey]) delete s.launchMitigations[launchKey];
4176
+ const i = (s.jobs || []).findIndex((x) => x.slug === job.slug);
4177
+ if (i >= 0 && s.jobs[i].status === 'running') {
4178
+ const prevCount = s.jobs[i].launchFailure?.count ?? 0;
4179
+ const msg = `launch failure (${lf.kind}${lf.httpStatus ? ` HTTP ${lf.httpStatus}` : ''}): ${lf.message}`;
4180
+ resetJobFields(s.jobs[i], msg, { source: 'spawnJob:launch-failure' });
4181
+ s.jobs[i].launchFailure = {
4182
+ kind: lf.kind, httpStatus: lf.httpStatus ?? null, message: lf.message,
4183
+ at: new Date(now).toISOString(), runId, count: prevCount + 1, mitigationApplied,
4184
+ };
4185
+ s.jobs[i].terminalReason = `launch_failure:${lf.kind}`;
4186
+ s.jobs[i].heldReason = armed.exhausted
4187
+ ? `launch blocked (${lf.kind}) after ${armed.attempts} failed probe(s) — ${armed.hint}`
4188
+ : `launch blocked (${lf.kind}) — re-probe at ${armed.until}. ${armed.hint}`;
4189
+ }
4190
+ return armed;
4191
+ }
4192
+
4193
+ async function handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion }) {
4194
+ const lf = res.launchFailure;
4195
+ const now = Date.now();
4196
+ const mitigationApplied = !!(launchEnv && Object.keys(launchEnv).length);
4197
+ let armed = null;
4198
+ await mutate((s) => {
4199
+ armed = applyLaunchFailure(s, { job, lf, runId, launchKey, mitigationApplied, claudeVersion, now });
4200
+ });
4201
+ launchFailure.writeOutcomeSidecar(runDir, job.slug, {
4202
+ runId,
4203
+ exitCode: res.exitCode,
4204
+ durationMs: res.durationMs ?? null,
4205
+ numTurns: res.resultStats?.numTurns ?? null,
4206
+ outputTokens: res.resultStats?.outputTokens ?? null,
4207
+ totalCostUsd: res.resultStats?.totalCostUsd ?? null,
4208
+ verdict: null,
4209
+ status: 'pending',
4210
+ terminalReason: `launch_failure:${lf.kind}`,
4211
+ launchFailure: { kind: lf.kind, httpStatus: lf.httpStatus ?? null, message: lf.message },
4212
+ launchEnvApplied: launchEnv ? Object.keys(launchEnv) : [],
4213
+ filesChanged: 0,
4214
+ landedCommit: null,
4215
+ });
4216
+ try {
4217
+ appendError({
4218
+ cwd: job.cwd || DEFAULT_PROJECT_CWD,
4219
+ scope: 'scheduler',
4220
+ level: 'error',
4221
+ message: `launch failure (${lf.kind}) for ${job.slug}: ${lf.message} — persona '${launchKey}' blocked, attempt ${armed?.attempts}${armed?.exhausted ? ' (exhausted; needs CLI update or Retry)' : ''}`,
4222
+ meta: { slug: job.slug, runId, kind: lf.kind, httpStatus: lf.httpStatus ?? null, claudeVersion: claudeVersion ?? null, mitigationApplied, hint: armed?.hint },
4223
+ });
4224
+ } catch { /* durable logging must never break the queue */ }
4225
+ console.error(`[scheduler] ${job.slug}: LAUNCH FAILURE (${lf.kind}${lf.httpStatus ? ` HTTP ${lf.httpStatus}` : ''}) — ${lf.message}. ` +
4226
+ `Persona '${launchKey}' blocked (attempt ${armed?.attempts}${armed?.until ? `, re-probe at ${armed.until}` : ', exhausted'}). ${armed?.hint}`);
4227
+ await broadcast({ flush: true });
4228
+ }
4229
+
4230
+ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
3395
4231
  // Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
3396
4232
  // — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
3397
4233
  // leaves the job pending; the next tick retries when a slot frees up.
@@ -3401,13 +4237,108 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3401
4237
  return;
3402
4238
  }
3403
4239
  runningSet.add(job.slug);
4240
+ // Exclusive quiet-machine lease (PRD 1107) — acquired here, in the same
4241
+ // slot-acquire/dispatch step as sessionSlots, and released in this
4242
+ // function's own finally below alongside sessionSlots.release, so every
4243
+ // exit path (normal exit, timeout, SIGTERM, crash) that already frees the
4244
+ // session slot also frees the lease. pickNextBatch only ever hands this
4245
+ // function a quietMachine job when the lease was free at pick time, so
4246
+ // acquire() here should never fail in practice — but check anyway rather
4247
+ // than assume, since a lease held by a stale slug would otherwise wedge
4248
+ // silently.
4249
+ const quietLeaseAcquired = job.quietMachine === true && quietMachineLease.acquire(job.slug);
3404
4250
  try {
4251
+ // Worktree isolation cap check (PRD 1112) — probed BEFORE the job is
4252
+ // marked 'running', so a job that can't get isolation right now is a
4253
+ // DEFERRAL, not a fallback: it stays 'pending' and is retried on the
4254
+ // next dispatch pass, exactly like the sessionSlots miss above, instead
4255
+ // of degrading into an in-place run in a tree a sibling job may be
4256
+ // actively writing to (the shared-tree collision this cap exists to
4257
+ // prevent). Every OTHER worktree.ok===false reason (not a git repo,
4258
+ // disabled, carry-over failure) keeps the existing in-place fallback —
4259
+ // only the cap-reached reason is a deferral, checked here via
4260
+ // createJobWorktree's own reason string so the two paths never
4261
+ // silently drift out of sync with gitWorktree.cjs's actual wording.
4262
+ const preflightWorktree = resumeTarget
4263
+ ? { ok: false, reason: 'resume-recovery: running in place to reuse the session\'s prior working tree' }
4264
+ : await jobWorktree.createJobWorktree({ cwd: job.cwd || defaultCwd, slug: job.slug });
4265
+ if (!preflightWorktree.ok && /^worktree cap reached\b/.test(preflightWorktree.reason || '')) {
4266
+ console.log(`[scheduler] ${job.slug}: deferring — ${preflightWorktree.reason}`);
4267
+ await mutate((s) => {
4268
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4269
+ if (idx >= 0) s.jobs[idx].heldReason = preflightWorktree.reason;
4270
+ });
4271
+ await broadcast({ flush: true });
4272
+ return;
4273
+ }
4274
+ // Launch circuit breaker (lib/launchFailure.cjs, issue #11). Re-evaluated
4275
+ // here, not just in tickQueue, because the block can change between the
4276
+ // pick and this dispatch (another job's probe just failed). 'blocked' →
4277
+ // hold the row; 'probe' → this job is the single half-open probe and is
4278
+ // stamped as such so no sibling probes the same broken persona at once.
4279
+ const launchKey = launchFailure.launchBlockKeyFor(job);
4280
+ const claudeVersionNow = await probeClaudeVersion();
4281
+ let launchEnv = null;
4282
+ let launchProbe = false;
4283
+ const launchGate = await mutate((s) => {
4284
+ s.launchBlocks = s.launchBlocks || {};
4285
+ s.launchMitigations = s.launchMitigations || {};
4286
+ const mitigation = s.launchMitigations[launchKey];
4287
+ if (mitigation && claudeVersionNow && mitigation.claudeVersion && mitigation.claudeVersion !== claudeVersionNow) {
4288
+ console.log(`[scheduler] launch gate: CLI version changed (${mitigation.claudeVersion} → ${claudeVersionNow}) — dropping ${launchKey} mitigation to retry a clean launch`);
4289
+ delete s.launchMitigations[launchKey];
4290
+ }
4291
+ const block = s.launchBlocks[launchKey];
4292
+ const gate = launchFailure.evaluateLaunchGate(block, { now: Date.now(), claudeVersion: claudeVersionNow });
4293
+ if (gate.state === 'open' && block) {
4294
+ console.log(`[scheduler] launch gate: clearing ${launchKey} block (${gate.reason})`);
4295
+ delete s.launchBlocks[launchKey];
4296
+ }
4297
+ if (gate.state === 'blocked') {
4298
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4299
+ if (idx >= 0) s.jobs[idx].heldReason = gate.reason;
4300
+ return gate;
4301
+ }
4302
+ if (gate.state === 'probe') {
4303
+ block.probing = { slug: job.slug, at: new Date().toISOString() };
4304
+ launchProbe = true;
4305
+ launchEnv = block.mitigationEnv || null;
4306
+ } else if (s.launchMitigations[launchKey]?.env) {
4307
+ launchEnv = { ...s.launchMitigations[launchKey].env };
4308
+ }
4309
+ return gate;
4310
+ });
4311
+ if (launchGate.state === 'blocked') {
4312
+ console.log(`[scheduler] ${job.slug}: deferring — ${launchGate.reason}`);
4313
+ await broadcast({ flush: true });
4314
+ return;
4315
+ }
4316
+ if (launchProbe) {
4317
+ console.log(`[scheduler] ${job.slug}: dispatching as launch probe for '${launchKey}'${launchEnv ? ` with mitigation ${JSON.stringify(launchEnv)}` : ''}`);
4318
+ }
4319
+
3405
4320
  await mutate((s) => {
3406
4321
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
3407
4322
  if (idx >= 0) {
3408
- transitionJob(s.jobs[idx], 'running', { reason: 'dispatched for execution', source: 'spawnJob:dispatch' });
4323
+ transitionJob(s.jobs[idx], 'running', {
4324
+ reason: resumeTarget ? 'dispatched for resume-recovery' : 'dispatched for execution',
4325
+ source: 'spawnJob:dispatch',
4326
+ });
4327
+ delete s.jobs[idx].heldReason;
3409
4328
  s.jobs[idx].runId = runId;
3410
4329
  s.jobs[idx].startedAt = new Date().toISOString();
4330
+ if (job.quietMachine === true) {
4331
+ s.jobs[idx].quietMachine = true;
4332
+ s.jobs[idx].quietLeaseDegraded = job.quietLeaseDegraded === true;
4333
+ }
4334
+ // Stamp the bounded one-attempt marker BEFORE the resume spawn, in
4335
+ // the SAME mutate as the 'running' transition, so an app crash
4336
+ // between here and the child actually spawning still leaves this
4337
+ // job un-retriable (selectResumeRecoveryTarget returns null once
4338
+ // this is true) rather than silently re-firing forever.
4339
+ if (resumeTarget) {
4340
+ s.jobs[idx].resumeRecoveryAttempted = true;
4341
+ }
3411
4342
  }
3412
4343
  });
3413
4344
  await broadcast({ flush: true });
@@ -3417,19 +4348,74 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3417
4348
  const guardCwd = job.cwd || defaultCwd;
3418
4349
  const guardBaseline = await uncommittedChanges(guardCwd);
3419
4350
  const guardHeadBefore = await gitHead(guardCwd);
4351
+ // Shared-tree stash guard baseline (incident 2026-09-01): captured
4352
+ // unconditionally, before worktree isolation is even attempted, so an
4353
+ // in-place run always has a true pre-run snapshot to diff against. See
4354
+ // checkSharedTreeGuard below, gated to in-place runs only.
4355
+ const stashBaseline = await stashList(guardCwd);
4356
+
4357
+ // Persist the pre-run baseline onto the row itself (not just the local
4358
+ // variable) so a finalizer that never reaches the rest of THIS function
4359
+ // — namely reapDeadRunningJobs, when the process vanishes mid-run — can
4360
+ // still compute a truthful newly-dirty delta instead of having no
4361
+ // baseline at all. `runtime` (unlike this) is deleted on finalize; this
4362
+ // survives until the finalize mutate below explicitly clears it.
4363
+ //
4364
+ // preRunDirtyPaths is the SAME snapshot, capped and reworked into the
4365
+ // executor-facing manifest (buildForeignWipSection) telling the job which
4366
+ // paths it does not own — unlike guardBaseline/guardHeadBefore, it is
4367
+ // deliberately left on the row through to history.jsonl (not deleted at
4368
+ // finalize) so a post-hoc reader can tell whether a completed job ran
4369
+ // against foreign WIP.
4370
+ const preRunDirtyPaths = capDirtyPaths(guardBaseline);
4371
+ await mutate((s) => {
4372
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4373
+ if (idx >= 0) {
4374
+ s.jobs[idx].guardBaseline = guardBaseline || [];
4375
+ s.jobs[idx].guardHeadBefore = guardHeadBefore || null;
4376
+ if (preRunDirtyPaths.length) s.jobs[idx].preRunDirtyPaths = preRunDirtyPaths;
4377
+ else delete s.jobs[idx].preRunDirtyPaths;
4378
+ }
4379
+ });
3420
4380
 
3421
4381
  // Worktree isolation (PRD 994): give this job its own linked `git worktree`
3422
4382
  // checkout so its edits/tests/commit never collide with a sibling job or
3423
4383
  // an interactive session in the SAME repo. `worktree.ok` is false (with a
3424
- // logged reason) for a non-git cwd, a dirty base tree, the cap being hit,
3425
- // or SM_JOB_WORKTREE_DISABLE=1 — every case falls back to running in place,
3426
- // never a hard failure. See jobWorktree.cjs's header comment for why
3427
- // job.cwd (guardCwd) itself is NEVER repointed at the worktree dir.
3428
- const worktree = await jobWorktree.createJobWorktree({ cwd: guardCwd, slug: job.slug });
4384
+ // logged reason) for a non-git cwd, a dirty base tree, or
4385
+ // SM_JOB_WORKTREE_DISABLE=1 — every case falls back to running in place,
4386
+ // never a hard failure. (The cap-reached reason was already handled above
4387
+ // as a pre-dispatch DEFERRAL — a job never reaches this point with that
4388
+ // reason.) See jobWorktree.cjs's header comment for why job.cwd
4389
+ // (guardCwd) itself is NEVER repointed at the worktree dir. Reuses
4390
+ // `preflightWorktree` computed above the 'running' transition — it
4391
+ // already IS this job's worktree attempt (or already-created checkout),
4392
+ // so calling createJobWorktree a second time here would double-create
4393
+ // (or double-count the cap) for the exact same job.
4394
+ const worktree = preflightWorktree;
3429
4395
  if (worktree.ok) {
3430
4396
  console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
3431
4397
  } else {
3432
4398
  console.log(`[scheduler] ${job.slug}: running in main tree (worktree not used: ${worktree.reason})`);
4399
+ // Surface any degraded-isolation fallback on the job row itself so it's
4400
+ // queryable from the queue instead of console-only — except the
4401
+ // deliberate env-disable flag, which is an intentional opt-out, not a
4402
+ // degradation worth flagging.
4403
+ if (!jobWorktree.isWorktreeDisabled()) {
4404
+ await mutate((s) => {
4405
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4406
+ if (idx >= 0) s.jobs[idx].worktreeFallbackReason = worktree.reason;
4407
+ });
4408
+ }
4409
+ }
4410
+ // Base-tree WIP carried into the worktree (createWorktree, PRD 1094) —
4411
+ // recorded on the job row so integration can exclude these paths from
4412
+ // the branch diff below, and so it's queryable from the queue.
4413
+ const carriedPaths = (worktree.ok && Array.isArray(worktree.carriedPaths)) ? worktree.carriedPaths : [];
4414
+ if (carriedPaths.length) {
4415
+ await mutate((s) => {
4416
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4417
+ if (idx >= 0) s.jobs[idx].carriedPaths = carriedPaths;
4418
+ });
3433
4419
  }
3434
4420
 
3435
4421
  // Integrate the job's branch back into guardCwd's own HEAD, THEN tear the
@@ -3448,6 +4434,17 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3448
4434
  let res;
3449
4435
  let worktreeLeftoverDirty = [];
3450
4436
  let worktreeIntegrationFailure = null;
4437
+ // A job's uncommitted-work patch, whichever isolation mode produced it —
4438
+ // set by EITHER branch below, never both (worktree.ok picks exactly one
4439
+ // shape for the whole run). Named generically (not "worktree...") because
4440
+ // an in-place run salvages one too (PRD 1098).
4441
+ let salvagePatch = null;
4442
+ // Which foreign-WIP shape applies to THIS run: an isolated worktree only
4443
+ // ever needs to disclose carriedPaths (its checkout starts clean apart
4444
+ // from those carried paths); an in-place/shared-tree run discloses the
4445
+ // raw pre-run dirty snapshot instead. Never both — see
4446
+ // buildForeignWipSection.
4447
+ const foreignWip = worktree.ok ? { carriedPaths } : { preRunDirtyPaths };
3451
4448
  try {
3452
4449
  res = await executeJob(job, runDir, defaultCwd, async (pid, sessionId, cwd) => {
3453
4450
  await mutate((s) => {
@@ -3458,11 +4455,27 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3458
4455
  }
3459
4456
  });
3460
4457
  await broadcast({ flush: true });
3461
- }, worktree.ok ? worktree.dir : undefined);
4458
+ }, worktree.ok ? worktree.dir : undefined, resumeTarget, foreignWip, launchEnv);
3462
4459
  } finally {
3463
4460
  if (worktree.ok) {
3464
4461
  worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
3465
- const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug });
4462
+ // Salvage the worktree's full diff (tracked + untracked) to the run
4463
+ // dir BEFORE the checkout is removed below — otherwise a job killed
4464
+ // before its finish-protocol commit loses that work outright, with
4465
+ // no branch, no stash, no patch anywhere. Best-effort: never blocks
4466
+ // integration/cleanup and never changes the job's verdict.
4467
+ if (worktreeLeftoverDirty.length) {
4468
+ const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
4469
+ const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
4470
+ if (salvage && salvage.ok) {
4471
+ salvagePatch = salvagePath;
4472
+ console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
4473
+ }
4474
+ }
4475
+ const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
4476
+ if (integration.ok && integration.reason === 'carried-wip-only') {
4477
+ console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
4478
+ }
3466
4479
  if (!integration.ok) {
3467
4480
  worktreeIntegrationFailure = integration.reason;
3468
4481
  console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
@@ -3475,9 +4488,86 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3475
4488
  branch: worktree.branch,
3476
4489
  keepBranch: !integration.ok,
3477
4490
  });
4491
+ } else {
4492
+ // In-place run (non-git cwd, cap reached, env-disabled, or a carry-over
4493
+ // failure) — there is no throwaway checkout to diff, so salvage only
4494
+ // the DELTA this job itself dirtied: paths in guardBaseline are a
4495
+ // human's or a sibling job's pre-existing WIP and must never appear in
4496
+ // this job's patch. Runs for every exit code (finally always fires
4497
+ // once `res` resolves, success or not) including signal deaths and the
4498
+ // rate-limited/halt path — a killed in-place run is exactly the case
4499
+ // this exists to cover. Never mutates guardCwd's index or stashes:
4500
+ // salvageDirtyDelta is read-only (git status + git diff only).
4501
+ try {
4502
+ const after = await uncommittedChanges(guardCwd);
4503
+ if (after) {
4504
+ const baseSet = new Set(guardBaseline || []);
4505
+ const deltaPaths = after.filter((p) => !baseSet.has(p));
4506
+ if (deltaPaths.length) {
4507
+ const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
4508
+ const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: guardCwd, paths: deltaPaths, outFile: salvagePath });
4509
+ if (salvage && salvage.ok) {
4510
+ salvagePatch = salvagePath;
4511
+ console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted in-place diff (${deltaPaths.length} path(s)) to ${salvagePath}`);
4512
+ }
4513
+ }
4514
+ }
4515
+ } catch (e) {
4516
+ console.error(`[scheduler] ${job.slug}: in-place salvage failed`, e);
4517
+ }
3478
4518
  }
3479
4519
  }
3480
4520
 
4521
+ // Newly-dirty leftover computation — hoisted OUT of the exit===0 branch
4522
+ // (below) so it runs for every terminal outcome: exit 0, any non-zero
4523
+ // exit including 137/143, and the rate-limited/halt path alike. This is
4524
+ // the exact same shape the exit=0 commit-guard and the transient-failure
4525
+ // classifier each used to compute independently (guardCwd's own
4526
+ // baseline-delta UNION worktreeLeftoverDirty, which is already
4527
+ // inherently-new since it came from a fresh worktree checkout with no
4528
+ // baseline to diff against) — computed once here and reused by both
4529
+ // below, plus by the terminal-finalize mutate for leftoverPaths/
4530
+ // leftoverCount. null only when git-status itself is unavailable
4531
+ // (non-git cwd / git errored) — NEVER treated as "left nothing", exactly
4532
+ // like every other best-effort git-state check in this function.
4533
+ const afterGuardCwd = await uncommittedChanges(guardCwd);
4534
+ const newlyDirtyAll = afterGuardCwd === null
4535
+ ? null
4536
+ : [...new Set([
4537
+ ...afterGuardCwd.filter((p) => !new Set(guardBaseline || []).has(p)),
4538
+ ...worktreeLeftoverDirty,
4539
+ ])];
4540
+
4541
+ if (res.launchFailure) {
4542
+ await handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion: claudeVersionNow });
4543
+ return;
4544
+ }
4545
+ if (launchFailure.resultShowsRealTurn(res.resultStats)) {
4546
+ // The launch worked (whatever happens to the run next) — close the
4547
+ // breaker for this persona. If the probe only got through thanks to a
4548
+ // mitigation env, keep applying that env to every later launch of the
4549
+ // persona until the CLI version changes; otherwise the very next job
4550
+ // would fail the same way and re-arm the block (a flap per job).
4551
+ await mutate((s) => {
4552
+ const block = s.launchBlocks?.[launchKey];
4553
+ if (!block) return;
4554
+ delete s.launchBlocks[launchKey];
4555
+ if (launchEnv && Object.keys(launchEnv).length) {
4556
+ s.launchMitigations = s.launchMitigations || {};
4557
+ s.launchMitigations[launchKey] = {
4558
+ kind: block.kind,
4559
+ env: { ...launchEnv },
4560
+ since: new Date().toISOString(),
4561
+ claudeVersion: claudeVersionNow ?? block.claudeVersion ?? null,
4562
+ hint: launchFailure.launchFailureHint(block.kind, { claudeVersion: claudeVersionNow ?? block.claudeVersion }),
4563
+ };
4564
+ console.log(`[scheduler] launch gate: ${launchKey} recovered via mitigation ${JSON.stringify(launchEnv)} — kept in force until the CLI version changes`);
4565
+ } else {
4566
+ console.log(`[scheduler] launch gate: ${launchKey} recovered — block cleared after ${block.attempts} failed probe(s)`);
4567
+ }
4568
+ });
4569
+ }
4570
+
3481
4571
  if (res.rateLimited) {
3482
4572
  const resetIso = await refreshNextReset().catch(() => cachedNextReset);
3483
4573
  await setPaused('rate_limit', resetIso);
@@ -3519,6 +4609,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3519
4609
  // pass it back into verifyRun as priorLandedCommit (see the
3520
4610
  // pass_no_commit_prior_run_verified exemption in runVerify.cjs).
3521
4611
  let jobLandedCommitThisRun = null;
4612
+ let sharedTreeGuard = null;
3522
4613
  if (res.exitCode === 0 && !res.rateLimited) {
3523
4614
  // Detect whether the job self-committed by comparing HEAD before/after.
3524
4615
  // Used by the sentinel override: SCHEDULER_VERDICT: PASS + a landed
@@ -3591,20 +4682,12 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3591
4682
  const guardWillRefire = verifyResult && verifyResult.downgradeTo === 'pending';
3592
4683
  const guardIsLegitimateNoOp = verifyResult && COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict);
3593
4684
  if (res.exitCode === 0 && !res.rateLimited && !guardWillRefire && !guardIsLegitimateNoOp) {
3594
- const after = await uncommittedChanges(guardCwd);
3595
- // after === null means non-git cwd (or git errored) — best-effort skip,
3596
- // same as always; only a git-status result (even an empty one) counts
3597
- // as evidence for the zero-edit path.
3598
- if (after !== null) {
3599
- const baseSet = new Set(guardBaseline || []);
3600
- // worktreeLeftoverDirty was captured from a FRESH checkout (no baseline
3601
- // to diff against — every path in it is inherently new) right before
3602
- // the worktree was torn down, so it must be counted here or a job's
3603
- // uncommitted leftovers silently vanish with the worktree.
3604
- const newlyDirty = [...new Set([
3605
- ...after.filter((p) => !baseSet.has(p)),
3606
- ...worktreeLeftoverDirty,
3607
- ])];
4685
+ // afterGuardCwd === null means non-git cwd (or git errored) —
4686
+ // best-effort skip, same as always; only a git-status result (even an
4687
+ // empty one) counts as evidence for the zero-edit path. newlyDirtyAll
4688
+ // was computed once, above, right after the try/finally.
4689
+ if (afterGuardCwd !== null) {
4690
+ const newlyDirty = newlyDirtyAll;
3608
4691
  const guardState = await readQueue().catch(() => ({ jobs: [] }));
3609
4692
  const siblingRunning = (guardState.jobs || []).some(
3610
4693
  (j) => j.slug !== job.slug && j.status === 'running' && (j.cwd || defaultCwd) === guardCwd,
@@ -3614,10 +4697,12 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3614
4697
  const guardVerdict = commitGuardVerdict({
3615
4698
  newlyDirty,
3616
4699
  siblingRunning,
4700
+ ranInWorktree: worktree.ok,
3617
4701
  jobSelfCommitted,
3618
4702
  legitimateNoOp: guardIsLegitimateNoOp,
3619
4703
  isFixPlanJob: isFixPlanSlug(job.slug),
3620
4704
  verifyResult,
4705
+ salvagePatch,
3621
4706
  });
3622
4707
  if (guardVerdict) {
3623
4708
  verifyResult = guardVerdict;
@@ -3641,6 +4726,36 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3641
4726
  };
3642
4727
  }
3643
4728
 
4729
+ // Shared-tree stash guard (incident 2026-09-01): only meaningful for an
4730
+ // IN-PLACE run — worktree.ok isolates the job's git state into its own
4731
+ // checkout, so nothing there can leak into guardCwd. Best-effort and run
4732
+ // regardless of exit code: a job can discard shared state on its way to
4733
+ // a non-zero exit just as easily as on a clean one.
4734
+ if (!worktree.ok) {
4735
+ sharedTreeGuard = await module.exports.checkSharedTreeGuard({
4736
+ cwd: guardCwd,
4737
+ stashBaseline,
4738
+ dirtyBaseline: guardBaseline,
4739
+ headBefore: guardHeadBefore,
4740
+ slug: job.slug,
4741
+ });
4742
+ // A restored stash alone isn't silence — it's logged loudly above and
4743
+ // surfaced on the job row below — but a path that's still missing
4744
+ // (restore failed, or two-plus stashes we refused to guess between, or
4745
+ // a revert with no stash to restore at all) must not finish green.
4746
+ if (sharedTreeGuard && (sharedTreeGuard.restoreFailed || sharedTreeGuard.ambiguousStashes || sharedTreeGuard.reverted)) {
4747
+ verifyResult = {
4748
+ verdict: 'shared_tree_reverted',
4749
+ reason: sharedTreeGuard.reverted
4750
+ ? `job discarded pre-existing state in the shared tree: ${sharedTreeGuard.reverted.length} path(s) reverted with no commit to explain it (${sharedTreeGuard.reverted.slice(0, 3).join(', ')})`
4751
+ : sharedTreeGuard.restoreFailed
4752
+ ? `job stashed the shared tree and the stash could not be auto-restored: ${sharedTreeGuard.restoreFailed}`
4753
+ : `job created ${sharedTreeGuard.ambiguousStashes.length} stashes in the shared tree — ambiguous, not auto-restored (${sharedTreeGuard.ambiguousStashes.join(', ')})`,
4754
+ downgradeTo: 'needs_review',
4755
+ };
4756
+ }
4757
+ }
4758
+
3644
4759
  // SIGTERM commit check: reuse the same commit-window scan the exit=0
3645
4760
  // guard uses above (one commit-detection path, not two) to see whether a
3646
4761
  // 143 (SIGTERM) run still landed a deliverable before it died. Scoped
@@ -3663,6 +4778,8 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3663
4778
  let needsInvestigationNow = false;
3664
4779
  let investigationJobSnapshot = null;
3665
4780
  let needsReviewRcaSnapshot = null;
4781
+ let resumeRecoveryJob = null;
4782
+ let resumeRecoveryTarget = null;
3666
4783
  let terminalNotifySnapshot = null;
3667
4784
  const newlyCompletedPrds = [];
3668
4785
  await mutate((s) => {
@@ -3709,6 +4826,12 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3709
4826
  transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
3710
4827
  s.jobs[i2].finishedAt = new Date().toISOString();
3711
4828
  s.jobs[i2].exitCode = res.exitCode;
4829
+ s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
4830
+ if (salvagePatch) {
4831
+ s.jobs[i2].salvagePatch = salvagePatch;
4832
+ } else {
4833
+ delete s.jobs[i2].salvagePatch;
4834
+ }
3712
4835
  s.jobs[i2].error = effectiveStatus === 'needs_review'
3713
4836
  ? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
3714
4837
  // A failed job (non-zero exit) never consults verifyResult above,
@@ -3731,6 +4854,26 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3731
4854
  } else {
3732
4855
  delete s.jobs[i2].verifierVerdict;
3733
4856
  }
4857
+ // Closed-set outcome taxonomy (issue #11 list A2) so a queue row
4858
+ // says WHY it ended without anyone opening the transcript.
4859
+ s.jobs[i2].terminalReason = launchFailure.deriveTerminalReason({
4860
+ effectiveStatus, exitCode: res.exitCode, verifyResult, sigtermOverride, worktreeIntegrationFailure,
4861
+ });
4862
+ delete s.jobs[i2].launchFailure;
4863
+ delete s.jobs[i2].heldReason;
4864
+ // Persist the commit-guard's exact dirty-path list (verdict
4865
+ // 'uncommitted_changes' only) so a later resume-recovery attempt
4866
+ // (selectResumeRecoveryTarget) can name these paths without
4867
+ // re-running `git status` against a tree that may have moved on.
4868
+ if (verifyResult?.verdict === 'uncommitted_changes' && Array.isArray(verifyResult.dirtyPaths)) {
4869
+ // Capped the same way preRunDirtyPaths/leftoverPaths are — an
4870
+ // uncapped list here would let a pathologically dirty tree bloat
4871
+ // queue.json/history.jsonl and the resume-recovery prompt built
4872
+ // from it (buildResumeRecoveryPreamble/selectResumeRecoveryTarget).
4873
+ s.jobs[i2].uncommittedPaths = capDirtyPaths(verifyResult.dirtyPaths);
4874
+ } else {
4875
+ delete s.jobs[i2].uncommittedPaths;
4876
+ }
3734
4877
  // Non-blocking notes (e.g. a recovered missing-dependency probe, or a
3735
4878
  // pattern hit demoted because a materially-checkable verdict outranked
3736
4879
  // it) — surfaced even on completed jobs so the signal isn't lost.
@@ -3741,7 +4884,29 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3741
4884
  } else {
3742
4885
  delete s.jobs[i2].verifierAnnotations;
3743
4886
  }
4887
+ // Shared-tree guard outcome (restored stash / unresolved revert /
4888
+ // ambiguous stashes) — visible on the row even when a restored
4889
+ // stash left the run otherwise green, so it's never silent.
4890
+ if (sharedTreeGuard) {
4891
+ s.jobs[i2].sharedTreeGuard = sharedTreeGuard;
4892
+ } else {
4893
+ delete s.jobs[i2].sharedTreeGuard;
4894
+ }
3744
4895
  delete s.jobs[i2].runtime;
4896
+ // Pre-run baseline no longer needed once this run has finalized —
4897
+ // its whole purpose (letting THIS finalize compute a truthful
4898
+ // delta) is done; a fresh one is captured at the next dispatch.
4899
+ delete s.jobs[i2].guardBaseline;
4900
+ delete s.jobs[i2].guardHeadBefore;
4901
+ // Leftover-attribution fields (PRD: capture+surface uncommitted
4902
+ // work on every terminal path, not just exit=0) — set for EVERY
4903
+ // terminal outcome above (completed/failed/needs_review alike),
4904
+ // not just the exit=0 commit-guard branch, so a bare `failed` row
4905
+ // is visually distinguishable from one that quietly left work
4906
+ // behind. newlyDirtyAll is null when git-status was unavailable
4907
+ // (non-git cwd) — applyLeftoverFields treats null like "nothing to
4908
+ // attribute" via its Array.isArray guard, same as an empty array.
4909
+ applyLeftoverFields(s.jobs[i2], newlyDirtyAll);
3745
4910
 
3746
4911
  if (isNotifiableTerminalStatus(effectiveStatus)) {
3747
4912
  terminalNotifySnapshot = { ...s.jobs[i2] };
@@ -3759,6 +4924,19 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3759
4924
  // takes the treatAsPending branch above and never reaches here).
3760
4925
  needsReviewRcaSnapshot = { ...s.jobs[i2] };
3761
4926
 
4927
+ // Resume-first recovery (PRD 1111): evaluated BEFORE the auto-fix
4928
+ // eligibility check below — a job whose verdict is
4929
+ // 'uncommitted_changes' with a live sessionId gets one bounded
4930
+ // `--resume` dispatch instead of a cold-read fix-plan
4931
+ // investigation. Snapshot only (no I/O inside mutate()); the
4932
+ // actual dispatch happens outside mutate(), below. Never sets
4933
+ // needsInvestigationNow — the two are mutually exclusive for the
4934
+ // same tick, mirroring the `else if` used outside mutate().
4935
+ const target = selectResumeRecoveryTarget(s.jobs[i2]);
4936
+ if (target) {
4937
+ resumeRecoveryJob = { ...s.jobs[i2] };
4938
+ resumeRecoveryTarget = target;
4939
+ } else {
3762
4940
  // Same-tick auto-fix (feedback 2026-07-12): rather than waiting up to
3763
4941
  // 10 min for reverifyNeedsReview()'s periodic pass, check right here
3764
4942
  // whether this job qualifies for auto-fix (same eligibility rule
@@ -3782,6 +4960,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3782
4960
  needsInvestigationNow = true;
3783
4961
  investigationJobSnapshot = { ...s.jobs[i2] };
3784
4962
  }
4963
+ }
3785
4964
  }
3786
4965
  // Auto-promote: when a fix-* PRD completes successfully, the original
3787
4966
  // failed PRD's work is logically done. Flip its status to 'completed'
@@ -3797,6 +4976,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3797
4976
  orig.exitCode = 0;
3798
4977
  orig.error = null;
3799
4978
  orig.completedBy = job.slug;
4979
+ delete orig.looksDone;
3800
4980
  if (priorStatus === 'needs_review') {
3801
4981
  delete orig.verifierVerdict;
3802
4982
  }
@@ -3811,6 +4991,24 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3811
4991
  }
3812
4992
  await broadcast({ flush: true });
3813
4993
 
4994
+ // Per-run outcome sidecar (issue #11 list B5): turns/tokens/verdict in one
4995
+ // small JSON next to the log so fleet health never needs a transcript parse.
4996
+ launchFailure.writeOutcomeSidecar(runDir, job.slug, {
4997
+ runId,
4998
+ exitCode: res.exitCode,
4999
+ durationMs: res.durationMs ?? null,
5000
+ numTurns: res.resultStats?.numTurns ?? null,
5001
+ outputTokens: res.resultStats?.outputTokens ?? null,
5002
+ totalCostUsd: res.resultStats?.totalCostUsd ?? null,
5003
+ verdict: verifyResult?.verdict ?? (res.exitCode === 0 ? 'clean' : null),
5004
+ status: terminalNotifySnapshot?.status ?? failedJobSnapshot?.status ?? null,
5005
+ terminalReason: terminalNotifySnapshot?.terminalReason ?? failedJobSnapshot?.terminalReason ?? null,
5006
+ launchFailure: null,
5007
+ launchEnvApplied: launchEnv ? Object.keys(launchEnv) : [],
5008
+ filesChanged: Array.isArray(newlyDirtyAll) ? newlyDirtyAll.length : null,
5009
+ landedCommit: jobLandedCommitThisRun ?? null,
5010
+ });
5011
+
3814
5012
  if (terminalNotifySnapshot) {
3815
5013
  notifyOriginatingTab(terminalNotifySnapshot).catch((e) => {
3816
5014
  console.error('[scheduler] notifyOriginatingTab error', job.slug, e);
@@ -3826,12 +5024,28 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3826
5024
  verdict: needsReviewRcaSnapshot.verifierVerdict,
3827
5025
  annotations: needsReviewRcaSnapshot.verifierAnnotations,
3828
5026
  })
3829
- .then((report) => notifyNeedsReview(needsReviewRcaSnapshot, report))
5027
+ .then(async (report) => {
5028
+ // Persist the classification onto the parked job row so the scheduler
5029
+ // can route on it (e.g. selectAutoFixTargets excluding 'archive')
5030
+ // without re-parsing the RCA markdown on every pass.
5031
+ await mutate((s) => {
5032
+ const j = s.jobs.find((x) => x.slug === needsReviewRcaSnapshot.slug);
5033
+ applyRcaClassification(j, report);
5034
+ }).catch(() => {});
5035
+ return notifyNeedsReview(needsReviewRcaSnapshot, report);
5036
+ })
3830
5037
  .catch((e) => {
3831
5038
  console.error('[scheduler] writeRcaReport error', job.slug, e);
3832
5039
  });
3833
5040
  }
3834
5041
 
5042
+ if (resumeRecoveryJob && resumeRecoveryTarget) {
5043
+ console.log(`[scheduler] needs_review ${job.slug} → resume-recovery (session ${resumeRecoveryTarget.sessionId}, ${resumeRecoveryTarget.dirtyPaths.length} dirty path(s))`);
5044
+ spawnResumeRecovery(resumeRecoveryJob, resumeRecoveryTarget).catch((e) => {
5045
+ console.error('[scheduler] spawnResumeRecovery error', job.slug, e);
5046
+ });
5047
+ }
5048
+
3835
5049
  if (actuallyFailed && failedJobSnapshot) {
3836
5050
  // Transient-failure detector. A 143/137 exit is ALWAYS a signal kill — the
3837
5051
  // agent never self-exits with those — so the only question is WHO killed it.
@@ -3849,25 +5063,11 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3849
5063
  // the threshold and still fall through to investigation.
3850
5064
  const ec = failedJobSnapshot.exitCode;
3851
5065
  const retries = failedJobSnapshot.transientRetries ?? 0;
3852
- // Only pay for the extra git status call when the failure is plausibly
3853
- // transient — a real code failure never needs the dirty-tree check.
3854
5066
  const maybeTransient = (ec === 143 || ec === 137) || res.networkError === true;
3855
- let newlyDirtyCount = 0;
3856
- let dirtySample = '';
3857
- if (maybeTransient) {
3858
- const afterFailure = await uncommittedChanges(guardCwd);
3859
- const baseSet = new Set(guardBaseline || []);
3860
- // See the commit-guard block above: worktreeLeftoverDirty was captured
3861
- // (and the checkout already torn down) before this point, so it must
3862
- // be folded in here too or a transiently-killed job's leftover WIP
3863
- // silently disappears with its worktree.
3864
- const newlyDirty = [...new Set([
3865
- ...(afterFailure || []).filter((p) => !baseSet.has(p)),
3866
- ...worktreeLeftoverDirty,
3867
- ])];
3868
- newlyDirtyCount = newlyDirty.length;
3869
- dirtySample = newlyDirty.slice(0, 3).join(', ');
3870
- }
5067
+ // newlyDirtyAll was computed once, above, right after the try/finally —
5068
+ // reused here rather than re-querying git status a third time.
5069
+ const newlyDirtyCount = maybeTransient ? (newlyDirtyAll || []).length : 0;
5070
+ const dirtySample = maybeTransient ? (newlyDirtyAll || []).slice(0, 3).join(', ') : '';
3871
5071
  const decision = classifyFailureOutcome({
3872
5072
  exitCode: ec,
3873
5073
  networkError: res.networkError,
@@ -3887,12 +5087,13 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3887
5087
  });
3888
5088
  await broadcast({ flush: true });
3889
5089
  } else if (decision.action === 'fail-dirty') {
3890
- console.log(`[scheduler] transient failure (${decision.transientKind}) for ${job.slug} left ${newlyDirtyCount} uncommitted file(s) (e.g. ${dirtySample}) — not auto-requeuing`);
5090
+ const salvageNote = salvagePatch ? ` — recoverable from salvage patch ${salvagePatch}` : '';
5091
+ console.log(`[scheduler] transient failure (${decision.transientKind}) for ${job.slug} left ${newlyDirtyCount} uncommitted file(s) (e.g. ${dirtySample})${salvageNote} — not auto-requeuing`);
3891
5092
  await mutate((s) => {
3892
5093
  const i = s.jobs.findIndex((x) => x.slug === job.slug);
3893
5094
  if (i >= 0) {
3894
5095
  transitionJob(s.jobs[i], 'failed', { reason: `transient failure (${decision.transientKind}) left uncommitted work — not auto-requeued`, source: 'spawnJob:fail-dirty' });
3895
- s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample}) — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
5096
+ s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample})${salvageNote} — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
3896
5097
  }
3897
5098
  });
3898
5099
  await broadcast({ flush: true });
@@ -3927,17 +5128,41 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3927
5128
  runningSet.delete(job.slug);
3928
5129
  // Slot release notifies subscribed pumps (chat lane) machine-wide.
3929
5130
  sessionSlots.release(slotToken);
5131
+ // Release the exclusive quiet-machine lease on EVERY exit path this
5132
+ // finally covers (normal exit, timeout, SIGTERM, crash) — see the
5133
+ // acquire-site comment above. Bounded: a lease this function never
5134
+ // acquired is simply a no-op release.
5135
+ if (quietLeaseAcquired) quietMachineLease.release(job.slug);
3930
5136
  // Each job completion is a signal to advance the queue.
3931
5137
  tickQueue().catch(() => {});
3932
5138
  }
3933
5139
  }
3934
5140
 
5141
+ /**
5142
+ * Dispatch a resume-recovery attempt (PRD 1111) for a job already found
5143
+ * eligible by selectResumeRecoveryTarget. Thin wrapper around spawnJob —
5144
+ * reuses its entire slot-acquire/worktree/verify/commit-guard/finalize
5145
+ * machinery unchanged, so a resume run that itself parks or fails falls
5146
+ * through to the SAME spawnInvestigation fallback any other run would, with
5147
+ * zero special-casing. `job` and `resumeTarget` must be snapshots taken
5148
+ * BEFORE this call (this function does no eligibility re-check — spawnJob's
5149
+ * own dispatch mutate is what stamps resumeRecoveryAttempted, atomically
5150
+ * with the 'running' transition).
5151
+ */
5152
+ async function spawnResumeRecovery(job, resumeTarget) {
5153
+ const { runId, dir: runDir } = pickRunDir();
5154
+ await spawnJob(job, runId, runDir, job.cwd || DEFAULT_PROJECT_CWD, resumeTarget);
5155
+ }
5156
+
3935
5157
  // Serialized ticker: prevents two concurrent tickQueue() calls from racing
3936
5158
  // on the same pending jobs. A simple promise tail suffices since pickNextBatch
3937
5159
  // is synchronous and spawnJob is fire-and-forget.
3938
5160
  let tickTail = Promise.resolve();
3939
5161
 
3940
- function tickQueue() {
5162
+ // `bypassLoadGate` is set only by the explicit human run-now / force-tick
5163
+ // paths (via runDueJobs): the human is asking, so the CPU-load gate yields
5164
+ // and logs that it did. Every automatic caller leaves it false.
5165
+ function tickQueue({ bypassLoadGate = false } = {}) {
3941
5166
  const next = tickTail.then(async () => {
3942
5167
  const state = await readQueue();
3943
5168
  // Never reconcile against an unreadable queue: reconcile() would see zero
@@ -3962,7 +5187,13 @@ function tickQueue() {
3962
5187
  // cap that sessionSlots.cjs was written to replace — which silently
3963
5188
  // ceilinged the queue at 3 while the pool the user configured said 5.
3964
5189
  const freeSlots = sessionSlots.available();
3965
- const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots);
5190
+ const heldSlugs = await computeLaunchHolds(state);
5191
+ const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
5192
+ leaseHeld: quietMachineLease.isHeld(),
5193
+ machineInUse: sessionSlots.inUse(),
5194
+ now: Date.now(),
5195
+ heldSlugs,
5196
+ });
3966
5197
  if (batch.length === 0 && freeSlots === 0) {
3967
5198
  const snap = sessionSlots.snapshot();
3968
5199
  const pendingCount = state.jobs.filter((j) => j.status === 'pending').length;
@@ -4025,6 +5256,35 @@ function tickQueue() {
4025
5256
  lastMemGate = null;
4026
5257
  }
4027
5258
 
5259
+ // Load gate (PRD 1085) — the INNERMOST launch predicate, evaluated only
5260
+ // once every outer gate (sessionSlots pool → per-project cap inside
5261
+ // pickNextBatch → memory above) has already admitted `gatedBatch`. It
5262
+ // never touches running jobs and never becomes a second pool: it only
5263
+ // withholds this tick's launches while the 1-minute loadavg per core is
5264
+ // over LOAD_GATE_PER_CORE. An explicit human Run now bypasses it.
5265
+ const load = loadGate.evaluate({ bypass: bypassLoadGate });
5266
+ if (load.bypassed) {
5267
+ console.log(`[scheduler] load gate: BYPASSED by run-now (loadavg1=${load.loadavg1} cores=${load.cores} ratio=${load.ratio} > ${load.threshold})`);
5268
+ } else if (load.gated) {
5269
+ const line = `[scheduler] load gate: loadavg1=${load.loadavg1} cores=${load.cores} ratio=${load.ratio} > ${load.threshold} — holding ${gatedBatch.length} eligible job(s)`;
5270
+ if (load.escalate) {
5271
+ const top = topCpuConsumers(3);
5272
+ console.warn(`${line} for ${Math.round(load.gatedSinceMs / 60_000)}m; top CPU: ${top.length ? top.join(' | ') : 'n/a'}`);
5273
+ } else {
5274
+ console.log(line);
5275
+ }
5276
+ if (load.shouldAudit) {
5277
+ appendAuditEvent('launch_load_gated', {
5278
+ loadavg1: load.loadavg1, cores: load.cores, ratio: load.ratio, threshold: load.threshold,
5279
+ held: gatedBatch.map((j) => j.slug), gatedSinceMs: load.gatedSinceMs,
5280
+ });
5281
+ }
5282
+ return recordTick(
5283
+ { fired: false, reason: 'load-deferred', deferredCount: gatedBatch.length, ratio: load.ratio, threshold: load.threshold },
5284
+ { detail: `load gate: ${load.loadavg1} / ${load.cores} cores = ${load.ratio} > ${load.threshold}`, holds },
5285
+ );
5286
+ }
5287
+
4028
5288
  await mutate((s) => { s.lastRunAt = new Date().toISOString(); });
4029
5289
  await broadcast();
4030
5290
 
@@ -4072,7 +5332,7 @@ function forceTickOutcome(result) {
4072
5332
  }
4073
5333
  }
4074
5334
 
4075
- async function runDueJobs() {
5335
+ async function runDueJobs({ bypassLoadGate = false } = {}) {
4076
5336
  const state = await readQueue();
4077
5337
  if (state.unreadable) {
4078
5338
  console.error('[scheduler] runDueJobs skipped: queue.json unreadable');
@@ -4083,7 +5343,7 @@ async function runDueJobs() {
4083
5343
  return { fired: false, reason: 'paused' };
4084
5344
  }
4085
5345
  cancelToken = { cancelled: false };
4086
- const result = await tickQueue();
5346
+ const result = await tickQueue({ bypassLoadGate });
4087
5347
  // Clear the one-shot scheduledFor without waiting for jobs to settle.
4088
5348
  await mutate((s) => { s.scheduledFor = null; });
4089
5349
  await broadcast();
@@ -4108,12 +5368,38 @@ async function maybeLaunchWhenAvailable(state) {
4108
5368
 
4109
5369
  // ---------- dead-process reaper ----------
4110
5370
 
5371
+ // Queue-health sweep cadence: hangs off reapDeadRunningJobs's own cycle
5372
+ // counter (it already runs once per poll tick) rather than a second timer,
5373
+ // so its cadence can never drift from the poll cadence or double-fire
5374
+ // across a backoff reset.
5375
+ let queueHealthSweepCycle = 0;
5376
+ const QUEUE_HEALTH_SWEEP_EVERY_N_CYCLES = 20;
5377
+
5378
+ /**
5379
+ * runQueueHealthSweep(jobs) — read-only reporting pass over the queue
5380
+ * snapshot reapDeadRunningJobs already read this cycle. Never transitions a
5381
+ * job, never archives a PRD, never spawns anything; only logs and appends
5382
+ * an audit event for any project with drift worth a human glance.
5383
+ */
5384
+ function runQueueHealthSweep(jobs) {
5385
+ try {
5386
+ for (const { cwd, neverRan, looksDone, stuck } of computeQueueHealth(jobs)) {
5387
+ console.log(`[scheduler] queue-health ${cwd}: ${neverRan} never_ran, ${looksDone} looks-done, ${stuck} stuck`);
5388
+ appendAuditEvent('scheduler_queue_health', { cwd, neverRan, looksDone, stuck });
5389
+ }
5390
+ } catch (e) {
5391
+ console.warn('[scheduler] queue-health sweep error', e?.message);
5392
+ }
5393
+ }
5394
+
4111
5395
  /**
4112
- * Scan running jobs, identify those whose claude process is provably dead, and
4113
- * finalize them to completed/failed by reading the run log. Called once per
4114
- * poll cycle. Conservative: a job with no runtime.pid yet (spawn mid-flight)
4115
- * is always skipped. A job whose pid is alive (claudePidAlive) is always skipped.
4116
- * Exported so unit tests can invoke it directly.
5396
+ * Scan running jobs, identify those whose claude process is provably dead OR
5397
+ * whose spawn never got far enough to record a runtime.pid in the first
5398
+ * place, and finalize them to completed/failed by reading the run log.
5399
+ * Called once per poll cycle. A job whose pid is alive (claudePidAlive) is
5400
+ * always skipped. A pidless job younger than PIDLESS_SPAWN_GRACE_MS is
5401
+ * skipped too (spawn may still be mid-flight) — see selectReapableJobs for
5402
+ * the full predicate. Exported so unit tests can invoke it directly.
4117
5403
  */
4118
5404
  async function reapDeadRunningJobs() {
4119
5405
  try {
@@ -4123,32 +5409,102 @@ async function reapDeadRunningJobs() {
4123
5409
  // status:"running" with no slug left in runningSet to trigger reconciliation.
4124
5410
  // queue.json is the source of truth for which jobs are actually running.
4125
5411
  const state = await readQueue();
5412
+ const { reapable, warnings } = selectReapableJobs(state.jobs, Date.now(), {
5413
+ pidAlive: claudePidAlive,
5414
+ grace: PIDLESS_SPAWN_GRACE_MS,
5415
+ });
5416
+ for (const w of warnings) {
5417
+ console.warn(`[scheduler] reapDeadRunningJobs: ${w.reason} slug=${w.slug} — leaving row alone`);
5418
+ }
5419
+
4126
5420
  const dead = [];
4127
- for (const j of state.jobs) {
4128
- if (j.status !== 'running') continue;
4129
- const pid = j.runtime?.pid;
4130
- if (!pid) continue; // spawn may be mid-flight; give it a cycle
4131
- if (claudePidAlive(pid)) continue;
4132
- const logPath = j.runId
5421
+ for (const { slug, pid, pidless, reason } of reapable) {
5422
+ const j = state.jobs.find((x) => x.slug === slug);
5423
+ const logPath = j?.runId
4133
5424
  ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`)
4134
5425
  : null;
5426
+ // Absent/empty run dir → classifyRunOutcome finds no result event →
5427
+ // 'no_result' → non-success below → filed as failed, never completed.
4135
5428
  const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
4136
- dead.push({ slug: j.slug, pid, outcome });
5429
+ // A pidless reap means the spawn never got far enough to record a
5430
+ // pid — the gate could not possibly have run, regardless of what
5431
+ // classifyRunOutcome makes of an absent/empty log.
5432
+ const gateOutcome = pidless ? 'never_ran' : mapOutcomeToGateOutcome(outcome);
5433
+ dead.push({ slug, pid, outcome, gateOutcome, pidless, reason });
4137
5434
  }
5435
+
5436
+ queueHealthSweepCycle += 1;
5437
+ if (queueHealthSweepCycle % QUEUE_HEALTH_SWEEP_EVERY_N_CYCLES === 0) {
5438
+ runQueueHealthSweep(state.jobs);
5439
+ }
5440
+
4138
5441
  if (dead.length === 0) return;
4139
5442
 
4140
- await mutate((s) => {
4141
- for (const { slug, pid, outcome } of dead) {
5443
+ await mutate(async (s) => {
5444
+ for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
4142
5445
  const idx = s.jobs.findIndex((x) => x.slug === slug);
4143
5446
  if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
4144
5447
  const success = outcome === 'success';
4145
- transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: `reaped: process gone (outcome=${outcome})`, source: 'reapDeadRunningJobs' });
5448
+
5449
+ // Best-effort in-place leftover computation: a job whose owning
5450
+ // process vanished without spawnJob()'s own finally block ever
5451
+ // running (the exact case this reaper exists for) never got that
5452
+ // block's salvage OR leftover-attribution pass either. Only
5453
+ // attempted when the row carries a persisted pre-run baseline
5454
+ // (guardBaseline, persisted by spawnJob at dispatch — see there).
5455
+ // With no baseline there is no safe way to tell this job's own dirt
5456
+ // from a human's or a sibling's pre-existing WIP, so this skips
5457
+ // rather than ever dumping/attributing the whole tree.
5458
+ let deltaPaths = null;
5459
+ if (Array.isArray(s.jobs[idx].guardBaseline) && s.jobs[idx].runId) {
5460
+ try {
5461
+ const rowCwd = s.jobs[idx].cwd || s.config?.defaultCwd || DEFAULT_PROJECT_CWD;
5462
+ const after = await uncommittedChanges(rowCwd);
5463
+ if (after) {
5464
+ const baseSet = new Set(s.jobs[idx].guardBaseline);
5465
+ deltaPaths = after.filter((p) => !baseSet.has(p));
5466
+ if (deltaPaths.length) {
5467
+ const salvagePath = path.join(RUNS_DIR, s.jobs[idx].runId, `${slug}.uncommitted.patch`);
5468
+ const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: rowCwd, paths: deltaPaths, outFile: salvagePath });
5469
+ if (salvage && salvage.ok) {
5470
+ s.jobs[idx].salvagePatch = salvagePath;
5471
+ console.log(`[scheduler] reapDeadRunningJobs: salvaged ${salvage.bytes} byte(s) of uncommitted in-place diff for ${slug} to ${salvagePath}`);
5472
+ }
5473
+ }
5474
+ }
5475
+ } catch (e) {
5476
+ console.error(`[scheduler] reapDeadRunningJobs: in-place salvage failed for ${slug}`, e);
5477
+ }
5478
+ }
5479
+ const leftoverSuffix = deltaPaths && deltaPaths.length
5480
+ ? ` — left ${deltaPaths.length} files uncommitted`
5481
+ : '';
5482
+ const transitionReason = (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
5483
+
5484
+ transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
4146
5485
  s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
4147
5486
  s.jobs[idx].finishedAt = new Date().toISOString();
4148
- s.jobs[idx].error = success ? null : `reaped: process gone, no success result in log (${outcome})`;
5487
+ s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
5488
+ s.jobs[idx].gateOutcome = gateOutcome;
4149
5489
  delete s.jobs[idx].runtime;
5490
+ delete s.jobs[idx].guardBaseline;
5491
+ delete s.jobs[idx].guardHeadBefore;
5492
+ applyLeftoverFields(s.jobs[idx], deltaPaths);
4150
5493
  runningSet.delete(slug);
4151
- console.log(`[scheduler] reaped dead job slug=${slug} pid=${pid} outcome=${outcome}`);
5494
+ // A dead job reaped here never reached spawnJob's own finally block
5495
+ // (that's this reaper's whole reason to exist — see its header
5496
+ // comment) — so if it held the quiet-machine lease, spawnJob never
5497
+ // got the chance to release it. Release it here too, or a
5498
+ // quietMachine job whose process silently vanished (OOM, a crash
5499
+ // with no exit event) wedges the lease held forever and stalls
5500
+ // dispatch for every project until the app restarts.
5501
+ if (s.jobs[idx].quietMachine === true) quietMachineLease.release(slug);
5502
+ if (pidless) {
5503
+ console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
5504
+ appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
5505
+ } else {
5506
+ console.log(`[scheduler] reaped dead job slug=${slug} pid=${pid} outcome=${outcome}`);
5507
+ }
4152
5508
  }
4153
5509
  });
4154
5510
 
@@ -4335,14 +5691,19 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
4335
5691
  // investigation jobs correctly found "nothing to fix" but were flagged
4336
5692
  // anyway). For non-fix-plan jobs the exemption never applies, so rescanning
4337
5693
  // their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
4338
- const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'pass_no_commit', 'pass_no_commit_already_shipped']);
4339
-
4340
- // Bounds fix-plan recursion: depth 1 = the original job, depth 2 = its fix
4341
- // (gets exactly one follow-up investigation if it also lands in
4342
- // needs_review), depth 3+ (a fix-of-a-fix-of-a-fix) is excluded. Shared by
4343
- // selectAutoFixTargets and spawnInvestigation so both call sites agree on
4344
- // one threshold.
4345
- const MAX_INVESTIGATION_DEPTH = 2;
5694
+ const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'abandoned_background_task', 'pass_no_commit', 'pass_no_commit_already_shipped']);
5695
+
5696
+ // Bounds fix-plan recursion: cap N permits at most N+1 fix jobs per original
5697
+ // slug (depth 1 = the original job, depth 2 = its `-fix`, depth 3+ is
5698
+ // excluded). With N=1 that's `<slug>-fix` and `<slug>-fix-fix`, never a third
5699
+ // `-fix-fix-fix`. Lowered from 2 to 1 on 2026-08-31 (starry-night-ships):
5700
+ // three concurrent chains (115-fix-fix, 113-fix-fix, 111-fix-fix-fix) were
5701
+ // riding the old cap, and 115-fix-fix's own root-cause section read "The
5702
+ // code was already CORRECT. Only verification and commit failed." — a third
5703
+ // auto-retry re-runs an entire PRD and test battery to redo a `git commit`,
5704
+ // at near-zero marginal success probability. Shared by selectAutoFixTargets
5705
+ // and spawnInvestigation so both call sites agree on one threshold.
5706
+ const MAX_INVESTIGATION_DEPTH = 1;
4346
5707
 
4347
5708
  /**
4348
5709
  * True when a fix-plan job's investigationDepth is at or past the recursion
@@ -4467,22 +5828,48 @@ function isPlanUnqueued(job, queuedSlugs) {
4467
5828
  * Bias to needs_review: a false yellow costs a human glance, a false green
4468
5829
  * costs a silently-unfixed bug — which is exactly what happened.
4469
5830
  */
5831
+ // abandoned_background_task shares no_verdict_sentinel's exact rescan path
5832
+ // (same "sentinel === null && !commitEvidence" gate in runVerify, same
5833
+ // committedDuringRun repo-wide-not-per-job attribution problem) — the PRD 983
5834
+ // incident mechanism above applies identically, so it gets the same guard
5835
+ // rather than a carve-out that would silently reopen the same false-heal hole.
5836
+ const NO_ATTRIBUTABLE_COMMIT_VERDICTS = new Set(['no_verdict_sentinel', 'abandoned_background_task']);
5837
+
4470
5838
  function healRefusalReason(job, verdict, committedDuringRun) {
4471
5839
  if (!job || !verdict) return null;
4472
5840
  if (!COMPLETED_EQUIVALENT_VERDICTS.has(verdict.verdict)) return null;
4473
- if (job.verifierVerdict !== 'no_verdict_sentinel') return null;
5841
+ if (!NO_ATTRIBUTABLE_COMMIT_VERDICTS.has(job.verifierVerdict)) return null;
4474
5842
  // A commit this job actually recorded as its own is real evidence; the
4475
5843
  // repo-wide window scan is not.
4476
5844
  if (job.landedCommit) return null;
4477
- return 'no_verdict_sentinel with no job-attributable commit — refusing to heal'
5845
+ return `${job.verifierVerdict} with no job-attributable commit — refusing to heal`
4478
5846
  + ` (committedInWindow=${committedDuringRun === true} is repo-wide, not proof this job delivered)`;
4479
5847
  }
4480
5848
 
5849
+ /**
5850
+ * True when a `failed` job's failure is unverified-shaped — no result event
5851
+ * was ever recorded for its run (classifyRunOutcome === 'no_result'), so no
5852
+ * SCHEDULER_VERDICT sentinel could have been parsed either, OR it already
5853
+ * carries a RESCANNABLE_VERDICTS verifierVerdict. A row that failed with a
5854
+ * real result event (classifyRunOutcome === 'failed', i.e. a genuine red
5855
+ * gate or a real non-zero-exit error) is excluded — that failure is
5856
+ * evidence, not silence, and must never become a heal candidate (PRD 1102).
5857
+ */
5858
+ function isFailedUnverifiedShaped(job) {
5859
+ if (!job || job.status !== 'failed') return false;
5860
+ if (job.verifierVerdict && RESCANNABLE_VERDICTS.has(job.verifierVerdict)) return true;
5861
+ const runId = job.runId || resolveRunId(job);
5862
+ if (!runId) return false;
5863
+ const logPath = path.join(RUNS_DIR, runId, `${job.slug}.log`);
5864
+ return classifyRunOutcome(logPath) === 'no_result';
5865
+ }
5866
+
4481
5867
  function isRescanCandidate(job) {
4482
- return !!job
4483
- && job.status === 'needs_review'
4484
- && !!(job.runId || resolveRunId(job))
4485
- && RESCANNABLE_VERDICTS.has(job.verifierVerdict);
5868
+ if (!job) return false;
5869
+ if (!(job.runId || resolveRunId(job))) return false;
5870
+ if (job.status === 'needs_review') return RESCANNABLE_VERDICTS.has(job.verifierVerdict);
5871
+ if (job.status === 'failed') return isFailedUnverifiedShaped(job);
5872
+ return false;
4486
5873
  }
4487
5874
 
4488
5875
  /**
@@ -4516,10 +5903,36 @@ function isRescanCandidate(job) {
4516
5903
  * exhausted retry is excluded
4517
5904
  * - no fix sibling on disk (fixSlugExists) or already in the queue
4518
5905
  */
5906
+ /**
5907
+ * Persist a writeRcaReport() result onto its job row — job.rcaFailureClass /
5908
+ * job.rcaRecoveryAction — so selectAutoFixTargets and future routing can read
5909
+ * the classification straight off the queue row instead of re-parsing the RCA
5910
+ * markdown. Pure mutation of the passed-in job object; no I/O. A no-op when
5911
+ * the job is missing, has moved off needs_review (e.g. resumed and completed
5912
+ * before this async write landed), or the report was never filed (disabled,
5913
+ * error, etc). Returns whether it applied, for callers/tests that want to
5914
+ * assert on it.
5915
+ */
5916
+ function applyRcaClassification(job, report) {
5917
+ if (!job || job.status !== 'needs_review' || !report?.filed) return false;
5918
+ job.rcaFailureClass = report.failureClass;
5919
+ job.rcaRecoveryAction = report.recoveryAction;
5920
+ return true;
5921
+ }
5922
+
4519
5923
  function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRunId }) {
4520
5924
  const slugsInQueue = new Set(jobs.map((j) => j.slug));
4521
5925
  return jobs.filter((job) => {
4522
5926
  if (job.status !== 'needs_review') return false;
5927
+ // A stale re-run whose work already shipped (rcaReport's 'already-shipped'
5928
+ // class) must never buy a fix-plan PRD — there is nothing to fix, and the
5929
+ // correct recovery (archiving the PRD) is a human/reconcile action, not
5930
+ // an investigation.
5931
+ if (job.rcaRecoveryAction === 'archive') return false;
5932
+ // Resume-first recovery (PRD 1111): a job still eligible for its one
5933
+ // bounded `--resume` attempt must never also become a fix-plan target
5934
+ // in the same pass — see spawnInvestigation's own identical guard.
5935
+ if (selectResumeRecoveryTarget(job)) return false;
4523
5936
  const runId = job.runId || resolveJobRunId(job);
4524
5937
  if (!runId) return false;
4525
5938
  if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth)) return false;
@@ -4555,12 +5968,52 @@ function isEligibleForImmediateAutoFix(job, allJobs, fixSlugExists) {
4555
5968
  return targets.some((t) => t.slug === job.slug);
4556
5969
  }
4557
5970
 
5971
+ /**
5972
+ * Widened evidence check (PRD 1102): does at least one commit land AFTER
5973
+ * this job's run window that touches a path the PRD itself declares? Scoped
5974
+ * to the PRD's own declared paths (never the whole repo) so a sibling job's
5975
+ * unrelated commit is not credited to this one — see healRefusalReason's own
5976
+ * rationale for why unscoped, repo-wide evidence is not attribution.
5977
+ *
5978
+ * Returns null (no annotation, never fabricated) when the PRD names no
5979
+ * paths — the caller then has only the existing, already-computed
5980
+ * committedInWindow signal to go on, same as before this PRD.
5981
+ *
5982
+ * @returns {Promise<{commits: string[], paths: string[], detectedAt: string} | null>}
5983
+ */
5984
+ async function computeLooksDone(job) {
5985
+ const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
5986
+ const paths = declaredPathsForPrd(prdPath);
5987
+ if (!paths.length) return null;
5988
+ await fetchAllRefs(job.cwd);
5989
+ const commits = await landedSinceRun(job.cwd, job.startedAt, paths);
5990
+ if (!commits.length) return null;
5991
+ return { commits, paths, detectedAt: new Date().toISOString() };
5992
+ }
5993
+
4558
5994
  async function reverifyNeedsReview() {
4559
5995
  const snap = await readQueue();
4560
5996
  const candidates = snap.jobs.filter(isRescanCandidate);
4561
5997
  const healed = [];
4562
5998
  const leftForReview = [];
5999
+ const looksDoneUpdates = [];
4563
6000
  for (const job of candidates) {
6001
+ if (job.status === 'failed') {
6002
+ // A failed row never runs the transcript-verifier rescan below — that
6003
+ // machinery (verifyRun/COMPLETED_EQUIVALENT_VERDICTS) exists to
6004
+ // auto-COMPLETE a stale needs_review row, and a failed row must never
6005
+ // auto-complete through this pass (see the AC's conservative-in-the-
6006
+ // completing-direction constraint). The only thing a failed candidate
6007
+ // can gain here is a looksDone annotation + a failed → needs_review
6008
+ // transition, for a human to confirm.
6009
+ const looksDone = await computeLooksDone(job);
6010
+ if (looksDone) {
6011
+ looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: true });
6012
+ } else {
6013
+ leftForReview.push({ slug: job.slug, reason: 'failed, unverified-shaped run — no post-window evidence on declared paths' });
6014
+ }
6015
+ continue;
6016
+ }
4564
6017
  const runDir = path.join(RUNS_DIR, job.runId || resolveRunId(job));
4565
6018
  const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
4566
6019
  // Derive committedDuringRun from the recorded run window. The live
@@ -4587,13 +6040,45 @@ async function reverifyNeedsReview() {
4587
6040
  });
4588
6041
  } catch { leftForReview.push({ slug: job.slug, reason: 'verifyRun threw' }); continue; }
4589
6042
  const refusal = healRefusalReason(job, v, committedDuringRun);
6043
+ let stillOpen = true;
4590
6044
  if (refusal) {
4591
6045
  leftForReview.push({ slug: job.slug, reason: refusal });
4592
6046
  } else if (v && COMPLETED_EQUIVALENT_VERDICTS.has(v.verdict)) {
4593
6047
  healed.push(job.slug);
6048
+ stillOpen = false;
4594
6049
  } else {
4595
6050
  leftForReview.push({ slug: job.slug, reason: v ? `${v.verdict}: ${v.reason}` : 'null verdict' });
4596
6051
  }
6052
+ // Still needs_review after the existing heal pass — widen the evidence
6053
+ // window before giving up on it entirely (unchanged heal semantics for
6054
+ // rows that already qualified above; this only adds an annotation).
6055
+ if (stillOpen) {
6056
+ const looksDone = await computeLooksDone(job);
6057
+ if (looksDone) {
6058
+ looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
6059
+ }
6060
+ }
6061
+ }
6062
+ if (looksDoneUpdates.length) {
6063
+ const bySlug = new Map(looksDoneUpdates.map((u) => [u.slug, u]));
6064
+ await mutate((s) => {
6065
+ for (const j of s.jobs) {
6066
+ const u = bySlug.get(j.slug);
6067
+ if (!u) continue;
6068
+ if (u.fromFailed) {
6069
+ transitionJob(j, 'needs_review', {
6070
+ reason: 'looks done — commit(s) since this run touch this PRD\'s declared paths; confirm before archiving',
6071
+ source: 'reverifyNeedsReview:looksDone',
6072
+ });
6073
+ }
6074
+ if (j.status !== 'needs_review') continue;
6075
+ j.looksDone = u.looksDone;
6076
+ const shaList = u.looksDone.commits.slice(0, 5).map((c) => c.slice(0, 7)).join(', ');
6077
+ j.error = `looks done — ${u.looksDone.commits.length} commit(s) since this run touch this PRD's paths (${shaList}); confirm before archiving`;
6078
+ }
6079
+ });
6080
+ console.log(`[scheduler] boot reverify: looksDone annotated for ${looksDoneUpdates.length} row(s): ${looksDoneUpdates.map((u) => u.slug).join(', ')}`);
6081
+ await broadcast();
4597
6082
  }
4598
6083
  if (healed.length) {
4599
6084
  const healSet = new Set(healed);
@@ -4604,6 +6089,7 @@ async function reverifyNeedsReview() {
4604
6089
  transitionJob(j, 'completed', { reason: 'boot reverify: stale needs_review healed', source: 'reverifyNeedsReview:heal' });
4605
6090
  j.error = null;
4606
6091
  delete j.verifierVerdict;
6092
+ delete j.looksDone;
4607
6093
  healedPrds.push({ slug: j.slug, cwd: j.cwd });
4608
6094
  }
4609
6095
  }
@@ -4643,6 +6129,7 @@ async function reverifyNeedsReview() {
4643
6129
  orig.exitCode = 0;
4644
6130
  orig.error = null;
4645
6131
  orig.completedBy = job.slug;
6132
+ delete orig.looksDone;
4646
6133
  if (priorStatus === 'needs_review') delete orig.verifierVerdict;
4647
6134
  promoted.push(`${orig.slug} (was ${priorStatus}, via ${job.slug})`);
4648
6135
  promotedPrds.push({ slug: orig.slug, cwd: orig.cwd });
@@ -4709,14 +6196,40 @@ async function reverifyNeedsReview() {
4709
6196
  await broadcast();
4710
6197
  }
4711
6198
 
6199
+ // The annotate mutate above only runs conditionally — when it didn't fire,
6200
+ // afterHealForAnnotate is still the current on-disk state, so reuse it
6201
+ // instead of re-reading queue.json twice more back-to-back for the
6202
+ // resume-recovery and auto-fix passes below (neither of which mutates
6203
+ // synchronously: spawnResumeRecovery/spawnJob's own writes land later).
6204
+ const queueForResumeAndAutofix = (unresolvable.length || exhaustedAutoFix.length || planUnqueued.length)
6205
+ ? await readQueue()
6206
+ : afterHealForAnnotate;
6207
+
6208
+ // Resume-first recovery (PRD 1111): before any fix-plan investigation is
6209
+ // authored below, offer the bounded one-attempt `--resume` dispatch to any
6210
+ // needs_review job this periodic pass finds still eligible — e.g. one the
6211
+ // same-tick check in spawnJob missed because the app restarted between
6212
+ // that job parking and this pass running. selectAutoFixTargets below
6213
+ // already excludes every job this loop dispatches, so a resumable job
6214
+ // never also gets a fix-plan PRD authored in the same pass.
6215
+ {
6216
+ for (const job of queueForResumeAndAutofix.jobs) {
6217
+ const target = selectResumeRecoveryTarget(job);
6218
+ if (!target) continue;
6219
+ console.log(`[scheduler] resume-recovery: needs_review ${job.slug} → resuming session ${target.sessionId}`);
6220
+ spawnResumeRecovery(job, target).catch((e) => {
6221
+ console.error('[scheduler] spawnResumeRecovery error', job.slug, e);
6222
+ });
6223
+ }
6224
+ }
6225
+
4712
6226
  // Auto-fix: spawn a fix-plan investigation for each job still in
4713
6227
  // needs_review after the heal pass (kill-switch: SM_AUTOFIX_DISABLE=1).
4714
6228
  // spawnInvestigation early-returns once investigationsInFlight reaches
4715
6229
  // MAX_CONCURRENT_INVESTIGATIONS (queues the rest for retry), so this loop
4716
6230
  // cannot fan out past the cap regardless of how many targets are selected.
4717
6231
  if (process.env.SM_AUTOFIX_DISABLE !== '1') {
4718
- const afterHeal = await readQueue();
4719
- const targets = selectAutoFixTargets(afterHeal.jobs, {
6232
+ const targets = selectAutoFixTargets(queueForResumeAndAutofix.jobs, {
4720
6233
  fixSlugExists: (s) => candidatePrdsDirs().some((dir) => fs.existsSync(path.join(dir, `${s}.md`))),
4721
6234
  });
4722
6235
  for (const job of targets) {
@@ -4745,7 +6258,7 @@ async function reverifyNeedsReview() {
4745
6258
  }
4746
6259
  }
4747
6260
 
4748
- return { rescanned: candidates.length, healed, leftForReview };
6261
+ return { rescanned: candidates.length, healed, leftForReview, looksDone: looksDoneUpdates.map((u) => u.slug) };
4749
6262
  }
4750
6263
 
4751
6264
  /**
@@ -4841,7 +6354,7 @@ function registerScheduleHandlers() {
4841
6354
  // Clears any existing pause first (same semantics as run-now).
4842
6355
  await clearPause('run-now');
4843
6356
  try {
4844
- const result = await runDueJobs();
6357
+ const result = await runDueJobs({ bypassLoadGate: true });
4845
6358
  return forceTickOutcome(result);
4846
6359
  } catch (e) {
4847
6360
  logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (force-tick)', meta: { error: e?.message } });
@@ -4917,7 +6430,7 @@ function registerScheduleHandlers() {
4917
6430
  ipcMain.handle('schedule:run-now', async () => {
4918
6431
  // Manual run-now overrides any auto-pause. Clear it first.
4919
6432
  await clearPause('run-now');
4920
- runDueJobs().catch((e) => logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (run-now)', meta: { error: e?.message } }));
6433
+ runDueJobs({ bypassLoadGate: true }).catch((e) => logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (run-now)', meta: { error: e?.message } }));
4921
6434
  return { ok: true };
4922
6435
  });
4923
6436
 
@@ -5286,6 +6799,49 @@ async function init() {
5286
6799
  slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
5287
6800
  });
5288
6801
  }
6802
+
6803
+ // Stranded-investigation restore. Unlike the two escalations above, this
6804
+ // one ACTS: 'investigating' is a transient status whose restore
6805
+ // (spawnInvestigation's onExit/catch) only runs inside the process that
6806
+ // spawned the probe, so an app restart mid-probe leaves the row frozen
6807
+ // there forever (see findStrandedInvestigations' header, and the
6808
+ // "'investigating' must never be the job's resting state" comment at
6809
+ // spawnInvestigation's onExit). This restores each stranded row to the
6810
+ // exact terminal status it already carried before the probe was
6811
+ // spawned — it never re-runs or re-investigates anything.
6812
+ const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
6813
+ if (stranded.length > 0) {
6814
+ mutate((ms) => {
6815
+ for (const st of stranded) {
6816
+ const j = ms.jobs.find((x) => x.slug === st.slug);
6817
+ if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
6818
+ transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
6819
+ delete j.runtime;
6820
+ console.warn(
6821
+ `[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
6822
+ + `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
6823
+ + `restored to '${st.restoreStatus}'`,
6824
+ );
6825
+ appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
6826
+ }
6827
+ })
6828
+ .then(() => broadcast({ flush: true }))
6829
+ .catch(() => {});
6830
+ }
6831
+
6832
+ // Per-project starvation (PRD 1087): a project with pending work that has
6833
+ // been passed over on every tick while OTHER projects dispatch. Nothing
6834
+ // else distinguishes "no pending work" from "pending work, never
6835
+ // started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
6836
+ // Escalation only, same shape as the quarantine/overrun warnings above.
6837
+ for (const sp of findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS)) {
6838
+ console.warn(
6839
+ `[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
6840
+ + `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
6841
+ + `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
6842
+ );
6843
+ appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
6844
+ }
5289
6845
  }, 10 * 60_000);
5290
6846
 
5291
6847
  // Self-rescheduling poll loop with exponential backoff. Replaces the
@@ -5412,8 +6968,8 @@ async function init() {
5412
6968
  // there" (parallelGroup/estimateMinutes/sourcePromptId/epicId/
5413
6969
  // archivedStatus); `fields=full` restores them.
5414
6970
  function toCompactPrdEntry(entry) {
5415
- const { slug, title, cwd, mtimeMs, archived, status } = entry;
5416
- return { slug, title, cwd, mtimeMs, archived, status };
6971
+ const { slug, title, cwd, mtimeMs, archived, status, agentType } = entry;
6972
+ return { slug, title, cwd, mtimeMs, archived, status, agentType };
5417
6973
  }
5418
6974
 
5419
6975
  /**
@@ -5460,6 +7016,7 @@ async function listPrdsInternal() {
5460
7016
  estimateMinutes: parsed.estimateMinutes,
5461
7017
  sourcePromptId: parsed.sourcePromptId,
5462
7018
  epicId: parsed.epicId ?? null,
7019
+ agentType: parsed.agentType ?? null,
5463
7020
  mtimeMs: stat.mtimeMs,
5464
7021
  archived,
5465
7022
  };
@@ -5638,9 +7195,16 @@ const remote = {
5638
7195
  },
5639
7196
 
5640
7197
  async resetJob(slug, opts = {}) {
5641
- if (!(await safeSlugPath(slug))) return { ok: false, error: 'invalid slug' };
7198
+ const resolved = await resolveSlugOrReason(slug, opts.cwd);
7199
+ if (!resolved.ok) {
7200
+ return { ok: false, error: resolved.reason === 'invalid-slug' ? 'invalid slug' : unknownSlugMessage(slug) };
7201
+ }
5642
7202
  const outcome = await mutate((state) => {
5643
- const idx = state.jobs.findIndex((j) => j.slug === slug);
7203
+ // Same cwd filter as resolveSlugOrReason's file lookup above — slugs are
7204
+ // derived from title text with no cwd salt, so two different projects
7205
+ // can independently produce the identical slug; an opts.cwd caller must
7206
+ // reset THAT project's job, not just any queue row matching the string.
7207
+ const idx = state.jobs.findIndex((j) => j.slug === slug && (!opts.cwd || j.cwd === opts.cwd));
5644
7208
  if (idx < 0) return { kind: 'not-found' };
5645
7209
  // Terminal-status guard lives in resetJobFields itself; force:true
5646
7210
  // threads through to override it.
@@ -5662,7 +7226,7 @@ const remote = {
5662
7226
 
5663
7227
  async listJobs() {
5664
7228
  const state = await readQueue();
5665
- return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd }));
7229
+ return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd, agentType: j.agentType ?? null }));
5666
7230
  },
5667
7231
 
5668
7232
  // Single queue row lookup, used by cancelJob/updatePrd's status guards and
@@ -5828,10 +7392,16 @@ const remote = {
5828
7392
  // cancelled job lands in 'failed' with an error naming the cause,
5829
7393
  // consistent with every other non-success terminal outcome. Refuses a
5830
7394
  // slug that's already terminal — nothing left to cancel.
5831
- async cancelJob(slug) {
7395
+ async cancelJob(slug, opts = {}) {
7396
+ if (!SCHEDULE_SLUG_RE.test(slug)) return { ok: false, error: 'invalid slug' };
5832
7397
  const state = await readQueue();
5833
- const job = state.jobs.find((j) => j.slug === slug);
5834
- if (!job) return { ok: false, error: 'not found' };
7398
+ const job = state.jobs.find((j) => j.slug === slug && (!opts.cwd || j.cwd === opts.cwd));
7399
+ if (!job) {
7400
+ return {
7401
+ ok: false,
7402
+ error: `unknown slug "${slug}": no queued job with that name${opts.cwd ? ` in cwd ${opts.cwd}` : ''} — call scheduler_list_jobs to see what exists`,
7403
+ };
7404
+ }
5835
7405
  if (job.status === 'completed' || job.status === 'failed' || job.status === 'needs_review' || job.status === 'skipped') {
5836
7406
  return { ok: false, error: `job already terminal (status: "${job.status}") — nothing to cancel` };
5837
7407
  }
@@ -5889,9 +7459,10 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
5889
7459
  return;
5890
7460
  }
5891
7461
  const force = parsed.force === true;
5892
- const result = await remoteObj.resetJob(slug, { force });
7462
+ const cwd = typeof parsed.cwd === 'string' ? parsed.cwd : undefined;
7463
+ const result = await remoteObj.resetJob(slug, { force, cwd });
5893
7464
  sendJson(res, 200, result);
5894
7465
  });
5895
7466
  }
5896
7467
 
5897
- module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims };
7468
+ module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure };