claude-code-session-manager 0.87.0 → 0.88.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (291) hide show
  1. package/README.md +26 -53
  2. package/dist/assets/{AgentLibrary-DyLWzZDf.js → AgentLibrary-kQF8_wAw.js} +1 -1
  3. package/dist/assets/{DataModel--mISIJ6h.js → DataModel-CY_mMnsr.js} +1 -1
  4. package/dist/assets/{History-C2ahUXTg.js → History-Sji2brb0.js} +1 -1
  5. package/dist/assets/{Hooks-BiC6oyR2.js → Hooks-CtqpO-p5.js} +1 -1
  6. package/dist/assets/{HostBilko-BPleEOld.js → HostBilko-DN0ydOCr.js} +1 -1
  7. package/dist/assets/{Library-Dc8Qst1R.js → Library-CoiLZTKh.js} +1 -1
  8. package/dist/assets/{ListDetail-DIXh-OLX.js → ListDetail-CmqiQ_io.js} +1 -1
  9. package/dist/assets/{MarkdownEditor-C90bkLXK.js → MarkdownEditor-HPoNeKhT.js} +1 -1
  10. package/dist/assets/{McpServers-DqcbLOLZ.js → McpServers-Bj1X3_NT.js} +1 -1
  11. package/dist/assets/{Memory-CW62MXlh.js → Memory-1PKjsM3b.js} +1 -1
  12. package/dist/assets/{Panel-Bw1FhRuF.js → Panel-I1s6ghPo.js} +1 -1
  13. package/dist/assets/{Permissions-BcUC-5y8.js → Permissions-BhiiZjHQ.js} +1 -1
  14. package/dist/assets/{Plugins-BnKx9flD.js → Plugins-gv3_fkPX.js} +2 -2
  15. package/dist/assets/{ProvenanceBadge-Bw5vNVPT.js → ProvenanceBadge-BGcgBd8V.js} +1 -1
  16. package/dist/assets/{SaveBar-CWr0O_w-.js → SaveBar-BGr9eZ4g.js} +1 -1
  17. package/dist/assets/{Scheduler-DYdLuUqq.js → Scheduler-Bdo_awPq.js} +7 -7
  18. package/dist/assets/{ScopeSwitcher-CrBLbg8s.js → ScopeSwitcher-ukl0qWaA.js} +1 -1
  19. package/dist/assets/{Settings-DluB-vN1.js → Settings-DJkFj22G.js} +1 -1
  20. package/dist/assets/{SkillReferenceGraph-CHLSseay.js → SkillReferenceGraph-0t1Qblp6.js} +1 -1
  21. package/dist/assets/{Skills-gNdo_HNK.js → Skills-CeVdBGE-.js} +1 -1
  22. package/dist/assets/{SystemPrompt-Cru05-Ia.js → SystemPrompt-CfKvNHi3.js} +1 -1
  23. package/dist/assets/{TagLibrary-DNHY0xou.js → TagLibrary-DAzNaUPd.js} +1 -1
  24. package/dist/assets/{TiptapBody-I4lmbCgP.js → TiptapBody--S_rGXoA.js} +1 -1
  25. package/dist/assets/{Toggle-bWMHjmRh.js → Toggle-Dszm0xnp.js} +1 -1
  26. package/dist/assets/{index-DV3PorRY.css → index-BHTX4OTc.css} +1 -1
  27. package/dist/assets/{index-fc_JjdxL.js → index-Dp4Rc--q.js} +356 -356
  28. package/dist/assets/{settingsSchema-BfhtZnGD.js → settingsSchema-BLbssMYo.js} +1 -1
  29. package/dist/assets/{whisperWorker-Dbia1OpC.js → whisperWorker-C7ZGQwKg.js} +7 -7
  30. package/dist/index.html +2 -2
  31. package/dist/vad/ort-wasm-simd-threaded.asyncify.mjs +106 -110
  32. package/dist/vad/ort-wasm-simd-threaded.asyncify.wasm +0 -0
  33. package/dist/vad/ort-wasm-simd-threaded.jsep.mjs +98 -98
  34. package/dist/vad/ort-wasm-simd-threaded.jsep.wasm +0 -0
  35. package/dist/vad/ort-wasm-simd-threaded.jspi.mjs +99 -102
  36. package/dist/vad/ort-wasm-simd-threaded.jspi.wasm +0 -0
  37. package/dist/vad/ort-wasm-simd-threaded.mjs +46 -46
  38. package/dist/vad/ort-wasm-simd-threaded.wasm +0 -0
  39. package/package.json +10 -13
  40. package/scripts/README.md +59 -0
  41. package/scripts/audit-ops-hygiene.cjs +350 -0
  42. package/scripts/hooks/guard-destructive-git.cjs +10 -34
  43. package/scripts/hooks/guard-inline-implementation.cjs +14 -8
  44. package/scripts/hooks/guard-prd-writes.cjs +7 -36
  45. package/scripts/hooks/guard-self-schedule.cjs +175 -0
  46. package/scripts/ops-sweep.cjs +355 -0
  47. package/scripts/scheduler-mcp-server.cjs +28 -71
  48. package/src/main/__tests__/bilkoHost-integration.test.cjs +3 -3
  49. package/src/main/__tests__/chat-cancel-terminal.test.cjs +6 -9
  50. package/src/main/__tests__/chat-exit-close-race.test.cjs +3 -3
  51. package/src/main/__tests__/chat-mcp-consent-notice.test.cjs +4 -5
  52. package/src/main/__tests__/chat-queue.test.cjs +2 -2
  53. package/src/main/__tests__/chat-stop-signal.test.cjs +2 -2
  54. package/src/main/__tests__/dep-orphan-archive-health.test.cjs +77 -0
  55. package/src/main/__tests__/dod-batchkey.test.cjs +2 -2
  56. package/src/main/__tests__/dod-drain-hook.test.cjs +2 -2
  57. package/src/main/__tests__/dod-report.test.cjs +2 -2
  58. package/src/main/__tests__/dod-reverify.test.cjs +2 -2
  59. package/src/main/__tests__/epicMint.test.cjs +2 -2
  60. package/src/main/__tests__/exchanges.test.cjs +2 -2
  61. package/src/main/__tests__/extractJson.test.cjs +2 -2
  62. package/src/main/__tests__/files-reject-credentials.test.cjs +1 -1
  63. package/src/main/__tests__/fixtures/1218-fo-01-move-scripts-lib-into-src-main-lib.log +556 -0
  64. package/src/main/__tests__/health-build-freshness.test.cjs +39 -0
  65. package/src/main/__tests__/health-delegation-chain.test.cjs +15 -1
  66. package/src/main/__tests__/health-queue-dispatch.test.cjs +58 -7
  67. package/src/main/__tests__/health-tick-liveness.test.cjs +27 -3
  68. package/src/main/__tests__/health-usage-poller.test.cjs +70 -23
  69. package/src/main/__tests__/historyRollup.test.cjs +2 -2
  70. package/src/main/__tests__/kg-augment.test.cjs +2 -2
  71. package/src/main/__tests__/mcpStatus.test.cjs +2 -2
  72. package/src/main/__tests__/memoryAggregate.test.cjs +1 -1
  73. package/src/main/__tests__/memoryStale.test.cjs +1 -1
  74. package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +25 -1
  75. package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +33 -3
  76. package/src/main/__tests__/prd-group-allocator.test.cjs +2 -2
  77. package/src/main/__tests__/prdAdminRouteParity.test.cjs +2 -0
  78. package/src/main/__tests__/prdAdminRoutes.test.cjs +14 -2
  79. package/src/main/__tests__/prdAuthoringSeed.test.cjs +39 -0
  80. package/src/main/__tests__/prdLocationsArchived.test.cjs +20 -13
  81. package/src/main/__tests__/proc-role-env.test.cjs +125 -0
  82. package/src/main/__tests__/procname-claude-spawn-sites.test.cjs +304 -0
  83. package/src/main/__tests__/procname-sm-processes.test.cjs +127 -0
  84. package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +81 -401
  85. package/src/main/__tests__/projectPages.test.cjs +63 -149
  86. package/src/main/__tests__/queue-health-verdict.test.cjs +9 -9
  87. package/src/main/__tests__/queue-starvation-dispatch-driver.test.cjs +117 -20
  88. package/src/main/__tests__/queueHistory.test.cjs +2 -2
  89. package/src/main/__tests__/rateLimitPollerStreak.test.cjs +38 -3
  90. package/src/main/__tests__/runVerify-landed-commit-outranks.test.cjs +181 -0
  91. package/src/main/__tests__/runVerify.test.cjs +5 -5
  92. package/src/main/__tests__/scheduleJobStatusDrift.test.cjs +3 -3
  93. package/src/main/__tests__/scheduleJobTransitions.test.cjs +2 -2
  94. package/src/main/__tests__/scheduler-adopted-run-supervision.test.cjs +143 -0
  95. package/src/main/__tests__/scheduler-autofix-select.test.cjs +2 -2
  96. package/src/main/__tests__/scheduler-autopromote.test.cjs +2 -2
  97. package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +3 -5
  98. package/src/main/__tests__/scheduler-boot-orphans.test.cjs +78 -97
  99. package/src/main/__tests__/scheduler-default-eligible-heal.test.cjs +161 -0
  100. package/src/main/__tests__/scheduler-dispatch-loop.test.cjs +58 -0
  101. package/src/main/__tests__/scheduler-epic-digest.test.cjs +3 -5
  102. package/src/main/__tests__/scheduler-force-tick-outcome.test.cjs +2 -2
  103. package/src/main/__tests__/scheduler-gate-shadow.test.cjs +119 -0
  104. package/src/main/__tests__/scheduler-guard-verdict-autoresolve.test.cjs +46 -0
  105. package/src/main/__tests__/scheduler-heartbeat-payload.test.cjs +80 -0
  106. package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +25 -18
  107. package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +4 -4
  108. package/src/main/__tests__/scheduler-launch-failure.test.cjs +3 -5
  109. package/src/main/__tests__/scheduler-looks-done.test.cjs +26 -7
  110. package/src/main/__tests__/scheduler-manual-pause.test.cjs +118 -0
  111. package/src/main/__tests__/scheduler-meta-code-sha.test.cjs +26 -3
  112. package/src/main/__tests__/scheduler-prd-missing-skip.test.cjs +17 -2
  113. package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +3 -5
  114. package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +41 -6
  115. package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +62 -6
  116. package/src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs +3 -5
  117. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +23 -4
  118. package/src/main/__tests__/scheduler-reconcile-cwd-preserve.test.cjs +100 -0
  119. package/src/main/__tests__/scheduler-reconcile-invalid-repair.test.cjs +3 -3
  120. package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +4 -4
  121. package/src/main/__tests__/scheduler-shard-quarantine.test.cjs +110 -0
  122. package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +81 -1
  123. package/src/main/__tests__/scheduler-stall-per-project.test.cjs +14 -0
  124. package/src/main/__tests__/scheduler-starve-escalation.test.cjs +4 -4
  125. package/src/main/__tests__/scheduler-stranded-autofix-park.test.cjs +245 -0
  126. package/src/main/__tests__/scheduler-supervisor-record.test.cjs +81 -0
  127. package/src/main/__tests__/scheduler-tick-cancel-token.test.cjs +2 -2
  128. package/src/main/__tests__/scheduler-tick-wedge.test.cjs +172 -0
  129. package/src/main/__tests__/scheduler-transient-failure.test.cjs +2 -2
  130. package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +37 -11
  131. package/src/main/__tests__/scheduler-worktree-exec-cwd.test.cjs +3 -5
  132. package/src/main/__tests__/usageSingleFlight.test.cjs +157 -0
  133. package/src/main/__tests__/workTypeLibrary.test.cjs +1 -1
  134. package/src/main/bilkoHost.cjs +16 -7
  135. package/src/main/build-info.json +8 -0
  136. package/src/main/chatRunner.cjs +4 -2
  137. package/src/main/config.cjs +2 -3
  138. package/src/main/docEdit.cjs +4 -2
  139. package/src/main/health.cjs +276 -52
  140. package/src/main/heapSnapshot.cjs +2 -2
  141. package/src/main/historyAggregator.cjs +1 -1
  142. package/src/main/index.cjs +61 -10
  143. package/src/main/ipcSchemas.cjs +6 -17
  144. package/src/main/lib/__tests__/auditLog.test.cjs +38 -0
  145. package/src/main/lib/__tests__/buildIdentity.test.cjs +121 -0
  146. package/src/main/lib/__tests__/cwdClassify.test.cjs +111 -0
  147. package/src/main/lib/__tests__/definitionOfDoneSequence.test.cjs +95 -0
  148. package/src/main/lib/__tests__/delegationReadiness.test.cjs +1 -1
  149. package/src/main/lib/__tests__/dispatchLoop.test.cjs +63 -0
  150. package/src/main/lib/__tests__/gateFixtures.json +20 -0
  151. package/src/main/lib/__tests__/gitWorktree.test.cjs +8 -1
  152. package/src/main/lib/__tests__/instanceLock.test.cjs +93 -7
  153. package/src/main/lib/__tests__/jobSupervisorRecord.test.cjs +78 -0
  154. package/src/main/lib/__tests__/jobWorktreeBootLive.test.cjs +11 -0
  155. package/src/main/lib/__tests__/localAdminHttp.test.cjs +1 -1
  156. package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +6 -1
  157. package/src/main/lib/__tests__/procIdentity.test.cjs +119 -0
  158. package/src/main/lib/__tests__/procName.test.cjs +92 -0
  159. package/src/main/lib/__tests__/queueStoreMachineStateRecovery.test.cjs +67 -0
  160. package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +6 -7
  161. package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +37 -204
  162. package/src/main/lib/__tests__/schedulerPaths.test.cjs +226 -0
  163. package/src/main/lib/__tests__/schedulerPathsWorktree.test.cjs +94 -0
  164. package/src/main/lib/__tests__/schedulerRuntimeState.test.cjs +56 -0
  165. package/src/main/lib/__tests__/sessionSlots.test.cjs +45 -0
  166. package/src/main/lib/__tests__/telemetryBacklog.test.cjs +1 -1
  167. package/src/main/lib/__tests__/upgradeDrain.test.cjs +130 -0
  168. package/src/main/lib/__tests__/usageCircuit.test.cjs +61 -0
  169. package/src/main/lib/__tests__/watchdog-helpers.test.cjs +63 -0
  170. package/src/main/lib/__tests__/watchdog-relaunch.test.cjs +73 -0
  171. package/src/main/lib/activeIndexRebuild.cjs +1 -3
  172. package/src/main/lib/activeSessions.cjs +20 -88
  173. package/src/main/lib/adoptedRunSupervisor.cjs +136 -0
  174. package/src/main/lib/agentModelResolve.cjs +33 -1
  175. package/src/main/lib/agentPersonaSchema.cjs +2 -2
  176. package/src/main/lib/auditLog.cjs +30 -5
  177. package/src/main/lib/buildIdentity.cjs +113 -0
  178. package/src/main/lib/classifyPromptTicket.cjs +3 -2
  179. package/src/main/lib/claudeBin.cjs +37 -2
  180. package/src/main/lib/cleanEnv.cjs +27 -1
  181. package/src/main/lib/credentials.cjs +4 -2
  182. package/src/main/lib/cwdClassify.cjs +185 -0
  183. package/src/main/lib/definitionOfDone.cjs +296 -52
  184. package/src/main/lib/dispatchLoop.cjs +40 -0
  185. package/src/main/lib/effectiveModelInfo.cjs +9 -10
  186. package/src/main/lib/ephemeralCwd.cjs +7 -28
  187. package/src/main/lib/gitWorktree.cjs +95 -8
  188. package/src/main/lib/guardShims.cjs +3 -3
  189. package/src/main/lib/historyRollup.cjs +6 -7
  190. package/src/main/lib/instanceLock.cjs +32 -5
  191. package/src/main/lib/jobSupervisorRecord.cjs +147 -0
  192. package/src/main/lib/jobWorktreeBootLive.cjs +9 -5
  193. package/src/main/lib/localAdminHttp.cjs +9 -27
  194. package/src/main/lib/mcpToolCatalog.cjs +29 -77
  195. package/src/main/lib/opsOwnership.cjs +27 -19
  196. package/src/main/lib/prdAuthoringSeed.cjs +36 -0
  197. package/src/main/lib/prdLocations.cjs +48 -1
  198. package/src/main/lib/procIdentity.cjs +126 -0
  199. package/src/main/lib/procName.cjs +98 -0
  200. package/src/main/lib/projectHomeAdminRoutes.cjs +56 -327
  201. package/src/main/lib/queueHistory.cjs +8 -7
  202. package/src/main/lib/queueStore.cjs +57 -37
  203. package/src/main/lib/quietMachineLease.cjs +19 -2
  204. package/src/main/lib/reservationExpiry.cjs +30 -0
  205. package/src/main/lib/runClaudeP.cjs +4 -2
  206. package/src/main/lib/runLogRetention.cjs +3 -2
  207. package/src/main/lib/scheduleJobSchema.cjs +2 -2
  208. package/src/main/lib/scheduleJobTransitions.cjs +29 -2
  209. package/src/main/lib/schedulerBatch.cjs +27 -2
  210. package/src/main/lib/schedulerPaths.cjs +175 -0
  211. package/src/main/lib/schedulerRuntimeState.cjs +59 -0
  212. package/src/main/lib/sessionSlots.cjs +47 -9
  213. package/src/main/lib/smProcNames.cjs +51 -0
  214. package/src/main/lib/upgradeDrain.cjs +188 -0
  215. package/src/main/lib/usageCircuit.cjs +53 -6
  216. package/src/main/lib/watchdogHelpers.cjs +70 -37
  217. package/src/main/lib/withTimeout.cjs +33 -0
  218. package/src/main/mcpStatus.cjs +4 -2
  219. package/src/main/pluginInstall.cjs +6 -2
  220. package/src/main/projectPages.cjs +41 -145
  221. package/src/main/pty.cjs +8 -1
  222. package/src/main/queueOps.cjs +10 -10
  223. package/src/main/runVerify.cjs +69 -3
  224. package/src/main/scheduler.cjs +1607 -383
  225. package/src/main/seedAgentPersonas.cjs +1 -1
  226. package/src/main/seedDevPlugin.cjs +1 -1
  227. package/src/main/seedSchedulerMcp.cjs +7 -6
  228. package/src/main/seedStatus.cjs +1 -1
  229. package/src/main/supervisor.cjs +5 -3
  230. package/src/main/usage.cjs +126 -62
  231. package/src/preload/api.d.ts +18 -64
  232. package/src/preload/index.cjs +2 -4
  233. package/src/seed/agents/project-home-builder.md +31 -48
  234. package/screenshots/.gitkeep +0 -0
  235. package/screenshots/README-screenshots.md +0 -13
  236. package/src/main/lib/projectPageSummarySchema.cjs +0 -181
  237. package/src/main/teams.cjs +0 -95
  238. package/src/main/templates/project-pages-catalog.json +0 -741
  239. package/src/main/templates/project-pages-default-home.html +0 -123
  240. package/src/main/templates/project-pages-pipeline.md +0 -417
  241. package/src/seed/prompts/code-review/ac-coverage-check.md +0 -8
  242. package/src/seed/prompts/code-review/correctness-only.md +0 -8
  243. package/src/seed/prompts/code-review/full-spectrum-high.md +0 -8
  244. package/src/seed/prompts/code-review/hallucination-check.md +0 -8
  245. package/src/seed/prompts/code-review/public-api-compat.md +0 -8
  246. package/src/seed/prompts/code-review/readability-naming.md +0 -8
  247. package/src/seed/prompts/debugging/bug-as-failing-test.md +0 -8
  248. package/src/seed/prompts/debugging/git-bisect-regression.md +0 -8
  249. package/src/seed/prompts/debugging/instrument-intermittent-bug.md +0 -8
  250. package/src/seed/prompts/debugging/localize-pipeline-failure.md +0 -8
  251. package/src/seed/prompts/debugging/reproduce-then-diagnose.md +0 -8
  252. package/src/seed/prompts/documentation/adr-from-change.md +0 -8
  253. package/src/seed/prompts/documentation/module-readme.md +0 -8
  254. package/src/seed/prompts/documentation/onboarding-plan.md +0 -8
  255. package/src/seed/prompts/documentation/refresh-claude-md.md +0 -8
  256. package/src/seed/prompts/documentation/tsdoc-public-exports.md +0 -8
  257. package/src/seed/prompts/git-pr/conventional-commit.md +0 -8
  258. package/src/seed/prompts/git-pr/draft-pr-title-body.md +0 -8
  259. package/src/seed/prompts/git-pr/pre-commit-safety-sweep.md +0 -8
  260. package/src/seed/prompts/git-pr/release-notes-block.md +0 -8
  261. package/src/seed/prompts/git-pr/split-large-pr.md +0 -8
  262. package/src/seed/prompts/performance/bundle-startup-audit.md +0 -8
  263. package/src/seed/prompts/performance/complexity-audit.md +0 -8
  264. package/src/seed/prompts/performance/cpu-profile-hot-path.md +0 -8
  265. package/src/seed/prompts/performance/db-query-plan-review.md +0 -8
  266. package/src/seed/prompts/performance/memory-leak-hunt.md +0 -8
  267. package/src/seed/prompts/qa/api-contract-tests.md +0 -8
  268. package/src/seed/prompts/qa/e2e-critical-path.md +0 -8
  269. package/src/seed/prompts/qa/failing-test-for-bug.md +0 -8
  270. package/src/seed/prompts/qa/find-missing-test-coverage.md +0 -8
  271. package/src/seed/prompts/qa/stabilize-flaky-test.md +0 -8
  272. package/src/seed/prompts/qa/tdd-red-first.md +0 -8
  273. package/src/seed/prompts/qa/visual-regression-review.md +0 -8
  274. package/src/seed/prompts/qa/wcag-axe-scan.md +0 -8
  275. package/src/seed/prompts/refactoring/dead-code-sweep.md +0 -8
  276. package/src/seed/prompts/refactoring/extract-duplicated-pattern.md +0 -8
  277. package/src/seed/prompts/refactoring/modernize-legacy-file.md +0 -8
  278. package/src/seed/prompts/refactoring/reduce-cyclomatic-complexity.md +0 -8
  279. package/src/seed/prompts/refactoring/tighten-module-boundaries.md +0 -8
  280. package/src/seed/prompts/security/authz-audit.md +0 -8
  281. package/src/seed/prompts/security/crypto-correctness.md +0 -8
  282. package/src/seed/prompts/security/cwe-top-25-hunt.md +0 -8
  283. package/src/seed/prompts/security/dependency-audit.md +0 -8
  284. package/src/seed/prompts/security/ipc-boundary-hardening.md +0 -8
  285. package/src/seed/prompts/security/owasp-top-10-staged-diff.md +0 -8
  286. package/src/seed/prompts/security/secret-credential-scan.md +0 -8
  287. package/web/README.md +0 -41
  288. package/web/project-pages/logic/dist/logic.cjs +0 -4709
  289. package/web/project-pages/render.cjs +0 -70
  290. package/web/project-pages/renderer/dist/renderer.cjs +0 -18900
  291. package/web/project-pages/validate-summary.cjs +0 -62
@@ -47,13 +47,15 @@ const fs = require('node:fs');
47
47
  const fsp = require('node:fs/promises');
48
48
  const path = require('node:path');
49
49
  const os = require('node:os');
50
+ const { startDispatchLoop } = require('./lib/dispatchLoop.cjs');
51
+ const schedulerPaths = require('./lib/schedulerPaths.cjs');
50
52
  const { randomUUID } = require('node:crypto');
51
53
  const { execFile, execFileSync } = require('node:child_process');
52
54
  const { ipcMain } = require('electron');
53
55
  const billing = require('./usage.cjs');
54
56
  const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
55
57
  const supervisor = require('./supervisor.cjs');
56
- const { resolveClaudeBin, probeClaudeVersion } = require('./lib/claudeBin.cjs');
58
+ const { resolveClaudeBin, claudeSpawnTarget, probeClaudeVersion } = require('./lib/claudeBin.cjs');
57
59
  const launchFailure = require('./lib/launchFailure.cjs');
58
60
  const { appendError } = require('./lib/opsErrorLog.cjs');
59
61
  const { readTail } = require('./lib/fileTail.cjs');
@@ -67,6 +69,7 @@ const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
67
69
  const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
68
70
  const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
69
71
  const { resolveBindingRateLimitReset } = require('./lib/rateLimitWindow.cjs');
72
+ const { isResetFresh, bindingWindow, degradedBudget } = require('./lib/usageCircuit.cjs');
70
73
  const { computeQueueHealth } = require('./lib/queueHealth.cjs');
71
74
  const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
72
75
  const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
@@ -83,6 +86,7 @@ const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require(
83
86
  const { isFixPlanSlug, classifyDiscoveredFixPlan, resolveIsFixPlan } = require('./lib/fixPlanSlug.cjs');
84
87
  const { landedSinceRun, landedOnMainSince } = require('./lib/landedSinceRun.cjs');
85
88
  const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
89
+ const { identity: procIdentityOf, isDifferentProcess } = require('./lib/procIdentity.cjs');
86
90
  const logs = require('./logs.cjs');
87
91
  const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
88
92
  const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
@@ -131,19 +135,21 @@ const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD,
131
135
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
132
136
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
133
137
  const queueHistory = require('./lib/queueHistory.cjs');
138
+ const { resolveGate, runGateSequence } = require('./lib/definitionOfDone.cjs');
134
139
  const queueOps = require('./queueOps.cjs');
135
140
  // Feedback-auto-PRD sweep — formerly only run by the external scheduler-watchdog
136
141
  // while the app was down (PRD 686 moved it in-app so it also runs while alive).
137
142
  // Plain Node module, no Electron dependency; queuePath/prdsDir defaults already
138
143
  // match ROOT/QUEUE_PATH below since both resolve the same ~/.claude/session-manager
139
144
  // home-dir layout.
140
- const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
145
+ const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs, deriveProjectCwdFromPrdPath } = require('./lib/prdLocations.cjs');
141
146
  const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
142
147
  const agentModelResolve = require('./lib/agentModelResolve.cjs');
143
148
  const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
144
149
  const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
145
150
  const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
146
151
  const { appendAuditEvent } = require('./lib/auditLog.cjs');
152
+ const { withTimeout } = require('./lib/withTimeout.cjs');
147
153
 
148
154
  // ---------- origin session resolution (PRD 832) ----------
149
155
  // An Epic IS a tagged claude session — job rows carry the originating
@@ -164,12 +170,15 @@ function resolveOriginSessionId(cwd, epicId) {
164
170
  }
165
171
  const sessionSlots = require('./lib/sessionSlots.cjs');
166
172
  const quietMachineLease = require('./lib/quietMachineLease.cjs');
173
+ const runtimeState = require('./lib/schedulerRuntimeState.cjs');
167
174
  const jobWorktree = require('./lib/jobWorktree.cjs');
168
175
  const gitWorktree = require('./lib/gitWorktree.cjs');
169
176
  const { buildJobWorktreeIsLive } = require('./lib/jobWorktreeBootLive.cjs');
170
177
  const { buildTerminalOrphanIsLive } = require('./lib/jobWorktreeTerminalOrphanLive.cjs');
171
178
  const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
172
179
  const queueStore = require('./lib/queueStore.cjs');
180
+ const supervisorRecord = require('./lib/jobSupervisorRecord.cjs');
181
+ const adoptedRunSupervisor = require('./lib/adoptedRunSupervisor.cjs');
173
182
  const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
174
183
  const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
175
184
  const { computeDispositionRewrite } = require('./lib/prdDisposition.cjs');
@@ -182,20 +191,26 @@ const { allProjectCwds } = require('./lib/activeSessions.cjs');
182
191
  // an exemption it should have applied landed on disk, and nothing in the
183
192
  // run record showed that; this is the fix).
184
193
  const SCHEDULER_BOOTED_AT = new Date().toISOString();
185
- // Resolves against __dirname (this app's OWN source checkout) — unaffected by
186
- // PRD 994's job worktrees, which live under a job's PROJECT cwd, never under
187
- // this app's install directory.
188
- const SCHEDULER_CODE_SHA = (() => {
189
- try {
190
- return execFileSync('git', ['-C', __dirname, 'rev-parse', '--short', 'HEAD'], {
191
- timeout: 5000,
192
- encoding: 'utf8',
193
- stdio: ['ignore', 'pipe', 'ignore'],
194
- }).trim();
195
- } catch {
196
- return null;
197
- }
198
- })();
194
+ // A production npx install ships no .git at all, so a runtime `git
195
+ // rev-parse` from __dirname was structurally always null there — every
196
+ // production run-meta sidecar recorded schedulerCodeSha: null. buildIdentity
197
+ // resolves build-info.json (baked at publish time) first, falling back to a
198
+ // non-walking git read only in a dev checkout / job worktree — see
199
+ // src/main/lib/buildIdentity.cjs's header.
200
+ const { resolveBuildIdentity, readInstalledBuildInfo } = require('./lib/buildIdentity.cjs');
201
+ const upgradeDrain = require('./lib/upgradeDrain.cjs');
202
+ const SCHEDULER_BUILD_IDENTITY = resolveBuildIdentity({ bootedAt: SCHEDULER_BOOTED_AT });
203
+ const SCHEDULER_CODE_SHA = SCHEDULER_BUILD_IDENTITY.codeSha;
204
+ // Spread into EVERY metaPath writer below (grep `metaPath` for the full
205
+ // list) — single source so a future field never lands in some sidecars and
206
+ // not others, the exact gap that left 3 of 5 writers silently missing
207
+ // schedulerBootedAt/schedulerCodeSha before this constant existed.
208
+ const SCHEDULER_META_IDENTITY = {
209
+ schedulerBootedAt: SCHEDULER_BOOTED_AT,
210
+ schedulerCodeSha: SCHEDULER_CODE_SHA,
211
+ schedulerVersion: SCHEDULER_BUILD_IDENTITY.version,
212
+ schedulerBuiltAt: SCHEDULER_BUILD_IDENTITY.builtAt,
213
+ };
199
214
 
200
215
  const MAX_INVESTIGATION_DURATION_MS = 30 * 60_000;
201
216
 
@@ -590,7 +605,7 @@ function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAf
590
605
  // executor-created stash (never guesses when there are 2+); reports anything
591
606
  // it can't safely resolve on the returned object so the caller can surface it
592
607
  // on the job row instead of finishing silently green.
593
- async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug }) {
608
+ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug, landedCommit }) {
594
609
  try {
595
610
  const [stashAfter, headAfter] = await Promise.all([
596
611
  module.exports.stashList(cwd),
@@ -655,7 +670,16 @@ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBef
655
670
  pathsCommittedDuringRun,
656
671
  existsAfter,
657
672
  });
658
- if (reverted.length) {
673
+ // Ground truth outranks the baseline diff (2026-09-18, 1229-fo-03): the
674
+ // dirty baseline is invalidated by ANY later writer (a human commit that
675
+ // sweeps the same paths), so it can't prove a revert on its own. The
676
+ // job's own landedCommit still being an ancestor of HEAD proves its work
677
+ // was not discarded — anchored to that sha, not to the baseline.
678
+ const workSurvives = reverted.length > 0
679
+ && await module.exports.landedCommitIsAncestorOfHead(cwd, landedCommit);
680
+ if (workSurvives) {
681
+ console.log(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} baseline path(s) went clean but landed commit ${String(landedCommit).slice(0, 7)} is still an ancestor of HEAD — not a revert`);
682
+ } else if (reverted.length) {
659
683
  result.reverted = reverted;
660
684
  console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
661
685
  }
@@ -895,13 +919,9 @@ function isQueueRowRegression({ statusBefore, statusAfter, historyLenBefore, his
895
919
  return statusBefore === 'running' && statusAfter === 'pending' && historyLenAfter < historyLenBefore;
896
920
  }
897
921
 
898
- const ROOT = path.join(os.homedir(), '.claude', 'session-manager', 'scheduled-plans');
899
- const PRDS_DIR = path.join(ROOT, 'prds');
900
- const RUNS_DIR = path.join(ROOT, 'runs');
901
- const PRDS_ARCHIVE_DIR = path.join(ROOT, 'prds-archived');
902
- const QUEUE_PATH = path.join(ROOT, 'queue.json');
903
- const SCHEDULER_STATE_PATH = path.join(os.homedir(), '.claude', 'session-manager', 'scheduler-state.json');
904
- const HEARTBEAT_PATH = path.join(os.homedir(), '.claude', 'session-manager', 'scheduler-heartbeat.log');
922
+ // Machine-wide roots resolve lazily via lib/schedulerPaths.cjs (SM_SCHEDULER_HOME
923
+ // override) — never module-scope consts. The ROOT/PRDS_DIR/RUNS_DIR/
924
+ // SCHEDULER_STATE_PATH exports below are lazy getters over the same resolver.
905
925
  const HEARTBEAT_MAX_BYTES = 1024 * 1024;
906
926
  // DEFAULT_PROJECT_CWD imported from lib/schedulerBatch.cjs (single source of truth).
907
927
 
@@ -1020,7 +1040,7 @@ function biasJobOomScore(pid) {
1020
1040
  * (reconcile, list-prds, lint, rescan).
1021
1041
  */
1022
1042
  function candidatePrdsDirs() {
1023
- return [PRDS_DIR, ...resolvePrdsDirs()];
1043
+ return [schedulerPaths.prdsRoot(), ...resolvePrdsDirs()];
1024
1044
  }
1025
1045
 
1026
1046
  /**
@@ -1141,7 +1161,7 @@ function prdArchivedSkipResult(job, cwd, sessionId, startedAt, safeLog, closeFd,
1141
1161
  const finishedAt = Date.now();
1142
1162
  config.writeJsonSync(metaPath, {
1143
1163
  slug: job.slug, cwd, sessionId, exitCode: 0, skipped: reason,
1144
- note: msg, startedAt, finishedAt, durationMs: 0,
1164
+ note: msg, startedAt, finishedAt, durationMs: 0, ...SCHEDULER_META_IDENTITY,
1145
1165
  });
1146
1166
  return { exitCode: 0, durationMs: 0, skipped: reason, note: msg, sessionId };
1147
1167
  }
@@ -1283,17 +1303,21 @@ async function retireCompletedSlugs(slugs) {
1283
1303
  // Bundled authoring guide seeded into the scheduler dir so the session-manager-dev
1284
1304
  // plugin's /develop and /prd skills — which reference this stable `~`-absolute
1285
1305
  // path — work on any user's machine, not just the author's.
1306
+ // Line 1 of the template is `<!-- PRD_AUTHORING.md vN -->`: bump vN whenever the
1307
+ // template changes, or existing installs never receive the update.
1286
1308
  const PRD_AUTHORING_TEMPLATE = path.join(__dirname, 'templates', 'PRD_AUTHORING.md');
1287
- const PRD_AUTHORING_DEST = path.join(ROOT, 'PRD_AUTHORING.md');
1288
1309
 
1289
1310
  function ensureDirs() {
1290
- fs.mkdirSync(PRDS_DIR, { recursive: true });
1291
- fs.mkdirSync(RUNS_DIR, { recursive: true });
1292
- // Seed the authoring guide once; never clobber a user's edited copy.
1311
+ fs.mkdirSync(schedulerPaths.prdsRoot(), { recursive: true });
1312
+ fs.mkdirSync(schedulerPaths.runsDir(), { recursive: true });
1313
+ // Re-seed the guide whenever the bundled template's version stamp differs.
1293
1314
  try {
1294
- if (!fs.existsSync(PRD_AUTHORING_DEST) && fs.existsSync(PRD_AUTHORING_TEMPLATE)) {
1295
- fs.copyFileSync(PRD_AUTHORING_TEMPLATE, PRD_AUTHORING_DEST);
1296
- }
1315
+ const authoringDest = path.join(schedulerPaths.scheduledPlansRoot(), 'PRD_AUTHORING.md');
1316
+ seedAuthoringGuide({
1317
+ src: PRD_AUTHORING_TEMPLATE,
1318
+ dest: authoringDest,
1319
+ write: (abs, text) => config.writeTextAtomic(abs, text, { writer: 'scheduler' }),
1320
+ }).catch(() => { /* non-fatal, same as below */ });
1297
1321
  } catch { /* non-fatal: the guide is a convenience, not load-bearing for a run */ }
1298
1322
  }
1299
1323
 
@@ -1321,8 +1345,9 @@ function ensureDirs() {
1321
1345
  * queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
1322
1346
  * sweep archives it before reconcile can ever turn it into a pending job.
1323
1347
  */
1324
- async function consolidateAllFlatPrds(cwds) {
1348
+ async function consolidateAllFlatPrds(cwds, skipCwds) {
1325
1349
  for (const cwd of cwds) {
1350
+ if (skipCwds?.has(cwd)) continue; // torn shard: its PRDs are not ours to touch this pass
1326
1351
  try {
1327
1352
  const c = await consolidateFlatPrds(cwd);
1328
1353
  if (c.moved > 0) {
@@ -1354,7 +1379,7 @@ async function consolidateAllFlatPrds(cwds) {
1354
1379
  async function runPrdMigration() {
1355
1380
  let result;
1356
1381
  try {
1357
- result = await migratePrds(PRDS_DIR);
1382
+ result = await migratePrds(schedulerPaths.prdsRoot());
1358
1383
  } catch (e) {
1359
1384
  logs.writeLine({ level: 'error', scope: 'scheduler', message: 'PRD migration failed', meta: { error: e?.message } });
1360
1385
  return null;
@@ -1365,7 +1390,7 @@ async function runPrdMigration() {
1365
1390
  level: 'warn',
1366
1391
  scope: 'scheduler',
1367
1392
  message: `PRD migration: ${result.unresolved.length} file(s) left in legacy dir`,
1368
- meta: { legacyDir: PRDS_DIR, unresolved: result.unresolved },
1393
+ meta: { legacyDir: schedulerPaths.prdsRoot(), unresolved: result.unresolved },
1369
1394
  });
1370
1395
  for (const u of result.unresolved) {
1371
1396
  console.warn(`[scheduler] PRD migration: left ${u.file} in legacy dir (${u.reason})`);
@@ -1432,7 +1457,7 @@ const QUEUE_BAK_KEEP = 5;
1432
1457
  async function sweepQueueBackups() {
1433
1458
  let entries;
1434
1459
  try {
1435
- entries = await fsp.readdir(ROOT);
1460
+ entries = await fsp.readdir(schedulerPaths.scheduledPlansRoot());
1436
1461
  } catch {
1437
1462
  return;
1438
1463
  }
@@ -1448,7 +1473,7 @@ async function sweepQueueBackups() {
1448
1473
  let removed = 0;
1449
1474
  for (const f of toDelete) {
1450
1475
  try {
1451
- await fsp.unlink(path.join(ROOT, f));
1476
+ await fsp.unlink(path.join(schedulerPaths.scheduledPlansRoot(), f));
1452
1477
  removed++;
1453
1478
  } catch (e) {
1454
1479
  console.warn('[scheduler] backup sweep: unlink failed', f, e?.message);
@@ -1464,20 +1489,24 @@ async function sweepQueueBackups() {
1464
1489
  // callback that must flush meta.json before resolving) — replacing with async
1465
1490
  // would deadlock the exit path.
1466
1491
  const config = require('./config.cjs');
1492
+ const { seedAuthoringGuide } = require('./lib/prdAuthoringSeed.cjs');
1467
1493
  const atomicWriteJsonSync = (p, data) => config.writeJsonSync(p, data);
1468
1494
 
1469
1495
  // ---------- scheduler-state.json (sidecar) ----------
1470
1496
 
1471
1497
  function loadSchedulerState() {
1472
1498
  try {
1473
- const raw = fs.readFileSync(SCHEDULER_STATE_PATH, 'utf8');
1499
+ const raw = fs.readFileSync(schedulerPaths.schedulerStatePath(), 'utf8');
1474
1500
  const s = JSON.parse(raw);
1475
1501
  if (s.lastObservedReset) cachedNextReset = s.lastObservedReset;
1502
+ if (typeof s.lastResetObservedAt === 'number') lastResetObservedAtMs = s.lastResetObservedAt;
1476
1503
  if (typeof s.consecutiveFailures === 'number') consecutiveFailures = s.consecutiveFailures;
1477
1504
  if (typeof s.backoffMs === 'number') backoffMs = s.backoffMs;
1478
1505
  if (typeof s.pauseClearedManuallyAt === 'number') pauseClearedManuallyAt = s.pauseClearedManuallyAt;
1479
1506
  if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
1480
1507
  if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
1508
+ if (typeof s.failureStreakWarnedAt === 'number') failureStreakWarnedAt = s.failureStreakWarnedAt;
1509
+ if (typeof s.lastEscalationAt === 'number') lastEscalationAtMs = s.lastEscalationAt;
1481
1510
  } catch { /* first boot or corrupt — start fresh */ }
1482
1511
  }
1483
1512
 
@@ -1487,10 +1516,14 @@ function persistSchedulerState() {
1487
1516
  // require threading awaits through pause/resume bookkeeping for negligible
1488
1517
  // benefit — the file is well under one page.
1489
1518
  try {
1490
- config.writeJsonSync(SCHEDULER_STATE_PATH, {
1519
+ config.writeJsonSync(schedulerPaths.schedulerStatePath(), {
1491
1520
  version: 1,
1492
1521
  lastObservedReset: cachedNextReset,
1493
- lastResetObservedAt: cachedNextReset ? Date.now() : null,
1522
+ // Only stamped at the moment a FRESH reset was actually observed (see
1523
+ // recordObservedReset) — never Date.now() on every persist call, which
1524
+ // used to make a stale cachedNextReset look freshly-confirmed on every
1525
+ // tick even when nothing new had been read.
1526
+ lastResetObservedAt: lastResetObservedAtMs,
1494
1527
  lastPollAt,
1495
1528
  consecutiveFailures,
1496
1529
  backoffMs,
@@ -1498,6 +1531,13 @@ function persistSchedulerState() {
1498
1531
  pausedSince: null,
1499
1532
  pauseClearedManuallyAt,
1500
1533
  failureStreakWarned,
1534
+ failureStreakWarnedAt,
1535
+ lastEscalationAt: lastEscalationAtMs,
1536
+ // Circuit fields are read fresh from the live shared breaker each
1537
+ // persist — health.cjs (a separate `npm run health` process) reads
1538
+ // THESE persisted values, since it never holds the in-memory circuit.
1539
+ usageCircuitState: billing.usageCircuit.state(),
1540
+ usageCircuitOpenedAt: billing.usageCircuit.openedAt(),
1501
1541
  });
1502
1542
  } catch (e) {
1503
1543
  console.warn('[scheduler] failed to persist scheduler state', e?.message);
@@ -1510,18 +1550,279 @@ function appendHeartbeat(entry) {
1510
1550
  try {
1511
1551
  const line = JSON.stringify(entry) + '\n';
1512
1552
  let size = 0;
1513
- try { size = fs.statSync(HEARTBEAT_PATH).size; } catch { /* new file */ }
1553
+ try { size = fs.statSync(schedulerPaths.heartbeatPath()).size; } catch { /* new file */ }
1514
1554
  if (size >= HEARTBEAT_MAX_BYTES) {
1515
- const rotated = HEARTBEAT_PATH + '.1';
1555
+ const rotated = schedulerPaths.heartbeatPath() + '.1';
1516
1556
  try { fs.unlinkSync(rotated); } catch { /* */ }
1517
- try { fs.renameSync(HEARTBEAT_PATH, rotated); } catch { /* */ }
1557
+ try { fs.renameSync(schedulerPaths.heartbeatPath(), rotated); } catch { /* */ }
1518
1558
  }
1519
- fs.appendFileSync(HEARTBEAT_PATH, line);
1559
+ fs.appendFileSync(schedulerPaths.heartbeatPath(), line);
1520
1560
  } catch (e) {
1521
1561
  console.warn('[scheduler] heartbeat write failed', e?.message);
1522
1562
  }
1523
1563
  }
1524
1564
 
1565
+ // Build identity stamped on every heartbeat line — memoized at boot
1566
+ // (SCHEDULER_BUILD_IDENTITY), so a tick costs no git or fs work.
1567
+ function heartbeatBuild() {
1568
+ const { version, codeSha, builtAt } = SCHEDULER_BUILD_IDENTITY;
1569
+ return { version, codeSha, builtAt };
1570
+ }
1571
+
1572
+ // ---------- upgrade drain driver (lib/upgradeDrain.cjs) ----------
1573
+
1574
+ // Set by index.cjs: performs the actual app teardown + relaunch + exit.
1575
+ let restartHandler = null;
1576
+ function setRestartHandler(fn) { restartHandler = typeof fn === 'function' ? fn : null; }
1577
+ let drainDriving = false;
1578
+
1579
+ function drainSnapshot(jobs) {
1580
+ const running = new Set();
1581
+ let investigating = 0;
1582
+ for (const j of jobs ?? []) {
1583
+ if (j?.status === 'running') running.add(j.slug);
1584
+ else if (j?.status === 'investigating') investigating++;
1585
+ }
1586
+ for (const slug of runningSet) running.add(slug);
1587
+ // Deferred investigations are not busy: while draining they never spawn.
1588
+ return { running: running.size, investigating: Math.max(investigating, runtimeState.investigationCount()) };
1589
+ }
1590
+
1591
+ /**
1592
+ * Restart is triggered automatically ONLY when the installed build-info.json
1593
+ * differs from the running codeSha (an install/update already happened) —
1594
+ * never by polling npm. SM_AUTO_UPGRADE_RESTART=0 disables it.
1595
+ */
1596
+ function maybeAutoRequestRestart() {
1597
+ if (process.env.SM_AUTO_UPGRADE_RESTART === '0' || process.env.SM_DEV === '1') return null;
1598
+ const installed = readInstalledBuildInfo();
1599
+ const installedSha = typeof installed?.gitShortSha === 'string' ? installed.gitShortSha : null;
1600
+ if (!upgradeDrain.installedBuildDiffers({ running: SCHEDULER_CODE_SHA, installed: installedSha })) return null;
1601
+ return upgradeDrain.requestRestart({ reason: `installed build ${installedSha} differs from running ${SCHEDULER_CODE_SHA}`, requestedBy: 'auto-upgrade' });
1602
+ }
1603
+
1604
+ async function driveUpgradeDrain(state) {
1605
+ if (drainDriving) return;
1606
+ drainDriving = true;
1607
+ try {
1608
+ let request = upgradeDrain.readRestartRequest();
1609
+ if (!request && !state.drain?.active) request = maybeAutoRequestRestart();
1610
+ const { action, reason } = upgradeDrain.evaluateDrain({
1611
+ request,
1612
+ queueSnapshot: drainSnapshot(state.jobs),
1613
+ drainState: state.drain,
1614
+ now: Date.now(),
1615
+ });
1616
+ if (action === 'none' || action === 'wait') { drainActive = Boolean(state.drain?.active); return; }
1617
+ if (action === 'pause') {
1618
+ await mutate((s) => { s.drain = { active: true, since: new Date().toISOString(), requestedAt: request.requestedAt }; });
1619
+ drainActive = true;
1620
+ appendAuditEvent('upgrade_drain_started', { reason: request.reason, requestedBy: request.requestedBy });
1621
+ await broadcast({ flush: true });
1622
+ return;
1623
+ }
1624
+ if (action === 'abort') {
1625
+ upgradeDrain.retireRestartRequest();
1626
+ await mutate((s) => { s.drain = null; });
1627
+ drainActive = false;
1628
+ appendAuditEvent('upgrade_drain_aborted', { reason });
1629
+ await broadcast({ flush: true });
1630
+ runDueJobs().catch(() => {});
1631
+ return;
1632
+ }
1633
+ // 'restart': the FINAL zero-busy check runs inside a mutate, immediately
1634
+ // before exit — the snapshot above may be stale by now.
1635
+ let go = false;
1636
+ await mutate((s) => {
1637
+ const snap = drainSnapshot(s.jobs);
1638
+ if (!s.drain?.active || snap.running + snap.investigating > 0) return;
1639
+ upgradeDrain.stampDrainCompleted();
1640
+ go = true;
1641
+ });
1642
+ if (!go) return;
1643
+ appendAuditEvent('upgrade_drain_restart', { reason: request.reason, requestedBy: request.requestedBy });
1644
+ try {
1645
+ if (!restartHandler) throw new Error('no restart handler registered');
1646
+ upgradeDrain.markRestarting();
1647
+ await restartHandler(request);
1648
+ // Only reached when the handler did NOT exit the process (dev-server
1649
+ // in-place reboot): the restart is done, so retire the drain here.
1650
+ upgradeDrain.clearRestartingMarker();
1651
+ upgradeDrain.retireRestartRequest();
1652
+ await mutate((s) => { s.drain = null; });
1653
+ drainActive = false;
1654
+ } catch (e) {
1655
+ // Never strand the queue drained-and-paused: fall back to an abort.
1656
+ console.error('[scheduler] drain restart failed — aborting drain:', e?.message ?? e);
1657
+ upgradeDrain.clearRestartingMarker();
1658
+ upgradeDrain.retireRestartRequest();
1659
+ await mutate((s) => { s.drain = null; });
1660
+ drainActive = false;
1661
+ runDueJobs().catch(() => {});
1662
+ }
1663
+ } finally {
1664
+ drainDriving = false;
1665
+ }
1666
+ }
1667
+
1668
+ /** Boot: clear a leftover drain whose request is complete (the restart happened) or gone. */
1669
+ async function clearStaleDrainAtBoot(boot) {
1670
+ const request = upgradeDrain.readRestartRequest();
1671
+ const action = upgradeDrain.bootDrainAction({ drainState: boot.drain, request });
1672
+ upgradeDrain.clearRestartingMarker();
1673
+ if (request?.drainCompletedAt) upgradeDrain.retireRestartRequest();
1674
+ if (action === 'clear') await mutate((s) => { s.drain = null; });
1675
+ drainActive = action === 'keep';
1676
+ }
1677
+
1678
+ /**
1679
+ * heartbeatTick(deps?) — one 60 s heartbeat interval body. Each subsystem
1680
+ * (queue read + starvation watchdog, stall detector, heartbeat write) runs in
1681
+ * its own try/catch so one throw can't silently skip the others. Any failure
1682
+ * makes the written line `degraded: true` + `errors`; watchdogHelpers'
1683
+ * heartbeatFresh() and health.cjs's readFreshHeartbeat() treat such a line as
1684
+ * NOT fresh, so a throw never disarms the external watchdog or fakes
1685
+ * utilization health never read.
1686
+ */
1687
+ function heartbeatTick(deps = {}) {
1688
+ const readQueue = deps.readQueueSync ?? readQueueSync;
1689
+ const errors = [];
1690
+ const guard = (subsystem, fn) => {
1691
+ try {
1692
+ return fn();
1693
+ } catch (e) {
1694
+ errors.push({ subsystem, error: e?.message ?? String(e) });
1695
+ console.error(`[scheduler] heartbeat subsystem "${subsystem}" failed`, e);
1696
+ return undefined;
1697
+ }
1698
+ };
1699
+
1700
+ const s = guard('queue-read-starvation-watchdog', () => {
1701
+ const q = readQueue();
1702
+ // NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
1703
+ // running, something must drive it. This is the only driver that does
1704
+ // not depend on the billing poll loop, a pause timer, or a completing
1705
+ // job to schedule the next tick — every one of which has failed at
1706
+ // least once. See classifyQueueStarvation.
1707
+ if (!q.unreadable) {
1708
+ runQueueStarvationWatchdog(q).catch((e) => console.error('[scheduler] starvation watchdog error', e));
1709
+ // Restart-request drain state machine rides this same 60 s interval —
1710
+ // no new driver.
1711
+ driveUpgradeDrain(q).catch((e) => console.error('[scheduler] upgrade drain error', e));
1712
+ }
1713
+ return q;
1714
+ });
1715
+
1716
+ let stall;
1717
+ if (s) {
1718
+ stall = guard('stall-detector', () => {
1719
+ const summary = computeStallSummary(s);
1720
+ // Per-project alerting (see computeStallSummary's header): a project
1721
+ // stalled while others are busy must still fire, and one project
1722
+ // recovering must not clear or suppress another's still-open episode —
1723
+ // that is exactly what a single module-level stallSince/stallToasted
1724
+ // flag masked before (the burrow-vs-others incident this PRD fixes).
1725
+ const now = Date.now();
1726
+ const stalledCwds = Object.keys(summary.byProject).filter((cwd) => summary.byProject[cwd].stalled);
1727
+ for (const cwd of [...stallSince.keys()]) {
1728
+ if (!stalledCwds.includes(cwd)) {
1729
+ stallSince.delete(cwd);
1730
+ stallToasted.delete(cwd);
1731
+ }
1732
+ }
1733
+ const toAlert = [];
1734
+ for (const cwd of stalledCwds) {
1735
+ if (!stallSince.has(cwd)) stallSince.set(cwd, now);
1736
+ if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
1737
+ stallToasted.set(cwd, true);
1738
+ toAlert.push(cwd);
1739
+ }
1740
+ }
1741
+ if (toAlert.length > 0) {
1742
+ console.error(
1743
+ `[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
1744
+ + `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
1745
+ summary.byProject,
1746
+ );
1747
+ appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: summary.total, byProject: summary.byProject });
1748
+ if (mainWindow && !mainWindow.isDestroyed()) {
1749
+ sendIfAlive(mainWindow, 'schedule:stall', {
1750
+ message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
1751
+ projects: toAlert,
1752
+ total: summary.total,
1753
+ byProject: summary.byProject,
1754
+ });
1755
+ }
1756
+ }
1757
+ return summary;
1758
+ });
1759
+ }
1760
+
1761
+ let entry = null;
1762
+ if (s && stall && errors.length === 0) {
1763
+ entry = guard('heartbeat-write', () => {
1764
+ // Initialise from the real status union (scheduleJobSchema.cjs) rather
1765
+ // than a hand-maintained subset — the old `{ pending, running, completed,
1766
+ // failed }` literal silently minted a NEW key for any other value, which
1767
+ // is how a heartbeat with a `queued: 2` bucket looked like "normal" 24h
1768
+ // visibility instead of the alarm it should have been. Any row whose
1769
+ // status isn't in JOB_STATUSES routes into `unknown`, never a
1770
+ // freshly-minted key.
1771
+ const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
1772
+ counts.unknown = 0;
1773
+ for (const j of s.jobs) {
1774
+ if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
1775
+ counts[j.status] += 1;
1776
+ } else {
1777
+ counts.unknown += 1;
1778
+ }
1779
+ }
1780
+ // Logical-liveness signal for the external watchdog (see watchdogHelpers
1781
+ // evaluateDispatchLiveness): computed once per heartbeat from the same
1782
+ // queue snapshot. pendingDispatchable = pending rows minus those
1783
+ // terminally blocked behind a failed/skipped dependency.
1784
+ const blockedPending = computeBlockedChains(s.jobs).reduce((n, c) => n + c.blocked, 0);
1785
+ const dispatch = {
1786
+ lastDispatchAttemptAt: s.lastDispatchAttemptAt ?? null,
1787
+ lastRunAt: s.lastRunAt ?? null,
1788
+ lastTickReason: lastTick?.reason ?? null,
1789
+ pendingDispatchable: Math.max(0, counts.pending - blockedPending),
1790
+ runningCount: counts.running,
1791
+ paused: Boolean(s.paused),
1792
+ drain: s.drain?.active ? { since: s.drain.since ?? null, requestedAt: s.drain.requestedAt ?? null } : null,
1793
+ };
1794
+ return {
1795
+ ts: Date.now(),
1796
+ pid: process.pid,
1797
+ build: heartbeatBuild(),
1798
+ counts,
1799
+ dispatch,
1800
+ stall: { stalled: stall.stalled, total: stall.total },
1801
+ paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
1802
+ quarantinedCwds: (s.unreadableCwds ?? []).map((u) => u.cwd),
1803
+ nextReset: cachedNextReset,
1804
+ utilization: cachedUtilization,
1805
+ consecutiveFailures,
1806
+ // State/consecutiveFailures/degraded-budget snapshot of the shared
1807
+ // usage-meter breaker, so a human reading only the heartbeat log can
1808
+ // see the meter's own health apart from the queue's.
1809
+ usageMeter: {
1810
+ state: billing.usageCircuit.state(),
1811
+ consecutiveFailures,
1812
+ degradedBudget: computeDegradedBudget(),
1813
+ },
1814
+ };
1815
+ });
1816
+ }
1817
+ if (!entry || errors.length > 0) {
1818
+ // Deliberately carries no utilization/counts: this line says "I ran but
1819
+ // could not read state", and consumers must not mistake it for a fresh read.
1820
+ entry = { ts: Date.now(), pid: process.pid, build: heartbeatBuild(), degraded: true, errors };
1821
+ }
1822
+ appendHeartbeat(entry);
1823
+ return entry;
1824
+ }
1825
+
1525
1826
  /**
1526
1827
  * computeStallSummary(state) → { stalled, total, running, pending, byProject }
1527
1828
  *
@@ -1564,7 +1865,12 @@ function computeStallSummary(state) {
1564
1865
  byProject[key].invalid = (byProject[key].invalid || 0) + 1;
1565
1866
  }
1566
1867
  const total = jobs.length + invalidJobs.length;
1567
- const stalled = total > 0 && running === 0 && pending === 0 && !state?.paused;
1868
+ // A drained queue (every row completed/skipped) is idle, not stalled — a
1869
+ // stall needs at least one parked problem row (failed/needs_review/
1870
+ // quarantined/invalid) that is waiting on someone.
1871
+ const isProblemStatus = (st) => st !== 'completed' && st !== 'skipped';
1872
+ const stalled = total > 0 && running === 0 && pending === 0 && !state?.paused
1873
+ && (invalidJobs.length > 0 || jobs.some((j) => isProblemStatus(j.status)));
1568
1874
  for (const key of Object.keys(byProject)) {
1569
1875
  const counts = byProject[key];
1570
1876
  const projRunning = counts.running || 0;
@@ -1572,7 +1878,9 @@ function computeStallSummary(state) {
1572
1878
  const projTotal = Object.keys(counts)
1573
1879
  .filter((k) => k !== 'stalled')
1574
1880
  .reduce((sum, k) => sum + counts[k], 0);
1575
- counts.stalled = projTotal > 0 && projRunning === 0 && projPending === 0 && !state?.paused;
1881
+ const projProblem = Object.keys(counts)
1882
+ .some((k) => k !== 'stalled' && k !== 'completed' && k !== 'skipped' && counts[k] > 0);
1883
+ counts.stalled = projTotal > 0 && projRunning === 0 && projPending === 0 && !state?.paused && projProblem;
1576
1884
  }
1577
1885
  return { stalled, total, running, pending, byProject };
1578
1886
  }
@@ -2015,6 +2323,22 @@ function findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
2015
2323
  // the .bak-* snapshots.
2016
2324
  const quarantinedPaths = new Set();
2017
2325
  function flagUnreadable(state) {
2326
+ // Per-shard quarantine (queueStore.unreadableCwds): snapshot each torn shard
2327
+ // once and name it, but never halt — other projects keep dispatching.
2328
+ for (const u of state.unreadableCwds ?? []) {
2329
+ if (!quarantinedPaths.has(u.file)) {
2330
+ quarantinedPaths.add(u.file);
2331
+ try {
2332
+ fs.copyFileSync(u.file, `${u.file}.corrupt-${Date.now()}`);
2333
+ } catch { /* best-effort: the read already failed, the copy may too */ }
2334
+ console.error(`[scheduler] project queue shard quarantined (${u.cwd}): ${u.error}`);
2335
+ logs.writeLine({
2336
+ level: 'error', scope: 'scheduler',
2337
+ message: `project queue shard unreadable — ${u.cwd} is quarantined, other projects keep dispatching`,
2338
+ meta: { cwd: u.cwd, path: u.file, error: u.error },
2339
+ });
2340
+ }
2341
+ }
2018
2342
  if (!state.unreadable) return state;
2019
2343
  if (state.unreadablePath && !quarantinedPaths.has(state.unreadablePath)) {
2020
2344
  quarantinedPaths.add(state.unreadablePath);
@@ -2073,8 +2397,35 @@ async function writeQueue(state) {
2073
2397
  // preceding mutate threw, so the chain never deadlocks.
2074
2398
  let mutateTail = Promise.resolve();
2075
2399
 
2400
+ // Observe-only watchdog: a mutate body over MUTATE_WATCHDOG_MS is logged and
2401
+ // audited once per episode (latched until a mutate completes). mutateTail is
2402
+ // deliberately NEVER reset — it is what enforces the single-writer law, and
2403
+ // abandoning a live writer would let two read-modify-writes interleave.
2404
+ const MUTATE_WATCHDOG_MS = 60_000;
2405
+ let mutateWedgeLatched = false;
2406
+
2076
2407
  function mutate(fn) {
2077
2408
  const next = mutateTail.then(async () => {
2409
+ const wedgeTimer = setTimeout(() => {
2410
+ if (mutateWedgeLatched) return;
2411
+ mutateWedgeLatched = true;
2412
+ console.warn(`[scheduler] MUTATE WEDGED: a queue mutation has run > ${MUTATE_WATCHDOG_MS}ms`);
2413
+ appendAuditEvent('mutate_wedged', { budgetMs: MUTATE_WATCHDOG_MS });
2414
+ }, MUTATE_WATCHDOG_MS);
2415
+ if (typeof wedgeTimer.unref === 'function') wedgeTimer.unref();
2416
+ try {
2417
+ return await mutateBody(fn);
2418
+ } finally {
2419
+ clearTimeout(wedgeTimer);
2420
+ mutateWedgeLatched = false;
2421
+ }
2422
+ });
2423
+ mutateTail = next.catch(() => {}); // keep chain alive on errors
2424
+ return next;
2425
+ }
2426
+
2427
+ async function mutateBody(fn) {
2428
+ {
2078
2429
  const state = await readQueue();
2079
2430
  // Bail BEFORE fn runs: a mutator handed an unreadable (therefore empty)
2080
2431
  // state would compute its result from a queue that isn't there, and
@@ -2120,9 +2471,7 @@ function mutate(fn) {
2120
2471
  }
2121
2472
  await writeQueue(state);
2122
2473
  return ret;
2123
- });
2124
- mutateTail = next.catch(() => {}); // keep chain alive on errors
2125
- return next;
2474
+ }
2126
2475
  }
2127
2476
 
2128
2477
  // ---------- PRD parsing ----------
@@ -2142,11 +2491,23 @@ const parsePrd = prdParser.parsePrd;
2142
2491
  // one — acceptable: PRD counts per project are bounded (hundreds, not
2143
2492
  // millions), and correctness across multiple project dirs matters more than
2144
2493
  // preserving the single-dir cache's steady-state zero-read fast path.
2145
- async function listPrdFiles() {
2494
+ async function listPrdFiles(skipCwds) {
2146
2495
  ensureDirs();
2147
2496
  const dirs = candidatePrdsDirs();
2148
2497
  const perDir = await Promise.all(dirs.map((dir) => prdParser.listPrdFiles(dir)));
2149
- return { files: perDir.flat().sort(), dirCount: dirs.length };
2498
+ let files = perDir.flat();
2499
+ // A quarantined project has no job rows this pass; scanning its PRDs would
2500
+ // mint fresh `pending` rows for work that may already be running.
2501
+ if (skipCwds && skipCwds.size > 0) {
2502
+ const prefixes = [...skipCwds].map((c) => c + path.sep);
2503
+ files = files.filter((f) => !prefixes.some((p) => f.startsWith(p)));
2504
+ }
2505
+ return { files: files.sort(), dirCount: dirs.length };
2506
+ }
2507
+
2508
+ /** Set of cwds whose shard is quarantined in this merged read. */
2509
+ function quarantinedCwdSet(state) {
2510
+ return new Set((state?.unreadableCwds ?? []).map((u) => u.cwd));
2150
2511
  }
2151
2512
 
2152
2513
  /**
@@ -2186,7 +2547,22 @@ async function allocateParallelGroup(cwd) {
2186
2547
  * Safety:
2187
2548
  * - PID-recycling: between app death and this call, another process may have
2188
2549
  * reused the PID. We read /proc/<pid>/cmdline (Linux) or `ps -p` (macOS)
2189
- * and only SIGTERM if the cmdline starts with the claude bin path.
2550
+ * and only SIGTERM if the cmdline matches /\bclaude\b/. Since procName
2551
+ * aliasing, cmdline[0] is the alias path (`.../procnames/sm-claude-job`) or
2552
+ * the smArgv0 label (`sm-claude-job:<slug>`), NOT the claude bin path —
2553
+ * both still contain the word `claude`. The macOS `ps -p <pid> -o command=`
2554
+ * branch has the same exposure and the same guarantee (ps shows argv0).
2555
+ * Migration: cmdline is fixed at exec, and no claude procIdentity is
2556
+ * persisted (job.runtime carries none), so a pre-aliasing process recorded
2557
+ * and compared after upgrade still compares equal to itself — no
2558
+ * tolerance needed; unaliased legacy cmdlines also match /\bclaude\b/.
2559
+ * - recordedIdentity (optional): when the caller has a COMPLETE prior
2560
+ * procIdentity for this pid (startTicks + cmdline), it is used only as a
2561
+ * VETO — if it provably differs from the pid's live identity right now,
2562
+ * the pid was recycled and the kill is refused ('mismatch') even before
2563
+ * the cmdline heuristic below runs. No recorded identity (today's only
2564
+ * case — job.runtime carries no identity yet) falls through unchanged to
2565
+ * the existing /\bclaude\b/ + `ps -p` heuristics.
2190
2566
  * - Detached process group: jobs are spawned with detached:true so we kill
2191
2567
  * -pid (the group). If the group leader is already gone, that fails
2192
2568
  * silently and we fall back to single-pid kill.
@@ -2194,12 +2570,17 @@ async function allocateParallelGroup(cwd) {
2194
2570
  * scheduled via setTimeout to clean up any process ignoring SIGTERM.
2195
2571
  *
2196
2572
  * Returns: 'killed' (cmdline matched + signal sent), 'gone' (pid not alive),
2197
- * 'mismatch' (pid alive but cmdline doesn't look like claude),
2573
+ * 'mismatch' (pid alive but cmdline doesn't look like claude, or a
2574
+ * complete recorded identity proves the pid was recycled),
2198
2575
  * 'unknown' (couldn't read cmdline — leave the pid alone).
2199
2576
  */
2200
- function killOrphanClaudePid(pid) {
2577
+ function killOrphanClaudePid(pid, recordedIdentity = null) {
2201
2578
  if (!pid || typeof pid !== 'number' || pid <= 1) return 'gone';
2202
2579
  try { process.kill(pid, 0); } catch { return 'gone'; }
2580
+ if (recordedIdentity && recordedIdentity.complete
2581
+ && isDifferentProcess(recordedIdentity, procIdentityOf(pid))) {
2582
+ return 'mismatch';
2583
+ }
2203
2584
  let cmdline = '';
2204
2585
  try {
2205
2586
  cmdline = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8').replace(/\0/g, ' ');
@@ -2307,25 +2688,48 @@ async function reconcile(state) {
2307
2688
  // has no queue row yet, so it is never "live" and gets archived here
2308
2689
  // instead of ever reaching the onDisk scan below.
2309
2690
  let phaseStartMs = Date.now();
2310
- await consolidateAllFlatPrds(allProjectCwds());
2691
+ const skipCwds = quarantinedCwdSet(state);
2692
+ await consolidateAllFlatPrds(allProjectCwds(), skipCwds);
2311
2693
  phaseMs.flatPrdSweep = Date.now() - phaseStartMs;
2312
2694
 
2313
2695
  phaseStartMs = Date.now();
2314
- const { files, dirCount } = await listPrdFiles();
2696
+ const { files, dirCount } = await listPrdFiles(skipCwds);
2315
2697
  phaseMs.prdDirResolve = Date.now() - phaseStartMs;
2316
2698
 
2317
2699
  phaseStartMs = Date.now();
2318
2700
  const onDisk = new Map();
2701
+ // Slugs derive from title text with no cwd salt, so two different projects
2702
+ // can legitimately queue an identically-slugged PRD — onDisk alone can
2703
+ // only hold ONE parsed PRD per slug (last-file-wins), which would silently
2704
+ // hand an EXISTING row the wrong project's PRD (or none at all) when two
2705
+ // projects collide on a slug. This side index lets the two existing-row
2706
+ // lookups below (job refresh + invalid-row repair) disambiguate by the
2707
+ // row's own cwd first; the fresh-discovery loop further down still reads
2708
+ // the bare `onDisk` (unscoped) since a same-slug NEW-PRD collision across
2709
+ // two projects is a rarer edge this reconcile pass doesn't yet resolve.
2710
+ const onDiskByCwd = new Map();
2319
2711
  for (const f of files) {
2320
2712
  try {
2321
2713
  // Per-file await: parsing is mtime-cached so steady-state hits zero
2322
2714
  // disk reads; on cold cache the awaits keep the main thread responsive.
2323
2715
  const p = await parsePrd(f);
2324
2716
  onDisk.set(p.slug, p);
2717
+ if (p.cwd) onDiskByCwd.set(`${p.slug}::${p.cwd}`, p);
2325
2718
  } catch (e) {
2326
2719
  console.warn('[scheduler] failed to parse', f, e?.message);
2327
2720
  }
2328
2721
  }
2722
+ // resolvePrdForJob(slug, cwd) — cwd-scoped PRD lookup for an EXISTING
2723
+ // queue row, falling back to the unscoped onDisk entry when this exact
2724
+ // (slug, cwd) pair has no PRD (e.g. cwd is null/stale) — same behavior as
2725
+ // a bare onDisk.get() for every slug that isn't cross-project-colliding.
2726
+ function resolvePrdForJob(slug, cwd) {
2727
+ if (cwd) {
2728
+ const scoped = onDiskByCwd.get(`${slug}::${cwd}`);
2729
+ if (scoped) return scoped;
2730
+ }
2731
+ return onDisk.get(slug);
2732
+ }
2329
2733
  phaseMs.parseLoop = Date.now() - phaseStartMs;
2330
2734
 
2331
2735
  const next = [];
@@ -2345,7 +2749,7 @@ async function reconcile(state) {
2345
2749
  // historyTerminalBySlug() below and backfilled before being dropped.
2346
2750
  const terminalDroppedNeedingHistoryCheck = [];
2347
2751
  for (const job of state.jobs) {
2348
- const p = onDisk.get(job.slug);
2752
+ const p = resolvePrdForJob(job.slug, job.cwd);
2349
2753
  if (!p) {
2350
2754
  // A terminal job whose .md is gone was archived on purpose — dropping
2351
2755
  // its row is the intended end of the auto-archive flow, PROVIDED it's
@@ -2379,10 +2783,24 @@ async function reconcile(state) {
2379
2783
  continue;
2380
2784
  }
2381
2785
  seen.add(job.slug);
2786
+ // p.cwd REFINES the row's existing cwd; it never erases one. A PRD
2787
+ // file with no `cwd:` frontmatter parses p.cwd as undefined — falling
2788
+ // through to a bare `cwd: p.cwd` here nulled the row's real cwd,
2789
+ // which queueStore.writeSplit then buckets into
2790
+ // schedulerBatch.js's DEFAULT_PROJECT_CWD, silently relocating the
2791
+ // row into the WRONG project's queue.json shard and emptying the
2792
+ // owning project's shard underneath it (2026-09 data-loss incident).
2793
+ // Resolved ONCE into a local so originSessionId's fallback below
2794
+ // resolves against the SAME cwd this row actually gets, not the raw
2795
+ // (possibly undefined) p.cwd — resolveOriginSessionId(undefined, ...)
2796
+ // returns null unconditionally, which silently dropped the origin link
2797
+ // for every PRD with no `cwd:` frontmatter even though a good cwd was
2798
+ // available one line below.
2799
+ const refreshedCwd = p.cwd ?? job.cwd ?? null;
2382
2800
  const updatedJob = {
2383
2801
  ...job,
2384
2802
  title: p.title,
2385
- cwd: p.cwd,
2803
+ cwd: refreshedCwd,
2386
2804
  parallelGroup: p.parallelGroup,
2387
2805
  estimateMinutes: p.estimateMinutes,
2388
2806
  sourcePromptId: reconcileSourcePromptId(job, p.sourcePromptId),
@@ -2396,7 +2814,7 @@ async function reconcile(state) {
2396
2814
  quietMachine: p.quietMachine === true,
2397
2815
  budgetExempt: p.budgetExempt === true,
2398
2816
  originSessionId: job.originSessionId
2399
- ?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
2817
+ ?? resolveOriginSessionId(refreshedCwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
2400
2818
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
2401
2819
  agentType: p.agentType ?? job.agentType ?? null,
2402
2820
  };
@@ -2479,7 +2897,7 @@ async function reconcile(state) {
2479
2897
  for (const inv of invalidJobs) {
2480
2898
  if (seen.has(inv.slug)) continue; // a valid row for this slug already exists
2481
2899
  const oldStatus = inv.row?.status;
2482
- const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir: RUNS_DIR });
2900
+ const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir: schedulerPaths.runsDir() });
2483
2901
  if (hist) {
2484
2902
  // Never resurrect: this slug already has a durable terminal record
2485
2903
  // elsewhere (history.jsonl or a run sidecar) — repairing its corrupted
@@ -2492,17 +2910,22 @@ async function reconcile(state) {
2492
2910
  });
2493
2911
  continue;
2494
2912
  }
2495
- const p = onDisk.get(inv.slug);
2913
+ const p = resolvePrdForJob(inv.slug, inv.row?.cwd);
2496
2914
  if (!p) {
2497
2915
  // PRD file also gone with no terminal record anywhere — nothing to
2498
2916
  // repair against. queueStore already logged the quarantine once.
2499
2917
  continue;
2500
2918
  }
2919
+ // Same cwd-refines-not-erases rule as the normal refresh path above, and
2920
+ // same reason for resolving it once into a local: originSessionId's
2921
+ // fallback must resolve against the cwd this row actually gets, not the
2922
+ // raw (possibly undefined) p.cwd.
2923
+ const repairedCwd = p.cwd ?? inv.row?.cwd ?? null;
2501
2924
  const job = {
2502
2925
  ...inv.row,
2503
2926
  slug: inv.slug,
2504
2927
  title: p.title,
2505
- cwd: p.cwd,
2928
+ cwd: repairedCwd,
2506
2929
  parallelGroup: p.parallelGroup,
2507
2930
  estimateMinutes: p.estimateMinutes,
2508
2931
  sourcePromptId: p.sourcePromptId ?? inv.row?.sourcePromptId ?? null,
@@ -2512,7 +2935,7 @@ async function reconcile(state) {
2512
2935
  disposition: p.disposition ?? null,
2513
2936
  quietMachine: p.quietMachine === true,
2514
2937
  budgetExempt: p.budgetExempt === true,
2515
- originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
2938
+ originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(repairedCwd, p.epicId ?? p.sourcePromptId),
2516
2939
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
2517
2940
  agentType: p.agentType ?? inv.row?.agentType ?? null,
2518
2941
  };
@@ -2616,17 +3039,27 @@ async function reconcile(state) {
2616
3039
  // guard above inert. Fall back to reading the slug's own newest run
2617
3040
  // sidecars straight off disk — same "don't resurrect an already-terminal
2618
3041
  // slug" intent, independent of history.jsonl's existence.
2619
- const fallback = latestTerminalOutcomeForSlug(slug, { runsDir: RUNS_DIR });
3042
+ const fallback = latestTerminalOutcomeForSlug(slug, { runsDir: schedulerPaths.runsDir() });
2620
3043
  if (fallback) {
2621
3044
  if (fallback.status === 'completed') {
2622
3045
  historyArchiveCandidates.push({ slug, status: fallback.status, finishedAt: fallback.finishedAt });
2623
3046
  }
2624
3047
  continue;
2625
3048
  }
3049
+ // No prior row exists to fall back to (this is a fresh discovery), so a
3050
+ // PRD file with no `cwd:` frontmatter falls back to the project root it
3051
+ // was actually found under (derived from its own file path) rather than
3052
+ // nulling out to schedulerBatch.js's DEFAULT_PROJECT_CWD. Resolved once
3053
+ // so originSessionId (below) resolves against this SAME cwd — passing
3054
+ // the raw p.cwd there instead would resolve against `undefined` for
3055
+ // exactly the no-frontmatter case this fallback exists to handle, since
3056
+ // resolveOriginSessionId(cwd, ...) returns null unconditionally when
3057
+ // `cwd` is falsy.
3058
+ const discoveredCwd = p.cwd ?? deriveProjectCwdFromPrdPath(p.path) ?? null;
2626
3059
  const entry = {
2627
3060
  slug,
2628
3061
  title: p.title,
2629
- cwd: p.cwd,
3062
+ cwd: discoveredCwd,
2630
3063
  parallelGroup: p.parallelGroup,
2631
3064
  estimateMinutes: p.estimateMinutes,
2632
3065
  sourcePromptId: p.sourcePromptId,
@@ -2636,7 +3069,7 @@ async function reconcile(state) {
2636
3069
  disposition: p.disposition ?? null,
2637
3070
  quietMachine: p.quietMachine === true,
2638
3071
  budgetExempt: p.budgetExempt === true,
2639
- originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
3072
+ originSessionId: resolveOriginSessionId(discoveredCwd, p.epicId ?? p.sourcePromptId),
2640
3073
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
2641
3074
  agentType: p.agentType ?? null,
2642
3075
  status: 'pending',
@@ -2787,14 +3220,73 @@ async function reconcile(state) {
2787
3220
  // ---------- next-reset detection ----------
2788
3221
 
2789
3222
  let cachedNextReset = null; // bare ISO string or null
2790
- let cachedUtilization = null; // five_hour utilization %, 0–100, or null if unknown
3223
+ let cachedUtilization = null; // binding-window utilization %, 0–100, or null if unknown
3224
+ // ms timestamp of the last FRESH reset observation (see recordObservedReset)
3225
+ // — distinct from Date.now(), so persistSchedulerState never re-stamps a
3226
+ // stale cachedNextReset as "just observed" on every poll cycle.
3227
+ let lastResetObservedAtMs = null;
3228
+ // Full usage payload (`{ five_hour, limits?, ... }`) from the last SUCCESSFUL
3229
+ // poll — the input degradedBudget() carries forward while the meter is down.
3230
+ // Never itself defaults to 0; absent (null) reads as "no signal yet" and
3231
+ // degradedBudget() treats that conservatively (100% / capped concurrency).
3232
+ let lastGoodUsagePayload = null;
3233
+ // Non-null while the meter is degraded (circuit open, or a poll otherwise
3234
+ // failed to return 'ok') — narrows tickQueue's freeSlots as a picker-side
3235
+ // hold instead of a second slot pool (see tickQueue's freeSlots computation).
3236
+ // Cleared to null the moment a poll succeeds or the meter is inapplicable.
3237
+ let degradedConcurrencyCapValue = null;
3238
+ // Deliberately a named constant, not a bare zero literal assigned straight
3239
+ // into cachedUtilization: a genuine "no consumer meter to poll" (enterprise
3240
+ // auth) is categorically different from "the meter is down and we don't
3241
+ // know" — the latter must never read as 0%/full-speed-ahead.
3242
+ const NO_METER_UTILIZATION = 0;
3243
+
3244
+ /**
3245
+ * Records a freshly-observed reset, stamping lastResetObservedAtMs only when
3246
+ * there actually WAS a reset to observe (never on every poll regardless of
3247
+ * payload content — see persistSchedulerState's header).
3248
+ */
3249
+ function recordObservedReset(resetIso) {
3250
+ if (resetIso) {
3251
+ cachedNextReset = resetIso;
3252
+ lastResetObservedAtMs = Date.now();
3253
+ } else {
3254
+ cachedNextReset = null;
3255
+ }
3256
+ }
3257
+
3258
+ /** Pure: this poll/executor cycle's conservative budget while the meter is degraded. */
3259
+ function computeDegradedBudget() {
3260
+ return degradedBudget(lastGoodUsagePayload, {
3261
+ observed429: lastFailureKind === 'meter_rate_limited',
3262
+ resetsAt: cachedNextReset,
3263
+ now: Date.now(),
3264
+ configuredCap: sessionSlots.snapshot().total,
3265
+ });
3266
+ }
3267
+
3268
+ /**
3269
+ * Computes AND applies this cycle's degraded budget to cachedUtilization /
3270
+ * degradedConcurrencyCapValue in one call — every "meter down" branch in
3271
+ * pollLoop/runQueueStarvationWatchdog needs the exact same
3272
+ * compute-then-assign-both-fields pair, so it lives once here rather than
3273
+ * being copy-pasted at each call site (where a future change to how the
3274
+ * budget applies would otherwise have to be made N times).
3275
+ */
3276
+ function applyDegradedBudget() {
3277
+ const budget = computeDegradedBudget();
3278
+ cachedUtilization = budget.utilization;
3279
+ degradedConcurrencyCapValue = budget.concurrencyCap;
3280
+ return budget;
3281
+ }
2791
3282
 
2792
3283
  /** Fetches latest usage from billing API. Throws on any error — callers handle it. */
2793
3284
  async function refreshNextReset() {
2794
3285
  const r = await billing.fetchUsage();
2795
3286
  if (r.kind !== 'ok') throw new Error(`usage fetch failed (${r.kind}): ${r.message ?? ''}`);
2796
- cachedNextReset = r.data?.usage?.five_hour?.resets_at ?? null;
2797
- cachedUtilization = r.data?.usage?.five_hour?.utilization ?? cachedUtilization;
3287
+ const window = bindingWindow(r.data?.usage);
3288
+ recordObservedReset(window.resets_at ?? null);
3289
+ cachedUtilization = Number.isFinite(window.utilization) ? window.utilization : cachedUtilization;
2798
3290
  return cachedNextReset;
2799
3291
  }
2800
3292
 
@@ -2805,15 +3297,38 @@ function getNextResetCached() {
2805
3297
  /**
2806
3298
  * Pure: picks the reset to pause against for a rate-limited run (PRD 1118).
2807
3299
  * Prefers the BINDING window read off the run's own log — refreshNextReset()
2808
- * only ever reports five_hour, which is the wrong clock when a
2809
- * seven_day/seven_day_overage_included window is what actually 429'd
2810
- * (five_hour can read 0% utilization at the very same moment). Falls back
2811
- * to the billing-endpoint-derived reset only when the log yields nothing.
2812
- */
2813
- function resolveRateLimitPauseReset(logPath, billingResetIso) {
3300
+ * only ever reports the binding window at POLL time, which can be a
3301
+ * different clock than what actually 429'd this run (five_hour can read 0%
3302
+ * utilization at the very same moment a seven_day window binds). Falls back
3303
+ * to the billing-endpoint-derived reset only when the log yields nothing —
3304
+ * and rejects EITHER source when it has already passed relative to `nowMs`
3305
+ * (usageCircuit.isResetFresh): a stale reset must never arm a resume timer
3306
+ * that already elapsed (that's the 30-second-nap bug computeEffectiveResumeAt
3307
+ * below also guards against), so a stale value here returns null and lets
3308
+ * the caller's 30-minute fallback take over instead.
3309
+ */
3310
+ function resolveRateLimitPauseReset(logPath, billingResetIso, nowMs = Date.now()) {
2814
3311
  const logReset = resolveBindingRateLimitReset(logPath);
2815
- if (logReset != null) return new Date(logReset * 1000).toISOString();
2816
- return billingResetIso ?? null;
3312
+ if (logReset != null) {
3313
+ const iso = new Date(logReset * 1000).toISOString();
3314
+ if (isResetFresh(iso, nowMs)) return iso;
3315
+ }
3316
+ if (billingResetIso && isResetFresh(billingResetIso, nowMs)) return billingResetIso;
3317
+ return null;
3318
+ }
3319
+
3320
+ /**
3321
+ * Shared by both rate-limit pause sites (spawnJob's rateLimited branch and
3322
+ * reapDeadRunningJobs): the billing-endpoint-derived fallback reset used
3323
+ * when the run's own log yields no binding window. While the shared
3324
+ * usageCircuit is OPEN, skip calling refreshNextReset() — it would just be
3325
+ * another request against the endpoint the breaker just decided is down —
3326
+ * and fall back to whatever was last cached instead (resolveRateLimitPauseReset
3327
+ * itself still rejects that cached value if it has since gone stale).
3328
+ */
3329
+ async function billingResetForPause() {
3330
+ if (billing.usageCircuit.state() === 'open') return cachedNextReset;
3331
+ return refreshNextReset().catch(() => cachedNextReset);
2817
3332
  }
2818
3333
 
2819
3334
  // ---------- health / poll state ----------
@@ -2828,10 +3343,27 @@ let firstFailureAt = null;
2828
3343
  let firstNon429FailureAt = null; // tracks only transient/config failures; 429s don't count toward network-pause threshold
2829
3344
  let lastFailureKind = null; // 'transient' | 'meter_rate_limited' | 'auth' | null
2830
3345
  let pauseClearedManuallyAt = null;
3346
+ // In-memory only (no new persisted field): when clearPause last actually lifted a pause — a
3347
+ // legitimate restart point of the dispatch-idleness clock (dispatchIdleMs).
3348
+ let lastPauseClearedAt = null;
2831
3349
  // PRD: the usage-poller silent-failure-streak WARN is emitted once per streak,
2832
3350
  // not once per failure (57 failures must produce ONE opsErrorLog line, not 57).
2833
3351
  // Reset alongside consecutiveFailures everywhere that resets to 0.
2834
3352
  let failureStreakWarned = false;
3353
+ // ms timestamp the initial WARN fired this streak — anchors the periodic
3354
+ // escalation cadence below. Reset to null alongside failureStreakWarned.
3355
+ let failureStreakWarnedAt = null;
3356
+ // ms timestamp of the last periodic escalation (audit event + opsErrorLog
3357
+ // line) this streak. Reset to null alongside failureStreakWarned so a LATER
3358
+ // streak re-arms both the initial WARN and the escalation cadence.
3359
+ let lastEscalationAtMs = null;
3360
+
3361
+ /** The failure-streak-WARN trio must always reset together — one helper, not 3 copies. */
3362
+ function resetFailureStreak() {
3363
+ failureStreakWarned = false;
3364
+ failureStreakWarnedAt = null;
3365
+ lastEscalationAtMs = null;
3366
+ }
2835
3367
  // Ceiling on pollLoop's exponential poll backoff (both the 'transient'/'config'
2836
3368
  // branch and the 'meter_rate_limited' branch below share this cap — a single
2837
3369
  // constant so the two never drift to different ceilings).
@@ -2840,6 +3372,11 @@ const BACKOFF_MAX_MS = 480_000; // 8 minutes
2840
3372
  // jitter and becomes worth a human's attention. health.cjs imports this so the
2841
3373
  // WARN and the `npm run health` non-GREEN trip at the exact same count.
2842
3374
  const FAILURE_STREAK_WARN_THRESHOLD = 5;
3375
+ // How often a PERSISTING failure streak re-escalates (audit event +
3376
+ // opsErrorLog line) after the initial WARN, and the health.cjs YELLOW->RED
3377
+ // threshold for how long the usageCircuit has been open — one constant so
3378
+ // the log cadence and the health-color flip never drift apart.
3379
+ const FAILURE_STREAK_ESCALATION_MS = 30 * 60_000; // 30 minutes
2843
3380
 
2844
3381
  /** Pure: exponential backoff with a cap, shared by every pollLoop failure branch. Exported for unit testing. */
2845
3382
  function nextBackoffMs(prevBackoffMs) {
@@ -2856,19 +3393,59 @@ function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold =
2856
3393
  return consecutiveFailures >= threshold && !alreadyWarned;
2857
3394
  }
2858
3395
 
2859
- /** Emits the one-time opsErrorLog WARN for a failure streak crossing the threshold, if not already warned this streak. */
3396
+ /**
3397
+ * Pure: does a PERSISTING failure streak warrant another escalation (audit
3398
+ * event + opsErrorLog line) at `nowMs`? Exported for unit testing. Only
3399
+ * relevant once the streak has already crossed `warnThreshold` (the initial
3400
+ * WARN); `lastEscalatedAtMs` null means no escalation has fired yet this
3401
+ * streak, so the first one is due immediately. Re-arms automatically once a
3402
+ * streak clears (the caller resets `lastEscalatedAtMs` to null alongside
3403
+ * `failureStreakWarned`), so a later independent streak escalates again.
3404
+ */
3405
+ function shouldEscalateFailureStreak(consecutiveFailures, lastEscalatedAtMs, nowMs, thresholdMs = FAILURE_STREAK_ESCALATION_MS, warnThreshold = FAILURE_STREAK_WARN_THRESHOLD) {
3406
+ if (consecutiveFailures < warnThreshold) return false;
3407
+ if (!lastEscalatedAtMs) return true;
3408
+ return nowMs - lastEscalatedAtMs >= thresholdMs;
3409
+ }
3410
+
3411
+ /**
3412
+ * Emits the one-time opsErrorLog WARN the moment a failure streak crosses
3413
+ * the threshold, then — while that streak PERSISTS — re-escalates (audit
3414
+ * event + another opsErrorLog line) every FAILURE_STREAK_ESCALATION_MS so a
3415
+ * human watching only the audit log still sees a live incident, not just the
3416
+ * single opening WARN from hours ago.
3417
+ */
2860
3418
  function warnFailureStreakIfNeeded() {
2861
- if (!shouldWarnFailureStreak(consecutiveFailures, failureStreakWarned)) return;
2862
- failureStreakWarned = true;
2863
- try {
2864
- appendError({
2865
- cwd: DEFAULT_PROJECT_CWD,
2866
- scope: 'scheduler',
2867
- level: 'warn',
2868
- message: `usage/rate-limit poller has failed ${consecutiveFailures} consecutive times (kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${SCHEDULER_STATE_PATH}`,
2869
- meta: { consecutiveFailures, backoffMs, lastFailureKind },
2870
- });
2871
- } catch { /* durable logging must never break the poll loop */ }
3419
+ const nowMs = Date.now();
3420
+ if (shouldWarnFailureStreak(consecutiveFailures, failureStreakWarned)) {
3421
+ failureStreakWarned = true;
3422
+ failureStreakWarnedAt = nowMs;
3423
+ lastEscalationAtMs = nowMs;
3424
+ try {
3425
+ appendError({
3426
+ cwd: DEFAULT_PROJECT_CWD,
3427
+ scope: 'scheduler',
3428
+ level: 'warn',
3429
+ message: `usage/rate-limit poller has failed ${consecutiveFailures} consecutive times (kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${schedulerPaths.schedulerStatePath()}`,
3430
+ meta: { consecutiveFailures, backoffMs, lastFailureKind },
3431
+ });
3432
+ } catch { /* durable logging must never break the poll loop */ }
3433
+ return;
3434
+ }
3435
+ if (failureStreakWarned && shouldEscalateFailureStreak(consecutiveFailures, lastEscalationAtMs, nowMs)) {
3436
+ lastEscalationAtMs = nowMs;
3437
+ const persistedMinutes = failureStreakWarnedAt ? Math.round((nowMs - failureStreakWarnedAt) / 60_000) : null;
3438
+ try {
3439
+ appendAuditEvent('usage_poller_failure_streak_persists', { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes });
3440
+ appendError({
3441
+ cwd: DEFAULT_PROJECT_CWD,
3442
+ scope: 'scheduler',
3443
+ level: 'warn',
3444
+ message: `usage/rate-limit poller streak still failing after ${persistedMinutes}m (${consecutiveFailures} consecutive, kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${schedulerPaths.schedulerStatePath()}`,
3445
+ meta: { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes },
3446
+ });
3447
+ } catch { /* durable logging must never break the poll loop */ }
3448
+ }
2872
3449
  }
2873
3450
  // PRD 1119: consecutive-rapid-rate-limit hard-pause tracking, keyed per slug.
2874
3451
  // See isCooldownSuppressed/nextRapidRateLimitCount below for the pure rules.
@@ -2882,6 +3459,7 @@ let resumeTimer = null;
2882
3459
  let pollLoopTimer = null;
2883
3460
  let rescheduleInterval = null;
2884
3461
  let heartbeatInterval = null;
3462
+ let dispatchLoopHandle = null;
2885
3463
  // Stall-detector state (computeStallSummary), read/written only inside the
2886
3464
  // heartbeat interval below. Keyed per-project cwd (never a single value) —
2887
3465
  // a single module-level flag would let one busy project's activity clear or
@@ -2905,12 +3483,13 @@ const runningSet = new Set();
2905
3483
  // N concurrent Opus processes — the >3-concurrent class that OOM-killed Electron.
2906
3484
  // Over-cap requests are QUEUED (not dropped) and drained as slots free, so a failed
2907
3485
  // PRD that never reaches 'needs_review' still eventually gets its fix-plan authored.
2908
- let investigationsInFlight = 0;
2909
3486
  const MAX_CONCURRENT_INVESTIGATIONS = 1;
2910
3487
  const deferredInvestigations = new Map(); // fixable-job slug -> { failedJob, runDir }
3488
+ // Mirror of machine `drain.active`, kept in memory so spawn sites need no queue read.
3489
+ let drainActive = false;
2911
3490
 
2912
3491
  function drainDeferredInvestigation() {
2913
- if (investigationsInFlight >= MAX_CONCURRENT_INVESTIGATIONS) return;
3492
+ if (drainActive || runtimeState.investigationCount() >= MAX_CONCURRENT_INVESTIGATIONS) return;
2914
3493
  const next = deferredInvestigations.entries().next();
2915
3494
  if (next.done) return;
2916
3495
  const [slug, ctx] = next.value;
@@ -3001,6 +3580,7 @@ function buildScheduleStatePayload(state) {
3001
3580
  lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
3002
3581
  nextReset: getNextResetCached(),
3003
3582
  paused: state.paused,
3583
+ drain: state.drain ?? null,
3004
3584
  // Launch circuit breaker (issue #11): which personas cannot launch right
3005
3585
  // now and why, plus any degraded-mode env in force. Empty objects when healthy.
3006
3586
  launchBlocks: state.launchBlocks ?? {},
@@ -3170,13 +3750,17 @@ function nextRapidRateLimitCount(prevCount, { rateLimited, durationMs }) {
3170
3750
  /**
3171
3751
  * Pure: decides the resumeAt actually armed for a pause. 'network' and
3172
3752
  * 'rate_limit' (PRD 1118) both get a bounded 30-minute fallback when no
3173
- * explicit resumeAt is supplied — the live rate_limit failure mode is the
3174
- * billing usage endpoint itself 429ing while the log yields no binding
3175
- * window either, which used to leave an indefinite pause with no resume
3176
- * timer at all (a queue that never comes back on its own).
3753
+ * explicit resumeAt is supplied, OR when the supplied resumeAtIso has
3754
+ * already passed (usageCircuit.isResetFresh) — a stale reset used to produce
3755
+ * `Math.max(30_000, <negative>)` downstream in computeResumeDelay, a
3756
+ * 30-SECOND nap instead of a real pause, spinning the queue right back into
3757
+ * the same still-active rate limit. The live failure mode is the billing
3758
+ * usage endpoint itself 429ing while the log yields no fresh binding window
3759
+ * either, which used to leave an indefinite pause with no resume timer at
3760
+ * all (a queue that never comes back on its own) — this covers both.
3177
3761
  */
3178
3762
  function computeEffectiveResumeAt(reason, resumeAtIso, nowMs = Date.now()) {
3179
- if (resumeAtIso) return resumeAtIso;
3763
+ if (resumeAtIso && isResetFresh(resumeAtIso, nowMs)) return resumeAtIso;
3180
3764
  if (reason === 'network' || reason === 'rate_limit') {
3181
3765
  return new Date(nowMs + 30 * 60_000).toISOString();
3182
3766
  }
@@ -3196,11 +3780,14 @@ function computeResumeDelay(effectiveResumeAtIso, nowMs = Date.now()) {
3196
3780
 
3197
3781
  async function setPaused(reason, resumeAtIso, opts = {}) {
3198
3782
  const { observedAt = null, force = false } = opts;
3783
+ const isManual = reason === 'manual';
3199
3784
  // Honor manual-override cooldown: if the user cleared a pause within the
3200
3785
  // last 5 minutes, suppress auto-pause re-engagement UNLESS this pause is
3201
3786
  // backed by a fresh observation (a run that started after the clear) or is
3202
3787
  // forced (the rapid-repeat hard pause, which the cooldown cannot suppress).
3203
- if (isCooldownSuppressed({ pauseClearedManuallyAt, now: Date.now(), observedAt, force })) {
3788
+ // A user-initiated 'manual' pause is never an auto-detection, so the cooldown
3789
+ // (which exists to ignore STALE auto-detections) does not apply to it.
3790
+ if (!isManual && isCooldownSuppressed({ pauseClearedManuallyAt, now: Date.now(), observedAt, force })) {
3204
3791
  console.log(`[scheduler] setPaused(${reason}) suppressed by manual override cooldown`);
3205
3792
  return;
3206
3793
  }
@@ -3210,15 +3797,24 @@ async function setPaused(reason, resumeAtIso, opts = {}) {
3210
3797
  console.log(`[scheduler] setPaused(${reason}) engaging despite manual override cooldown — triggering run started after the manual clear`);
3211
3798
  }
3212
3799
 
3213
- const effectiveResumeAt = computeEffectiveResumeAt(reason, resumeAtIso);
3800
+ // 'manual' never arms a resume timer: only schedule:resume / run-now clears it.
3801
+ const effectiveResumeAt = isManual ? null : computeEffectiveResumeAt(reason, resumeAtIso);
3214
3802
 
3215
- await mutate((s) => {
3803
+ const kept = await mutate((s) => {
3804
+ // A user pause outranks every auto-pause (rate_limit/auth/network): the
3805
+ // auto path must not overwrite it, or its resume timer would auto-clear it.
3806
+ if (!isManual && s.paused && s.paused.reason === 'manual') return true;
3216
3807
  if (s.paused && s.paused.reason === reason) {
3217
3808
  if (effectiveResumeAt) s.paused.resumeAt = effectiveResumeAt;
3218
3809
  } else {
3219
3810
  s.paused = { reason, since: new Date().toISOString(), resumeAt: effectiveResumeAt || null };
3220
3811
  }
3812
+ return false;
3221
3813
  });
3814
+ if (kept) {
3815
+ console.log(`[scheduler] setPaused(${reason}) ignored: a manual pause is in force`);
3816
+ return;
3817
+ }
3222
3818
  await broadcast({ flush: true });
3223
3819
  cancelToken.cancelled = true;
3224
3820
  if (resumeTimer) { clearTimeout(resumeTimer); resumeTimer = null; }
@@ -3256,6 +3852,7 @@ async function clearPause(source) {
3256
3852
  });
3257
3853
  // Un-cancel the tick guard on every recovery path, not just runDueJobs().
3258
3854
  applyPauseCleared(wasPaused, cancelToken);
3855
+ if (wasPaused) lastPauseClearedAt = Date.now();
3259
3856
  // Track manual clears for the auto-pause cooldown.
3260
3857
  if (source === 'manual' || source === 'run-now') {
3261
3858
  pauseClearedManuallyAt = Date.now();
@@ -3268,7 +3865,7 @@ async function clearPause(source) {
3268
3865
  firstFailureAt = null;
3269
3866
  firstNon429FailureAt = null;
3270
3867
  lastFailureKind = null;
3271
- failureStreakWarned = false;
3868
+ resetFailureStreak();
3272
3869
  persistSchedulerState();
3273
3870
  }
3274
3871
  if (wasPaused) await broadcast({ flush: true });
@@ -3351,37 +3948,65 @@ function resetJobFields(job, errorMsg, opts = {}) {
3351
3948
  return true;
3352
3949
  }
3353
3950
 
3354
- // Grace period between a boot orphan's SIGTERM and reading its log to
3355
- // classify the outcome — matches killOrphanClaudePid's own internal 5s
3356
- // SIGKILL follow-up delay, plus a small margin so classification always runs
3357
- // after that SIGKILL has had a chance to land.
3358
- const BOOT_ORPHAN_KILL_GRACE_MS = 6000;
3359
-
3360
3951
  /**
3361
- * partitionBootOrphans(jobs, isAlive?) → { immediate: string[], deferred: string[] }
3952
+ * partitionBootOrphans(jobs, liveness) → { immediate: string[], adopted: string[] }
3362
3953
  *
3363
- * Pure decision split for boot reconciliation. A 'running' job whose recorded
3364
- * pid is still alive must NOT be classified from its log yet — the orphaned
3365
- * process may still be writing to it, so reading now risks misclassifying a
3366
- * job that is about to emit result:success as no_result and double-running it.
3367
- * Ported from reconcileQueueOffline's cross-tick escalation (see
3368
- * src/main/lib/watchdogHelpers.cjs) — here it's a single deferred window since
3369
- * this process stays up to revisit it, rather than a separate short-lived
3370
- * watchdog process needing another tick.
3371
- */
3372
- function partitionBootOrphans(jobs, isAlive = claudePidAlive) {
3954
+ * Pure decision split for boot reconciliation of 'running' rows. A row PROVEN
3955
+ * ALIVE is `adopted`: left `running`, never signalled — the steady-state
3956
+ * reaper (reapDeadRunningJobs) finishes it on exit, exactly as it does for any
3957
+ * pidless-recovered row. Only rows proven dead or exited are `immediate` and go
3958
+ * through applyOrphanOutcome. `liveness` is the same injected set
3959
+ * selectReapableJobs takes (plus readRecord/runsDir/identityOf):
3960
+ * { pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess,
3961
+ * readRecord(runDir, slug) }.
3962
+ *
3963
+ * classifyAdoption runs FIRST when the row has a supervisor record: 'adopt' and
3964
+ * 'over-budget' are alive (budget re-arm across restart is a separate PRD, so
3965
+ * an over-budget row is spared, not killed); 'exited' / 'dead' / 'foreign-pid'
3966
+ * are not. A row with no record falls to the reaper's own ladder: recorded
3967
+ * pid alive, then fresh log, log-pid alive, /proc cwd scan.
3968
+ * Complexity: O(jobs) plus one /proc probe per running row.
3969
+ */
3970
+ function partitionBootOrphans(jobs, {
3971
+ pidAlive = claudePidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess,
3972
+ readRecord = supervisorRecord.readSupervisorRecord, runsDir = null, identityOf = procIdentityOf,
3973
+ now = Date.now(),
3974
+ } = {}) {
3373
3975
  const immediate = [];
3374
- const deferred = [];
3976
+ const adopted = [];
3375
3977
  for (const j of jobs) {
3376
3978
  if (j.status !== 'running') continue;
3377
- const pid = j.runtime?.pid;
3378
- if (pid && isAlive(pid)) {
3379
- deferred.push(j.slug);
3380
- } else {
3381
- immediate.push(j.slug);
3382
- }
3979
+ if (isBootRowAlive(j, {
3980
+ pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess, readRecord, runsDir, identityOf, now,
3981
+ })) adopted.push(j.slug);
3982
+ else immediate.push(j.slug);
3983
+ }
3984
+ return { immediate, adopted };
3985
+ }
3986
+
3987
+ function isBootRowAlive(j, {
3988
+ pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess, readRecord, runsDir, identityOf, now,
3989
+ }) {
3990
+ const logMtimeMs = typeof getLogMtimeMs === 'function' ? getLogMtimeMs(j) : null;
3991
+ const runDir = j.runId ? path.join(runsDir || schedulerPaths.runsDir(), j.runId) : null;
3992
+ const record = runDir && typeof readRecord === 'function' ? readRecord(runDir, j.slug) : null;
3993
+ if (record && record.pid) {
3994
+ const alive = !!pidAlive(record.pid);
3995
+ const verdict = supervisorRecord.classifyAdoption(record, {
3996
+ identity: alive ? identityOf(record.pid) : null,
3997
+ pidAlive: alive,
3998
+ logMtimeMs,
3999
+ exitMarker: supervisorRecord.hasExitMarker(runDir, record),
4000
+ now,
4001
+ });
4002
+ return verdict === 'adopt' || verdict === 'over-budget';
3383
4003
  }
3384
- return { immediate, deferred };
4004
+ const pid = j.runtime?.pid;
4005
+ if (pid && pidAlive(pid)) return true;
4006
+ if (typeof logFreshWindowMs === 'number' && Number.isFinite(logMtimeMs) && now - logMtimeMs <= logFreshWindowMs) return true;
4007
+ const logPid = typeof getLogPid === 'function' ? getLogPid(j) : null;
4008
+ if (logPid && pidAlive(logPid)) return true;
4009
+ return !!(typeof findLiveProcess === 'function' && findLiveProcess(j));
3385
4010
  }
3386
4011
 
3387
4012
  /**
@@ -3647,7 +4272,7 @@ async function notifyOriginatingTab(job, {
3647
4272
  const epicIdForTranscript = prd?.sourcePromptId || prd?.sourceTabId || job.epicId || null;
3648
4273
  if (epicIdForTranscript && job.cwd) {
3649
4274
  try {
3650
- const logPath = job.runId ? path.join(RUNS_DIR, job.runId, `${job.slug}.log`) : null;
4275
+ const logPath = job.runId ? path.join(schedulerPaths.runsDir(), job.runId, `${job.slug}.log`) : null;
3651
4276
  const resultText = readResultFromLog(logPath);
3652
4277
  await appendTranscriptTurn(job.cwd, epicIdForTranscript, {
3653
4278
  role: 'assistant',
@@ -4378,6 +5003,39 @@ async function resolveLandedCommitEvidence(cwd, sha, sinceIso) {
4378
5003
  }
4379
5004
  }
4380
5005
 
5006
+ /**
5007
+ * True when `sha` is a non-empty commit that is an ancestor of (or equal to)
5008
+ * HEAD in the repo at `cwd` (`git merge-base --is-ancestor`). Bounded, never
5009
+ * throws: an empty sha, an unknown sha, or any git failure is `false`, so the
5010
+ * caller's safe default is "cannot prove the work survived".
5011
+ */
5012
+ async function landedCommitIsAncestorOfHead(cwd, sha) {
5013
+ if (!sha || typeof sha !== 'string' || !cwd) return false;
5014
+ try {
5015
+ await execGitAt(resolveProjectRoot(cwd), ['merge-base', '--is-ancestor', sha, 'HEAD'], { timeout: 10_000 });
5016
+ return true;
5017
+ } catch {
5018
+ return false;
5019
+ }
5020
+ }
5021
+
5022
+ /**
5023
+ * Pure predicate: a needs_review row parked as shared_tree_reverted that
5024
+ * carries a landedCommit — the only shape reverifyNeedsReview can re-check
5025
+ * against ground truth (landedCommitIsAncestorOfHead). Deliberately
5026
+ * independent of autoFixAttempted: 1229-fo-03 was parked with
5027
+ * autoFixAttempted:true and no autoFixOutcome (isStrandedAutoFixPark shape),
5028
+ * yet isStrandedAutoFixPark could not release it — that ladder only resolves
5029
+ * once job.looksDone is set, and reverifyNeedsReview computes looksDone only
5030
+ * for isRescanCandidate / isGuardParkedWithoutAutoFix rows, neither of which
5031
+ * a shared_tree_reverted + autoFixAttempted row is.
5032
+ */
5033
+ function isStaleSharedTreeRevertedPark(job) {
5034
+ return !!job && job.status === 'needs_review'
5035
+ && job.verifierVerdict === 'shared_tree_reverted'
5036
+ && typeof job.landedCommit === 'string' && job.landedCommit.length > 0;
5037
+ }
5038
+
4381
5039
  /**
4382
5040
  * Commit exactly `paths` (must already be dirty on disk) onto a dedicated
4383
5041
  * `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
@@ -4557,7 +5215,7 @@ function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }
4557
5215
  // create it — `recursive: true` makes that race safe.
4558
5216
  function pickRunDir() {
4559
5217
  const ts = new Date().toISOString().replace(/[:.]/g, '-');
4560
- const dir = path.join(RUNS_DIR, ts);
5218
+ const dir = path.join(schedulerPaths.runsDir(), ts);
4561
5219
  return { runId: ts, dir };
4562
5220
  }
4563
5221
 
@@ -4617,7 +5275,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4617
5275
  // Sync write: this is an early-exit error path inside an async function,
4618
5276
  // so we could await, but using the sync variant keeps the error path
4619
5277
  // ordering identical to the spawn-failed branch below (also sync).
4620
- config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs: 0 });
5278
+ config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs: 0, ...SCHEDULER_META_IDENTITY });
4621
5279
  return { exitCode: -1, durationMs: 0, error: errMsg, sessionId };
4622
5280
  }
4623
5281
 
@@ -4778,7 +5436,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4778
5436
  if (!promptCheck.ok) {
4779
5437
  safeLog(`[scheduler] ${promptCheck.error}\n`);
4780
5438
  closeFd();
4781
- config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: promptCheck.error, startedAt, finishedAt: Date.now(), durationMs: 0 });
5439
+ config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: promptCheck.error, startedAt, finishedAt: Date.now(), durationMs: 0, ...SCHEDULER_META_IDENTITY });
4782
5440
  return { exitCode: -1, durationMs: 0, error: promptCheck.error, sessionId };
4783
5441
  }
4784
5442
 
@@ -4972,13 +5630,17 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4972
5630
 
4973
5631
  // ---------- spawn ----------
4974
5632
 
5633
+ // Distinct comm (`sm-claude-job`) + slug-labelled argv0 for System Monitor.
5634
+ // Both keep the word `claude`, which the /\bclaude\b/ reaper gates need.
5635
+ const jobSpawn = claudeSpawnTarget('job', job.slug, claudeBin);
5636
+
4975
5637
  const { child } = withChildAndLog({
4976
5638
  fd,
4977
5639
  logPath,
4978
5640
  safeLog,
4979
5641
  closeFd,
4980
5642
  spawn: {
4981
- command: claudeBin,
5643
+ command: jobSpawn.command,
4982
5644
  // Resume mode passes `--resume <sessionId>` (reconnect to the SAME
4983
5645
  // session) INSTEAD of `--session-id <sessionId>` (mint a new one) —
4984
5646
  // never both, see buildClaudeSpawnArgs.
@@ -4992,6 +5654,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4992
5654
  options: {
4993
5655
  cwd: spawnCwd,
4994
5656
  env: childEnv,
5657
+ ...(jobSpawn.argv0 ? { argv0: jobSpawn.argv0 } : {}),
4995
5658
  // detached:true puts the child in its own process group so we can kill
4996
5659
  // the entire descendant tree (including any stray background bashes the
4997
5660
  // agent spawned) with `process.kill(-pid)`. Without this, child.kill()
@@ -5017,7 +5680,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5017
5680
  sl(`\n[scheduler] ${errMsg}\n`);
5018
5681
  // Sync write: inside a Promise executor callback; must flush meta
5019
5682
  // before resolve() so the spawnJob mutate() that follows sees it.
5020
- config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA, originSessionId, contextDigestApplied });
5683
+ config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked, ...SCHEDULER_META_IDENTITY, originSessionId, contextDigestApplied });
5021
5684
  resolve({ exitCode: -1, durationMs, error: errMsg, leakedDescendants: leaked, sessionId });
5022
5685
  return;
5023
5686
  }
@@ -5079,7 +5742,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5079
5742
  startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
5080
5743
  agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
5081
5744
  killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
5082
- schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
5745
+ ...SCHEDULER_META_IDENTITY,
5083
5746
  originSessionId, contextDigestApplied,
5084
5747
  });
5085
5748
  resolve({
@@ -5091,6 +5754,22 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5091
5754
 
5092
5755
  if (child) {
5093
5756
  safeLog(`[scheduler] spawned pid=${child.pid} sessionId=${sessionId} (process group)\n\n`);
5757
+ // The one synchronous, authoritative dispatch record (see
5758
+ // jobSupervisorRecord.cjs). detached:true → setsid → pgid === pid, so
5759
+ // no process.getpgid. A write failure must never fail the dispatch.
5760
+ try {
5761
+ supervisorRecord.writeSupervisorRecord({
5762
+ runDir, slug: job.slug, cwd, runId: path.basename(runDir), pid: child.pid, pgid: child.pid,
5763
+ identity: procIdentityOf(child.pid), execCwd: spawnCwd,
5764
+ worktreeDir: execCwd || null, worktreeBranch: execCwd ? `sm-job/${job.slug}` : null,
5765
+ sessionId, startedAt, budgetMs: budgetExempt ? null : jobBudgetMs, maxDurationMs: null,
5766
+ idleKillMs: IDLE_OUTPUT_KILL_MS, schedulerPid: process.pid, codeSha: SCHEDULER_CODE_SHA,
5767
+ });
5768
+ } catch (e) {
5769
+ const message = e?.message ?? String(e);
5770
+ console.error(`[scheduler] FAILED to write supervisor record for ${job.slug} pid=${child.pid}: ${message}`);
5771
+ appendAuditEvent('supervisor_record_write_failed', { slug: job.slug, cwd, pid: child.pid, error: message });
5772
+ }
5094
5773
  // Make this job the OOM killer's preferred victim over Electron.
5095
5774
  biasJobOomScore(child.pid);
5096
5775
  // Persist runtime.pid with one retry — still fire-and-forget (must
@@ -5380,7 +6059,8 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
5380
6059
  return { deferred: false };
5381
6060
  }
5382
6061
  }
5383
- if (investigationsInFlight >= MAX_CONCURRENT_INVESTIGATIONS) {
6062
+ // A drain (lib/upgradeDrain.cjs) admits no NEW work: queue instead of spawning.
6063
+ if (drainActive || runtimeState.investigationCount() >= MAX_CONCURRENT_INVESTIGATIONS) {
5384
6064
  // Queue for retry when a slot frees rather than dropping — otherwise a failed
5385
6065
  // job (never 'needs_review', so reverifyNeedsReview won't retry it) would
5386
6066
  // silently never get an auto-authored fix-plan.
@@ -5394,12 +6074,12 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
5394
6074
  // both pass the cap check. Released in onExit, on any pre-spawn early return, or
5395
6075
  // on a synchronous throw (try/catch below) — and releasing hands the slot to a
5396
6076
  // queued investigation so none are stranded.
5397
- investigationsInFlight++;
6077
+ runtimeState.reserveInvestigation(failedJob.slug);
5398
6078
  let slotReleased = false;
5399
6079
  const releaseSlot = () => {
5400
6080
  if (slotReleased) return;
5401
6081
  slotReleased = true;
5402
- investigationsInFlight--;
6082
+ runtimeState.releaseInvestigation(failedJob.slug);
5403
6083
  drainDeferredInvestigation();
5404
6084
  };
5405
6085
  try {
@@ -5504,13 +6184,14 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
5504
6184
  };
5505
6185
 
5506
6186
  // Phase 2: spawn with lifecycle managed by withChildAndLog.
6187
+ const probeSpawn = claudeSpawnTarget('aux', 'investigate', claudeBin);
5507
6188
  const { child } = withChildAndLog({
5508
6189
  fd,
5509
6190
  logPath: investigationLogPath,
5510
6191
  safeLog,
5511
6192
  closeFd,
5512
6193
  spawn: {
5513
- command: claudeBin,
6194
+ command: probeSpawn.command,
5514
6195
  args: [
5515
6196
  '-p', prompt,
5516
6197
  '--model', 'opus',
@@ -5519,7 +6200,7 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
5519
6200
  '--verbose',
5520
6201
  '--session-id', sessionId,
5521
6202
  ],
5522
- options: { cwd, env: childEnv },
6203
+ options: { cwd, env: childEnv, ...(probeSpawn.argv0 ? { argv0: probeSpawn.argv0 } : {}) },
5523
6204
  },
5524
6205
  watchdogs: [deadmanWatchdog],
5525
6206
  onExit({ exitCode, error, spawnFailed, safeLog: sl }) {
@@ -5621,6 +6302,20 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
5621
6302
 
5622
6303
  if (child) {
5623
6304
  safeLog(`[scheduler] investigation pid=${child.pid}\n\n`);
6305
+ try {
6306
+ supervisorRecord.writeSupervisorRecord({
6307
+ runDir, slug: `${failedJob.slug}.investigation`, kind: 'investigation', cwd: failedJob.cwd ?? null,
6308
+ runId: path.basename(runDir), pid: child.pid, pgid: child.pid, identity: procIdentityOf(child.pid),
6309
+ execCwd: failedJob.cwd ?? null, worktreeDir: null, worktreeBranch: null, sessionId: null,
6310
+ startedAt: Date.now(), budgetMs: null, maxDurationMs: MAX_INVESTIGATION_DURATION_MS, idleKillMs: null,
6311
+ schedulerPid: process.pid, codeSha: SCHEDULER_CODE_SHA,
6312
+ });
6313
+ } catch (e) {
6314
+ const message = e?.message ?? String(e);
6315
+ console.error(`[scheduler] FAILED to write supervisor record for investigation ${failedJob.slug} pid=${child.pid}: ${message}`);
6316
+ appendAuditEvent('supervisor_record_write_failed', { slug: failedJob.slug, cwd: failedJob.cwd, pid: child.pid, error: message, kind: 'investigation' });
6317
+ }
6318
+ runtimeState.stampInvestigationPid(failedJob.slug, child.pid);
5624
6319
  // Recorded so findStrandedInvestigations (a post-restart maintenance
5625
6320
  // sweep — the live process has no other way to know a probe is still
5626
6321
  // running) can tell a live probe apart from one whose owning process is
@@ -5722,9 +6417,22 @@ async function computeDepHistorySatisfaction(state) {
5722
6417
  for (const slug of await queueHistory.completedSlugsForCwd(cwd)) satisfied.add(slug);
5723
6418
  for (const dir of listArchivedPrdDirs(cwd)) {
5724
6419
  let entries;
5725
- try { entries = await fsp.readdir(dir); } catch { continue; }
5726
- for (const name of entries) {
5727
- if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
6420
+ try { entries = await fsp.readdir(dir, { withFileTypes: true }); } catch { continue; }
6421
+ for (const ent of entries) {
6422
+ if (ent.isFile() && ent.name.endsWith('.md')) { satisfied.add(ent.name.slice(0, -3)); continue; }
6423
+ // Option (b) of PRD 1286: a manual archive (queueOps.archiveOne, the
6424
+ // schedule:archive-prd route and scheduler_archive_prd MCP tool) files the PRD
6425
+ // under prds-archived/<ISO-ts>/<slug>.md — one level DEEPER than the auto-archive
6426
+ // layout. Reading only the top level made every manually-archived slug invisible
6427
+ // here, so once its row aged out its dependents held forever as 'unresolved'.
6428
+ // A dep whose PRD file was archived is satisfied; a dep with NO file, row or
6429
+ // history record (never ran, or a typo) still holds — nothing is dropped.
6430
+ if (!ent.isDirectory()) continue;
6431
+ let inner;
6432
+ try { inner = await fsp.readdir(path.join(dir, ent.name)); } catch { continue; }
6433
+ for (const name of inner) {
6434
+ if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
6435
+ }
5728
6436
  }
5729
6437
  }
5730
6438
  } catch (e) {
@@ -5821,7 +6529,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5821
6529
  // Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
5822
6530
  // — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
5823
6531
  // leaves the job pending; the next tick retries when a slot frees up.
5824
- const slotToken = sessionSlots.acquire(`scheduler:${job.slug}`);
6532
+ const slotToken = sessionSlots.acquire(`scheduler:${job.slug}`, { claimedAt: Date.now() });
5825
6533
  if (!slotToken) {
5826
6534
  console.log(`[scheduler] no session slot free for ${job.slug} — deferring (${JSON.stringify(sessionSlots.snapshot().holders.map((h) => h.owner))})`);
5827
6535
  return;
@@ -5947,7 +6655,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5947
6655
  // targets a specific prior session on purpose) and for anything not
5948
6656
  // currently 'pending' (e.g. a needs_review->running recovery row).
5949
6657
  if (!resumeTarget && s.jobs[idx].status === 'pending') {
5950
- const outcome = latestTerminalOutcomeForSlug(job.slug, { runsDir: RUNS_DIR });
6658
+ const outcome = latestTerminalOutcomeForSlug(job.slug, { runsDir: schedulerPaths.runsDir() });
5951
6659
  const reconcileDecision = evaluateDispatchSidecarReconcile({
5952
6660
  rowStatus: s.jobs[idx].status,
5953
6661
  rowRunId: s.jobs[idx].runId ?? null,
@@ -5956,7 +6664,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5956
6664
  outcome,
5957
6665
  });
5958
6666
  if (reconcileDecision.skip) {
5959
- const sidecar = readRunOutcomeSidecars(path.join(RUNS_DIR, reconcileDecision.runId), job.slug);
6667
+ const sidecar = readRunOutcomeSidecars(path.join(schedulerPaths.runsDir(), reconcileDecision.runId), job.slug);
5960
6668
  transitionJob(s.jobs[idx], 'completed', {
5961
6669
  reason: `prior run ${reconcileDecision.runId} already completed this slug (sidecar-reconciled)`,
5962
6670
  source: 'spawnJob:dispatch-sidecar-reconcile',
@@ -5986,7 +6694,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5986
6694
  // pass_no_commit_prior_run_verified exemption can fire on this
5987
6695
  // run if it turns out to be another no-op re-verification.
5988
6696
  if (!s.jobs[idx].landedCommit && outcome?.runId) {
5989
- const sidecar = readRunOutcomeSidecars(path.join(RUNS_DIR, outcome.runId), job.slug);
6697
+ const sidecar = readRunOutcomeSidecars(path.join(schedulerPaths.runsDir(), outcome.runId), job.slug);
5990
6698
  if (sidecar.outcome?.landedCommit) {
5991
6699
  s.jobs[idx].landedCommit = sidecar.outcome.landedCommit;
5992
6700
  }
@@ -6169,6 +6877,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6169
6877
  const foreignWip = worktree.ok ? { carriedPaths } : { preRunDirtyPaths };
6170
6878
  try {
6171
6879
  res = await executeJob(job, runDir, defaultCwd, async (pid, sessionId, cwd) => {
6880
+ sessionSlots.stampPid(slotToken, pid);
6172
6881
  await mutate((s) => {
6173
6882
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
6174
6883
  if (idx >= 0) {
@@ -6312,8 +7021,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6312
7021
  }
6313
7022
 
6314
7023
  if (res.rateLimited) {
7024
+ // The executor itself observed a 429 — open the shared circuit even if
7025
+ // the billing poller has been reporting 'ok' all along (AC3: a
7026
+ // different window can 429 the executor than the one binding the
7027
+ // poller's own reads).
7028
+ billing.usageCircuit.recordFailure('executor_429');
6315
7029
  const logPath = path.join(runDir, `${job.slug}.log`);
6316
- const billingResetIso = await refreshNextReset().catch(() => cachedNextReset);
7030
+ const billingResetIso = await billingResetForPause();
6317
7031
  const resetIso = resolveRateLimitPauseReset(logPath, billingResetIso);
6318
7032
  const observedAt = dispatchStartedAtMs;
6319
7033
  const prevCount = consecutiveRapidRateLimitsBySlug.get(job.slug) || 0;
@@ -6401,6 +7115,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6401
7115
  allJobs: stateForDeps.jobs,
6402
7116
  committedDuringRun,
6403
7117
  priorLandedCommit,
7118
+ jobLandedCommitThisRun,
7119
+ exitCode: res.exitCode,
6404
7120
  }).catch((e) => ({
6405
7121
  verdict: 'verify_unavailable',
6406
7122
  reason: `verifier threw: ${e?.message ?? String(e)}`,
@@ -6506,6 +7222,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6506
7222
  dirtyBaseline: guardBaselineEntries,
6507
7223
  headBefore: guardHeadBefore,
6508
7224
  slug: job.slug,
7225
+ landedCommit: jobLandedCommitThisRun,
6509
7226
  });
6510
7227
  // A restored stash alone isn't silence — it's logged loudly above and
6511
7228
  // surfaced on the job row below — but a path that's still missing
@@ -7251,36 +7968,122 @@ async function spawnResumeRecovery(job, resumeTarget) {
7251
7968
  // is synchronous and spawnJob is fire-and-forget.
7252
7969
  let tickTail = Promise.resolve();
7253
7970
 
7971
+ // Tick watchdog. The tick BODY (not enqueue-to-settle) is bounded: a body
7972
+ // that never settles (hung reconcile / git walk) is declared wedged, the
7973
+ // chain is reset, and `tickGeneration` is bumped so the abandoned body —
7974
+ // which may resume much later — fails its generation re-check after every
7975
+ // await and returns without spawning or mutating. mutateTail is NOT touched.
7976
+ let tickGeneration = 0;
7977
+ let tickWedgeLatched = false;
7978
+ function tickWatchdogMs() {
7979
+ const raw = process.env.SM_TICK_WATCHDOG_MS;
7980
+ if (raw === undefined || raw === '') return 120_000;
7981
+ const n = Number(raw);
7982
+ return Number.isFinite(n) && n >= 0 ? n : 120_000;
7983
+ }
7984
+
7254
7985
  // `bypassLoadGate` is set only by the explicit human run-now / force-tick
7255
7986
  // paths (via runDueJobs): the human is asking, so the CPU-load gate yields
7256
7987
  // and logs that it did. Every automatic caller leaves it false.
7988
+ /**
7989
+ * Fail-CLOSED expiry of runtime reservations (slot tokens, quiet-machine
7990
+ * lease, investigation set, worktree accounting) whose owner is provably
7991
+ * dead. `now` is the pass start: reservations claimed after it are never
7992
+ * expired. Accounting only — never signals or removes anything.
7993
+ */
7994
+ function runReservationExpiryPass(jobs, now = Date.now()) {
7995
+ try {
7996
+ const liveSlugs = new Set(runningSet);
7997
+ const terminal = new Set();
7998
+ for (const j of jobs || []) {
7999
+ if (j.status === 'running' || j.status === 'investigating') liveSlugs.add(j.slug);
8000
+ else if (j.status === 'completed' || j.status === 'failed' || j.status === 'skipped') terminal.add(j.slug);
8001
+ }
8002
+ const pidAlive = (pid) => claudePidAlive(pid);
8003
+ sessionSlots.expireDead({ liveSlugs, pidAlive, now });
8004
+ quietMachineLease.expireDead({ liveSlugs, now });
8005
+ runtimeState.expireDeadInvestigations({ liveSlugs, pidAlive, now });
8006
+ gitWorktree.expireDeadWorktreeRegistrations({
8007
+ isTerminalBranch: (branch) => {
8008
+ const key = gitWorktree.keyFromBranch('job', branch);
8009
+ return !!key && !liveSlugs.has(key) && terminal.has(key);
8010
+ },
8011
+ });
8012
+ } catch (e) {
8013
+ console.warn('[scheduler] reservation expiry pass failed', e?.message);
8014
+ }
8015
+ }
8016
+
7257
8017
  function tickQueue({ bypassLoadGate = false } = {}) {
7258
- const next = tickTail.then(async () => {
8018
+ const budgetMs = tickWatchdogMs();
8019
+ const next = tickTail.then(() => {
8020
+ const gen = tickGeneration;
8021
+ return withTimeout(() => tickBody(gen, { bypassLoadGate }), budgetMs, () => {
8022
+ tickGeneration++; // fence the abandoned body
8023
+ console.warn(`[scheduler] TICK WEDGED: tick body exceeded ${budgetMs}ms — resetting tick chain`);
8024
+ if (!tickWedgeLatched) {
8025
+ tickWedgeLatched = true;
8026
+ appendAuditEvent('tick_wedged', { budgetMs });
8027
+ }
8028
+ // CAS: only reset if nothing has queued behind this wedged link.
8029
+ if (tickTail === tail) tickTail = Promise.resolve();
8030
+ return recordTick({ fired: false, reason: 'wedged' }, { detail: `tick body exceeded ${budgetMs}ms` });
8031
+ }).then((r) => {
8032
+ if (r?.reason !== 'wedged') tickWedgeLatched = false;
8033
+ return r;
8034
+ });
8035
+ });
8036
+ const tail = next.catch(() => {});
8037
+ tickTail = tail;
8038
+ return next;
8039
+ }
8040
+
8041
+ // The stale sentinel a fenced body returns: never recorded, never acted on.
8042
+ const STALE_TICK = Object.freeze({ fired: false, reason: 'stale-generation' });
8043
+
8044
+ async function tickBody(gen, { bypassLoadGate }) {
8045
+ const tickStartedAt = Date.now();
8046
+ {
8047
+ const stale = () => gen !== tickGeneration;
7259
8048
  const state = await readQueue();
8049
+ if (stale()) return STALE_TICK;
7260
8050
  // Never reconcile against an unreadable queue: reconcile() would see zero
7261
8051
  // job rows for every PRD on disk and resurrect the lot as 'pending'.
7262
8052
  if (state.unreadable) {
7263
8053
  console.error('[scheduler] tickQueue skipped: queue.json unreadable');
7264
8054
  return { fired: false, reason: 'unreadable' };
7265
8055
  }
8056
+ // Supervision of an adopted executor is independent of dispatch: it runs
8057
+ // even while paused, before any early return below.
8058
+ await superviseAdoptedRunsPass(state.jobs);
8059
+ if (stale()) return STALE_TICK;
7266
8060
  if (state.paused) {
7267
8061
  console.log('[scheduler] tickQueue skipped: paused');
7268
8062
  return recordTick({ fired: false, reason: 'paused' }, { detail: 'scheduler paused' });
7269
8063
  }
8064
+ // Upgrade drain (lib/upgradeDrain.cjs): a separate field from `paused`, so a
8065
+ // rate-limit pause can't overwrite it. Nothing new dispatches; running and
8066
+ // investigating rows finish.
8067
+ if (state.drain?.active) {
8068
+ return recordTick({ fired: false, reason: 'draining' }, { detail: 'draining for restart' });
8069
+ }
7270
8070
  if (cancelToken.cancelled) return { fired: false, reason: 'cancelled' };
7271
8071
 
7272
8072
  // Stamped here — the moment tickQueue actually reaches the picker,
7273
8073
  // regardless of whether this pass ends in a launch — so
7274
8074
  // classifyQueueStarvation can tell "the engine keeps evaluating the
7275
8075
  // queue" apart from "nothing has invoked tickQueue in a long time".
7276
- // Distinct from `lastRunAt` below, which stays true to its existing
7277
- // meaning (a batch actually launched) since other readers depend on that.
8076
+ // Distinct from `lastRunAt` below (a batch actually launched): this one
8077
+ // means only "the loop is alive" (heartbeat/health) and must NEVER feed
8078
+ // the idle clock — see dispatchIdleMs.
7278
8079
  await mutate((s) => { s.lastDispatchAttemptAt = new Date().toISOString(); });
8080
+ if (stale()) return STALE_TICK;
7279
8081
 
7280
8082
  // The retired-flat-dir sweep now lives inside reconcile() itself (see its
7281
8083
  // own comment) so every caller of reconcile — not just this tick — gets
7282
8084
  // the guarantee.
7283
8085
  await reconcile(state);
8086
+ if (stale()) return STALE_TICK;
7284
8087
  // Reclaim any job-kind worktree whose owning row already resolved
7285
8088
  // (completed/failed/skipped) without the run ever reaching
7286
8089
  // cleanupWorktree — a leaked checkout that would otherwise sit until the
@@ -7297,9 +8100,19 @@ function tickQueue({ bypassLoadGate = false } = {}) {
7297
8100
  // to also carry a private `concurrencyCap` of 3 — the exact per-consumer
7298
8101
  // cap that sessionSlots.cjs was written to replace — which silently
7299
8102
  // ceilinged the queue at 3 while the pool the user configured said 5.
7300
- const freeSlots = sessionSlots.available();
8103
+ // While the usage meter is degraded (degradedConcurrencyCapValue set by
8104
+ // pollLoop — circuit open, or a poll otherwise failed), a picker-side
8105
+ // hold narrows this SAME freeSlots figure instead of standing up a
8106
+ // second pool: the row count admitted this tick simply can't exceed the
8107
+ // degraded cap minus what's already running.
8108
+ runReservationExpiryPass(state.jobs, tickStartedAt);
8109
+ const freeSlots = degradedConcurrencyCapValue != null
8110
+ ? Math.max(0, Math.min(sessionSlots.available(), degradedConcurrencyCapValue - runningSet.size))
8111
+ : sessionSlots.available();
7301
8112
  const heldSlugs = await computeLaunchHolds(state);
8113
+ if (stale()) return STALE_TICK;
7302
8114
  const satisfiedSlugsByCwd = await computeDepHistorySatisfaction(state);
8115
+ if (stale()) return STALE_TICK;
7303
8116
  const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
7304
8117
  leaseHeld: quietMachineLease.isHeld(),
7305
8118
  machineInUse: sessionSlots.inUse(),
@@ -7399,18 +8212,18 @@ function tickQueue({ bypassLoadGate = false } = {}) {
7399
8212
  }
7400
8213
 
7401
8214
  await mutate((s) => { s.lastRunAt = new Date().toISOString(); });
8215
+ if (stale()) return STALE_TICK;
7402
8216
  await broadcast();
8217
+ if (stale()) return STALE_TICK;
7403
8218
 
7404
8219
  const { runId, dir: runDir } = pickRunDir();
7405
8220
  for (const job of gatedBatch) {
7406
- if (cancelToken.cancelled) break;
8221
+ if (cancelToken.cancelled || stale()) break;
7407
8222
  // spawnJob is fire-and-forget; it calls tickQueue() on completion.
7408
8223
  spawnJob(job, runId, runDir, state.config.defaultCwd).catch(() => {});
7409
8224
  }
7410
8225
  return recordTick({ fired: true, count: gatedBatch.length, group: gatedBatch[0]?.parallelGroup }, { holds });
7411
- });
7412
- tickTail = next.catch(() => {});
7413
- return next;
8226
+ }
7414
8227
  }
7415
8228
 
7416
8229
  // Translates a tickQueue()/runDueJobs() outcome descriptor into a renderer-facing
@@ -7486,6 +8299,40 @@ async function maybeLaunchWhenAvailable(state) {
7486
8299
  * second scheduler. */
7487
8300
  const QUEUE_STARVATION_MS = 10 * 60_000;
7488
8301
 
8302
+ /**
8303
+ * dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now }) → ms
8304
+ *
8305
+ * Pure. The ONE dispatch-idleness clock: `now - max(lastRunAt, lastPauseClearedAt,
8306
+ * schedulerBootedAt)`. `lastRunAt` has exactly one writer (tickQueue, right
8307
+ * before the spawn loop) so it already means "a batch actually launched";
8308
+ * a pause clear and a scheduler boot are the other two moments the queue
8309
+ * legitimately (re)starts. Deliberately NOT `lastDispatchAttemptAt`, which
8310
+ * tickQueue stamps before every gate — a queue that ticks every 30 s and
8311
+ * launches nothing (leaked slot, stuck hold) would refresh that stamp
8312
+ * forever and never look idle (the structural repeat of the f18e161 bug
8313
+ * where lastRunAt was refreshed every poll). No finite input → Infinity.
8314
+ */
8315
+ function dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now } = {}) {
8316
+ const stamps = [lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs].filter(Number.isFinite);
8317
+ return stamps.length ? now - Math.max(...stamps) : Infinity;
8318
+ }
8319
+
8320
+ /**
8321
+ * launchBlockedSlugs(jobs, launchBlocks) → Set<slug>
8322
+ *
8323
+ * Pure, sync (health.cjs runs as a cold process). Pending rows whose persona
8324
+ * has ANY launch-breaker entry — a superset of computeLaunchHolds, which
8325
+ * additionally lets one half-open probe row through per persona.
8326
+ */
8327
+ function launchBlockedSlugs(jobs, launchBlocks) {
8328
+ const out = new Set();
8329
+ if (!launchBlocks || !Object.keys(launchBlocks).length) return out;
8330
+ for (const j of Array.isArray(jobs) ? jobs : []) {
8331
+ if (j && j.status === 'pending' && launchBlocks[launchFailure.launchBlockKeyFor(j)]) out.add(j.slug);
8332
+ }
8333
+ return out;
8334
+ }
8335
+
7489
8336
  /**
7490
8337
  * classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs })
7491
8338
  * → null | { kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }
@@ -7513,22 +8360,28 @@ const QUEUE_STARVATION_MS = 10 * 60_000;
7513
8360
  * Returns null when the queue is healthy (work running, nothing pending,
7514
8361
  * paused on purpose, or simply not idle long enough yet).
7515
8362
  */
7516
- function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
8363
+ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, heldSlugs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
7517
8364
  if (paused) return null; // paused is a DECISION, not a stall
7518
8365
  if (runningCount > 0) return null; // work is flowing
7519
8366
  const rows = Array.isArray(jobs) ? jobs : [];
7520
8367
  const pending = rows.filter((j) => j && j.status === 'pending');
7521
8368
  if (pending.length === 0) return null; // nothing to run — not a stall
7522
8369
 
7523
- const idleMs = Number.isFinite(lastRunAtMs) ? now - lastRunAtMs : Infinity;
8370
+ const idleMs = dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now });
7524
8371
  if (idleMs < thresholdMs) return null; // give the normal path its chance first
7525
8372
 
7526
8373
  // Which pending rows could actually dispatch? Anything NOT named by a
7527
- // blocked chain. computeBlockedChains already walks dependsOn with the
7528
- // picker's own resolution, so the two can never disagree.
8374
+ // blocked chain (computeBlockedChains walks dependsOn with the picker's own
8375
+ // resolution, so the two can never disagree) and NOT held by an open launch
8376
+ // breaker / the quietMachine lease (`heldSlugs` — a separate input, never
8377
+ // folded into the dependsOn walker). Held rows can't launch no matter how
8378
+ // often we tick, so they read as 'blocked' (needs a human), not 'starved'.
7529
8379
  const blockedChains = computeBlockedChains(rows);
7530
- const blockedTotal = blockedChains.reduce((n, c) => n + c.blocked, 0);
7531
- const dispatchable = pending.length - blockedTotal;
8380
+ const held = heldSlugs?.has ? heldSlugs : new Set(heldSlugs ?? []);
8381
+ const open = held.size > 0 ? rows.filter((j) => !(j.status === 'pending' && held.has(j.slug))) : rows;
8382
+ const openBlocked = held.size > 0 ? computeBlockedChains(open) : blockedChains;
8383
+ const openPending = open.filter((j) => j.status === 'pending').length;
8384
+ const dispatchable = openPending - openBlocked.reduce((n, c) => n + c.blocked, 0);
7532
8385
 
7533
8386
  return {
7534
8387
  kind: dispatchable > 0 ? 'starved' : 'blocked',
@@ -7551,14 +8404,14 @@ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now,
7551
8404
  * projects sat starved/blocked for hours, and the watchdog never fired once
7552
8405
  * because "work is flowing" was true somewhere else. Partitioning by cwd
7553
8406
  * (the same grouping computeBlockedChains already does) fixes DETECTION only
7554
- * — the idle clock (`lastRunAtMs`) stays machine-wide, since
7555
- * `lastDispatchAttemptAt` is machine-level state, and only one tick is ever
8407
+ * — the idle clock (see dispatchIdleMs) stays machine-wide, since
8408
+ * `lastRunAt` is machine-level state, and only one tick is ever
7556
8409
  * forced per watchdog pass regardless of how many cwds are starved.
7557
8410
  *
7558
8411
  * Pure, no IO. Returns [] when paused (a DECISION, not a stall) or when no
7559
8412
  * project has a verdict.
7560
8413
  */
7561
- function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
8414
+ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, heldSlugs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
7562
8415
  if (paused) return [];
7563
8416
  const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
7564
8417
  const byCwd = new Map();
@@ -7580,6 +8433,9 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
7580
8433
  paused: false,
7581
8434
  runningCount: projRunningCount,
7582
8435
  lastRunAtMs,
8436
+ lastPauseClearedAtMs,
8437
+ schedulerBootedAtMs,
8438
+ heldSlugs,
7583
8439
  now,
7584
8440
  thresholdMs,
7585
8441
  });
@@ -7590,7 +8446,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
7590
8446
 
7591
8447
  /**
7592
8448
  * classifyQueueHealth({ jobs, paused, launchBlocks, runningSet, freeSlots,
7593
- * totalSlots, lastDispatchAttemptAtMs, now, cwd, thresholdMs })
8449
+ * totalSlots, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, cwd, thresholdMs })
7594
8450
  * → { kind, cwd, pending, dispatchable, blockedChains, needsReviewCount, runningCount, ... }
7595
8451
  *
7596
8452
  * Single source of truth for the Scheduler page's queue-health header: the
@@ -7601,7 +8457,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
7601
8457
  *
7602
8458
  * Reuses classifyQueueStarvation for the blocked/stalled read so the header
7603
8459
  * can never disagree with runQueueStarvationWatchdog's own decision to force
7604
- * a tick: both are handed the same lastDispatchAttemptAt-based idle clock and
8460
+ * a tick: both are handed the same launch-keyed idle clock (dispatchIdleMs) and
7605
8461
  * the same computeBlockedChains walk under the hood. Called here with
7606
8462
  * `thresholdMs: 0` first (a live header must say "blocked" the instant every
7607
8463
  * pending row is dependency-stuck, not wait out the watchdog's own 10-minute
@@ -7638,7 +8494,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
7638
8494
  */
7639
8495
  function classifyQueueHealth({
7640
8496
  jobs, paused, launchBlocks, runningSet: runningSlugs, freeSlots, totalSlots,
7641
- lastDispatchAttemptAtMs, now, cwd = null, thresholdMs = QUEUE_STARVATION_MS,
8497
+ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, cwd = null, thresholdMs = QUEUE_STARVATION_MS,
7642
8498
  } = {}) {
7643
8499
  const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
7644
8500
  const projectJobs = cwd ? rows.filter((j) => j.cwd === cwd) : rows;
@@ -7688,19 +8544,22 @@ function classifyQueueHealth({
7688
8544
  // over the same rows), so `base` already carries them.
7689
8545
  const immediate = classifyQueueStarvation({
7690
8546
  jobs: projectJobs, paused: false, runningCount: 0,
7691
- lastRunAtMs: lastDispatchAttemptAtMs, now, thresholdMs: 0,
8547
+ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, thresholdMs: 0,
7692
8548
  });
7693
8549
  // pending.length is already > 0 above, so `immediate` can only be null when
7694
- // lastDispatchAttemptAtMs is itself in the future (clock skew) — fall back
7695
- // to computing idleMs the same way rather than asserting a kind we can't
8550
+ // the clock stamp is itself in the future (clock skew) — fall back to
8551
+ // computing idleMs the same way rather than asserting a kind we can't
7696
8552
  // back up with a real number.
7697
8553
  const idleMs = immediate ? immediate.idleMs
7698
- : (Number.isFinite(lastDispatchAttemptAtMs) ? now - lastDispatchAttemptAtMs : Infinity);
8554
+ : dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now });
7699
8555
  if (dispatchable === 0) return { ...base, kind: 'blocked', idleMs };
7700
8556
  const kind = idleMs >= thresholdMs ? 'stalled' : 'running';
7701
8557
  return { ...base, kind, idleMs };
7702
8558
  }
7703
8559
 
8560
+ // Per-cwd latch for runQueueStarvationWatchdog: cwd → { kind, at, running }.
8561
+ const starvationLatch = new Map();
8562
+
7704
8563
  /**
7705
8564
  * The watchdog half: acts on classifyQueueStarvationByProject. Called from
7706
8565
  * the heartbeat, which already runs on its own timer independent of the
@@ -7713,33 +8572,55 @@ function classifyQueueHealth({
7713
8572
  * the tick itself is machine-wide (it drives whatever the picker finds
7714
8573
  * across every project), only the DETECTION is per-project.
7715
8574
  */
7716
- async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
7717
- // lastDispatchAttemptAt, not lastRunAt: the latter only advances when a
7718
- // batch actually launches, so a poll that keeps succeeding while dispatch
7719
- // itself never gets invoked would otherwise mask a stall behind a fresh-
7720
- // looking timestamp that was never actually tracking dispatch liveness.
8575
+ async function runQueueStarvationWatchdog(state, {
8576
+ now = Date.now(), thresholdMs = QUEUE_STARVATION_MS,
8577
+ bootedAtMs = Date.parse(SCHEDULER_BOOTED_AT), pauseClearedAtMs = lastPauseClearedAt,
8578
+ } = {}) {
8579
+ // The idle clock is launch-keyed (dispatchIdleMs): NOT lastDispatchAttemptAt,
8580
+ // which tickQueue stamps before every gate — a queue whose 30 s loop ticks
8581
+ // and launches nothing would refresh it forever and the watchdog would
8582
+ // never fire. Rows held by an open launch breaker or the quietMachine lease
8583
+ // can't launch however often we tick, so they're passed in as `heldSlugs`
8584
+ // and read as 'blocked' (needs a human) rather than a false 'starved'.
8585
+ const heldSlugs = new Set((await computeLaunchHolds(state)).keys());
8586
+ if (quietMachineLease.isHeld()) {
8587
+ for (const j of state?.jobs ?? []) if (j?.status === 'pending' && j.quietMachine === true) heldSlugs.add(j.slug);
8588
+ }
7721
8589
  const verdicts = classifyQueueStarvationByProject({
7722
8590
  jobs: state?.jobs,
7723
- paused: state?.paused,
8591
+ paused: upgradeDrain.effectivePaused(state),
7724
8592
  runningSet,
7725
- lastRunAtMs: Date.parse(state?.lastDispatchAttemptAt ?? ''),
8593
+ lastRunAtMs: Date.parse(state?.lastRunAt ?? ''),
8594
+ lastPauseClearedAtMs: pauseClearedAtMs,
8595
+ schedulerBootedAtMs: bootedAtMs,
8596
+ heldSlugs,
7726
8597
  now,
7727
8598
  thresholdMs,
7728
8599
  });
8600
+ // Latch: one episode per (cwd, kind) — re-arms only once QUEUE_STARVATION_MS
8601
+ // has elapsed again or the project's running count changes; forgotten the
8602
+ // moment the cwd stops having a verdict at all.
8603
+ const activeKeys = new Set(verdicts.map((v) => v.cwd));
8604
+ for (const cwd of [...starvationLatch.keys()]) if (!activeKeys.has(cwd)) starvationLatch.delete(cwd);
7729
8605
  if (verdicts.length === 0) return null;
7730
8606
 
7731
8607
  let anyStarved = false;
7732
8608
  let primary = null;
7733
8609
  for (const verdict of verdicts) {
7734
8610
  const mins = Math.round(verdict.idleMs / 60_000);
8611
+ const running = (state?.jobs ?? []).filter((j) => j?.cwd === verdict.cwd && (j.status === 'running' || runningSet.has(j.slug))).length;
8612
+ const latched = starvationLatch.get(verdict.cwd);
8613
+ const suppressed = !!latched && latched.kind === verdict.kind && latched.running === running && now - latched.at < QUEUE_STARVATION_MS;
8614
+ if (!primary) primary = verdict;
8615
+ if (suppressed) continue;
8616
+ starvationLatch.set(verdict.cwd, { kind: verdict.kind, at: now, running });
7735
8617
  if (verdict.kind === 'blocked') {
7736
8618
  console.warn(
7737
8619
  `[scheduler] QUEUE BLOCKED (${verdict.cwd}): ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
7738
- + `terminal or parked dependency, so ticking cannot help. Blockers: `
8620
+ + `terminal or parked dependency, an open launch breaker, or the quietMachine lease, so ticking cannot help. Blockers: `
7739
8621
  + verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
7740
8622
  );
7741
8623
  appendAuditEvent('queue_blocked_stall', { cwd: verdict.cwd, pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
7742
- if (!primary) primary = verdict;
7743
8624
  continue;
7744
8625
  }
7745
8626
 
@@ -7756,9 +8637,12 @@ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs
7756
8637
 
7757
8638
  // A never-populated utilization reading is itself one of the ways the
7758
8639
  // when-available path silently never fires (maybeLaunchWhenAvailable
7759
- // returns early on null). Treat unknown as safe here, exactly as the
7760
- // billing meter's own 429 fallback already does.
7761
- if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
8640
+ // returns early on null). Absence of information, not a green light: fall
8641
+ // back to the same conservative degraded budget the poll loop itself uses
8642
+ // rather than a blind cachedUtilization=0.
8643
+ if (cachedUtilization === null || cachedUtilization === undefined) {
8644
+ applyDegradedBudget();
8645
+ }
7762
8646
  // The in-process cancelToken is only ever reset by runDueJobs() (force-tick
7763
8647
  // / run-now / resume-timer) — every other path that clears a pause
7764
8648
  // (clearPause(), the poll loop's own auto-recovery) leaves it untouched
@@ -7924,6 +8808,72 @@ async function runBranchSweep(jobs) {
7924
8808
  * skipped too (spawn may still be mid-flight) — see selectReapableJobs for
7925
8809
  * the full predicate. Exported so unit tests can invoke it directly.
7926
8810
  */
8811
+ // Adopted-run supervision (see lib/adoptedRunSupervisor.cjs). In-memory by
8812
+ // design: the supervisors die with this process, and the next boot re-arms.
8813
+ const adoptedSupervisors = new Map();
8814
+
8815
+ /** Pid of a boot-time running row via the same ladder isBootRowAlive uses:
8816
+ * supervisor record → runtime.pid → pid spawned per the run log. */
8817
+ function bootRowPid(j, logPathOf) {
8818
+ const runDir = j.runId ? path.join(schedulerPaths.runsDir(), j.runId) : null;
8819
+ const record = runDir ? supervisorRecord.readSupervisorRecord(runDir, j.slug) : null;
8820
+ return record?.pid || j.runtime?.pid || readSpawnedPidFromLog(logPathOf(j)) || null;
8821
+ }
8822
+
8823
+ function signalAdoptedGroup(pgid, signal, pid) {
8824
+ try { process.kill(-pgid, signal); } catch {
8825
+ try { process.kill(pid, signal); } catch { /* already dead */ }
8826
+ }
8827
+ }
8828
+
8829
+ async function superviseAdoptedRunsPass(jobs) {
8830
+ try {
8831
+ const rows = jobs || (await readQueue()).jobs;
8832
+ const runDirOf = (j) => (j.runId ? path.join(schedulerPaths.runsDir(), j.runId) : null);
8833
+ return await adoptedRunSupervisor.superviseAdoptedRuns(rows, {
8834
+ registry: adoptedSupervisors,
8835
+ runDir: runDirOf,
8836
+ readRecord: supervisorRecord.readSupervisorRecord,
8837
+ lease: quietMachineLease,
8838
+ markSupervised: async (row) => {
8839
+ await mutate((s) => {
8840
+ const j = s.jobs.find((x) => x.slug === row.slug);
8841
+ if (j && j.status === 'running' && (j.runId ?? null) === (row.runId ?? null)) j.supervisedAt = new Date().toISOString();
8842
+ });
8843
+ },
8844
+ makeDeps: (row, record) => {
8845
+ const logPath = path.join(runDirOf(row), `${row.slug}.log`);
8846
+ return {
8847
+ logPath,
8848
+ statLogMtimeMs: readLogMtimeMs,
8849
+ pidAlive: claudePidAlive,
8850
+ identityOf: procIdentityOf,
8851
+ isDifferentProcess,
8852
+ killGroup: signalAdoptedGroup,
8853
+ // Stamped BEFORE the signal so reapDeadRunningJobs, which finalizes
8854
+ // the row once the process is gone, always sees why it died.
8855
+ stampKill: async (kind, reason) => {
8856
+ await mutate((s) => {
8857
+ const j = s.jobs.find((x) => x.slug === row.slug);
8858
+ if (j && j.status === 'running' && (j.runId ?? null) === (row.runId ?? null)) {
8859
+ j.adoptedKill = { watchdog: kind, reason, at: new Date().toISOString() };
8860
+ }
8861
+ });
8862
+ try { fs.appendFileSync(logPath, `\n[scheduler] adopted-run ${kind} watchdog: ${reason}\n`); } catch { /* best-effort */ }
8863
+ },
8864
+ log: (msg) => console.log(`[scheduler] ${row.slug}: ${msg}`),
8865
+ checkIntervalMs: IDLE_CHECK_INTERVAL_MS,
8866
+ sigkillAfterMs: POST_RESULT_KILL_MS,
8867
+ defaultMaxDurationMs: MAX_JOB_DURATION_MS,
8868
+ };
8869
+ },
8870
+ });
8871
+ } catch (e) {
8872
+ console.warn('[scheduler] adopted-run supervision pass failed', e?.message);
8873
+ return [];
8874
+ }
8875
+ }
8876
+
7927
8877
  async function reapDeadRunningJobs() {
7928
8878
  try {
7929
8879
  // Do NOT gate on runningSet: spawnJob()'s finally block unconditionally
@@ -7932,9 +8882,13 @@ async function reapDeadRunningJobs() {
7932
8882
  // status:"running" with no slug left in runningSet to trigger reconciliation.
7933
8883
  // queue.json is the source of truth for which jobs are actually running.
7934
8884
  const state = await readQueue();
8885
+ // A quarantined shard's rows never loaded; the filter is defence in depth
8886
+ // so a reap can never terminalize a row of a project we cannot persist.
8887
+ const reapSkip = quarantinedCwdSet(state);
8888
+ if (reapSkip.size > 0) state.jobs = state.jobs.filter((j) => !reapSkip.has(j.cwd));
7935
8889
  // Shared by the log-evidence injections below and the reapable-processing
7936
8890
  // loop further down — same `j.runId` → run log path formula either way.
7937
- const logPathForJob = (j) => (j?.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null);
8891
+ const logPathForJob = (j) => (j?.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null);
7938
8892
  const { reapable, warnings, recovered } = selectReapableJobs(state.jobs, Date.now(), {
7939
8893
  pidAlive: claudePidAlive,
7940
8894
  grace: PIDLESS_SPAWN_GRACE_MS,
@@ -8008,8 +8962,12 @@ async function reapDeadRunningJobs() {
8008
8962
  // the same still-active rate limit — the spin loop this PRD exists to
8009
8963
  // stop. Done once, outside mutate(), before finalizing any row below.
8010
8964
  if (dead.some((d) => d.outcome === 'rate_limited')) {
8965
+ // Same rationale as spawnJob's own rateLimited branch: a dead-process
8966
+ // reap that classifies as rate-limited is just as much an executor-
8967
+ // observed 429 as a live one, and must open the same shared circuit.
8968
+ billing.usageCircuit.recordFailure('executor_429');
8011
8969
  const triggering = dead.find((d) => d.outcome === 'rate_limited');
8012
- const billingResetIso = await refreshNextReset().catch(() => cachedNextReset);
8970
+ const billingResetIso = await billingResetForPause();
8013
8971
  const resetIso = resolveRateLimitPauseReset(triggering.logPath, billingResetIso);
8014
8972
  const triggeringRow = triggering ? state.jobs.find((x) => x.slug === triggering.slug) : null;
8015
8973
  const observedAtMs = triggeringRow?.startedAt ? Date.parse(triggeringRow.startedAt) : null;
@@ -8115,9 +9073,10 @@ async function reapDeadRunningJobs() {
8115
9073
  // this runs the whole dead-job batch concurrently rather than one
8116
9074
  // dispatch's git-spawn latency at a time.
8117
9075
  await Promise.all(dead.map(async (d) => {
8118
- if (d.outcome === 'rate_limited' || d.outcome === 'success') return;
8119
- if (d.pidless && d.failureOverride) return;
8120
9076
  const row = state.jobs.find((x) => x.slug === d.slug);
9077
+ const adoptedBudgetKill = row?.adoptedKill?.watchdog === 'budget';
9078
+ if (d.outcome === 'rate_limited' || (d.outcome === 'success' && !adoptedBudgetKill)) return;
9079
+ if (d.pidless && d.failureOverride) return;
8121
9080
  if (!row?.landedCommit) return;
8122
9081
  const rowCwd = row.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD;
8123
9082
  const resolved = await resolveLandedCommitEvidence(rowCwd, row.landedCommit, row.startedAt);
@@ -8150,7 +9109,7 @@ async function reapDeadRunningJobs() {
8150
9109
  const baseSet = new Set(s.jobs[idx].guardBaseline);
8151
9110
  deltaPaths = after.filter((p) => !baseSet.has(p));
8152
9111
  if (deltaPaths.length) {
8153
- const salvagePath = path.join(RUNS_DIR, s.jobs[idx].runId, `${slug}.uncommitted.patch`);
9112
+ const salvagePath = path.join(schedulerPaths.runsDir(), s.jobs[idx].runId, `${slug}.uncommitted.patch`);
8154
9113
  const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: rowCwd, paths: deltaPaths, outFile: salvagePath });
8155
9114
  if (salvage && salvage.ok) {
8156
9115
  s.jobs[idx].salvagePatch = salvagePath;
@@ -8198,6 +9157,17 @@ async function reapDeadRunningJobs() {
8198
9157
  if (pidless && !effectiveSuccess && !rateLimited && !notLandedInfo && failureOverride) {
8199
9158
  notLandedInfo = { verdict: failureOverride.verdict, reason: failureOverride.reason };
8200
9159
  }
9160
+ // An adopted run the budget watchdog killed (adoptedKill, stamped
9161
+ // before the signal) parks exactly like a native budget kill: the
9162
+ // shared classifyBudgetKill decides, and it wins over success/failure.
9163
+ const adoptedBudgetKill = rateLimited ? null : classifyBudgetKill({
9164
+ killedByWatchdog: s.jobs[idx].adoptedKill?.watchdog,
9165
+ budgetKillReason: s.jobs[idx].adoptedKill?.reason,
9166
+ }, landedCommitEvidence.get(slug) || null);
9167
+ if (adoptedBudgetKill) {
9168
+ effectiveSuccess = false;
9169
+ notLandedInfo = { verdict: 'budget_exceeded', reason: adoptedBudgetKill.reason };
9170
+ }
8201
9171
 
8202
9172
  const leftoverSuffix = deltaPaths && deltaPaths.length
8203
9173
  ? ` — left ${deltaPaths.length} files uncommitted`
@@ -8249,6 +9219,7 @@ async function reapDeadRunningJobs() {
8249
9219
  s.jobs[idx].gateOutcome = gateOutcome;
8250
9220
  if (confirmedLandedCommit) s.jobs[idx].landedCommit = confirmedLandedCommit;
8251
9221
  if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
9222
+ if (adoptedBudgetKill?.landedCommit) s.jobs[idx].landedCommit = adoptedBudgetKill.landedCommit;
8252
9223
  }
8253
9224
  // A pidless spawn that never wrote its own '<slug>.log' into the
8254
9225
  // batch runId dir it was stamped with must not keep that runId —
@@ -8263,6 +9234,7 @@ async function reapDeadRunningJobs() {
8263
9234
  s.jobs[idx].runId = null;
8264
9235
  }
8265
9236
  delete s.jobs[idx].runtime;
9237
+ delete s.jobs[idx].adoptedKill;
8266
9238
  delete s.jobs[idx].dispatchPhase;
8267
9239
  delete s.jobs[idx].dispatchPhaseAt;
8268
9240
  delete s.jobs[idx].overrun;
@@ -8327,14 +9299,21 @@ async function pollLoop() {
8327
9299
  // 404/time-out and eventually pause the queue on 'network' — treat usage as
8328
9300
  // wide-open and fire on pending + memory alone. (Blackrock-style machines.)
8329
9301
  if (!billing.usageMeterApplicable()) {
8330
- cachedUtilization = 0;
9302
+ // Close the shared circuit if a PRIOR consumer-auth session left it
9303
+ // open/half_open — this process has stopped polling the meter
9304
+ // entirely, so nothing else will ever call recordSuccess() to clear
9305
+ // it, and health.cjs would otherwise read a stale open circuit as
9306
+ // YELLOW/RED forever even though nothing is actually degraded.
9307
+ if (billing.usageCircuit.state() !== 'closed') billing.usageCircuit.recordSuccess({});
9308
+ cachedUtilization = NO_METER_UTILIZATION;
9309
+ degradedConcurrencyCapValue = null;
8331
9310
  consecutiveFailures = 0;
8332
9311
  backoffMs = 0;
8333
9312
  backoffNextAt = null;
8334
9313
  firstFailureAt = null;
8335
9314
  firstNon429FailureAt = null;
8336
9315
  lastFailureKind = null;
8337
- failureStreakWarned = false;
9316
+ resetFailureStreak();
8338
9317
  lastPollAt = Date.now();
8339
9318
  lastPollOk = true;
8340
9319
  persistSchedulerState();
@@ -8350,18 +9329,40 @@ async function pollLoop() {
8350
9329
  return; // finally re-arms the timer
8351
9330
  }
8352
9331
 
9332
+ // Shared breaker over the meter (AC1): while it is OPEN, no request is
9333
+ // made except the half-open probe below (billing.fetchUsage() is only
9334
+ // ever reached, further down, from the closed/half_open paths). "Meter
9335
+ // down" reads as absence of information, not a green light — the
9336
+ // conservative degraded budget stands in for both the utilization-
9337
+ // threshold gate (maybeLaunchWhenAvailable) and the concurrency cap
9338
+ // (tickQueue's freeSlots), never a blind cachedUtilization=0.
9339
+ if (billing.usageCircuit.state() === 'open') {
9340
+ applyDegradedBudget();
9341
+ lastPollAt = Date.now();
9342
+ lastPollOk = false;
9343
+ warnFailureStreakIfNeeded();
9344
+ persistSchedulerState();
9345
+ const cur = await readQueue();
9346
+ await maybeLaunchWhenAvailable(cur);
9347
+ await broadcast();
9348
+ return;
9349
+ }
9350
+
8353
9351
  const r = await billing.fetchUsage();
8354
9352
 
8355
9353
  if (r.kind === 'ok') {
8356
- cachedNextReset = r.data?.usage?.five_hour?.resets_at ?? cachedNextReset;
8357
- cachedUtilization = r.data?.usage?.five_hour?.utilization ?? cachedUtilization;
9354
+ const window = bindingWindow(r.data?.usage);
9355
+ recordObservedReset(window.resets_at ?? null);
9356
+ cachedUtilization = Number.isFinite(window.utilization) ? window.utilization : cachedUtilization;
9357
+ lastGoodUsagePayload = r.data?.usage ?? lastGoodUsagePayload;
9358
+ degradedConcurrencyCapValue = null;
8358
9359
  consecutiveFailures = 0;
8359
9360
  backoffMs = 0;
8360
9361
  backoffNextAt = null;
8361
9362
  firstFailureAt = null;
8362
9363
  firstNon429FailureAt = null;
8363
9364
  lastFailureKind = null;
8364
- failureStreakWarned = false;
9365
+ resetFailureStreak();
8365
9366
  lastPollAt = Date.now();
8366
9367
  lastPollOk = true;
8367
9368
  persistSchedulerState();
@@ -8377,14 +9378,14 @@ async function pollLoop() {
8377
9378
  await maybeLaunchWhenAvailable(cur);
8378
9379
  await broadcast();
8379
9380
  } else if (r.kind === 'meter_rate_limited') {
8380
- // Billing meter is itself being rate-limited. Treat as "utilization unknown but safe":
8381
- // fire available jobs anyway at utilization=0 rather than pausing the queue.
8382
- // Still back off the POLL cadence itself (same curve/cap as the transient
8383
- // branch) and persist state every cycle — without this, a sustained 429
8384
- // streak hammered the already-rate-limited endpoint every POLL_INTERVAL_MS
8385
- // forever AND never wrote lastPollAt/consecutiveFailures back to
8386
- // scheduler-state.json, so the sidecar froze stale while the loop kept
8387
- // failing silently underneath it (the 57-consecutive-failure incident).
9381
+ // Billing meter is itself being rate-limited — absence of information,
9382
+ // not a green light. Still back off the POLL cadence itself (same
9383
+ // curve/cap as the transient branch) and persist state every cycle —
9384
+ // without this, a sustained 429 streak hammered the already-rate-
9385
+ // limited endpoint every POLL_INTERVAL_MS forever AND never wrote
9386
+ // lastPollAt/consecutiveFailures back to scheduler-state.json, so the
9387
+ // sidecar froze stale while the loop kept failing silently underneath
9388
+ // it (the 57-consecutive-failure incident).
8388
9389
  lastPollAt = Date.now();
8389
9390
  lastPollOk = false;
8390
9391
  consecutiveFailures++;
@@ -8392,8 +9393,8 @@ async function pollLoop() {
8392
9393
  // Don't update firstNon429FailureAt — 429s don't count toward the 30-min network-pause threshold.
8393
9394
  backoffMs = nextBackoffMs(backoffMs);
8394
9395
  backoffNextAt = Date.now() + backoffMs;
8395
- cachedUtilization = 0; // assume safe; fire any pending work
8396
- console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on heuristic (failure #${consecutiveFailures}); retry in ${backoffMs / 1000}s`);
9396
+ applyDegradedBudget();
9397
+ console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on degraded budget (util=${cachedUtilization}%, cap=${degradedConcurrencyCapValue}) (failure #${consecutiveFailures}); retry in ${backoffMs / 1000}s`);
8397
9398
  warnFailureStreakIfNeeded();
8398
9399
  persistSchedulerState();
8399
9400
  const cur = await readQueue();
@@ -8433,13 +9434,12 @@ async function pollLoop() {
8433
9434
  // 'ok' and 'meter_rate_limited' branches used to reach
8434
9435
  // maybeLaunchWhenAvailable, so auth/transient failures left ready
8435
9436
  // pending work untouched until either the queue-starvation watchdog's
8436
- // 10-minute safety net fired or the poll itself recovered. Utilization
8437
- // is unknown during a failed poll, not unsafe — treated the same way
8438
- // the meter_rate_limited branch above already treats a 429 as safe to
8439
- // fire through. maybeLaunchWhenAvailable itself still honors an
8440
- // 'auth'/'network' pause (state.paused), so this is a no-op whenever
8441
- // setPaused() above actually engaged one.
8442
- if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
9437
+ // 10-minute safety net fired or the poll itself recovered. Absence of
9438
+ // information, not a green light: fall back to the degraded budget
9439
+ // rather than a blind cachedUtilization=0. maybeLaunchWhenAvailable
9440
+ // itself still honors an 'auth'/'network' pause (state.paused), so
9441
+ // this is a no-op whenever setPaused() above actually engaged one.
9442
+ applyDegradedBudget();
8443
9443
  await maybeLaunchWhenAvailable(await readQueue());
8444
9444
  await broadcast();
8445
9445
  }
@@ -8458,7 +9458,7 @@ async function pollLoop() {
8458
9458
  // Same rationale as the auth/transient branch above: the outer catch
8459
9459
  // must not be a silent dispatch dead-end either.
8460
9460
  try {
8461
- if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
9461
+ applyDegradedBudget();
8462
9462
  await maybeLaunchWhenAvailable(await readQueue());
8463
9463
  await broadcast();
8464
9464
  } catch { /* best-effort — the poll loop must still re-arm below */ }
@@ -8521,6 +9521,34 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
8521
9521
  // anyway). For non-fix-plan jobs the exemption never applies, so rescanning
8522
9522
  // their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
8523
9523
  const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'abandoned_background_task', 'pass_no_commit', 'pass_no_commit_already_shipped']);
9524
+ // RESCANNABLE_VERDICTS is a HINT, not a gate: it names the verdicts whose
9525
+ // recovery rung is a transcript re-verification (verifyRun). Every other
9526
+ // needs_review verdict is still a heal candidate (isRescanCandidate) — it just
9527
+ // gets the evidence-only rung (computeLooksDone) instead of a transcript
9528
+ // rescan, because verifyRun cannot see a commit-guard / shared-tree verdict and
9529
+ // would return 'clean' and falsely heal it.
9530
+
9531
+ // The ONLY needs_review verdicts NOT eligible for the periodic heal ladder.
9532
+ // An allow-list here was reopened three times (2026-09-12 x2, 2026-09-18
9533
+ // shared_tree_reverted) because a new park reason was born invisible to
9534
+ // self-healing. Add a verdict here only with a one-line proof that no
9535
+ // re-verification or evidence scan can ever change it.
9536
+ const RESCAN_EXCLUDED_VERDICTS = new Set([
9537
+ // Commit-guard verdict verifyRun never inspects: a rescan returns 'clean' and would heal genuinely unfinished work.
9538
+ 'uncommitted_changes',
9539
+ // Its damage IS a commit stranded on an unmerged sm-job branch — a landedCommit restates it; selectMechanicalRecoveryTarget owns the real re-merge.
9540
+ 'worktree_integration_failed',
9541
+ // The run overran its own time/cost estimate; no transcript or git evidence can un-overrun it (selectAutoFixTargets excludes it too).
9542
+ 'budget_exceeded',
9543
+ ]);
9544
+
9545
+ // Per-pass / per-row bounds on the evidence-only rung (the widened candidate
9546
+ // set). Each scan costs one computeLooksDone: a per-cwd-deduped `git fetch`
9547
+ // (<=~20s) + a git log. Unbounded, a backlog of N parked rows would pay N of
9548
+ // those every 10 minutes forever.
9549
+ const REVERIFY_INTERVAL_MS = 10 * 60_000;
9550
+ const EVIDENCE_SCAN_MAX_PER_PASS = 20;
9551
+ const EVIDENCE_SCAN_MIN_INTERVAL_MS = 6 * REVERIFY_INTERVAL_MS;
8524
9552
 
8525
9553
  // Bounds fix-plan recursion: cap N permits at most N+1 fix jobs per original
8526
9554
  // slug (depth 1 = the original job, depth 2 = its `-fix`, depth 3+ is
@@ -8563,7 +9591,7 @@ function isFixPlanBeyondDepthCap(slug, investigationDepth, isFixPlan) {
8563
9591
  * check, no nested loop over user-scaled data. Dir names are ISO timestamps,
8564
9592
  * so lexical-descending sort picks the newest match. Exported for tests.
8565
9593
  */
8566
- function resolveRunId(job, { runsDir = RUNS_DIR } = {}) {
9594
+ function resolveRunId(job, { runsDir = schedulerPaths.runsDir() } = {}) {
8567
9595
  if (!job || job.runId) return job?.runId || null;
8568
9596
  if (!job.slug) return null;
8569
9597
  let dirs;
@@ -8665,18 +9693,71 @@ function isGuardParkedWithoutAutoFix(job) {
8665
9693
  return GUARD_VERDICT_EVIDENCE_ELIGIBLE.has(job.verifierVerdict);
8666
9694
  }
8667
9695
 
9696
+ /**
9697
+ * Pure predicate, no I/O: a needs_review row whose auto-fix investigation
9698
+ * genuinely ran (autoFixAttempted === true) but whose outcome was NEVER
9699
+ * durably stamped at all — and that has nothing left in flight to wait on:
9700
+ * no live/queued fix-plan row at fixSlugFor(job).
9701
+ *
9702
+ * Distinct from isExhaustedAutoFix, which requires autoFixRetries >= 1 to
9703
+ * have already accumulated. spawnInvestigation's onExit handler restores the
9704
+ * job's status from 'investigating' back to needs_review in ONE mutate()
9705
+ * call (scheduler.cjs's spawnInvestigation, source
9706
+ * 'spawnInvestigation:onExit') and stamps autoFixOutcome ('plan' / 'no-plan'
9707
+ * / 'error') in a SEPARATE, later mutate() call — an app restart or process
9708
+ * death between the two leaves autoFixOutcome permanently unset, with
9709
+ * autoFixRetries never incremented either, so isExhaustedAutoFix never fires
9710
+ * and the row falls through every existing resolving door forever, re-scanned
9711
+ * by the periodic reverify pass against the same frozen transcript with no
9712
+ * new outcome to observe.
9713
+ *
9714
+ * Job 1218-fo-01 (2026-09-13, findings filed at
9715
+ * session-manager-operations/reviews/2026-09-13-scheduler-stability-investigation.md,
9716
+ * "post-run adjudication" section) sat exactly in this state: needs_review,
9717
+ * verifierVerdict transcript_errors, autoFixAttempted: true, autoFixOutcome:
9718
+ * undefined, autoFixRetries: undefined, statusHistory ending in
9719
+ * "investigation probe exited — restoring prior status" — with a landed
9720
+ * commit no existing ladder rung would credit.
9721
+ *
9722
+ * Deliberately narrower than "unset, 'error', or 'no-plan'": a row that DID
9723
+ * get a durably-stamped 'error'/'no-plan' outcome with its one bounded retry
9724
+ * still unspent (autoFixRetries < 1) is exactly the row
9725
+ * selectAutoFixTargets's own retryEligible check still owns and will retry
9726
+ * on its own — pulling it into THIS ladder instead would race it away from
9727
+ * that retry (scheduler-needs-review-autoresolve.test.cjs's "a non-exhausted
9728
+ * needs_review row … is left alone" guards exactly this). Only the
9729
+ * outcome-truly-never-stamped case is structurally unrecoverable by any
9730
+ * OTHER existing door, because nothing ever wrote a value selectAutoFixTargets
9731
+ * or isExhaustedAutoFix could act on.
9732
+ * Exported for tests.
9733
+ */
9734
+ function isStrandedAutoFixPark(job, jobsInProject) {
9735
+ if (!job || job.status !== 'needs_review') return false;
9736
+ if (job.autoFixAttempted !== true) return false;
9737
+ if (job.autoFixOutcome != null) return false;
9738
+ const fixSlug = fixSlugFor(job);
9739
+ const liveOrQueuedChild = (jobsInProject || []).some(
9740
+ (j) => j.slug === fixSlug && j.status !== 'completed' && !DEAD_FIX_CHILD_STATUSES.has(j.status),
9741
+ );
9742
+ return !liveOrQueuedChild;
9743
+ }
9744
+
8668
9745
  /**
8669
9746
  * Pure predicate, no I/O: is this needs_review row eligible for the bounded
8670
9747
  * auto-resolve ladder at all — either because its auto-fix path is genuinely
8671
- * spent (isExhaustedAutoFix), or because it was parked by a GUARD verdict
8672
- * that never entered auto-fix in the first place (isGuardParkedWithoutAutoFix).
8673
- * Both classes share ONE ladder (applyNeedsReviewAutoResolve) rather than a
8674
- * duplicated one — the ladder itself doesn't care which door a row came
8675
- * through, only whether it now carries completion evidence (job.looksDone).
9748
+ * spent (isExhaustedAutoFix), because it was parked by a GUARD verdict that
9749
+ * never entered auto-fix in the first place (isGuardParkedWithoutAutoFix),
9750
+ * or because its auto-fix investigation ran but was stranded before
9751
+ * recording any outcome (isStrandedAutoFixPark). All three classes share ONE
9752
+ * ladder (applyNeedsReviewAutoResolve) rather than a duplicated one — the
9753
+ * ladder itself doesn't care which door a row came through, only whether it
9754
+ * now carries completion evidence (job.looksDone). `jobsInProject` is only
9755
+ * consulted by isStrandedAutoFixPark (to check for a live/queued fix-plan
9756
+ * child) and defaults to empty so existing single-arg callers are unaffected.
8676
9757
  * Exported for tests.
8677
9758
  */
8678
- function isEligibleForNeedsReviewAutoResolve(job) {
8679
- return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job);
9759
+ function isEligibleForNeedsReviewAutoResolve(job, jobsInProject = []) {
9760
+ return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job) || isStrandedAutoFixPark(job, jobsInProject);
8680
9761
  }
8681
9762
 
8682
9763
  /**
@@ -8789,18 +9870,55 @@ function isFailedUnverifiedShaped(job) {
8789
9870
  if (job.verifierVerdict && RESCANNABLE_VERDICTS.has(job.verifierVerdict)) return true;
8790
9871
  const runId = job.runId || resolveRunId(job);
8791
9872
  if (!runId) return false;
8792
- const logPath = path.join(RUNS_DIR, runId, `${job.slug}.log`);
9873
+ const logPath = path.join(schedulerPaths.runsDir(), runId, `${job.slug}.log`);
8793
9874
  return classifyRunOutcome(logPath) === 'no_result';
8794
9875
  }
8795
9876
 
8796
9877
  function isRescanCandidate(job) {
8797
9878
  if (!job) return false;
9879
+ // Default-ELIGIBLE: every needs_review row is a heal candidate unless its
9880
+ // verdict is in RESCAN_EXCLUDED_VERDICTS. No runId requirement here — a row
9881
+ // without one still gets the evidence rung and the unresolvable annotation.
9882
+ if (job.status === 'needs_review') return !RESCAN_EXCLUDED_VERDICTS.has(job.verifierVerdict);
8798
9883
  if (!(job.runId || resolveRunId(job))) return false;
8799
- if (job.status === 'needs_review') return RESCANNABLE_VERDICTS.has(job.verifierVerdict);
8800
9884
  if (job.status === 'failed') return isFailedUnverifiedShaped(job);
8801
9885
  return false;
8802
9886
  }
8803
9887
 
9888
+ /**
9889
+ * Which rung a needs_review candidate gets (RESCANNABLE_VERDICTS as a hint):
9890
+ * true = transcript re-verification (needs a run dir to read); false = the
9891
+ * evidence-only rung. I/O only when a rescannable-verdict row lacks a runId.
9892
+ */
9893
+ function isTranscriptRescannable(job) {
9894
+ return !!job && RESCANNABLE_VERDICTS.has(job.verifierVerdict) && !!(job.runId || resolveRunId(job));
9895
+ }
9896
+
9897
+ /**
9898
+ * Pure, no I/O: the bounded subset of evidence-only needs_review candidates
9899
+ * reverifyNeedsReview scans this pass. Skips rows already carrying looksDone,
9900
+ * rows scanned within EVIDENCE_SCAN_MIN_INTERVAL_MS (evidenceScannedAt), and —
9901
+ * PRD 1136 — rows with a live auto-fix history unless they are an
9902
+ * auto-resolve door (isEligibleForNeedsReviewAutoResolve, which is what
9903
+ * consumes looksDone). Never-scanned rows go first, then least-recently
9904
+ * scanned; capped at EVIDENCE_SCAN_MAX_PER_PASS. O(n log n) in needs_review rows.
9905
+ * Per-pass cost ceiling: EVIDENCE_SCAN_MAX_PER_PASS computeLooksDone calls.
9906
+ */
9907
+ function selectEvidenceScanTargets(jobs, now = Date.now()) {
9908
+ const due = [];
9909
+ for (const j of jobs ?? []) {
9910
+ if (j.status !== 'needs_review' || !isRescanCandidate(j)) continue;
9911
+ if (isTranscriptRescannable(j)) continue;
9912
+ if (j.looksDone) continue;
9913
+ if (j.autoFixAttempted === true && !isEligibleForNeedsReviewAutoResolve(j, jobs)) continue;
9914
+ const last = Date.parse(j.evidenceScannedAt ?? '');
9915
+ if (!Number.isNaN(last) && now - last < EVIDENCE_SCAN_MIN_INTERVAL_MS) continue;
9916
+ due.push({ j, last: Number.isNaN(last) ? 0 : last });
9917
+ }
9918
+ due.sort((a, b) => a.last - b.last);
9919
+ return due.slice(0, EVIDENCE_SCAN_MAX_PER_PASS).map((d) => d.j);
9920
+ }
9921
+
8804
9922
  /**
8805
9923
  * Cheap-guard for the 10-minute periodic reverify tick. MUST be expressed in
8806
9924
  * terms of isRescanCandidate — not a hand-written status test — because the
@@ -8834,6 +9952,14 @@ function isRescanCandidate(job) {
8834
9952
  * that function). Same rule as always: never let this guard be narrower than
8835
9953
  * the work reverifyNeedsReview actually performs.
8836
9954
  *
9955
+ * Reopened a THIRD time 2026-09-18 (shared_tree_reverted parked 1229-fo-03
9956
+ * falsely, 19 of 20 pending rows held): the fix was not another OR-clause but
9957
+ * inverting the default — isRescanCandidate is now default-ELIGIBLE for every
9958
+ * needs_review row (RESCAN_EXCLUDED_VERDICTS names the few exceptions), so a
9959
+ * new park reason can never again be born unhealable. The OR-clauses below
9960
+ * are now redundant for needs_review rows and kept only for their
9961
+ * non-needs_review inputs.
9962
+ *
8837
9963
  * Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget and
8838
9964
  * isGuardParkedWithoutAutoFix are pure (no I/O). selectAutoFixTargets is
8839
9965
  * called with an injected fixSlugExists that always returns false — cheap
@@ -8845,7 +9971,7 @@ function isRescanCandidate(job) {
8845
9971
  */
8846
9972
  function shouldRunPeriodicReverify(jobs) {
8847
9973
  if (!Array.isArray(jobs)) return false;
8848
- if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j))) return true;
9974
+ if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j) || isStaleSharedTreeRevertedPark(j))) return true;
8849
9975
  if (jobs.some((j) => selectMechanicalRecoveryTarget(j) || selectResumeRecoveryTarget(j))) return true;
8850
9976
  return selectAutoFixTargets(jobs, { fixSlugExists: () => false }).length > 0;
8851
9977
  }
@@ -9012,11 +10138,12 @@ function needsReviewAutoResolveDisabled() {
9012
10138
  * [{ slug, cwd, ageMs, attempts }]
9013
10139
  *
9014
10140
  * Pure selector — no IO. Selects `needs_review` rows eligible for the
9015
- * bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve — either
9016
- * auto-fix genuinely spent, or parked by a GUARD verdict that never entered
9017
- * auto-fix at all), whose newest statusHistory entry with `to ===
9018
- * 'needs_review'` is older than `thresholdMs`, and whose
9019
- * exhaustedResolveAttempts counter has not yet spent its cap.
10141
+ * bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve — auto-fix
10142
+ * genuinely spent, parked by a GUARD verdict that never entered auto-fix at
10143
+ * all, or a stranded auto-fix park with no outcome ever recorded), whose
10144
+ * newest statusHistory entry with `to === 'needs_review'` is older than
10145
+ * `thresholdMs`, and whose exhaustedResolveAttempts counter has not yet
10146
+ * spent its cap.
9020
10147
  *
9021
10148
  * The inclusion bound is inclusive of the cap itself (`<= CAP`, not `<
9022
10149
  * CAP`): NEEDS_REVIEW_RESOLVE_CAP counts REQUEUE attempts already spent, and
@@ -9029,7 +10156,7 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
9029
10156
  const targets = [];
9030
10157
  for (const j of jobs ?? []) {
9031
10158
  if (j.status !== 'needs_review') continue;
9032
- if (!isEligibleForNeedsReviewAutoResolve(j)) continue;
10159
+ if (!isEligibleForNeedsReviewAutoResolve(j, jobs)) continue;
9033
10160
  if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) continue;
9034
10161
  const history = j.statusHistory || [];
9035
10162
  let entry = null;
@@ -9073,8 +10200,8 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
9073
10200
  * reason text (and the Queue UI's job.error) name the RIGHT evidence — a
9074
10201
  * guard-parked row was never "exhausted auto-fix" and must never claim to be.
9075
10202
  */
9076
- function applyNeedsReviewAutoResolve(j) {
9077
- if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j)) return null;
10203
+ function applyNeedsReviewAutoResolve(j, jobsInProject = []) {
10204
+ if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j, jobsInProject)) return null;
9078
10205
  if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) return null;
9079
10206
  const originIsGuardParked = !isExhaustedAutoFix(j) && isGuardParkedWithoutAutoFix(j);
9080
10207
 
@@ -9356,6 +10483,55 @@ async function computeLooksDone(job, fetchedCwds) {
9356
10483
  return { commits: attributed.commits, paths, detectedAt: new Date().toISOString(), rule: attributed.rule };
9357
10484
  }
9358
10485
 
10486
+ /**
10487
+ * Shadow gate (observation only): run a needs_review row's authored gate at
10488
+ * the project's current HEAD and record what it WOULD have decided as
10489
+ * `gateShadow` on the verdicts sidecar and the row. Changes NO status, takes
10490
+ * no slot (not a claude -p run — runGateSequence keeps one shadow gate in
10491
+ * flight machine-wide). Never called from finalize: only the reverify pass.
10492
+ * Returns the recorded gateShadow, or null when nothing was recorded (already
10493
+ * recorded at this HEAD, PRD unreadable, or another shadow gate is running).
10494
+ */
10495
+ async function runGateShadow(job) {
10496
+ if (!job || !job.slug || !job.cwd) return null;
10497
+ const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
10498
+ let prdText;
10499
+ try { prdText = fs.readFileSync(prdPath, 'utf8'); } catch { return null; }
10500
+ const head = await gitHead(job.cwd);
10501
+ if (job.gateShadow && job.gateShadow.head === head) return null;
10502
+ const gate = resolveGate(prdText);
10503
+ let outcome;
10504
+ if (gate.source === 'none') outcome = { status: 'unavailable', reason: 'gate-opt-out', results: [] };
10505
+ else if (!gate.sequence.length) outcome = { status: 'unavailable', reason: 'no-parseable-gate', results: [] };
10506
+ else {
10507
+ const r = await runGateSequence(gate.sequence, { cwd: job.cwd });
10508
+ if (r.status === 'busy') return null;
10509
+ outcome = r;
10510
+ }
10511
+ const gateShadow = { ...outcome, head, source: gate.source, ranAt: new Date().toISOString() };
10512
+ const runId = job.runId || resolveRunId(job);
10513
+ if (runId) {
10514
+ const verdictsPath = path.join(schedulerPaths.runsDir(), runId, `${job.slug}.verdicts.json`);
10515
+ // Read-merge (single-writer law: runVerify owns the sidecar's other keys).
10516
+ // Only merge into an existing run dir — never conjure one.
10517
+ if (fs.existsSync(path.dirname(verdictsPath))) {
10518
+ let existing = {};
10519
+ try { existing = JSON.parse(fs.readFileSync(verdictsPath, 'utf8')) || {}; } catch { /* absent/unparseable → fresh */ }
10520
+ try { atomicWriteJsonSync(verdictsPath, { ...existing, gateShadow }); } catch { /* best-effort */ }
10521
+ }
10522
+ }
10523
+ await mutate((s) => {
10524
+ for (const j of s.jobs) {
10525
+ if (j.slug === job.slug && j.status === 'needs_review') j.gateShadow = gateShadow;
10526
+ }
10527
+ });
10528
+ await broadcast();
10529
+ return gateShadow;
10530
+ }
10531
+
10532
+ // Tail of the last background shadow gate — lets tests (and only tests) await it.
10533
+ let gateShadowPending = null;
10534
+
9359
10535
  async function reverifyNeedsReview() {
9360
10536
  const snap = await readQueue();
9361
10537
  // isGuardParkedWithoutAutoFix rows are NOT isRescanCandidate (their
@@ -9365,7 +10541,7 @@ async function reverifyNeedsReview() {
9365
10541
  // guard-verdict auto-resolve gap this PRD closes. Handled in its own
9366
10542
  // branch below (no transcript rescan — there is no transcript verdict to
9367
10543
  // rescan) rather than through the isRescanCandidate machinery.
9368
- const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j));
10544
+ const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j) || isStaleSharedTreeRevertedPark(j));
9369
10545
  const healed = [];
9370
10546
  const leftForReview = [];
9371
10547
  const looksDoneUpdates = [];
@@ -9373,12 +10549,33 @@ async function reverifyNeedsReview() {
9373
10549
  // the `git fetch --all --prune` per distinct cwd (see computeLooksDone's
9374
10550
  // header) rather than re-fetching the same repo once per candidate row.
9375
10551
  const fetchedCwds = new Set();
10552
+ const evidenceSlugs = new Set(selectEvidenceScanTargets(snap.jobs).map((j) => j.slug));
10553
+ const evidenceScanned = [];
9376
10554
  for (const job of candidates) {
9377
- if (!isRescanCandidate(job) && isGuardParkedWithoutAutoFix(job)) {
9378
- // Guard-verdict park, never auto-fixed: only evidence gathering, never
9379
- // a transcript rescan (there was never a transcript-verifier verdict
9380
- // here) and never a direct heal — applyNeedsReviewAutoResolve is the
9381
- // sole place that turns this annotation into a status change.
10555
+ if (isStaleSharedTreeRevertedPark(job)) {
10556
+ // Re-apply the corrected shared-tree check: the row's own landedCommit
10557
+ // (this dispatch's, per resolveLandedCommitEvidence) still being an
10558
+ // ancestor of HEAD means the park was a false positive — heal it.
10559
+ const cwd = job.cwd || DEFAULT_PROJECT_CWD;
10560
+ if (await resolveLandedCommitEvidence(cwd, job.landedCommit, job.startedAt)
10561
+ && await module.exports.landedCommitIsAncestorOfHead(cwd, job.landedCommit)) {
10562
+ healed.push(job.slug);
10563
+ continue;
10564
+ }
10565
+ if (!isRescanCandidate(job) && !isGuardParkedWithoutAutoFix(job)) {
10566
+ leftForReview.push({ slug: job.slug, reason: 'shared_tree_reverted: landed commit not an ancestor of HEAD' });
10567
+ continue;
10568
+ }
10569
+ }
10570
+ if (job.status === 'needs_review' && !isTranscriptRescannable(job)) {
10571
+ // Any needs_review row whose verdict is not a transcript-verifier one
10572
+ // (a guard verdict, a not-yet-invented verdict, a stranded auto-fix
10573
+ // park): only evidence gathering, never a transcript rescan (verifyRun
10574
+ // would call it clean) and never a direct heal —
10575
+ // applyNeedsReviewAutoResolve is the sole place that turns this
10576
+ // annotation into a status change. Bounded by selectEvidenceScanTargets.
10577
+ if (!evidenceSlugs.has(job.slug)) continue;
10578
+ evidenceScanned.push(job.slug);
9382
10579
  const looksDone = await computeLooksDone(job, fetchedCwds);
9383
10580
  if (looksDone) {
9384
10581
  looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
@@ -9401,7 +10598,7 @@ async function reverifyNeedsReview() {
9401
10598
  }
9402
10599
  continue;
9403
10600
  }
9404
- const runDir = path.join(RUNS_DIR, job.runId || resolveRunId(job));
10601
+ const runDir = path.join(schedulerPaths.runsDir(), job.runId || resolveRunId(job));
9405
10602
  const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
9406
10603
  // Derive committedDuringRun from the recorded run window. The live
9407
10604
  // commit-guard uses gitHead() (before/after HEAD diff); here the run is
@@ -9424,6 +10621,13 @@ async function reverifyNeedsReview() {
9424
10621
  committedDuringRun,
9425
10622
  allowPreSentinelHeal: true,
9426
10623
  priorLandedCommit,
10624
+ // job.landedCommit is THIS row's own last-run attribution (stamped by
10625
+ // spawnJob's finalize, survives resetJobFields) — the same
10626
+ // ground-truth-outranks-heuristics evidence spawnJob passes live,
10627
+ // just read back post-hoc since there is no in-flight guardHeadBefore/
10628
+ // headAtExit pair to recompute for an already-terminal row.
10629
+ jobLandedCommitThisRun: job.landedCommit ?? null,
10630
+ exitCode: job.exitCode ?? null,
9427
10631
  });
9428
10632
  } catch { leftForReview.push({ slug: job.slug, reason: 'verifyRun threw' }); continue; }
9429
10633
  const refusal = healRefusalReason(job, v, committedDuringRun);
@@ -9454,6 +10658,23 @@ async function reverifyNeedsReview() {
9454
10658
  }
9455
10659
  }
9456
10660
  }
10661
+ // Shadow gate (observation only): at most ONE needs_review row per pass,
10662
+ // fired in the background so a 15-minute gate never stalls this pass.
10663
+ if (!gateShadowPending && process.env.SM_GATE_SHADOW_DISABLE !== '1') {
10664
+ const gateTarget = snap.jobs.find((j) => j.status === 'needs_review' && !j.gateShadow);
10665
+ if (gateTarget) {
10666
+ gateShadowPending = runGateShadow(gateTarget)
10667
+ .catch((e) => { console.error('[scheduler] gate shadow error', gateTarget.slug, e); })
10668
+ .finally(() => { gateShadowPending = null; });
10669
+ }
10670
+ }
10671
+ if (evidenceScanned.length) {
10672
+ const scannedSet = new Set(evidenceScanned);
10673
+ const stamp = new Date().toISOString();
10674
+ await mutate((s) => {
10675
+ for (const j of s.jobs) if (scannedSet.has(j.slug) && j.status === 'needs_review') j.evidenceScannedAt = stamp;
10676
+ });
10677
+ }
9457
10678
  if (looksDoneUpdates.length) {
9458
10679
  const bySlug = new Map(looksDoneUpdates.map((u) => [u.slug, u]));
9459
10680
  await mutate((s) => {
@@ -9666,7 +10887,7 @@ async function reverifyNeedsReview() {
9666
10887
  });
9667
10888
  for (const job of targets) {
9668
10889
  const runId = job.runId || resolveRunId(job);
9669
- const runDir = path.join(RUNS_DIR, runId);
10890
+ const runDir = path.join(schedulerPaths.runsDir(), runId);
9670
10891
  const isRetryAttempt = job.autoFixAttempted === true;
9671
10892
  const isDeadFixPlanReopen = isFixPlanDead(job, queueForResumeAndAutofix.jobs);
9672
10893
  const deadChild = isDeadFixPlanReopen
@@ -9813,12 +11034,14 @@ function registerScheduleHandlers() {
9813
11034
  const freeSlots = Math.max(0, slotSnapshot.total - slotSnapshot.inUse);
9814
11035
  const verdict = classifyQueueHealth({
9815
11036
  jobs: state.jobs,
9816
- paused: state.paused,
11037
+ paused: upgradeDrain.effectivePaused(state),
9817
11038
  launchBlocks: state.launchBlocks,
9818
11039
  runningSet,
9819
11040
  freeSlots,
9820
11041
  totalSlots: slotSnapshot.total,
9821
- lastDispatchAttemptAtMs: Date.parse(state.lastDispatchAttemptAt ?? ''),
11042
+ lastRunAtMs: Date.parse(state.lastRunAt ?? ''),
11043
+ lastPauseClearedAtMs: lastPauseClearedAt,
11044
+ schedulerBootedAtMs: Date.parse(SCHEDULER_BOOTED_AT),
9822
11045
  now,
9823
11046
  cwd,
9824
11047
  });
@@ -9949,6 +11172,11 @@ function registerScheduleHandlers() {
9949
11172
  return { ok: true };
9950
11173
  });
9951
11174
 
11175
+ ipcMain.handle('schedule:pause', async () => {
11176
+ await setPaused('manual', null);
11177
+ return { ok: true };
11178
+ });
11179
+
9952
11180
  ipcMain.handle('schedule:resume', async () => {
9953
11181
  await clearPause('manual');
9954
11182
  return { ok: true };
@@ -9980,7 +11208,7 @@ function registerScheduleHandlers() {
9980
11208
  ipcMain.handle('schedule:clear-queue', async () => {
9981
11209
  ensureDirs();
9982
11210
  const ts = new Date().toISOString().replace(/[:.]/g, '-');
9983
- const archiveDir = path.join(PRDS_ARCHIVE_DIR, ts);
11211
+ const archiveDir = path.join(schedulerPaths.scheduledPlansRoot(), 'prds-archived', ts);
9984
11212
  const state = await readQueue();
9985
11213
  const victims = state.jobs.filter((j) => j.status !== 'running');
9986
11214
  if (victims.length === 0) {
@@ -10027,7 +11255,7 @@ function registerScheduleHandlers() {
10027
11255
 
10028
11256
  ipcMain.handle('schedule:open-folder', async () => {
10029
11257
  const { shell } = require('electron');
10030
- await shell.openPath(ROOT);
11258
+ await shell.openPath(schedulerPaths.scheduledPlansRoot());
10031
11259
  return { ok: true };
10032
11260
  });
10033
11261
 
@@ -10045,8 +11273,8 @@ function registerScheduleHandlers() {
10045
11273
  ipcMain.handle('schedule:read-log', validated(schemas.scheduleReadLog, async ({ slug, runId }) => {
10046
11274
  // Defense-in-depth: re-check containment after path.resolve even though
10047
11275
  // SLUG_RE / RUN_ID_RE already forbid path separators.
10048
- const logPath = path.resolve(path.join(RUNS_DIR, runId, `${slug}.log`));
10049
- if (!logPath.startsWith(RUNS_DIR + path.sep)) {
11276
+ const logPath = path.resolve(path.join(schedulerPaths.runsDir(), runId, `${slug}.log`));
11277
+ if (!logPath.startsWith(schedulerPaths.runsDir() + path.sep)) {
10050
11278
  return { ok: false, error: 'invalid slug or runId' };
10051
11279
  }
10052
11280
  try {
@@ -10063,8 +11291,8 @@ function registerScheduleHandlers() {
10063
11291
  // template, authored before the user fills in `cwd`) falls back to the
10064
11292
  // legacy global dir until it's re-saved with a real cwd and migrated by
10065
11293
  // the next reconcile-driven scan.
10066
- const dir = (await findPrdDir(data.slug)) ?? PRDS_DIR;
10067
- if (dir === PRDS_DIR) ensureDirs();
11294
+ const dir = (await findPrdDir(data.slug)) ?? schedulerPaths.prdsRoot();
11295
+ if (dir === schedulerPaths.prdsRoot()) ensureDirs();
10068
11296
  const resolved = safeSlugPathIn(dir, data.slug);
10069
11297
  if (!resolved) return { ok: false, error: 'invalid slug' };
10070
11298
  try {
@@ -10096,6 +11324,15 @@ function registerScheduleHandlers() {
10096
11324
  });
10097
11325
  }
10098
11326
 
11327
+ function stopDispatchLoop() {
11328
+ if (dispatchLoopHandle) { dispatchLoopHandle.stop(); dispatchLoopHandle = null; }
11329
+ }
11330
+
11331
+ /** Shutdown path: stop the timers this module owns. */
11332
+ function stop() {
11333
+ stopDispatchLoop();
11334
+ }
11335
+
10099
11336
  async function init() {
10100
11337
  ensureDirs();
10101
11338
  // Boot phase — reconciliation, migrations, self-heal, first reset probe.
@@ -10111,6 +11348,9 @@ async function init() {
10111
11348
  // A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
10112
11349
  // batch — advance the queue without waiting for the next 60s poll.
10113
11350
  sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
11351
+ // Boot-time expiry pass (process-local state is empty after a restart, so
11352
+ // this is a cheap belt-and-braces run against the freshly read queue).
11353
+ try { runReservationExpiryPass((await readQueue()).jobs); } catch { /* best-effort */ }
10114
11354
  // Retire the global queue.json: split its rows into per-project shards
10115
11355
  // BEFORE the first read below, so boot reconciliation sees the shards.
10116
11356
  try {
@@ -10131,17 +11371,17 @@ async function init() {
10131
11371
  // Boot reconciliation: finalize any job that was 'running' when the app died.
10132
11372
  // Check the run log first — a job that emitted result/success before the crash
10133
11373
  // should be marked 'completed', not 'failed', so it doesn't wedge the queue
10134
- // via the failure-gate. Also kill any still-live orphan claude child to prevent
10135
- // it from continuing to write to the project unsupervised (2026-05-21 incident).
11374
+ // via the failure-gate. A still-live executor is spared, not killed.
10136
11375
  //
10137
11376
  // classifyRunOutcome calls readTail → fs.readFileSync (up to 64 KB per job).
10138
11377
  // Pre-compute all outcomes BEFORE entering the mutate lock so the blocking I/O
10139
11378
  // does not stall the event loop or hold the mutateTail chain during startup.
10140
11379
  //
10141
- // Jobs whose recorded pid is still alive are deferred (not classified here) —
10142
- // see partitionBootOrphans. Everything else (dead pid or no pid) is safe to
11380
+ // Rows proven alive are adopted (left running, never killed) — see
11381
+ // partitionBootOrphans. Everything else is proven dead/exited and is safe to
10143
11382
  // classify immediately below.
10144
11383
  const bootSnap = readQueueSync();
11384
+ const bootLogPath = (j) => (j?.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null);
10145
11385
 
10146
11386
  // Worktree boot reconciliation (PRD 994): a job worktree that survives an
10147
11387
  // app crash/host reboot must not leak disk or a dangling branch forever —
@@ -10156,11 +11396,15 @@ async function init() {
10156
11396
  // itself, proof its run already died. isLive checks the already-read
10157
11397
  // bootSnap (no extra queue read) for a live running-row pid, OR a live
10158
11398
  // /proc cwd holder under the checkout itself. See jobWorktreeBootLive.cjs.
11399
+ // rowPid walks the SAME record → runtime.pid → log-pid ladder
11400
+ // partitionBootOrphans uses, so the sweep and the partition can never
11401
+ // disagree about which executor is alive.
10159
11402
  const isLive = buildJobWorktreeIsLive({
10160
11403
  bootJobs: bootSnap.jobs,
10161
11404
  claudePidAlive,
10162
11405
  hasLiveHolder: gitWorktree.hasLiveHolder,
10163
11406
  cwdHolders: gitWorktree.listCwdHolders(),
11407
+ rowPid: (j) => bootRowPid(j, bootLogPath),
10164
11408
  });
10165
11409
  await jobWorktree.reconcileWorktreesOnBoot([...worktreeCwds], { isLive });
10166
11410
  } catch (e) {
@@ -10180,12 +11424,30 @@ async function init() {
10180
11424
  console.error('[scheduler] boot epic-worktree reconciliation failed', e?.message);
10181
11425
  }
10182
11426
 
10183
- const { immediate: immediateSlugs, deferred: deferredSlugs } = partitionBootOrphans(bootSnap.jobs);
11427
+ const { immediate: immediateSlugs, adopted: adoptedSlugs } = partitionBootOrphans(bootSnap.jobs, {
11428
+ pidAlive: claudePidAlive,
11429
+ getLogPid: (j) => readSpawnedPidFromLog(bootLogPath(j)),
11430
+ getLogMtimeMs: (j) => readLogMtimeMs(bootLogPath(j)),
11431
+ logFreshWindowMs: IDLE_OUTPUT_KILL_MS,
11432
+ findLiveProcess: (j) => findLiveProcessForJob(j, {
11433
+ worktreeDir: jobWorktree.worktreeDirFor(j.cwd || DEFAULT_PROJECT_CWD, j.slug),
11434
+ runCwd: j.runtime?.cwd || j.cwd,
11435
+ }),
11436
+ readRecord: supervisorRecord.readSupervisorRecord,
11437
+ });
10184
11438
  const bootOutcomes = new Map();
10185
11439
  for (const j of bootSnap.jobs) {
10186
11440
  if (!immediateSlugs.includes(j.slug)) continue;
10187
- const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
10188
- bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
11441
+ const logPath = j.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null;
11442
+ let outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
11443
+ // A row whose run already wrote its exit marker (meta.json) is finalized
11444
+ // from that meta when the log tail alone can't say (killed/torn tail).
11445
+ if (outcome === 'unknown' || outcome === 'no_result') {
11446
+ let meta = null;
11447
+ try { meta = j.runId ? JSON.parse(fs.readFileSync(path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.meta.json`), 'utf8')) : null; } catch { /* no/torn meta — keep the log outcome */ }
11448
+ if (meta && typeof meta.exitCode === 'number') outcome = meta.exitCode === 0 ? 'success' : 'failed';
11449
+ }
11450
+ bootOutcomes.set(j.slug, outcome);
10189
11451
  }
10190
11452
  // Same evidence-before-failure gate reapDeadRunningJobs applies, resolved
10191
11453
  // BEFORE mutate() for the same reason (git spawn work must never run
@@ -10215,53 +11477,32 @@ async function init() {
10215
11477
  await archiveCompletedPrd(slug, cwd);
10216
11478
  }
10217
11479
 
10218
- // Still-alive orphans: SIGTERM (+ killOrphanClaudePid's own deferred SIGKILL
10219
- // follow-up) now, but classification waits until BOOT_ORPHAN_KILL_GRACE_MS
10220
- // later — reading the log while the orphan might still be writing to it
10221
- // could misclassify an about-to-succeed run as no_result and double-run the
10222
- // same PRD (2026-05-21 incident this guard exists for).
10223
- for (const slug of deferredSlugs) {
10224
- const j = bootSnap.jobs.find((x) => x.slug === slug);
10225
- const pid = j?.runtime?.pid;
10226
- const bootRunId = j?.runId ?? null; // captured now — guards against reconciling a DIFFERENT later run of the same slug
10227
- if (!pid) continue;
10228
- const result = killOrphanClaudePid(pid);
10229
- const killNote = ` (orphan pid=${pid}: ${result})`;
10230
- if (result === 'killed') {
10231
- console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
10232
- }
10233
- setTimeout(async () => {
10234
- const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
10235
- const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
10236
- // Same evidence-before-failure gate as the immediate-orphan path
10237
- // above, resolved before mutate() for the same reason (git spawn
10238
- // work must never run inside mutate()'s serialization chain). Uses
10239
- // the captured pre-kill snapshot's landedCommit/cwd/startedAt — the
10240
- // race guard below already confirms `cur` is still this same run
10241
- // (runId === bootRunId) before this evidence is applied.
10242
- const confirmedLandedCommit = (outcome !== 'success' && j.landedCommit)
10243
- ? (await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt) ? j.landedCommit : null)
10244
- : null;
10245
- let deferredCompletedCwd;
10246
- mutate((state) => {
10247
- const cur = state.jobs.find((x) => x.slug === slug);
10248
- // Race guard: bail if the job already resolved, OR if it's already been
10249
- // re-picked into a NEW run (different runId) within the grace window —
10250
- // that new run is not the boot orphan we SIGTERM'd and must not be
10251
- // touched by this stale classification.
10252
- if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
10253
- applyOrphanOutcome(cur, outcome, killNote, confirmedLandedCommit);
10254
- console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
10255
- deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
10256
- }).then(() => {
10257
- if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
10258
- }).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
10259
- }, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
11480
+ // Proven-alive rows stay `running` and are never signalled (the boot worktree
11481
+ // sweep above already spares their checkout). No sessionSlots token is
11482
+ // acquired: pickNextBatch's untrackedRunning correction counts the row
11483
+ // against the pool, and reapDeadRunningJobs finalizes it on exit.
11484
+ if (adoptedSlugs.length) {
11485
+ const adoptedAtBoot = new Date().toISOString();
11486
+ const adoptedRunIds = new Map(bootSnap.jobs.filter((j) => adoptedSlugs.includes(j.slug)).map((j) => [j.slug, j.runId ?? null]));
11487
+ await mutate((state) => {
11488
+ for (const j of state.jobs) {
11489
+ // runId guard: never stamp a DIFFERENT later run of the same slug.
11490
+ if (j.status !== 'running' || !adoptedRunIds.has(j.slug) || (j.runId ?? null) !== adoptedRunIds.get(j.slug)) continue;
11491
+ j.adoptedAtBoot = adoptedAtBoot;
11492
+ delete j.supervisedAt; // a prior process's supervisor died with it
11493
+ console.log(`[scheduler] boot: adopted live executor for ${j.slug} (pid=${j.runtime?.pid ?? 'unknown'}) — left running, no signal`);
11494
+ }
11495
+ });
10260
11496
  }
10261
11497
 
11498
+ // Re-arm budget/idle/deadman + the quietMachine lease for the adopted rows
11499
+ // (a dispatch-loop pass repeats this for any row left without a supervisor).
11500
+ await superviseAdoptedRunsPass();
11501
+
10262
11502
  // If we boot up while paused with a resumeAt in the past, clear it. This
10263
11503
  // happens when the app was closed across the reset window.
10264
11504
  const boot = await readQueue();
11505
+ await clearStaleDrainAtBoot(boot);
10265
11506
  if (boot.paused && boot.paused.resumeAt && new Date(boot.paused.resumeAt).getTime() <= Date.now()) {
10266
11507
  await clearPause('boot-elapsed');
10267
11508
  } else if (boot.paused && boot.paused.resumeAt) {
@@ -10483,7 +11724,7 @@ async function init() {
10483
11724
  }
10484
11725
  for (const target of exhaustedNeedsReviewTargets) {
10485
11726
  const j = ms.jobs.find((x) => x.slug === target.slug);
10486
- const outcome = applyNeedsReviewAutoResolve(j);
11727
+ const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
10487
11728
  if (outcome) {
10488
11729
  console.warn(
10489
11730
  `[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
@@ -10504,7 +11745,7 @@ async function init() {
10504
11745
  }
10505
11746
  }).catch(() => {});
10506
11747
  }
10507
- }, 10 * 60_000);
11748
+ }, REVERIFY_INTERVAL_MS);
10508
11749
 
10509
11750
  // Self-rescheduling poll loop with exponential backoff. Replaces the
10510
11751
  // old fixed-interval pollTimer + initialPollTimeout.
@@ -10525,87 +11766,18 @@ async function init() {
10525
11766
  // setInterval callback is sync; readQueueSync stays sync to avoid awaiting
10526
11767
  // inside the timer body (and the 60s cadence makes the cost moot).
10527
11768
  if (heartbeatInterval) clearInterval(heartbeatInterval);
10528
- heartbeatInterval = setInterval(() => {
10529
- const s = readQueueSync();
10530
- // NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
10531
- // running, something must drive it. This is the only driver that does
10532
- // not depend on the billing poll loop, a pause timer, or a completing
10533
- // job to schedule the next tick — every one of which has failed at
10534
- // least once. See classifyQueueStarvation.
10535
- if (!s.unreadable) {
10536
- runQueueStarvationWatchdog(s).catch((e) => console.error('[scheduler] starvation watchdog error', e));
10537
- }
10538
- // Initialise from the real status union (scheduleJobSchema.cjs) rather
10539
- // than a hand-maintained subset — the old `{ pending, running, completed,
10540
- // failed }` literal silently minted a NEW key for any other value
10541
- // (`counts[j.status] = (counts[j.status]||0)+1`), which is exactly how a
10542
- // heartbeat with a `queued: 2` bucket looked like "normal" 24h
10543
- // visibility instead of the alarm it should have been. Any row whose
10544
- // status isn't in JOB_STATUSES (shouldn't happen post-quarantine, but
10545
- // this is the last line of defence) routes into `unknown`, never a
10546
- // freshly-minted key.
10547
- const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
10548
- counts.unknown = 0;
10549
- for (const j of s.jobs) {
10550
- if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
10551
- counts[j.status] += 1;
10552
- } else {
10553
- counts.unknown += 1;
10554
- }
10555
- }
10556
-
10557
- const stall = computeStallSummary(s);
10558
- // Per-project alerting (see computeStallSummary's header): a project
10559
- // stalled while others are busy must still fire, and one project
10560
- // recovering must not clear or suppress another's still-open episode —
10561
- // that is exactly what a single module-level stallSince/stallToasted
10562
- // flag masked before (the burrow-vs-others incident this PRD fixes).
10563
- const now = Date.now();
10564
- const stalledCwds = Object.keys(stall.byProject).filter((cwd) => stall.byProject[cwd].stalled);
10565
- for (const cwd of [...stallSince.keys()]) {
10566
- if (!stalledCwds.includes(cwd)) {
10567
- stallSince.delete(cwd);
10568
- stallToasted.delete(cwd);
10569
- }
10570
- }
10571
- const toAlert = [];
10572
- for (const cwd of stalledCwds) {
10573
- if (!stallSince.has(cwd)) stallSince.set(cwd, now);
10574
- if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
10575
- stallToasted.set(cwd, true);
10576
- toAlert.push(cwd);
10577
- }
10578
- }
10579
- if (toAlert.length > 0) {
10580
- console.error(
10581
- `[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
10582
- + `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
10583
- stall.byProject,
10584
- );
10585
- appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: stall.total, byProject: stall.byProject });
10586
- if (mainWindow && !mainWindow.isDestroyed()) {
10587
- sendIfAlive(mainWindow, 'schedule:stall', {
10588
- message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
10589
- projects: toAlert,
10590
- total: stall.total,
10591
- byProject: stall.byProject,
10592
- });
10593
- }
10594
- }
10595
-
10596
- appendHeartbeat({
10597
- ts: Date.now(),
10598
- pid: process.pid,
10599
- counts,
10600
- stall: { stalled: stall.stalled, total: stall.total },
10601
- paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
10602
- nextReset: cachedNextReset,
10603
- utilization: cachedUtilization,
10604
- consecutiveFailures,
10605
- });
10606
- }, 60_000);
11769
+ heartbeatInterval = setInterval(() => heartbeatTick(), 60_000);
10607
11770
  if (heartbeatInterval.unref) heartbeatInterval.unref();
10608
11771
 
11772
+ // Dispatch's own periodic driver: cadence is independent of pollLoop's billing
11773
+ // backoff. A loop tick meeting a cancelled cancelToken returns 'cancelled' from
11774
+ // tickBody; clearing stays with the starvation watchdog's existing force-clear.
11775
+ stopDispatchLoop();
11776
+ dispatchLoopHandle = startDispatchLoop({
11777
+ tick: () => tickQueue(),
11778
+ onError: (e) => console.warn('[scheduler] dispatch loop tick failed', e?.message),
11779
+ });
11780
+
10609
11781
  // Wake-from-sleep: immediately re-poll and re-evaluate the queue.
10610
11782
  try {
10611
11783
  const { powerMonitor } = require('electron');
@@ -10828,8 +12000,8 @@ const remote = {
10828
12000
  }
10829
12001
  await fsp.mkdir(dir, { recursive: true });
10830
12002
  } else {
10831
- dir = (await findPrdDir(slug)) ?? PRDS_DIR;
10832
- if (dir === PRDS_DIR) ensureDirs();
12003
+ dir = (await findPrdDir(slug)) ?? schedulerPaths.prdsRoot();
12004
+ if (dir === schedulerPaths.prdsRoot()) ensureDirs();
10833
12005
  }
10834
12006
 
10835
12007
  // writePrd only ever JOINS an existing Epic now (no mintAuthority
@@ -10866,6 +12038,19 @@ const remote = {
10866
12038
  }
10867
12039
  },
10868
12040
 
12041
+ // User-initiated pause/resume — the admin-route/MCP twins of the
12042
+ // schedule:pause / schedule:resume IPC handlers, through the same setPaused /
12043
+ // clearPause. Pause stops NEW dispatch only; running jobs are never touched.
12044
+ async pause() {
12045
+ await setPaused('manual', null);
12046
+ return { ok: true };
12047
+ },
12048
+
12049
+ async resume() {
12050
+ await clearPause('manual');
12051
+ return { ok: true };
12052
+ },
12053
+
10869
12054
  async resetJob(slug, opts = {}) {
10870
12055
  const resolved = await resolveSlugOrReason(slug, opts.cwd);
10871
12056
  if (!resolved.ok) {
@@ -11193,6 +12378,14 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
11193
12378
  sendJson(res, 200, jobs);
11194
12379
  });
11195
12380
 
12381
+ adminHttp.registerRoute('POST', '/admin/scheduler/pause', async (req, res) => {
12382
+ sendJson(res, 200, await remoteObj.pause());
12383
+ });
12384
+
12385
+ adminHttp.registerRoute('POST', '/admin/scheduler/resume', async (req, res) => {
12386
+ sendJson(res, 200, await remoteObj.resume());
12387
+ });
12388
+
11196
12389
  adminHttp.registerRoute('POST', '/admin/scheduler/reset-job', async (req, res) => {
11197
12390
  const raw = await readBody(req);
11198
12391
  let parsed;
@@ -11217,6 +12410,8 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
11217
12410
  module.exports = {
11218
12411
  classifyQueueStarvation,
11219
12412
  classifyQueueStarvationByProject,
12413
+ dispatchIdleMs,
12414
+ launchBlockedSlugs,
11220
12415
  classifyQueueHealth,
11221
12416
  runQueueStarvationWatchdog,
11222
12417
  QUEUE_STARVATION_MS,
@@ -11240,13 +12435,13 @@ module.exports = {
11240
12435
  registerScheduleHandlers,
11241
12436
  attachWindow,
11242
12437
  init,
11243
- ROOT,
11244
- PRDS_DIR,
11245
- SCHEDULER_STATE_PATH,
11246
12438
  BACKOFF_MAX_MS,
11247
12439
  FAILURE_STREAK_WARN_THRESHOLD,
12440
+ FAILURE_STREAK_ESCALATION_MS,
11248
12441
  nextBackoffMs,
11249
12442
  shouldWarnFailureStreak,
12443
+ shouldEscalateFailureStreak,
12444
+ computeDegradedBudget,
11250
12445
  healRefusalReason,
11251
12446
  writeQueue,
11252
12447
  reconcile,
@@ -11269,6 +12464,8 @@ module.exports = {
11269
12464
  memoryLimitedBatchSize,
11270
12465
  availableForJobs,
11271
12466
  reverifyNeedsReview,
12467
+ runGateShadow,
12468
+ awaitGateShadowIdle: async () => { while (gateShadowPending) await gateShadowPending; },
11272
12469
  shouldRunPeriodicReverify,
11273
12470
  findStuckFailedJobs,
11274
12471
  STUCK_FAILED_ESCALATE_MS,
@@ -11283,6 +12480,13 @@ module.exports = {
11283
12480
  NEEDS_REVIEW_RESOLVE_MS,
11284
12481
  needsReviewAutoResolveDisabled,
11285
12482
  isRescanCandidate,
12483
+ isTranscriptRescannable,
12484
+ selectEvidenceScanTargets,
12485
+ RESCAN_EXCLUDED_VERDICTS,
12486
+ RESCANNABLE_VERDICTS,
12487
+ EVIDENCE_SCAN_MAX_PER_PASS,
12488
+ EVIDENCE_SCAN_MIN_INTERVAL_MS,
12489
+ REVERIFY_INTERVAL_MS,
11286
12490
  isFailedUnverifiedShaped,
11287
12491
  computeLooksDone,
11288
12492
  attributeLandedCommits,
@@ -11295,6 +12499,9 @@ module.exports = {
11295
12499
  isExhaustedAutoFix,
11296
12500
  GUARD_VERDICT_EVIDENCE_ELIGIBLE,
11297
12501
  isGuardParkedWithoutAutoFix,
12502
+ isStrandedAutoFixPark,
12503
+ isStaleSharedTreeRevertedPark,
12504
+ landedCommitIsAncestorOfHead,
11298
12505
  isEligibleForNeedsReviewAutoResolve,
11299
12506
  isPlanUnqueued,
11300
12507
  isFixPlanDead,
@@ -11334,7 +12541,6 @@ module.exports = {
11334
12541
  buildScheduleStatePayload,
11335
12542
  partitionBootOrphans,
11336
12543
  applyOrphanOutcome,
11337
- BOOT_ORPHAN_KILL_GRACE_MS,
11338
12544
  registerAdminRoutes,
11339
12545
  notifyOriginatingTab,
11340
12546
  notifyNeedsReview,
@@ -11360,10 +12566,13 @@ module.exports = {
11360
12566
  SCHEDULER_CODE_SHA,
11361
12567
  resetJobFields,
11362
12568
  executeJob,
12569
+ killOrphanClaudePid,
11363
12570
  prdArchivedSkipResult,
11364
12571
  spawnJob,
11365
12572
  listPrdsInternal,
11366
12573
  computeStallSummary,
12574
+ heartbeatTick,
12575
+ appendHeartbeat,
11367
12576
  findStaleQuarantinedJobs,
11368
12577
  QUARANTINE_ESCALATE_MS,
11369
12578
  selectQuarantineAutoResolveTargets,
@@ -11402,6 +12611,10 @@ module.exports = {
11402
12611
  setPaused,
11403
12612
  clearPause,
11404
12613
  tickQueue,
12614
+ setRestartHandler,
12615
+ driveUpgradeDrain,
12616
+ clearStaleDrainAtBoot,
12617
+ stop,
11405
12618
  runDueJobs,
11406
12619
  pollLoop,
11407
12620
  maybeLaunchWhenAvailable,
@@ -11410,12 +12623,23 @@ module.exports = {
11410
12623
  CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
11411
12624
  RAPID_RATE_LIMIT_WINDOW_MS,
11412
12625
  MANUAL_PAUSE_COOLDOWN_MS,
11413
- RUNS_DIR,
11414
12626
  pickRunDir,
11415
12627
  resolveRateLimitPauseReset,
12628
+ billingResetForPause,
11416
12629
  computeEffectiveResumeAt,
11417
12630
  computeResumeDelay,
11418
12631
  FOREIGN_WIP_BLOCK_STREAK_LIMIT,
11419
12632
  validateForeignWipBlockClaim,
11420
12633
  requeueForeignWipBlockedJobs,
11421
12634
  };
12635
+
12636
+ // Lazy path getters: resolved from SM_SCHEDULER_HOME at each read, never frozen
12637
+ // at require time (see lib/schedulerPaths.cjs).
12638
+ for (const [name, resolve] of [
12639
+ ['ROOT', schedulerPaths.scheduledPlansRoot],
12640
+ ['PRDS_DIR', schedulerPaths.prdsRoot],
12641
+ ['RUNS_DIR', schedulerPaths.runsDir],
12642
+ ['SCHEDULER_STATE_PATH', schedulerPaths.schedulerStatePath],
12643
+ ]) {
12644
+ Object.defineProperty(module.exports, name, { get: resolve, enumerable: true });
12645
+ }