claude-code-session-manager 0.87.0 → 0.88.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (290) hide show
  1. package/README.md +26 -53
  2. package/dist/assets/{AgentLibrary-DyLWzZDf.js → AgentLibrary-BZ6IY2g1.js} +1 -1
  3. package/dist/assets/{DataModel--mISIJ6h.js → DataModel-dAoGAODX.js} +1 -1
  4. package/dist/assets/{History-C2ahUXTg.js → History-CF157HyX.js} +1 -1
  5. package/dist/assets/{Hooks-BiC6oyR2.js → Hooks-CCPPipBk.js} +1 -1
  6. package/dist/assets/{HostBilko-BPleEOld.js → HostBilko-Dlmz2m06.js} +1 -1
  7. package/dist/assets/{Library-Dc8Qst1R.js → Library-BaXcboWX.js} +1 -1
  8. package/dist/assets/{ListDetail-DIXh-OLX.js → ListDetail-CoUCzG5S.js} +1 -1
  9. package/dist/assets/{MarkdownEditor-C90bkLXK.js → MarkdownEditor-DTg3JlOT.js} +1 -1
  10. package/dist/assets/{McpServers-DqcbLOLZ.js → McpServers-DdjY1sA3.js} +1 -1
  11. package/dist/assets/{Memory-CW62MXlh.js → Memory-BJ3Jlw0C.js} +1 -1
  12. package/dist/assets/{Panel-Bw1FhRuF.js → Panel-ygRSgc41.js} +1 -1
  13. package/dist/assets/{Permissions-BcUC-5y8.js → Permissions-m0KAnccF.js} +1 -1
  14. package/dist/assets/{Plugins-BnKx9flD.js → Plugins-CJRl6B3Q.js} +2 -2
  15. package/dist/assets/{ProvenanceBadge-Bw5vNVPT.js → ProvenanceBadge-w8cnlVvW.js} +1 -1
  16. package/dist/assets/{SaveBar-CWr0O_w-.js → SaveBar-BpyPcUIk.js} +1 -1
  17. package/dist/assets/{Scheduler-DYdLuUqq.js → Scheduler-D2ppxSfy.js} +7 -7
  18. package/dist/assets/{ScopeSwitcher-CrBLbg8s.js → ScopeSwitcher-EyuIwt-j.js} +1 -1
  19. package/dist/assets/{Settings-DluB-vN1.js → Settings-BHSGGFEc.js} +1 -1
  20. package/dist/assets/{SkillReferenceGraph-CHLSseay.js → SkillReferenceGraph-BNaZ2JCF.js} +1 -1
  21. package/dist/assets/{Skills-gNdo_HNK.js → Skills-Dl3q75QV.js} +1 -1
  22. package/dist/assets/{SystemPrompt-Cru05-Ia.js → SystemPrompt-Bg6F-Cyp.js} +1 -1
  23. package/dist/assets/{TagLibrary-DNHY0xou.js → TagLibrary-BSutflCy.js} +1 -1
  24. package/dist/assets/{TiptapBody-I4lmbCgP.js → TiptapBody-CXih4xrc.js} +1 -1
  25. package/dist/assets/{Toggle-bWMHjmRh.js → Toggle-2qtUHhA9.js} +1 -1
  26. package/dist/assets/{index-fc_JjdxL.js → index-B3emy2oI.js} +356 -356
  27. package/dist/assets/{index-DV3PorRY.css → index-BHTX4OTc.css} +1 -1
  28. package/dist/assets/{settingsSchema-BfhtZnGD.js → settingsSchema-CdzxaqwM.js} +1 -1
  29. package/dist/assets/{whisperWorker-Dbia1OpC.js → whisperWorker-C7ZGQwKg.js} +7 -7
  30. package/dist/index.html +2 -2
  31. package/dist/vad/ort-wasm-simd-threaded.asyncify.mjs +106 -110
  32. package/dist/vad/ort-wasm-simd-threaded.asyncify.wasm +0 -0
  33. package/dist/vad/ort-wasm-simd-threaded.jsep.mjs +98 -98
  34. package/dist/vad/ort-wasm-simd-threaded.jsep.wasm +0 -0
  35. package/dist/vad/ort-wasm-simd-threaded.jspi.mjs +99 -102
  36. package/dist/vad/ort-wasm-simd-threaded.jspi.wasm +0 -0
  37. package/dist/vad/ort-wasm-simd-threaded.mjs +46 -46
  38. package/dist/vad/ort-wasm-simd-threaded.wasm +0 -0
  39. package/package.json +10 -13
  40. package/scripts/README.md +59 -0
  41. package/scripts/audit-ops-hygiene.cjs +350 -0
  42. package/scripts/hooks/guard-destructive-git.cjs +10 -34
  43. package/scripts/hooks/guard-inline-implementation.cjs +14 -8
  44. package/scripts/hooks/guard-prd-writes.cjs +7 -36
  45. package/scripts/hooks/guard-self-schedule.cjs +175 -0
  46. package/scripts/ops-sweep.cjs +355 -0
  47. package/scripts/scheduler-mcp-server.cjs +28 -71
  48. package/src/main/__tests__/bilkoHost-integration.test.cjs +3 -3
  49. package/src/main/__tests__/chat-cancel-terminal.test.cjs +6 -9
  50. package/src/main/__tests__/chat-exit-close-race.test.cjs +3 -3
  51. package/src/main/__tests__/chat-mcp-consent-notice.test.cjs +4 -5
  52. package/src/main/__tests__/chat-queue.test.cjs +2 -2
  53. package/src/main/__tests__/chat-stop-signal.test.cjs +2 -2
  54. package/src/main/__tests__/dep-orphan-archive-health.test.cjs +77 -0
  55. package/src/main/__tests__/dod-batchkey.test.cjs +2 -2
  56. package/src/main/__tests__/dod-drain-hook.test.cjs +2 -2
  57. package/src/main/__tests__/dod-report.test.cjs +2 -2
  58. package/src/main/__tests__/dod-reverify.test.cjs +2 -2
  59. package/src/main/__tests__/epicMint.test.cjs +2 -2
  60. package/src/main/__tests__/exchanges.test.cjs +2 -2
  61. package/src/main/__tests__/extractJson.test.cjs +2 -2
  62. package/src/main/__tests__/files-reject-credentials.test.cjs +1 -1
  63. package/src/main/__tests__/fixtures/1218-fo-01-move-scripts-lib-into-src-main-lib.log +556 -0
  64. package/src/main/__tests__/health-build-freshness.test.cjs +39 -0
  65. package/src/main/__tests__/health-delegation-chain.test.cjs +15 -1
  66. package/src/main/__tests__/health-queue-dispatch.test.cjs +58 -7
  67. package/src/main/__tests__/health-tick-liveness.test.cjs +11 -3
  68. package/src/main/__tests__/health-usage-poller.test.cjs +70 -23
  69. package/src/main/__tests__/historyRollup.test.cjs +2 -2
  70. package/src/main/__tests__/kg-augment.test.cjs +2 -2
  71. package/src/main/__tests__/mcpStatus.test.cjs +2 -2
  72. package/src/main/__tests__/memoryAggregate.test.cjs +1 -1
  73. package/src/main/__tests__/memoryStale.test.cjs +1 -1
  74. package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +25 -1
  75. package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +33 -3
  76. package/src/main/__tests__/prd-group-allocator.test.cjs +2 -2
  77. package/src/main/__tests__/prdAdminRouteParity.test.cjs +2 -0
  78. package/src/main/__tests__/prdAdminRoutes.test.cjs +14 -2
  79. package/src/main/__tests__/prdAuthoringSeed.test.cjs +39 -0
  80. package/src/main/__tests__/prdLocationsArchived.test.cjs +20 -13
  81. package/src/main/__tests__/proc-role-env.test.cjs +125 -0
  82. package/src/main/__tests__/procname-claude-spawn-sites.test.cjs +304 -0
  83. package/src/main/__tests__/procname-sm-processes.test.cjs +127 -0
  84. package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +81 -401
  85. package/src/main/__tests__/projectPages.test.cjs +63 -149
  86. package/src/main/__tests__/queue-health-verdict.test.cjs +9 -9
  87. package/src/main/__tests__/queue-starvation-dispatch-driver.test.cjs +117 -20
  88. package/src/main/__tests__/queueHistory.test.cjs +2 -2
  89. package/src/main/__tests__/rateLimitPollerStreak.test.cjs +38 -3
  90. package/src/main/__tests__/runVerify-landed-commit-outranks.test.cjs +181 -0
  91. package/src/main/__tests__/runVerify.test.cjs +5 -5
  92. package/src/main/__tests__/scheduleJobStatusDrift.test.cjs +3 -3
  93. package/src/main/__tests__/scheduleJobTransitions.test.cjs +2 -2
  94. package/src/main/__tests__/scheduler-adopted-run-supervision.test.cjs +143 -0
  95. package/src/main/__tests__/scheduler-autofix-select.test.cjs +2 -2
  96. package/src/main/__tests__/scheduler-autopromote.test.cjs +2 -2
  97. package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +3 -5
  98. package/src/main/__tests__/scheduler-boot-orphans.test.cjs +78 -97
  99. package/src/main/__tests__/scheduler-default-eligible-heal.test.cjs +161 -0
  100. package/src/main/__tests__/scheduler-dispatch-loop.test.cjs +58 -0
  101. package/src/main/__tests__/scheduler-epic-digest.test.cjs +3 -5
  102. package/src/main/__tests__/scheduler-force-tick-outcome.test.cjs +2 -2
  103. package/src/main/__tests__/scheduler-gate-shadow.test.cjs +119 -0
  104. package/src/main/__tests__/scheduler-guard-verdict-autoresolve.test.cjs +46 -0
  105. package/src/main/__tests__/scheduler-heartbeat-payload.test.cjs +80 -0
  106. package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +25 -18
  107. package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +4 -4
  108. package/src/main/__tests__/scheduler-launch-failure.test.cjs +3 -5
  109. package/src/main/__tests__/scheduler-looks-done.test.cjs +26 -7
  110. package/src/main/__tests__/scheduler-manual-pause.test.cjs +118 -0
  111. package/src/main/__tests__/scheduler-meta-code-sha.test.cjs +26 -3
  112. package/src/main/__tests__/scheduler-prd-missing-skip.test.cjs +17 -2
  113. package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +3 -5
  114. package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +41 -6
  115. package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +62 -6
  116. package/src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs +3 -5
  117. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +23 -4
  118. package/src/main/__tests__/scheduler-reconcile-cwd-preserve.test.cjs +100 -0
  119. package/src/main/__tests__/scheduler-reconcile-invalid-repair.test.cjs +3 -3
  120. package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +4 -4
  121. package/src/main/__tests__/scheduler-shard-quarantine.test.cjs +110 -0
  122. package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +81 -1
  123. package/src/main/__tests__/scheduler-starve-escalation.test.cjs +4 -4
  124. package/src/main/__tests__/scheduler-stranded-autofix-park.test.cjs +245 -0
  125. package/src/main/__tests__/scheduler-supervisor-record.test.cjs +81 -0
  126. package/src/main/__tests__/scheduler-tick-cancel-token.test.cjs +2 -2
  127. package/src/main/__tests__/scheduler-tick-wedge.test.cjs +172 -0
  128. package/src/main/__tests__/scheduler-transient-failure.test.cjs +2 -2
  129. package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +37 -11
  130. package/src/main/__tests__/scheduler-worktree-exec-cwd.test.cjs +3 -5
  131. package/src/main/__tests__/usageSingleFlight.test.cjs +157 -0
  132. package/src/main/__tests__/workTypeLibrary.test.cjs +1 -1
  133. package/src/main/bilkoHost.cjs +16 -7
  134. package/src/main/build-info.json +8 -0
  135. package/src/main/chatRunner.cjs +4 -2
  136. package/src/main/config.cjs +2 -3
  137. package/src/main/docEdit.cjs +4 -2
  138. package/src/main/health.cjs +267 -51
  139. package/src/main/heapSnapshot.cjs +2 -2
  140. package/src/main/historyAggregator.cjs +1 -1
  141. package/src/main/index.cjs +61 -10
  142. package/src/main/ipcSchemas.cjs +6 -17
  143. package/src/main/lib/__tests__/auditLog.test.cjs +38 -0
  144. package/src/main/lib/__tests__/buildIdentity.test.cjs +121 -0
  145. package/src/main/lib/__tests__/cwdClassify.test.cjs +111 -0
  146. package/src/main/lib/__tests__/definitionOfDoneSequence.test.cjs +95 -0
  147. package/src/main/lib/__tests__/delegationReadiness.test.cjs +1 -1
  148. package/src/main/lib/__tests__/dispatchLoop.test.cjs +63 -0
  149. package/src/main/lib/__tests__/gateFixtures.json +20 -0
  150. package/src/main/lib/__tests__/gitWorktree.test.cjs +8 -1
  151. package/src/main/lib/__tests__/instanceLock.test.cjs +93 -7
  152. package/src/main/lib/__tests__/jobSupervisorRecord.test.cjs +78 -0
  153. package/src/main/lib/__tests__/jobWorktreeBootLive.test.cjs +11 -0
  154. package/src/main/lib/__tests__/localAdminHttp.test.cjs +1 -1
  155. package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +6 -1
  156. package/src/main/lib/__tests__/procIdentity.test.cjs +119 -0
  157. package/src/main/lib/__tests__/procName.test.cjs +92 -0
  158. package/src/main/lib/__tests__/queueStoreMachineStateRecovery.test.cjs +67 -0
  159. package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +6 -7
  160. package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +37 -204
  161. package/src/main/lib/__tests__/schedulerPaths.test.cjs +226 -0
  162. package/src/main/lib/__tests__/schedulerPathsWorktree.test.cjs +94 -0
  163. package/src/main/lib/__tests__/schedulerRuntimeState.test.cjs +56 -0
  164. package/src/main/lib/__tests__/sessionSlots.test.cjs +45 -0
  165. package/src/main/lib/__tests__/telemetryBacklog.test.cjs +1 -1
  166. package/src/main/lib/__tests__/upgradeDrain.test.cjs +130 -0
  167. package/src/main/lib/__tests__/usageCircuit.test.cjs +61 -0
  168. package/src/main/lib/__tests__/watchdog-helpers.test.cjs +63 -0
  169. package/src/main/lib/__tests__/watchdog-relaunch.test.cjs +73 -0
  170. package/src/main/lib/activeIndexRebuild.cjs +1 -3
  171. package/src/main/lib/activeSessions.cjs +20 -88
  172. package/src/main/lib/adoptedRunSupervisor.cjs +136 -0
  173. package/src/main/lib/agentModelResolve.cjs +33 -1
  174. package/src/main/lib/agentPersonaSchema.cjs +2 -2
  175. package/src/main/lib/auditLog.cjs +30 -5
  176. package/src/main/lib/buildIdentity.cjs +113 -0
  177. package/src/main/lib/classifyPromptTicket.cjs +3 -2
  178. package/src/main/lib/claudeBin.cjs +37 -2
  179. package/src/main/lib/cleanEnv.cjs +27 -1
  180. package/src/main/lib/credentials.cjs +4 -2
  181. package/src/main/lib/cwdClassify.cjs +185 -0
  182. package/src/main/lib/definitionOfDone.cjs +296 -52
  183. package/src/main/lib/dispatchLoop.cjs +40 -0
  184. package/src/main/lib/effectiveModelInfo.cjs +9 -10
  185. package/src/main/lib/ephemeralCwd.cjs +7 -28
  186. package/src/main/lib/gitWorktree.cjs +95 -8
  187. package/src/main/lib/guardShims.cjs +3 -3
  188. package/src/main/lib/historyRollup.cjs +6 -7
  189. package/src/main/lib/instanceLock.cjs +32 -5
  190. package/src/main/lib/jobSupervisorRecord.cjs +147 -0
  191. package/src/main/lib/jobWorktreeBootLive.cjs +9 -5
  192. package/src/main/lib/localAdminHttp.cjs +9 -27
  193. package/src/main/lib/mcpToolCatalog.cjs +29 -77
  194. package/src/main/lib/opsOwnership.cjs +27 -19
  195. package/src/main/lib/prdAuthoringSeed.cjs +36 -0
  196. package/src/main/lib/prdLocations.cjs +48 -1
  197. package/src/main/lib/procIdentity.cjs +126 -0
  198. package/src/main/lib/procName.cjs +98 -0
  199. package/src/main/lib/projectHomeAdminRoutes.cjs +56 -327
  200. package/src/main/lib/queueHistory.cjs +8 -7
  201. package/src/main/lib/queueStore.cjs +57 -37
  202. package/src/main/lib/quietMachineLease.cjs +19 -2
  203. package/src/main/lib/reservationExpiry.cjs +30 -0
  204. package/src/main/lib/runClaudeP.cjs +4 -2
  205. package/src/main/lib/runLogRetention.cjs +3 -2
  206. package/src/main/lib/scheduleJobSchema.cjs +2 -2
  207. package/src/main/lib/scheduleJobTransitions.cjs +29 -2
  208. package/src/main/lib/schedulerBatch.cjs +27 -2
  209. package/src/main/lib/schedulerPaths.cjs +175 -0
  210. package/src/main/lib/schedulerRuntimeState.cjs +59 -0
  211. package/src/main/lib/sessionSlots.cjs +47 -9
  212. package/src/main/lib/smProcNames.cjs +51 -0
  213. package/src/main/lib/upgradeDrain.cjs +188 -0
  214. package/src/main/lib/usageCircuit.cjs +53 -6
  215. package/src/main/lib/watchdogHelpers.cjs +70 -37
  216. package/src/main/lib/withTimeout.cjs +33 -0
  217. package/src/main/mcpStatus.cjs +4 -2
  218. package/src/main/pluginInstall.cjs +6 -2
  219. package/src/main/projectPages.cjs +41 -145
  220. package/src/main/pty.cjs +8 -1
  221. package/src/main/queueOps.cjs +10 -10
  222. package/src/main/runVerify.cjs +69 -3
  223. package/src/main/scheduler.cjs +1598 -381
  224. package/src/main/seedAgentPersonas.cjs +1 -1
  225. package/src/main/seedDevPlugin.cjs +1 -1
  226. package/src/main/seedSchedulerMcp.cjs +7 -6
  227. package/src/main/seedStatus.cjs +1 -1
  228. package/src/main/supervisor.cjs +5 -3
  229. package/src/main/usage.cjs +126 -62
  230. package/src/preload/api.d.ts +18 -64
  231. package/src/preload/index.cjs +2 -4
  232. package/src/seed/agents/project-home-builder.md +31 -48
  233. package/screenshots/.gitkeep +0 -0
  234. package/screenshots/README-screenshots.md +0 -13
  235. package/src/main/lib/projectPageSummarySchema.cjs +0 -181
  236. package/src/main/teams.cjs +0 -95
  237. package/src/main/templates/project-pages-catalog.json +0 -741
  238. package/src/main/templates/project-pages-default-home.html +0 -123
  239. package/src/main/templates/project-pages-pipeline.md +0 -417
  240. package/src/seed/prompts/code-review/ac-coverage-check.md +0 -8
  241. package/src/seed/prompts/code-review/correctness-only.md +0 -8
  242. package/src/seed/prompts/code-review/full-spectrum-high.md +0 -8
  243. package/src/seed/prompts/code-review/hallucination-check.md +0 -8
  244. package/src/seed/prompts/code-review/public-api-compat.md +0 -8
  245. package/src/seed/prompts/code-review/readability-naming.md +0 -8
  246. package/src/seed/prompts/debugging/bug-as-failing-test.md +0 -8
  247. package/src/seed/prompts/debugging/git-bisect-regression.md +0 -8
  248. package/src/seed/prompts/debugging/instrument-intermittent-bug.md +0 -8
  249. package/src/seed/prompts/debugging/localize-pipeline-failure.md +0 -8
  250. package/src/seed/prompts/debugging/reproduce-then-diagnose.md +0 -8
  251. package/src/seed/prompts/documentation/adr-from-change.md +0 -8
  252. package/src/seed/prompts/documentation/module-readme.md +0 -8
  253. package/src/seed/prompts/documentation/onboarding-plan.md +0 -8
  254. package/src/seed/prompts/documentation/refresh-claude-md.md +0 -8
  255. package/src/seed/prompts/documentation/tsdoc-public-exports.md +0 -8
  256. package/src/seed/prompts/git-pr/conventional-commit.md +0 -8
  257. package/src/seed/prompts/git-pr/draft-pr-title-body.md +0 -8
  258. package/src/seed/prompts/git-pr/pre-commit-safety-sweep.md +0 -8
  259. package/src/seed/prompts/git-pr/release-notes-block.md +0 -8
  260. package/src/seed/prompts/git-pr/split-large-pr.md +0 -8
  261. package/src/seed/prompts/performance/bundle-startup-audit.md +0 -8
  262. package/src/seed/prompts/performance/complexity-audit.md +0 -8
  263. package/src/seed/prompts/performance/cpu-profile-hot-path.md +0 -8
  264. package/src/seed/prompts/performance/db-query-plan-review.md +0 -8
  265. package/src/seed/prompts/performance/memory-leak-hunt.md +0 -8
  266. package/src/seed/prompts/qa/api-contract-tests.md +0 -8
  267. package/src/seed/prompts/qa/e2e-critical-path.md +0 -8
  268. package/src/seed/prompts/qa/failing-test-for-bug.md +0 -8
  269. package/src/seed/prompts/qa/find-missing-test-coverage.md +0 -8
  270. package/src/seed/prompts/qa/stabilize-flaky-test.md +0 -8
  271. package/src/seed/prompts/qa/tdd-red-first.md +0 -8
  272. package/src/seed/prompts/qa/visual-regression-review.md +0 -8
  273. package/src/seed/prompts/qa/wcag-axe-scan.md +0 -8
  274. package/src/seed/prompts/refactoring/dead-code-sweep.md +0 -8
  275. package/src/seed/prompts/refactoring/extract-duplicated-pattern.md +0 -8
  276. package/src/seed/prompts/refactoring/modernize-legacy-file.md +0 -8
  277. package/src/seed/prompts/refactoring/reduce-cyclomatic-complexity.md +0 -8
  278. package/src/seed/prompts/refactoring/tighten-module-boundaries.md +0 -8
  279. package/src/seed/prompts/security/authz-audit.md +0 -8
  280. package/src/seed/prompts/security/crypto-correctness.md +0 -8
  281. package/src/seed/prompts/security/cwe-top-25-hunt.md +0 -8
  282. package/src/seed/prompts/security/dependency-audit.md +0 -8
  283. package/src/seed/prompts/security/ipc-boundary-hardening.md +0 -8
  284. package/src/seed/prompts/security/owasp-top-10-staged-diff.md +0 -8
  285. package/src/seed/prompts/security/secret-credential-scan.md +0 -8
  286. package/web/README.md +0 -41
  287. package/web/project-pages/logic/dist/logic.cjs +0 -4709
  288. package/web/project-pages/render.cjs +0 -70
  289. package/web/project-pages/renderer/dist/renderer.cjs +0 -18900
  290. package/web/project-pages/validate-summary.cjs +0 -62
@@ -47,13 +47,15 @@ const fs = require('node:fs');
47
47
  const fsp = require('node:fs/promises');
48
48
  const path = require('node:path');
49
49
  const os = require('node:os');
50
+ const { startDispatchLoop } = require('./lib/dispatchLoop.cjs');
51
+ const schedulerPaths = require('./lib/schedulerPaths.cjs');
50
52
  const { randomUUID } = require('node:crypto');
51
53
  const { execFile, execFileSync } = require('node:child_process');
52
54
  const { ipcMain } = require('electron');
53
55
  const billing = require('./usage.cjs');
54
56
  const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
55
57
  const supervisor = require('./supervisor.cjs');
56
- const { resolveClaudeBin, probeClaudeVersion } = require('./lib/claudeBin.cjs');
58
+ const { resolveClaudeBin, claudeSpawnTarget, probeClaudeVersion } = require('./lib/claudeBin.cjs');
57
59
  const launchFailure = require('./lib/launchFailure.cjs');
58
60
  const { appendError } = require('./lib/opsErrorLog.cjs');
59
61
  const { readTail } = require('./lib/fileTail.cjs');
@@ -67,6 +69,7 @@ const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
67
69
  const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
68
70
  const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
69
71
  const { resolveBindingRateLimitReset } = require('./lib/rateLimitWindow.cjs');
72
+ const { isResetFresh, bindingWindow, degradedBudget } = require('./lib/usageCircuit.cjs');
70
73
  const { computeQueueHealth } = require('./lib/queueHealth.cjs');
71
74
  const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
72
75
  const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
@@ -83,6 +86,7 @@ const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require(
83
86
  const { isFixPlanSlug, classifyDiscoveredFixPlan, resolveIsFixPlan } = require('./lib/fixPlanSlug.cjs');
84
87
  const { landedSinceRun, landedOnMainSince } = require('./lib/landedSinceRun.cjs');
85
88
  const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
89
+ const { identity: procIdentityOf, isDifferentProcess } = require('./lib/procIdentity.cjs');
86
90
  const logs = require('./logs.cjs');
87
91
  const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
88
92
  const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
@@ -131,19 +135,21 @@ const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD,
131
135
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
132
136
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
133
137
  const queueHistory = require('./lib/queueHistory.cjs');
138
+ const { resolveGate, runGateSequence } = require('./lib/definitionOfDone.cjs');
134
139
  const queueOps = require('./queueOps.cjs');
135
140
  // Feedback-auto-PRD sweep — formerly only run by the external scheduler-watchdog
136
141
  // while the app was down (PRD 686 moved it in-app so it also runs while alive).
137
142
  // Plain Node module, no Electron dependency; queuePath/prdsDir defaults already
138
143
  // match ROOT/QUEUE_PATH below since both resolve the same ~/.claude/session-manager
139
144
  // home-dir layout.
140
- const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
145
+ const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs, deriveProjectCwdFromPrdPath } = require('./lib/prdLocations.cjs');
141
146
  const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
142
147
  const agentModelResolve = require('./lib/agentModelResolve.cjs');
143
148
  const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
144
149
  const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
145
150
  const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
146
151
  const { appendAuditEvent } = require('./lib/auditLog.cjs');
152
+ const { withTimeout } = require('./lib/withTimeout.cjs');
147
153
 
148
154
  // ---------- origin session resolution (PRD 832) ----------
149
155
  // An Epic IS a tagged claude session — job rows carry the originating
@@ -164,12 +170,15 @@ function resolveOriginSessionId(cwd, epicId) {
164
170
  }
165
171
  const sessionSlots = require('./lib/sessionSlots.cjs');
166
172
  const quietMachineLease = require('./lib/quietMachineLease.cjs');
173
+ const runtimeState = require('./lib/schedulerRuntimeState.cjs');
167
174
  const jobWorktree = require('./lib/jobWorktree.cjs');
168
175
  const gitWorktree = require('./lib/gitWorktree.cjs');
169
176
  const { buildJobWorktreeIsLive } = require('./lib/jobWorktreeBootLive.cjs');
170
177
  const { buildTerminalOrphanIsLive } = require('./lib/jobWorktreeTerminalOrphanLive.cjs');
171
178
  const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
172
179
  const queueStore = require('./lib/queueStore.cjs');
180
+ const supervisorRecord = require('./lib/jobSupervisorRecord.cjs');
181
+ const adoptedRunSupervisor = require('./lib/adoptedRunSupervisor.cjs');
173
182
  const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
174
183
  const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
175
184
  const { computeDispositionRewrite } = require('./lib/prdDisposition.cjs');
@@ -182,20 +191,26 @@ const { allProjectCwds } = require('./lib/activeSessions.cjs');
182
191
  // an exemption it should have applied landed on disk, and nothing in the
183
192
  // run record showed that; this is the fix).
184
193
  const SCHEDULER_BOOTED_AT = new Date().toISOString();
185
- // Resolves against __dirname (this app's OWN source checkout) — unaffected by
186
- // PRD 994's job worktrees, which live under a job's PROJECT cwd, never under
187
- // this app's install directory.
188
- const SCHEDULER_CODE_SHA = (() => {
189
- try {
190
- return execFileSync('git', ['-C', __dirname, 'rev-parse', '--short', 'HEAD'], {
191
- timeout: 5000,
192
- encoding: 'utf8',
193
- stdio: ['ignore', 'pipe', 'ignore'],
194
- }).trim();
195
- } catch {
196
- return null;
197
- }
198
- })();
194
+ // A production npx install ships no .git at all, so a runtime `git
195
+ // rev-parse` from __dirname was structurally always null there — every
196
+ // production run-meta sidecar recorded schedulerCodeSha: null. buildIdentity
197
+ // resolves build-info.json (baked at publish time) first, falling back to a
198
+ // non-walking git read only in a dev checkout / job worktree — see
199
+ // src/main/lib/buildIdentity.cjs's header.
200
+ const { resolveBuildIdentity, readInstalledBuildInfo } = require('./lib/buildIdentity.cjs');
201
+ const upgradeDrain = require('./lib/upgradeDrain.cjs');
202
+ const SCHEDULER_BUILD_IDENTITY = resolveBuildIdentity({ bootedAt: SCHEDULER_BOOTED_AT });
203
+ const SCHEDULER_CODE_SHA = SCHEDULER_BUILD_IDENTITY.codeSha;
204
+ // Spread into EVERY metaPath writer below (grep `metaPath` for the full
205
+ // list) — single source so a future field never lands in some sidecars and
206
+ // not others, the exact gap that left 3 of 5 writers silently missing
207
+ // schedulerBootedAt/schedulerCodeSha before this constant existed.
208
+ const SCHEDULER_META_IDENTITY = {
209
+ schedulerBootedAt: SCHEDULER_BOOTED_AT,
210
+ schedulerCodeSha: SCHEDULER_CODE_SHA,
211
+ schedulerVersion: SCHEDULER_BUILD_IDENTITY.version,
212
+ schedulerBuiltAt: SCHEDULER_BUILD_IDENTITY.builtAt,
213
+ };
199
214
 
200
215
  const MAX_INVESTIGATION_DURATION_MS = 30 * 60_000;
201
216
 
@@ -590,7 +605,7 @@ function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAf
590
605
  // executor-created stash (never guesses when there are 2+); reports anything
591
606
  // it can't safely resolve on the returned object so the caller can surface it
592
607
  // on the job row instead of finishing silently green.
593
- async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug }) {
608
+ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug, landedCommit }) {
594
609
  try {
595
610
  const [stashAfter, headAfter] = await Promise.all([
596
611
  module.exports.stashList(cwd),
@@ -655,7 +670,16 @@ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBef
655
670
  pathsCommittedDuringRun,
656
671
  existsAfter,
657
672
  });
658
- if (reverted.length) {
673
+ // Ground truth outranks the baseline diff (2026-09-18, 1229-fo-03): the
674
+ // dirty baseline is invalidated by ANY later writer (a human commit that
675
+ // sweeps the same paths), so it can't prove a revert on its own. The
676
+ // job's own landedCommit still being an ancestor of HEAD proves its work
677
+ // was not discarded — anchored to that sha, not to the baseline.
678
+ const workSurvives = reverted.length > 0
679
+ && await module.exports.landedCommitIsAncestorOfHead(cwd, landedCommit);
680
+ if (workSurvives) {
681
+ console.log(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} baseline path(s) went clean but landed commit ${String(landedCommit).slice(0, 7)} is still an ancestor of HEAD — not a revert`);
682
+ } else if (reverted.length) {
659
683
  result.reverted = reverted;
660
684
  console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
661
685
  }
@@ -895,13 +919,9 @@ function isQueueRowRegression({ statusBefore, statusAfter, historyLenBefore, his
895
919
  return statusBefore === 'running' && statusAfter === 'pending' && historyLenAfter < historyLenBefore;
896
920
  }
897
921
 
898
- const ROOT = path.join(os.homedir(), '.claude', 'session-manager', 'scheduled-plans');
899
- const PRDS_DIR = path.join(ROOT, 'prds');
900
- const RUNS_DIR = path.join(ROOT, 'runs');
901
- const PRDS_ARCHIVE_DIR = path.join(ROOT, 'prds-archived');
902
- const QUEUE_PATH = path.join(ROOT, 'queue.json');
903
- const SCHEDULER_STATE_PATH = path.join(os.homedir(), '.claude', 'session-manager', 'scheduler-state.json');
904
- const HEARTBEAT_PATH = path.join(os.homedir(), '.claude', 'session-manager', 'scheduler-heartbeat.log');
922
+ // Machine-wide roots resolve lazily via lib/schedulerPaths.cjs (SM_SCHEDULER_HOME
923
+ // override) — never module-scope consts. The ROOT/PRDS_DIR/RUNS_DIR/
924
+ // SCHEDULER_STATE_PATH exports below are lazy getters over the same resolver.
905
925
  const HEARTBEAT_MAX_BYTES = 1024 * 1024;
906
926
  // DEFAULT_PROJECT_CWD imported from lib/schedulerBatch.cjs (single source of truth).
907
927
 
@@ -1020,7 +1040,7 @@ function biasJobOomScore(pid) {
1020
1040
  * (reconcile, list-prds, lint, rescan).
1021
1041
  */
1022
1042
  function candidatePrdsDirs() {
1023
- return [PRDS_DIR, ...resolvePrdsDirs()];
1043
+ return [schedulerPaths.prdsRoot(), ...resolvePrdsDirs()];
1024
1044
  }
1025
1045
 
1026
1046
  /**
@@ -1141,7 +1161,7 @@ function prdArchivedSkipResult(job, cwd, sessionId, startedAt, safeLog, closeFd,
1141
1161
  const finishedAt = Date.now();
1142
1162
  config.writeJsonSync(metaPath, {
1143
1163
  slug: job.slug, cwd, sessionId, exitCode: 0, skipped: reason,
1144
- note: msg, startedAt, finishedAt, durationMs: 0,
1164
+ note: msg, startedAt, finishedAt, durationMs: 0, ...SCHEDULER_META_IDENTITY,
1145
1165
  });
1146
1166
  return { exitCode: 0, durationMs: 0, skipped: reason, note: msg, sessionId };
1147
1167
  }
@@ -1283,17 +1303,21 @@ async function retireCompletedSlugs(slugs) {
1283
1303
  // Bundled authoring guide seeded into the scheduler dir so the session-manager-dev
1284
1304
  // plugin's /develop and /prd skills — which reference this stable `~`-absolute
1285
1305
  // path — work on any user's machine, not just the author's.
1306
+ // Line 1 of the template is `<!-- PRD_AUTHORING.md vN -->`: bump vN whenever the
1307
+ // template changes, or existing installs never receive the update.
1286
1308
  const PRD_AUTHORING_TEMPLATE = path.join(__dirname, 'templates', 'PRD_AUTHORING.md');
1287
- const PRD_AUTHORING_DEST = path.join(ROOT, 'PRD_AUTHORING.md');
1288
1309
 
1289
1310
  function ensureDirs() {
1290
- fs.mkdirSync(PRDS_DIR, { recursive: true });
1291
- fs.mkdirSync(RUNS_DIR, { recursive: true });
1292
- // Seed the authoring guide once; never clobber a user's edited copy.
1311
+ fs.mkdirSync(schedulerPaths.prdsRoot(), { recursive: true });
1312
+ fs.mkdirSync(schedulerPaths.runsDir(), { recursive: true });
1313
+ // Re-seed the guide whenever the bundled template's version stamp differs.
1293
1314
  try {
1294
- if (!fs.existsSync(PRD_AUTHORING_DEST) && fs.existsSync(PRD_AUTHORING_TEMPLATE)) {
1295
- fs.copyFileSync(PRD_AUTHORING_TEMPLATE, PRD_AUTHORING_DEST);
1296
- }
1315
+ const authoringDest = path.join(schedulerPaths.scheduledPlansRoot(), 'PRD_AUTHORING.md');
1316
+ seedAuthoringGuide({
1317
+ src: PRD_AUTHORING_TEMPLATE,
1318
+ dest: authoringDest,
1319
+ write: (abs, text) => config.writeTextAtomic(abs, text, { writer: 'scheduler' }),
1320
+ }).catch(() => { /* non-fatal, same as below */ });
1297
1321
  } catch { /* non-fatal: the guide is a convenience, not load-bearing for a run */ }
1298
1322
  }
1299
1323
 
@@ -1321,8 +1345,9 @@ function ensureDirs() {
1321
1345
  * queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
1322
1346
  * sweep archives it before reconcile can ever turn it into a pending job.
1323
1347
  */
1324
- async function consolidateAllFlatPrds(cwds) {
1348
+ async function consolidateAllFlatPrds(cwds, skipCwds) {
1325
1349
  for (const cwd of cwds) {
1350
+ if (skipCwds?.has(cwd)) continue; // torn shard: its PRDs are not ours to touch this pass
1326
1351
  try {
1327
1352
  const c = await consolidateFlatPrds(cwd);
1328
1353
  if (c.moved > 0) {
@@ -1354,7 +1379,7 @@ async function consolidateAllFlatPrds(cwds) {
1354
1379
  async function runPrdMigration() {
1355
1380
  let result;
1356
1381
  try {
1357
- result = await migratePrds(PRDS_DIR);
1382
+ result = await migratePrds(schedulerPaths.prdsRoot());
1358
1383
  } catch (e) {
1359
1384
  logs.writeLine({ level: 'error', scope: 'scheduler', message: 'PRD migration failed', meta: { error: e?.message } });
1360
1385
  return null;
@@ -1365,7 +1390,7 @@ async function runPrdMigration() {
1365
1390
  level: 'warn',
1366
1391
  scope: 'scheduler',
1367
1392
  message: `PRD migration: ${result.unresolved.length} file(s) left in legacy dir`,
1368
- meta: { legacyDir: PRDS_DIR, unresolved: result.unresolved },
1393
+ meta: { legacyDir: schedulerPaths.prdsRoot(), unresolved: result.unresolved },
1369
1394
  });
1370
1395
  for (const u of result.unresolved) {
1371
1396
  console.warn(`[scheduler] PRD migration: left ${u.file} in legacy dir (${u.reason})`);
@@ -1432,7 +1457,7 @@ const QUEUE_BAK_KEEP = 5;
1432
1457
  async function sweepQueueBackups() {
1433
1458
  let entries;
1434
1459
  try {
1435
- entries = await fsp.readdir(ROOT);
1460
+ entries = await fsp.readdir(schedulerPaths.scheduledPlansRoot());
1436
1461
  } catch {
1437
1462
  return;
1438
1463
  }
@@ -1448,7 +1473,7 @@ async function sweepQueueBackups() {
1448
1473
  let removed = 0;
1449
1474
  for (const f of toDelete) {
1450
1475
  try {
1451
- await fsp.unlink(path.join(ROOT, f));
1476
+ await fsp.unlink(path.join(schedulerPaths.scheduledPlansRoot(), f));
1452
1477
  removed++;
1453
1478
  } catch (e) {
1454
1479
  console.warn('[scheduler] backup sweep: unlink failed', f, e?.message);
@@ -1464,20 +1489,24 @@ async function sweepQueueBackups() {
1464
1489
  // callback that must flush meta.json before resolving) — replacing with async
1465
1490
  // would deadlock the exit path.
1466
1491
  const config = require('./config.cjs');
1492
+ const { seedAuthoringGuide } = require('./lib/prdAuthoringSeed.cjs');
1467
1493
  const atomicWriteJsonSync = (p, data) => config.writeJsonSync(p, data);
1468
1494
 
1469
1495
  // ---------- scheduler-state.json (sidecar) ----------
1470
1496
 
1471
1497
  function loadSchedulerState() {
1472
1498
  try {
1473
- const raw = fs.readFileSync(SCHEDULER_STATE_PATH, 'utf8');
1499
+ const raw = fs.readFileSync(schedulerPaths.schedulerStatePath(), 'utf8');
1474
1500
  const s = JSON.parse(raw);
1475
1501
  if (s.lastObservedReset) cachedNextReset = s.lastObservedReset;
1502
+ if (typeof s.lastResetObservedAt === 'number') lastResetObservedAtMs = s.lastResetObservedAt;
1476
1503
  if (typeof s.consecutiveFailures === 'number') consecutiveFailures = s.consecutiveFailures;
1477
1504
  if (typeof s.backoffMs === 'number') backoffMs = s.backoffMs;
1478
1505
  if (typeof s.pauseClearedManuallyAt === 'number') pauseClearedManuallyAt = s.pauseClearedManuallyAt;
1479
1506
  if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
1480
1507
  if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
1508
+ if (typeof s.failureStreakWarnedAt === 'number') failureStreakWarnedAt = s.failureStreakWarnedAt;
1509
+ if (typeof s.lastEscalationAt === 'number') lastEscalationAtMs = s.lastEscalationAt;
1481
1510
  } catch { /* first boot or corrupt — start fresh */ }
1482
1511
  }
1483
1512
 
@@ -1487,10 +1516,14 @@ function persistSchedulerState() {
1487
1516
  // require threading awaits through pause/resume bookkeeping for negligible
1488
1517
  // benefit — the file is well under one page.
1489
1518
  try {
1490
- config.writeJsonSync(SCHEDULER_STATE_PATH, {
1519
+ config.writeJsonSync(schedulerPaths.schedulerStatePath(), {
1491
1520
  version: 1,
1492
1521
  lastObservedReset: cachedNextReset,
1493
- lastResetObservedAt: cachedNextReset ? Date.now() : null,
1522
+ // Only stamped at the moment a FRESH reset was actually observed (see
1523
+ // recordObservedReset) — never Date.now() on every persist call, which
1524
+ // used to make a stale cachedNextReset look freshly-confirmed on every
1525
+ // tick even when nothing new had been read.
1526
+ lastResetObservedAt: lastResetObservedAtMs,
1494
1527
  lastPollAt,
1495
1528
  consecutiveFailures,
1496
1529
  backoffMs,
@@ -1498,6 +1531,13 @@ function persistSchedulerState() {
1498
1531
  pausedSince: null,
1499
1532
  pauseClearedManuallyAt,
1500
1533
  failureStreakWarned,
1534
+ failureStreakWarnedAt,
1535
+ lastEscalationAt: lastEscalationAtMs,
1536
+ // Circuit fields are read fresh from the live shared breaker each
1537
+ // persist — health.cjs (a separate `npm run health` process) reads
1538
+ // THESE persisted values, since it never holds the in-memory circuit.
1539
+ usageCircuitState: billing.usageCircuit.state(),
1540
+ usageCircuitOpenedAt: billing.usageCircuit.openedAt(),
1501
1541
  });
1502
1542
  } catch (e) {
1503
1543
  console.warn('[scheduler] failed to persist scheduler state', e?.message);
@@ -1510,18 +1550,279 @@ function appendHeartbeat(entry) {
1510
1550
  try {
1511
1551
  const line = JSON.stringify(entry) + '\n';
1512
1552
  let size = 0;
1513
- try { size = fs.statSync(HEARTBEAT_PATH).size; } catch { /* new file */ }
1553
+ try { size = fs.statSync(schedulerPaths.heartbeatPath()).size; } catch { /* new file */ }
1514
1554
  if (size >= HEARTBEAT_MAX_BYTES) {
1515
- const rotated = HEARTBEAT_PATH + '.1';
1555
+ const rotated = schedulerPaths.heartbeatPath() + '.1';
1516
1556
  try { fs.unlinkSync(rotated); } catch { /* */ }
1517
- try { fs.renameSync(HEARTBEAT_PATH, rotated); } catch { /* */ }
1557
+ try { fs.renameSync(schedulerPaths.heartbeatPath(), rotated); } catch { /* */ }
1518
1558
  }
1519
- fs.appendFileSync(HEARTBEAT_PATH, line);
1559
+ fs.appendFileSync(schedulerPaths.heartbeatPath(), line);
1520
1560
  } catch (e) {
1521
1561
  console.warn('[scheduler] heartbeat write failed', e?.message);
1522
1562
  }
1523
1563
  }
1524
1564
 
1565
+ // Build identity stamped on every heartbeat line — memoized at boot
1566
+ // (SCHEDULER_BUILD_IDENTITY), so a tick costs no git or fs work.
1567
+ function heartbeatBuild() {
1568
+ const { version, codeSha, builtAt } = SCHEDULER_BUILD_IDENTITY;
1569
+ return { version, codeSha, builtAt };
1570
+ }
1571
+
1572
+ // ---------- upgrade drain driver (lib/upgradeDrain.cjs) ----------
1573
+
1574
+ // Set by index.cjs: performs the actual app teardown + relaunch + exit.
1575
+ let restartHandler = null;
1576
+ function setRestartHandler(fn) { restartHandler = typeof fn === 'function' ? fn : null; }
1577
+ let drainDriving = false;
1578
+
1579
+ function drainSnapshot(jobs) {
1580
+ const running = new Set();
1581
+ let investigating = 0;
1582
+ for (const j of jobs ?? []) {
1583
+ if (j?.status === 'running') running.add(j.slug);
1584
+ else if (j?.status === 'investigating') investigating++;
1585
+ }
1586
+ for (const slug of runningSet) running.add(slug);
1587
+ // Deferred investigations are not busy: while draining they never spawn.
1588
+ return { running: running.size, investigating: Math.max(investigating, runtimeState.investigationCount()) };
1589
+ }
1590
+
1591
+ /**
1592
+ * Restart is triggered automatically ONLY when the installed build-info.json
1593
+ * differs from the running codeSha (an install/update already happened) —
1594
+ * never by polling npm. SM_AUTO_UPGRADE_RESTART=0 disables it.
1595
+ */
1596
+ function maybeAutoRequestRestart() {
1597
+ if (process.env.SM_AUTO_UPGRADE_RESTART === '0' || process.env.SM_DEV === '1') return null;
1598
+ const installed = readInstalledBuildInfo();
1599
+ const installedSha = typeof installed?.gitShortSha === 'string' ? installed.gitShortSha : null;
1600
+ if (!upgradeDrain.installedBuildDiffers({ running: SCHEDULER_CODE_SHA, installed: installedSha })) return null;
1601
+ return upgradeDrain.requestRestart({ reason: `installed build ${installedSha} differs from running ${SCHEDULER_CODE_SHA}`, requestedBy: 'auto-upgrade' });
1602
+ }
1603
+
1604
+ async function driveUpgradeDrain(state) {
1605
+ if (drainDriving) return;
1606
+ drainDriving = true;
1607
+ try {
1608
+ let request = upgradeDrain.readRestartRequest();
1609
+ if (!request && !state.drain?.active) request = maybeAutoRequestRestart();
1610
+ const { action, reason } = upgradeDrain.evaluateDrain({
1611
+ request,
1612
+ queueSnapshot: drainSnapshot(state.jobs),
1613
+ drainState: state.drain,
1614
+ now: Date.now(),
1615
+ });
1616
+ if (action === 'none' || action === 'wait') { drainActive = Boolean(state.drain?.active); return; }
1617
+ if (action === 'pause') {
1618
+ await mutate((s) => { s.drain = { active: true, since: new Date().toISOString(), requestedAt: request.requestedAt }; });
1619
+ drainActive = true;
1620
+ appendAuditEvent('upgrade_drain_started', { reason: request.reason, requestedBy: request.requestedBy });
1621
+ await broadcast({ flush: true });
1622
+ return;
1623
+ }
1624
+ if (action === 'abort') {
1625
+ upgradeDrain.retireRestartRequest();
1626
+ await mutate((s) => { s.drain = null; });
1627
+ drainActive = false;
1628
+ appendAuditEvent('upgrade_drain_aborted', { reason });
1629
+ await broadcast({ flush: true });
1630
+ runDueJobs().catch(() => {});
1631
+ return;
1632
+ }
1633
+ // 'restart': the FINAL zero-busy check runs inside a mutate, immediately
1634
+ // before exit — the snapshot above may be stale by now.
1635
+ let go = false;
1636
+ await mutate((s) => {
1637
+ const snap = drainSnapshot(s.jobs);
1638
+ if (!s.drain?.active || snap.running + snap.investigating > 0) return;
1639
+ upgradeDrain.stampDrainCompleted();
1640
+ go = true;
1641
+ });
1642
+ if (!go) return;
1643
+ appendAuditEvent('upgrade_drain_restart', { reason: request.reason, requestedBy: request.requestedBy });
1644
+ try {
1645
+ if (!restartHandler) throw new Error('no restart handler registered');
1646
+ upgradeDrain.markRestarting();
1647
+ await restartHandler(request);
1648
+ // Only reached when the handler did NOT exit the process (dev-server
1649
+ // in-place reboot): the restart is done, so retire the drain here.
1650
+ upgradeDrain.clearRestartingMarker();
1651
+ upgradeDrain.retireRestartRequest();
1652
+ await mutate((s) => { s.drain = null; });
1653
+ drainActive = false;
1654
+ } catch (e) {
1655
+ // Never strand the queue drained-and-paused: fall back to an abort.
1656
+ console.error('[scheduler] drain restart failed — aborting drain:', e?.message ?? e);
1657
+ upgradeDrain.clearRestartingMarker();
1658
+ upgradeDrain.retireRestartRequest();
1659
+ await mutate((s) => { s.drain = null; });
1660
+ drainActive = false;
1661
+ runDueJobs().catch(() => {});
1662
+ }
1663
+ } finally {
1664
+ drainDriving = false;
1665
+ }
1666
+ }
1667
+
1668
+ /** Boot: clear a leftover drain whose request is complete (the restart happened) or gone. */
1669
+ async function clearStaleDrainAtBoot(boot) {
1670
+ const request = upgradeDrain.readRestartRequest();
1671
+ const action = upgradeDrain.bootDrainAction({ drainState: boot.drain, request });
1672
+ upgradeDrain.clearRestartingMarker();
1673
+ if (request?.drainCompletedAt) upgradeDrain.retireRestartRequest();
1674
+ if (action === 'clear') await mutate((s) => { s.drain = null; });
1675
+ drainActive = action === 'keep';
1676
+ }
1677
+
1678
+ /**
1679
+ * heartbeatTick(deps?) — one 60 s heartbeat interval body. Each subsystem
1680
+ * (queue read + starvation watchdog, stall detector, heartbeat write) runs in
1681
+ * its own try/catch so one throw can't silently skip the others. Any failure
1682
+ * makes the written line `degraded: true` + `errors`; watchdogHelpers'
1683
+ * heartbeatFresh() and health.cjs's readFreshHeartbeat() treat such a line as
1684
+ * NOT fresh, so a throw never disarms the external watchdog or fakes
1685
+ * utilization health never read.
1686
+ */
1687
+ function heartbeatTick(deps = {}) {
1688
+ const readQueue = deps.readQueueSync ?? readQueueSync;
1689
+ const errors = [];
1690
+ const guard = (subsystem, fn) => {
1691
+ try {
1692
+ return fn();
1693
+ } catch (e) {
1694
+ errors.push({ subsystem, error: e?.message ?? String(e) });
1695
+ console.error(`[scheduler] heartbeat subsystem "${subsystem}" failed`, e);
1696
+ return undefined;
1697
+ }
1698
+ };
1699
+
1700
+ const s = guard('queue-read-starvation-watchdog', () => {
1701
+ const q = readQueue();
1702
+ // NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
1703
+ // running, something must drive it. This is the only driver that does
1704
+ // not depend on the billing poll loop, a pause timer, or a completing
1705
+ // job to schedule the next tick — every one of which has failed at
1706
+ // least once. See classifyQueueStarvation.
1707
+ if (!q.unreadable) {
1708
+ runQueueStarvationWatchdog(q).catch((e) => console.error('[scheduler] starvation watchdog error', e));
1709
+ // Restart-request drain state machine rides this same 60 s interval —
1710
+ // no new driver.
1711
+ driveUpgradeDrain(q).catch((e) => console.error('[scheduler] upgrade drain error', e));
1712
+ }
1713
+ return q;
1714
+ });
1715
+
1716
+ let stall;
1717
+ if (s) {
1718
+ stall = guard('stall-detector', () => {
1719
+ const summary = computeStallSummary(s);
1720
+ // Per-project alerting (see computeStallSummary's header): a project
1721
+ // stalled while others are busy must still fire, and one project
1722
+ // recovering must not clear or suppress another's still-open episode —
1723
+ // that is exactly what a single module-level stallSince/stallToasted
1724
+ // flag masked before (the burrow-vs-others incident this PRD fixes).
1725
+ const now = Date.now();
1726
+ const stalledCwds = Object.keys(summary.byProject).filter((cwd) => summary.byProject[cwd].stalled);
1727
+ for (const cwd of [...stallSince.keys()]) {
1728
+ if (!stalledCwds.includes(cwd)) {
1729
+ stallSince.delete(cwd);
1730
+ stallToasted.delete(cwd);
1731
+ }
1732
+ }
1733
+ const toAlert = [];
1734
+ for (const cwd of stalledCwds) {
1735
+ if (!stallSince.has(cwd)) stallSince.set(cwd, now);
1736
+ if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
1737
+ stallToasted.set(cwd, true);
1738
+ toAlert.push(cwd);
1739
+ }
1740
+ }
1741
+ if (toAlert.length > 0) {
1742
+ console.error(
1743
+ `[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
1744
+ + `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
1745
+ summary.byProject,
1746
+ );
1747
+ appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: summary.total, byProject: summary.byProject });
1748
+ if (mainWindow && !mainWindow.isDestroyed()) {
1749
+ sendIfAlive(mainWindow, 'schedule:stall', {
1750
+ message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
1751
+ projects: toAlert,
1752
+ total: summary.total,
1753
+ byProject: summary.byProject,
1754
+ });
1755
+ }
1756
+ }
1757
+ return summary;
1758
+ });
1759
+ }
1760
+
1761
+ let entry = null;
1762
+ if (s && stall && errors.length === 0) {
1763
+ entry = guard('heartbeat-write', () => {
1764
+ // Initialise from the real status union (scheduleJobSchema.cjs) rather
1765
+ // than a hand-maintained subset — the old `{ pending, running, completed,
1766
+ // failed }` literal silently minted a NEW key for any other value, which
1767
+ // is how a heartbeat with a `queued: 2` bucket looked like "normal" 24h
1768
+ // visibility instead of the alarm it should have been. Any row whose
1769
+ // status isn't in JOB_STATUSES routes into `unknown`, never a
1770
+ // freshly-minted key.
1771
+ const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
1772
+ counts.unknown = 0;
1773
+ for (const j of s.jobs) {
1774
+ if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
1775
+ counts[j.status] += 1;
1776
+ } else {
1777
+ counts.unknown += 1;
1778
+ }
1779
+ }
1780
+ // Logical-liveness signal for the external watchdog (see watchdogHelpers
1781
+ // evaluateDispatchLiveness): computed once per heartbeat from the same
1782
+ // queue snapshot. pendingDispatchable = pending rows minus those
1783
+ // terminally blocked behind a failed/skipped dependency.
1784
+ const blockedPending = computeBlockedChains(s.jobs).reduce((n, c) => n + c.blocked, 0);
1785
+ const dispatch = {
1786
+ lastDispatchAttemptAt: s.lastDispatchAttemptAt ?? null,
1787
+ lastRunAt: s.lastRunAt ?? null,
1788
+ lastTickReason: lastTick?.reason ?? null,
1789
+ pendingDispatchable: Math.max(0, counts.pending - blockedPending),
1790
+ runningCount: counts.running,
1791
+ paused: Boolean(s.paused),
1792
+ drain: s.drain?.active ? { since: s.drain.since ?? null, requestedAt: s.drain.requestedAt ?? null } : null,
1793
+ };
1794
+ return {
1795
+ ts: Date.now(),
1796
+ pid: process.pid,
1797
+ build: heartbeatBuild(),
1798
+ counts,
1799
+ dispatch,
1800
+ stall: { stalled: stall.stalled, total: stall.total },
1801
+ paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
1802
+ quarantinedCwds: (s.unreadableCwds ?? []).map((u) => u.cwd),
1803
+ nextReset: cachedNextReset,
1804
+ utilization: cachedUtilization,
1805
+ consecutiveFailures,
1806
+ // State/consecutiveFailures/degraded-budget snapshot of the shared
1807
+ // usage-meter breaker, so a human reading only the heartbeat log can
1808
+ // see the meter's own health apart from the queue's.
1809
+ usageMeter: {
1810
+ state: billing.usageCircuit.state(),
1811
+ consecutiveFailures,
1812
+ degradedBudget: computeDegradedBudget(),
1813
+ },
1814
+ };
1815
+ });
1816
+ }
1817
+ if (!entry || errors.length > 0) {
1818
+ // Deliberately carries no utilization/counts: this line says "I ran but
1819
+ // could not read state", and consumers must not mistake it for a fresh read.
1820
+ entry = { ts: Date.now(), pid: process.pid, build: heartbeatBuild(), degraded: true, errors };
1821
+ }
1822
+ appendHeartbeat(entry);
1823
+ return entry;
1824
+ }
1825
+
1525
1826
  /**
1526
1827
  * computeStallSummary(state) → { stalled, total, running, pending, byProject }
1527
1828
  *
@@ -2015,6 +2316,22 @@ function findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
2015
2316
  // the .bak-* snapshots.
2016
2317
  const quarantinedPaths = new Set();
2017
2318
  function flagUnreadable(state) {
2319
+ // Per-shard quarantine (queueStore.unreadableCwds): snapshot each torn shard
2320
+ // once and name it, but never halt — other projects keep dispatching.
2321
+ for (const u of state.unreadableCwds ?? []) {
2322
+ if (!quarantinedPaths.has(u.file)) {
2323
+ quarantinedPaths.add(u.file);
2324
+ try {
2325
+ fs.copyFileSync(u.file, `${u.file}.corrupt-${Date.now()}`);
2326
+ } catch { /* best-effort: the read already failed, the copy may too */ }
2327
+ console.error(`[scheduler] project queue shard quarantined (${u.cwd}): ${u.error}`);
2328
+ logs.writeLine({
2329
+ level: 'error', scope: 'scheduler',
2330
+ message: `project queue shard unreadable — ${u.cwd} is quarantined, other projects keep dispatching`,
2331
+ meta: { cwd: u.cwd, path: u.file, error: u.error },
2332
+ });
2333
+ }
2334
+ }
2018
2335
  if (!state.unreadable) return state;
2019
2336
  if (state.unreadablePath && !quarantinedPaths.has(state.unreadablePath)) {
2020
2337
  quarantinedPaths.add(state.unreadablePath);
@@ -2073,8 +2390,35 @@ async function writeQueue(state) {
2073
2390
  // preceding mutate threw, so the chain never deadlocks.
2074
2391
  let mutateTail = Promise.resolve();
2075
2392
 
2393
+ // Observe-only watchdog: a mutate body over MUTATE_WATCHDOG_MS is logged and
2394
+ // audited once per episode (latched until a mutate completes). mutateTail is
2395
+ // deliberately NEVER reset — it is what enforces the single-writer law, and
2396
+ // abandoning a live writer would let two read-modify-writes interleave.
2397
+ const MUTATE_WATCHDOG_MS = 60_000;
2398
+ let mutateWedgeLatched = false;
2399
+
2076
2400
  function mutate(fn) {
2077
2401
  const next = mutateTail.then(async () => {
2402
+ const wedgeTimer = setTimeout(() => {
2403
+ if (mutateWedgeLatched) return;
2404
+ mutateWedgeLatched = true;
2405
+ console.warn(`[scheduler] MUTATE WEDGED: a queue mutation has run > ${MUTATE_WATCHDOG_MS}ms`);
2406
+ appendAuditEvent('mutate_wedged', { budgetMs: MUTATE_WATCHDOG_MS });
2407
+ }, MUTATE_WATCHDOG_MS);
2408
+ if (typeof wedgeTimer.unref === 'function') wedgeTimer.unref();
2409
+ try {
2410
+ return await mutateBody(fn);
2411
+ } finally {
2412
+ clearTimeout(wedgeTimer);
2413
+ mutateWedgeLatched = false;
2414
+ }
2415
+ });
2416
+ mutateTail = next.catch(() => {}); // keep chain alive on errors
2417
+ return next;
2418
+ }
2419
+
2420
+ async function mutateBody(fn) {
2421
+ {
2078
2422
  const state = await readQueue();
2079
2423
  // Bail BEFORE fn runs: a mutator handed an unreadable (therefore empty)
2080
2424
  // state would compute its result from a queue that isn't there, and
@@ -2120,9 +2464,7 @@ function mutate(fn) {
2120
2464
  }
2121
2465
  await writeQueue(state);
2122
2466
  return ret;
2123
- });
2124
- mutateTail = next.catch(() => {}); // keep chain alive on errors
2125
- return next;
2467
+ }
2126
2468
  }
2127
2469
 
2128
2470
  // ---------- PRD parsing ----------
@@ -2142,11 +2484,23 @@ const parsePrd = prdParser.parsePrd;
2142
2484
  // one — acceptable: PRD counts per project are bounded (hundreds, not
2143
2485
  // millions), and correctness across multiple project dirs matters more than
2144
2486
  // preserving the single-dir cache's steady-state zero-read fast path.
2145
- async function listPrdFiles() {
2487
+ async function listPrdFiles(skipCwds) {
2146
2488
  ensureDirs();
2147
2489
  const dirs = candidatePrdsDirs();
2148
2490
  const perDir = await Promise.all(dirs.map((dir) => prdParser.listPrdFiles(dir)));
2149
- return { files: perDir.flat().sort(), dirCount: dirs.length };
2491
+ let files = perDir.flat();
2492
+ // A quarantined project has no job rows this pass; scanning its PRDs would
2493
+ // mint fresh `pending` rows for work that may already be running.
2494
+ if (skipCwds && skipCwds.size > 0) {
2495
+ const prefixes = [...skipCwds].map((c) => c + path.sep);
2496
+ files = files.filter((f) => !prefixes.some((p) => f.startsWith(p)));
2497
+ }
2498
+ return { files: files.sort(), dirCount: dirs.length };
2499
+ }
2500
+
2501
+ /** Set of cwds whose shard is quarantined in this merged read. */
2502
+ function quarantinedCwdSet(state) {
2503
+ return new Set((state?.unreadableCwds ?? []).map((u) => u.cwd));
2150
2504
  }
2151
2505
 
2152
2506
  /**
@@ -2186,7 +2540,22 @@ async function allocateParallelGroup(cwd) {
2186
2540
  * Safety:
2187
2541
  * - PID-recycling: between app death and this call, another process may have
2188
2542
  * reused the PID. We read /proc/<pid>/cmdline (Linux) or `ps -p` (macOS)
2189
- * and only SIGTERM if the cmdline starts with the claude bin path.
2543
+ * and only SIGTERM if the cmdline matches /\bclaude\b/. Since procName
2544
+ * aliasing, cmdline[0] is the alias path (`.../procnames/sm-claude-job`) or
2545
+ * the smArgv0 label (`sm-claude-job:<slug>`), NOT the claude bin path —
2546
+ * both still contain the word `claude`. The macOS `ps -p <pid> -o command=`
2547
+ * branch has the same exposure and the same guarantee (ps shows argv0).
2548
+ * Migration: cmdline is fixed at exec, and no claude procIdentity is
2549
+ * persisted (job.runtime carries none), so a pre-aliasing process recorded
2550
+ * and compared after upgrade still compares equal to itself — no
2551
+ * tolerance needed; unaliased legacy cmdlines also match /\bclaude\b/.
2552
+ * - recordedIdentity (optional): when the caller has a COMPLETE prior
2553
+ * procIdentity for this pid (startTicks + cmdline), it is used only as a
2554
+ * VETO — if it provably differs from the pid's live identity right now,
2555
+ * the pid was recycled and the kill is refused ('mismatch') even before
2556
+ * the cmdline heuristic below runs. No recorded identity (today's only
2557
+ * case — job.runtime carries no identity yet) falls through unchanged to
2558
+ * the existing /\bclaude\b/ + `ps -p` heuristics.
2190
2559
  * - Detached process group: jobs are spawned with detached:true so we kill
2191
2560
  * -pid (the group). If the group leader is already gone, that fails
2192
2561
  * silently and we fall back to single-pid kill.
@@ -2194,12 +2563,17 @@ async function allocateParallelGroup(cwd) {
2194
2563
  * scheduled via setTimeout to clean up any process ignoring SIGTERM.
2195
2564
  *
2196
2565
  * Returns: 'killed' (cmdline matched + signal sent), 'gone' (pid not alive),
2197
- * 'mismatch' (pid alive but cmdline doesn't look like claude),
2566
+ * 'mismatch' (pid alive but cmdline doesn't look like claude, or a
2567
+ * complete recorded identity proves the pid was recycled),
2198
2568
  * 'unknown' (couldn't read cmdline — leave the pid alone).
2199
2569
  */
2200
- function killOrphanClaudePid(pid) {
2570
+ function killOrphanClaudePid(pid, recordedIdentity = null) {
2201
2571
  if (!pid || typeof pid !== 'number' || pid <= 1) return 'gone';
2202
2572
  try { process.kill(pid, 0); } catch { return 'gone'; }
2573
+ if (recordedIdentity && recordedIdentity.complete
2574
+ && isDifferentProcess(recordedIdentity, procIdentityOf(pid))) {
2575
+ return 'mismatch';
2576
+ }
2203
2577
  let cmdline = '';
2204
2578
  try {
2205
2579
  cmdline = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8').replace(/\0/g, ' ');
@@ -2307,25 +2681,48 @@ async function reconcile(state) {
2307
2681
  // has no queue row yet, so it is never "live" and gets archived here
2308
2682
  // instead of ever reaching the onDisk scan below.
2309
2683
  let phaseStartMs = Date.now();
2310
- await consolidateAllFlatPrds(allProjectCwds());
2684
+ const skipCwds = quarantinedCwdSet(state);
2685
+ await consolidateAllFlatPrds(allProjectCwds(), skipCwds);
2311
2686
  phaseMs.flatPrdSweep = Date.now() - phaseStartMs;
2312
2687
 
2313
2688
  phaseStartMs = Date.now();
2314
- const { files, dirCount } = await listPrdFiles();
2689
+ const { files, dirCount } = await listPrdFiles(skipCwds);
2315
2690
  phaseMs.prdDirResolve = Date.now() - phaseStartMs;
2316
2691
 
2317
2692
  phaseStartMs = Date.now();
2318
2693
  const onDisk = new Map();
2694
+ // Slugs derive from title text with no cwd salt, so two different projects
2695
+ // can legitimately queue an identically-slugged PRD — onDisk alone can
2696
+ // only hold ONE parsed PRD per slug (last-file-wins), which would silently
2697
+ // hand an EXISTING row the wrong project's PRD (or none at all) when two
2698
+ // projects collide on a slug. This side index lets the two existing-row
2699
+ // lookups below (job refresh + invalid-row repair) disambiguate by the
2700
+ // row's own cwd first; the fresh-discovery loop further down still reads
2701
+ // the bare `onDisk` (unscoped) since a same-slug NEW-PRD collision across
2702
+ // two projects is a rarer edge this reconcile pass doesn't yet resolve.
2703
+ const onDiskByCwd = new Map();
2319
2704
  for (const f of files) {
2320
2705
  try {
2321
2706
  // Per-file await: parsing is mtime-cached so steady-state hits zero
2322
2707
  // disk reads; on cold cache the awaits keep the main thread responsive.
2323
2708
  const p = await parsePrd(f);
2324
2709
  onDisk.set(p.slug, p);
2710
+ if (p.cwd) onDiskByCwd.set(`${p.slug}::${p.cwd}`, p);
2325
2711
  } catch (e) {
2326
2712
  console.warn('[scheduler] failed to parse', f, e?.message);
2327
2713
  }
2328
2714
  }
2715
+ // resolvePrdForJob(slug, cwd) — cwd-scoped PRD lookup for an EXISTING
2716
+ // queue row, falling back to the unscoped onDisk entry when this exact
2717
+ // (slug, cwd) pair has no PRD (e.g. cwd is null/stale) — same behavior as
2718
+ // a bare onDisk.get() for every slug that isn't cross-project-colliding.
2719
+ function resolvePrdForJob(slug, cwd) {
2720
+ if (cwd) {
2721
+ const scoped = onDiskByCwd.get(`${slug}::${cwd}`);
2722
+ if (scoped) return scoped;
2723
+ }
2724
+ return onDisk.get(slug);
2725
+ }
2329
2726
  phaseMs.parseLoop = Date.now() - phaseStartMs;
2330
2727
 
2331
2728
  const next = [];
@@ -2345,7 +2742,7 @@ async function reconcile(state) {
2345
2742
  // historyTerminalBySlug() below and backfilled before being dropped.
2346
2743
  const terminalDroppedNeedingHistoryCheck = [];
2347
2744
  for (const job of state.jobs) {
2348
- const p = onDisk.get(job.slug);
2745
+ const p = resolvePrdForJob(job.slug, job.cwd);
2349
2746
  if (!p) {
2350
2747
  // A terminal job whose .md is gone was archived on purpose — dropping
2351
2748
  // its row is the intended end of the auto-archive flow, PROVIDED it's
@@ -2379,10 +2776,24 @@ async function reconcile(state) {
2379
2776
  continue;
2380
2777
  }
2381
2778
  seen.add(job.slug);
2779
+ // p.cwd REFINES the row's existing cwd; it never erases one. A PRD
2780
+ // file with no `cwd:` frontmatter parses p.cwd as undefined — falling
2781
+ // through to a bare `cwd: p.cwd` here nulled the row's real cwd,
2782
+ // which queueStore.writeSplit then buckets into
2783
+ // schedulerBatch.js's DEFAULT_PROJECT_CWD, silently relocating the
2784
+ // row into the WRONG project's queue.json shard and emptying the
2785
+ // owning project's shard underneath it (2026-09 data-loss incident).
2786
+ // Resolved ONCE into a local so originSessionId's fallback below
2787
+ // resolves against the SAME cwd this row actually gets, not the raw
2788
+ // (possibly undefined) p.cwd — resolveOriginSessionId(undefined, ...)
2789
+ // returns null unconditionally, which silently dropped the origin link
2790
+ // for every PRD with no `cwd:` frontmatter even though a good cwd was
2791
+ // available one line below.
2792
+ const refreshedCwd = p.cwd ?? job.cwd ?? null;
2382
2793
  const updatedJob = {
2383
2794
  ...job,
2384
2795
  title: p.title,
2385
- cwd: p.cwd,
2796
+ cwd: refreshedCwd,
2386
2797
  parallelGroup: p.parallelGroup,
2387
2798
  estimateMinutes: p.estimateMinutes,
2388
2799
  sourcePromptId: reconcileSourcePromptId(job, p.sourcePromptId),
@@ -2396,7 +2807,7 @@ async function reconcile(state) {
2396
2807
  quietMachine: p.quietMachine === true,
2397
2808
  budgetExempt: p.budgetExempt === true,
2398
2809
  originSessionId: job.originSessionId
2399
- ?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
2810
+ ?? resolveOriginSessionId(refreshedCwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
2400
2811
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
2401
2812
  agentType: p.agentType ?? job.agentType ?? null,
2402
2813
  };
@@ -2479,7 +2890,7 @@ async function reconcile(state) {
2479
2890
  for (const inv of invalidJobs) {
2480
2891
  if (seen.has(inv.slug)) continue; // a valid row for this slug already exists
2481
2892
  const oldStatus = inv.row?.status;
2482
- const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir: RUNS_DIR });
2893
+ const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir: schedulerPaths.runsDir() });
2483
2894
  if (hist) {
2484
2895
  // Never resurrect: this slug already has a durable terminal record
2485
2896
  // elsewhere (history.jsonl or a run sidecar) — repairing its corrupted
@@ -2492,17 +2903,22 @@ async function reconcile(state) {
2492
2903
  });
2493
2904
  continue;
2494
2905
  }
2495
- const p = onDisk.get(inv.slug);
2906
+ const p = resolvePrdForJob(inv.slug, inv.row?.cwd);
2496
2907
  if (!p) {
2497
2908
  // PRD file also gone with no terminal record anywhere — nothing to
2498
2909
  // repair against. queueStore already logged the quarantine once.
2499
2910
  continue;
2500
2911
  }
2912
+ // Same cwd-refines-not-erases rule as the normal refresh path above, and
2913
+ // same reason for resolving it once into a local: originSessionId's
2914
+ // fallback must resolve against the cwd this row actually gets, not the
2915
+ // raw (possibly undefined) p.cwd.
2916
+ const repairedCwd = p.cwd ?? inv.row?.cwd ?? null;
2501
2917
  const job = {
2502
2918
  ...inv.row,
2503
2919
  slug: inv.slug,
2504
2920
  title: p.title,
2505
- cwd: p.cwd,
2921
+ cwd: repairedCwd,
2506
2922
  parallelGroup: p.parallelGroup,
2507
2923
  estimateMinutes: p.estimateMinutes,
2508
2924
  sourcePromptId: p.sourcePromptId ?? inv.row?.sourcePromptId ?? null,
@@ -2512,7 +2928,7 @@ async function reconcile(state) {
2512
2928
  disposition: p.disposition ?? null,
2513
2929
  quietMachine: p.quietMachine === true,
2514
2930
  budgetExempt: p.budgetExempt === true,
2515
- originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
2931
+ originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(repairedCwd, p.epicId ?? p.sourcePromptId),
2516
2932
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
2517
2933
  agentType: p.agentType ?? inv.row?.agentType ?? null,
2518
2934
  };
@@ -2616,17 +3032,27 @@ async function reconcile(state) {
2616
3032
  // guard above inert. Fall back to reading the slug's own newest run
2617
3033
  // sidecars straight off disk — same "don't resurrect an already-terminal
2618
3034
  // slug" intent, independent of history.jsonl's existence.
2619
- const fallback = latestTerminalOutcomeForSlug(slug, { runsDir: RUNS_DIR });
3035
+ const fallback = latestTerminalOutcomeForSlug(slug, { runsDir: schedulerPaths.runsDir() });
2620
3036
  if (fallback) {
2621
3037
  if (fallback.status === 'completed') {
2622
3038
  historyArchiveCandidates.push({ slug, status: fallback.status, finishedAt: fallback.finishedAt });
2623
3039
  }
2624
3040
  continue;
2625
3041
  }
3042
+ // No prior row exists to fall back to (this is a fresh discovery), so a
3043
+ // PRD file with no `cwd:` frontmatter falls back to the project root it
3044
+ // was actually found under (derived from its own file path) rather than
3045
+ // nulling out to schedulerBatch.js's DEFAULT_PROJECT_CWD. Resolved once
3046
+ // so originSessionId (below) resolves against this SAME cwd — passing
3047
+ // the raw p.cwd there instead would resolve against `undefined` for
3048
+ // exactly the no-frontmatter case this fallback exists to handle, since
3049
+ // resolveOriginSessionId(cwd, ...) returns null unconditionally when
3050
+ // `cwd` is falsy.
3051
+ const discoveredCwd = p.cwd ?? deriveProjectCwdFromPrdPath(p.path) ?? null;
2626
3052
  const entry = {
2627
3053
  slug,
2628
3054
  title: p.title,
2629
- cwd: p.cwd,
3055
+ cwd: discoveredCwd,
2630
3056
  parallelGroup: p.parallelGroup,
2631
3057
  estimateMinutes: p.estimateMinutes,
2632
3058
  sourcePromptId: p.sourcePromptId,
@@ -2636,7 +3062,7 @@ async function reconcile(state) {
2636
3062
  disposition: p.disposition ?? null,
2637
3063
  quietMachine: p.quietMachine === true,
2638
3064
  budgetExempt: p.budgetExempt === true,
2639
- originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
3065
+ originSessionId: resolveOriginSessionId(discoveredCwd, p.epicId ?? p.sourcePromptId),
2640
3066
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
2641
3067
  agentType: p.agentType ?? null,
2642
3068
  status: 'pending',
@@ -2787,14 +3213,73 @@ async function reconcile(state) {
2787
3213
  // ---------- next-reset detection ----------
2788
3214
 
2789
3215
  let cachedNextReset = null; // bare ISO string or null
2790
- let cachedUtilization = null; // five_hour utilization %, 0–100, or null if unknown
3216
+ let cachedUtilization = null; // binding-window utilization %, 0–100, or null if unknown
3217
+ // ms timestamp of the last FRESH reset observation (see recordObservedReset)
3218
+ // — distinct from Date.now(), so persistSchedulerState never re-stamps a
3219
+ // stale cachedNextReset as "just observed" on every poll cycle.
3220
+ let lastResetObservedAtMs = null;
3221
+ // Full usage payload (`{ five_hour, limits?, ... }`) from the last SUCCESSFUL
3222
+ // poll — the input degradedBudget() carries forward while the meter is down.
3223
+ // Never itself defaults to 0; absent (null) reads as "no signal yet" and
3224
+ // degradedBudget() treats that conservatively (100% / capped concurrency).
3225
+ let lastGoodUsagePayload = null;
3226
+ // Non-null while the meter is degraded (circuit open, or a poll otherwise
3227
+ // failed to return 'ok') — narrows tickQueue's freeSlots as a picker-side
3228
+ // hold instead of a second slot pool (see tickQueue's freeSlots computation).
3229
+ // Cleared to null the moment a poll succeeds or the meter is inapplicable.
3230
+ let degradedConcurrencyCapValue = null;
3231
+ // Deliberately a named constant, not a bare zero literal assigned straight
3232
+ // into cachedUtilization: a genuine "no consumer meter to poll" (enterprise
3233
+ // auth) is categorically different from "the meter is down and we don't
3234
+ // know" — the latter must never read as 0%/full-speed-ahead.
3235
+ const NO_METER_UTILIZATION = 0;
3236
+
3237
+ /**
3238
+ * Records a freshly-observed reset, stamping lastResetObservedAtMs only when
3239
+ * there actually WAS a reset to observe (never on every poll regardless of
3240
+ * payload content — see persistSchedulerState's header).
3241
+ */
3242
+ function recordObservedReset(resetIso) {
3243
+ if (resetIso) {
3244
+ cachedNextReset = resetIso;
3245
+ lastResetObservedAtMs = Date.now();
3246
+ } else {
3247
+ cachedNextReset = null;
3248
+ }
3249
+ }
3250
+
3251
+ /** Pure: this poll/executor cycle's conservative budget while the meter is degraded. */
3252
+ function computeDegradedBudget() {
3253
+ return degradedBudget(lastGoodUsagePayload, {
3254
+ observed429: lastFailureKind === 'meter_rate_limited',
3255
+ resetsAt: cachedNextReset,
3256
+ now: Date.now(),
3257
+ configuredCap: sessionSlots.snapshot().total,
3258
+ });
3259
+ }
3260
+
3261
+ /**
3262
+ * Computes AND applies this cycle's degraded budget to cachedUtilization /
3263
+ * degradedConcurrencyCapValue in one call — every "meter down" branch in
3264
+ * pollLoop/runQueueStarvationWatchdog needs the exact same
3265
+ * compute-then-assign-both-fields pair, so it lives once here rather than
3266
+ * being copy-pasted at each call site (where a future change to how the
3267
+ * budget applies would otherwise have to be made N times).
3268
+ */
3269
+ function applyDegradedBudget() {
3270
+ const budget = computeDegradedBudget();
3271
+ cachedUtilization = budget.utilization;
3272
+ degradedConcurrencyCapValue = budget.concurrencyCap;
3273
+ return budget;
3274
+ }
2791
3275
 
2792
3276
  /** Fetches latest usage from billing API. Throws on any error — callers handle it. */
2793
3277
  async function refreshNextReset() {
2794
3278
  const r = await billing.fetchUsage();
2795
3279
  if (r.kind !== 'ok') throw new Error(`usage fetch failed (${r.kind}): ${r.message ?? ''}`);
2796
- cachedNextReset = r.data?.usage?.five_hour?.resets_at ?? null;
2797
- cachedUtilization = r.data?.usage?.five_hour?.utilization ?? cachedUtilization;
3280
+ const window = bindingWindow(r.data?.usage);
3281
+ recordObservedReset(window.resets_at ?? null);
3282
+ cachedUtilization = Number.isFinite(window.utilization) ? window.utilization : cachedUtilization;
2798
3283
  return cachedNextReset;
2799
3284
  }
2800
3285
 
@@ -2805,15 +3290,38 @@ function getNextResetCached() {
2805
3290
  /**
2806
3291
  * Pure: picks the reset to pause against for a rate-limited run (PRD 1118).
2807
3292
  * Prefers the BINDING window read off the run's own log — refreshNextReset()
2808
- * only ever reports five_hour, which is the wrong clock when a
2809
- * seven_day/seven_day_overage_included window is what actually 429'd
2810
- * (five_hour can read 0% utilization at the very same moment). Falls back
2811
- * to the billing-endpoint-derived reset only when the log yields nothing.
2812
- */
2813
- function resolveRateLimitPauseReset(logPath, billingResetIso) {
3293
+ * only ever reports the binding window at POLL time, which can be a
3294
+ * different clock than what actually 429'd this run (five_hour can read 0%
3295
+ * utilization at the very same moment a seven_day window binds). Falls back
3296
+ * to the billing-endpoint-derived reset only when the log yields nothing —
3297
+ * and rejects EITHER source when it has already passed relative to `nowMs`
3298
+ * (usageCircuit.isResetFresh): a stale reset must never arm a resume timer
3299
+ * that already elapsed (that's the 30-second-nap bug computeEffectiveResumeAt
3300
+ * below also guards against), so a stale value here returns null and lets
3301
+ * the caller's 30-minute fallback take over instead.
3302
+ */
3303
+ function resolveRateLimitPauseReset(logPath, billingResetIso, nowMs = Date.now()) {
2814
3304
  const logReset = resolveBindingRateLimitReset(logPath);
2815
- if (logReset != null) return new Date(logReset * 1000).toISOString();
2816
- return billingResetIso ?? null;
3305
+ if (logReset != null) {
3306
+ const iso = new Date(logReset * 1000).toISOString();
3307
+ if (isResetFresh(iso, nowMs)) return iso;
3308
+ }
3309
+ if (billingResetIso && isResetFresh(billingResetIso, nowMs)) return billingResetIso;
3310
+ return null;
3311
+ }
3312
+
3313
+ /**
3314
+ * Shared by both rate-limit pause sites (spawnJob's rateLimited branch and
3315
+ * reapDeadRunningJobs): the billing-endpoint-derived fallback reset used
3316
+ * when the run's own log yields no binding window. While the shared
3317
+ * usageCircuit is OPEN, skip calling refreshNextReset() — it would just be
3318
+ * another request against the endpoint the breaker just decided is down —
3319
+ * and fall back to whatever was last cached instead (resolveRateLimitPauseReset
3320
+ * itself still rejects that cached value if it has since gone stale).
3321
+ */
3322
+ async function billingResetForPause() {
3323
+ if (billing.usageCircuit.state() === 'open') return cachedNextReset;
3324
+ return refreshNextReset().catch(() => cachedNextReset);
2817
3325
  }
2818
3326
 
2819
3327
  // ---------- health / poll state ----------
@@ -2828,10 +3336,27 @@ let firstFailureAt = null;
2828
3336
  let firstNon429FailureAt = null; // tracks only transient/config failures; 429s don't count toward network-pause threshold
2829
3337
  let lastFailureKind = null; // 'transient' | 'meter_rate_limited' | 'auth' | null
2830
3338
  let pauseClearedManuallyAt = null;
3339
+ // In-memory only (no new persisted field): when clearPause last actually lifted a pause — a
3340
+ // legitimate restart point of the dispatch-idleness clock (dispatchIdleMs).
3341
+ let lastPauseClearedAt = null;
2831
3342
  // PRD: the usage-poller silent-failure-streak WARN is emitted once per streak,
2832
3343
  // not once per failure (57 failures must produce ONE opsErrorLog line, not 57).
2833
3344
  // Reset alongside consecutiveFailures everywhere that resets to 0.
2834
3345
  let failureStreakWarned = false;
3346
+ // ms timestamp the initial WARN fired this streak — anchors the periodic
3347
+ // escalation cadence below. Reset to null alongside failureStreakWarned.
3348
+ let failureStreakWarnedAt = null;
3349
+ // ms timestamp of the last periodic escalation (audit event + opsErrorLog
3350
+ // line) this streak. Reset to null alongside failureStreakWarned so a LATER
3351
+ // streak re-arms both the initial WARN and the escalation cadence.
3352
+ let lastEscalationAtMs = null;
3353
+
3354
+ /** The failure-streak-WARN trio must always reset together — one helper, not 3 copies. */
3355
+ function resetFailureStreak() {
3356
+ failureStreakWarned = false;
3357
+ failureStreakWarnedAt = null;
3358
+ lastEscalationAtMs = null;
3359
+ }
2835
3360
  // Ceiling on pollLoop's exponential poll backoff (both the 'transient'/'config'
2836
3361
  // branch and the 'meter_rate_limited' branch below share this cap — a single
2837
3362
  // constant so the two never drift to different ceilings).
@@ -2840,6 +3365,11 @@ const BACKOFF_MAX_MS = 480_000; // 8 minutes
2840
3365
  // jitter and becomes worth a human's attention. health.cjs imports this so the
2841
3366
  // WARN and the `npm run health` non-GREEN trip at the exact same count.
2842
3367
  const FAILURE_STREAK_WARN_THRESHOLD = 5;
3368
+ // How often a PERSISTING failure streak re-escalates (audit event +
3369
+ // opsErrorLog line) after the initial WARN, and the health.cjs YELLOW->RED
3370
+ // threshold for how long the usageCircuit has been open — one constant so
3371
+ // the log cadence and the health-color flip never drift apart.
3372
+ const FAILURE_STREAK_ESCALATION_MS = 30 * 60_000; // 30 minutes
2843
3373
 
2844
3374
  /** Pure: exponential backoff with a cap, shared by every pollLoop failure branch. Exported for unit testing. */
2845
3375
  function nextBackoffMs(prevBackoffMs) {
@@ -2856,19 +3386,59 @@ function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold =
2856
3386
  return consecutiveFailures >= threshold && !alreadyWarned;
2857
3387
  }
2858
3388
 
2859
- /** Emits the one-time opsErrorLog WARN for a failure streak crossing the threshold, if not already warned this streak. */
3389
+ /**
3390
+ * Pure: does a PERSISTING failure streak warrant another escalation (audit
3391
+ * event + opsErrorLog line) at `nowMs`? Exported for unit testing. Only
3392
+ * relevant once the streak has already crossed `warnThreshold` (the initial
3393
+ * WARN); `lastEscalatedAtMs` null means no escalation has fired yet this
3394
+ * streak, so the first one is due immediately. Re-arms automatically once a
3395
+ * streak clears (the caller resets `lastEscalatedAtMs` to null alongside
3396
+ * `failureStreakWarned`), so a later independent streak escalates again.
3397
+ */
3398
+ function shouldEscalateFailureStreak(consecutiveFailures, lastEscalatedAtMs, nowMs, thresholdMs = FAILURE_STREAK_ESCALATION_MS, warnThreshold = FAILURE_STREAK_WARN_THRESHOLD) {
3399
+ if (consecutiveFailures < warnThreshold) return false;
3400
+ if (!lastEscalatedAtMs) return true;
3401
+ return nowMs - lastEscalatedAtMs >= thresholdMs;
3402
+ }
3403
+
3404
+ /**
3405
+ * Emits the one-time opsErrorLog WARN the moment a failure streak crosses
3406
+ * the threshold, then — while that streak PERSISTS — re-escalates (audit
3407
+ * event + another opsErrorLog line) every FAILURE_STREAK_ESCALATION_MS so a
3408
+ * human watching only the audit log still sees a live incident, not just the
3409
+ * single opening WARN from hours ago.
3410
+ */
2860
3411
  function warnFailureStreakIfNeeded() {
2861
- if (!shouldWarnFailureStreak(consecutiveFailures, failureStreakWarned)) return;
2862
- failureStreakWarned = true;
2863
- try {
2864
- appendError({
2865
- cwd: DEFAULT_PROJECT_CWD,
2866
- scope: 'scheduler',
2867
- level: 'warn',
2868
- message: `usage/rate-limit poller has failed ${consecutiveFailures} consecutive times (kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${SCHEDULER_STATE_PATH}`,
2869
- meta: { consecutiveFailures, backoffMs, lastFailureKind },
2870
- });
2871
- } catch { /* durable logging must never break the poll loop */ }
3412
+ const nowMs = Date.now();
3413
+ if (shouldWarnFailureStreak(consecutiveFailures, failureStreakWarned)) {
3414
+ failureStreakWarned = true;
3415
+ failureStreakWarnedAt = nowMs;
3416
+ lastEscalationAtMs = nowMs;
3417
+ try {
3418
+ appendError({
3419
+ cwd: DEFAULT_PROJECT_CWD,
3420
+ scope: 'scheduler',
3421
+ level: 'warn',
3422
+ message: `usage/rate-limit poller has failed ${consecutiveFailures} consecutive times (kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${schedulerPaths.schedulerStatePath()}`,
3423
+ meta: { consecutiveFailures, backoffMs, lastFailureKind },
3424
+ });
3425
+ } catch { /* durable logging must never break the poll loop */ }
3426
+ return;
3427
+ }
3428
+ if (failureStreakWarned && shouldEscalateFailureStreak(consecutiveFailures, lastEscalationAtMs, nowMs)) {
3429
+ lastEscalationAtMs = nowMs;
3430
+ const persistedMinutes = failureStreakWarnedAt ? Math.round((nowMs - failureStreakWarnedAt) / 60_000) : null;
3431
+ try {
3432
+ appendAuditEvent('usage_poller_failure_streak_persists', { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes });
3433
+ appendError({
3434
+ cwd: DEFAULT_PROJECT_CWD,
3435
+ scope: 'scheduler',
3436
+ level: 'warn',
3437
+ message: `usage/rate-limit poller streak still failing after ${persistedMinutes}m (${consecutiveFailures} consecutive, kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${schedulerPaths.schedulerStatePath()}`,
3438
+ meta: { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes },
3439
+ });
3440
+ } catch { /* durable logging must never break the poll loop */ }
3441
+ }
2872
3442
  }
2873
3443
  // PRD 1119: consecutive-rapid-rate-limit hard-pause tracking, keyed per slug.
2874
3444
  // See isCooldownSuppressed/nextRapidRateLimitCount below for the pure rules.
@@ -2882,6 +3452,7 @@ let resumeTimer = null;
2882
3452
  let pollLoopTimer = null;
2883
3453
  let rescheduleInterval = null;
2884
3454
  let heartbeatInterval = null;
3455
+ let dispatchLoopHandle = null;
2885
3456
  // Stall-detector state (computeStallSummary), read/written only inside the
2886
3457
  // heartbeat interval below. Keyed per-project cwd (never a single value) —
2887
3458
  // a single module-level flag would let one busy project's activity clear or
@@ -2905,12 +3476,13 @@ const runningSet = new Set();
2905
3476
  // N concurrent Opus processes — the >3-concurrent class that OOM-killed Electron.
2906
3477
  // Over-cap requests are QUEUED (not dropped) and drained as slots free, so a failed
2907
3478
  // PRD that never reaches 'needs_review' still eventually gets its fix-plan authored.
2908
- let investigationsInFlight = 0;
2909
3479
  const MAX_CONCURRENT_INVESTIGATIONS = 1;
2910
3480
  const deferredInvestigations = new Map(); // fixable-job slug -> { failedJob, runDir }
3481
+ // Mirror of machine `drain.active`, kept in memory so spawn sites need no queue read.
3482
+ let drainActive = false;
2911
3483
 
2912
3484
  function drainDeferredInvestigation() {
2913
- if (investigationsInFlight >= MAX_CONCURRENT_INVESTIGATIONS) return;
3485
+ if (drainActive || runtimeState.investigationCount() >= MAX_CONCURRENT_INVESTIGATIONS) return;
2914
3486
  const next = deferredInvestigations.entries().next();
2915
3487
  if (next.done) return;
2916
3488
  const [slug, ctx] = next.value;
@@ -3001,6 +3573,7 @@ function buildScheduleStatePayload(state) {
3001
3573
  lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
3002
3574
  nextReset: getNextResetCached(),
3003
3575
  paused: state.paused,
3576
+ drain: state.drain ?? null,
3004
3577
  // Launch circuit breaker (issue #11): which personas cannot launch right
3005
3578
  // now and why, plus any degraded-mode env in force. Empty objects when healthy.
3006
3579
  launchBlocks: state.launchBlocks ?? {},
@@ -3170,13 +3743,17 @@ function nextRapidRateLimitCount(prevCount, { rateLimited, durationMs }) {
3170
3743
  /**
3171
3744
  * Pure: decides the resumeAt actually armed for a pause. 'network' and
3172
3745
  * 'rate_limit' (PRD 1118) both get a bounded 30-minute fallback when no
3173
- * explicit resumeAt is supplied — the live rate_limit failure mode is the
3174
- * billing usage endpoint itself 429ing while the log yields no binding
3175
- * window either, which used to leave an indefinite pause with no resume
3176
- * timer at all (a queue that never comes back on its own).
3746
+ * explicit resumeAt is supplied, OR when the supplied resumeAtIso has
3747
+ * already passed (usageCircuit.isResetFresh) — a stale reset used to produce
3748
+ * `Math.max(30_000, <negative>)` downstream in computeResumeDelay, a
3749
+ * 30-SECOND nap instead of a real pause, spinning the queue right back into
3750
+ * the same still-active rate limit. The live failure mode is the billing
3751
+ * usage endpoint itself 429ing while the log yields no fresh binding window
3752
+ * either, which used to leave an indefinite pause with no resume timer at
3753
+ * all (a queue that never comes back on its own) — this covers both.
3177
3754
  */
3178
3755
  function computeEffectiveResumeAt(reason, resumeAtIso, nowMs = Date.now()) {
3179
- if (resumeAtIso) return resumeAtIso;
3756
+ if (resumeAtIso && isResetFresh(resumeAtIso, nowMs)) return resumeAtIso;
3180
3757
  if (reason === 'network' || reason === 'rate_limit') {
3181
3758
  return new Date(nowMs + 30 * 60_000).toISOString();
3182
3759
  }
@@ -3196,11 +3773,14 @@ function computeResumeDelay(effectiveResumeAtIso, nowMs = Date.now()) {
3196
3773
 
3197
3774
  async function setPaused(reason, resumeAtIso, opts = {}) {
3198
3775
  const { observedAt = null, force = false } = opts;
3776
+ const isManual = reason === 'manual';
3199
3777
  // Honor manual-override cooldown: if the user cleared a pause within the
3200
3778
  // last 5 minutes, suppress auto-pause re-engagement UNLESS this pause is
3201
3779
  // backed by a fresh observation (a run that started after the clear) or is
3202
3780
  // forced (the rapid-repeat hard pause, which the cooldown cannot suppress).
3203
- if (isCooldownSuppressed({ pauseClearedManuallyAt, now: Date.now(), observedAt, force })) {
3781
+ // A user-initiated 'manual' pause is never an auto-detection, so the cooldown
3782
+ // (which exists to ignore STALE auto-detections) does not apply to it.
3783
+ if (!isManual && isCooldownSuppressed({ pauseClearedManuallyAt, now: Date.now(), observedAt, force })) {
3204
3784
  console.log(`[scheduler] setPaused(${reason}) suppressed by manual override cooldown`);
3205
3785
  return;
3206
3786
  }
@@ -3210,15 +3790,24 @@ async function setPaused(reason, resumeAtIso, opts = {}) {
3210
3790
  console.log(`[scheduler] setPaused(${reason}) engaging despite manual override cooldown — triggering run started after the manual clear`);
3211
3791
  }
3212
3792
 
3213
- const effectiveResumeAt = computeEffectiveResumeAt(reason, resumeAtIso);
3793
+ // 'manual' never arms a resume timer: only schedule:resume / run-now clears it.
3794
+ const effectiveResumeAt = isManual ? null : computeEffectiveResumeAt(reason, resumeAtIso);
3214
3795
 
3215
- await mutate((s) => {
3796
+ const kept = await mutate((s) => {
3797
+ // A user pause outranks every auto-pause (rate_limit/auth/network): the
3798
+ // auto path must not overwrite it, or its resume timer would auto-clear it.
3799
+ if (!isManual && s.paused && s.paused.reason === 'manual') return true;
3216
3800
  if (s.paused && s.paused.reason === reason) {
3217
3801
  if (effectiveResumeAt) s.paused.resumeAt = effectiveResumeAt;
3218
3802
  } else {
3219
3803
  s.paused = { reason, since: new Date().toISOString(), resumeAt: effectiveResumeAt || null };
3220
3804
  }
3805
+ return false;
3221
3806
  });
3807
+ if (kept) {
3808
+ console.log(`[scheduler] setPaused(${reason}) ignored: a manual pause is in force`);
3809
+ return;
3810
+ }
3222
3811
  await broadcast({ flush: true });
3223
3812
  cancelToken.cancelled = true;
3224
3813
  if (resumeTimer) { clearTimeout(resumeTimer); resumeTimer = null; }
@@ -3256,6 +3845,7 @@ async function clearPause(source) {
3256
3845
  });
3257
3846
  // Un-cancel the tick guard on every recovery path, not just runDueJobs().
3258
3847
  applyPauseCleared(wasPaused, cancelToken);
3848
+ if (wasPaused) lastPauseClearedAt = Date.now();
3259
3849
  // Track manual clears for the auto-pause cooldown.
3260
3850
  if (source === 'manual' || source === 'run-now') {
3261
3851
  pauseClearedManuallyAt = Date.now();
@@ -3268,7 +3858,7 @@ async function clearPause(source) {
3268
3858
  firstFailureAt = null;
3269
3859
  firstNon429FailureAt = null;
3270
3860
  lastFailureKind = null;
3271
- failureStreakWarned = false;
3861
+ resetFailureStreak();
3272
3862
  persistSchedulerState();
3273
3863
  }
3274
3864
  if (wasPaused) await broadcast({ flush: true });
@@ -3351,37 +3941,65 @@ function resetJobFields(job, errorMsg, opts = {}) {
3351
3941
  return true;
3352
3942
  }
3353
3943
 
3354
- // Grace period between a boot orphan's SIGTERM and reading its log to
3355
- // classify the outcome — matches killOrphanClaudePid's own internal 5s
3356
- // SIGKILL follow-up delay, plus a small margin so classification always runs
3357
- // after that SIGKILL has had a chance to land.
3358
- const BOOT_ORPHAN_KILL_GRACE_MS = 6000;
3359
-
3360
3944
  /**
3361
- * partitionBootOrphans(jobs, isAlive?) → { immediate: string[], deferred: string[] }
3945
+ * partitionBootOrphans(jobs, liveness) → { immediate: string[], adopted: string[] }
3362
3946
  *
3363
- * Pure decision split for boot reconciliation. A 'running' job whose recorded
3364
- * pid is still alive must NOT be classified from its log yet — the orphaned
3365
- * process may still be writing to it, so reading now risks misclassifying a
3366
- * job that is about to emit result:success as no_result and double-running it.
3367
- * Ported from reconcileQueueOffline's cross-tick escalation (see
3368
- * src/main/lib/watchdogHelpers.cjs) — here it's a single deferred window since
3369
- * this process stays up to revisit it, rather than a separate short-lived
3370
- * watchdog process needing another tick.
3371
- */
3372
- function partitionBootOrphans(jobs, isAlive = claudePidAlive) {
3947
+ * Pure decision split for boot reconciliation of 'running' rows. A row PROVEN
3948
+ * ALIVE is `adopted`: left `running`, never signalled — the steady-state
3949
+ * reaper (reapDeadRunningJobs) finishes it on exit, exactly as it does for any
3950
+ * pidless-recovered row. Only rows proven dead or exited are `immediate` and go
3951
+ * through applyOrphanOutcome. `liveness` is the same injected set
3952
+ * selectReapableJobs takes (plus readRecord/runsDir/identityOf):
3953
+ * { pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess,
3954
+ * readRecord(runDir, slug) }.
3955
+ *
3956
+ * classifyAdoption runs FIRST when the row has a supervisor record: 'adopt' and
3957
+ * 'over-budget' are alive (budget re-arm across restart is a separate PRD, so
3958
+ * an over-budget row is spared, not killed); 'exited' / 'dead' / 'foreign-pid'
3959
+ * are not. A row with no record falls to the reaper's own ladder: recorded
3960
+ * pid alive, then fresh log, log-pid alive, /proc cwd scan.
3961
+ * Complexity: O(jobs) plus one /proc probe per running row.
3962
+ */
3963
+ function partitionBootOrphans(jobs, {
3964
+ pidAlive = claudePidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess,
3965
+ readRecord = supervisorRecord.readSupervisorRecord, runsDir = null, identityOf = procIdentityOf,
3966
+ now = Date.now(),
3967
+ } = {}) {
3373
3968
  const immediate = [];
3374
- const deferred = [];
3969
+ const adopted = [];
3375
3970
  for (const j of jobs) {
3376
3971
  if (j.status !== 'running') continue;
3377
- const pid = j.runtime?.pid;
3378
- if (pid && isAlive(pid)) {
3379
- deferred.push(j.slug);
3380
- } else {
3381
- immediate.push(j.slug);
3382
- }
3972
+ if (isBootRowAlive(j, {
3973
+ pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess, readRecord, runsDir, identityOf, now,
3974
+ })) adopted.push(j.slug);
3975
+ else immediate.push(j.slug);
3976
+ }
3977
+ return { immediate, adopted };
3978
+ }
3979
+
3980
+ function isBootRowAlive(j, {
3981
+ pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess, readRecord, runsDir, identityOf, now,
3982
+ }) {
3983
+ const logMtimeMs = typeof getLogMtimeMs === 'function' ? getLogMtimeMs(j) : null;
3984
+ const runDir = j.runId ? path.join(runsDir || schedulerPaths.runsDir(), j.runId) : null;
3985
+ const record = runDir && typeof readRecord === 'function' ? readRecord(runDir, j.slug) : null;
3986
+ if (record && record.pid) {
3987
+ const alive = !!pidAlive(record.pid);
3988
+ const verdict = supervisorRecord.classifyAdoption(record, {
3989
+ identity: alive ? identityOf(record.pid) : null,
3990
+ pidAlive: alive,
3991
+ logMtimeMs,
3992
+ exitMarker: supervisorRecord.hasExitMarker(runDir, record),
3993
+ now,
3994
+ });
3995
+ return verdict === 'adopt' || verdict === 'over-budget';
3383
3996
  }
3384
- return { immediate, deferred };
3997
+ const pid = j.runtime?.pid;
3998
+ if (pid && pidAlive(pid)) return true;
3999
+ if (typeof logFreshWindowMs === 'number' && Number.isFinite(logMtimeMs) && now - logMtimeMs <= logFreshWindowMs) return true;
4000
+ const logPid = typeof getLogPid === 'function' ? getLogPid(j) : null;
4001
+ if (logPid && pidAlive(logPid)) return true;
4002
+ return !!(typeof findLiveProcess === 'function' && findLiveProcess(j));
3385
4003
  }
3386
4004
 
3387
4005
  /**
@@ -3647,7 +4265,7 @@ async function notifyOriginatingTab(job, {
3647
4265
  const epicIdForTranscript = prd?.sourcePromptId || prd?.sourceTabId || job.epicId || null;
3648
4266
  if (epicIdForTranscript && job.cwd) {
3649
4267
  try {
3650
- const logPath = job.runId ? path.join(RUNS_DIR, job.runId, `${job.slug}.log`) : null;
4268
+ const logPath = job.runId ? path.join(schedulerPaths.runsDir(), job.runId, `${job.slug}.log`) : null;
3651
4269
  const resultText = readResultFromLog(logPath);
3652
4270
  await appendTranscriptTurn(job.cwd, epicIdForTranscript, {
3653
4271
  role: 'assistant',
@@ -4378,6 +4996,39 @@ async function resolveLandedCommitEvidence(cwd, sha, sinceIso) {
4378
4996
  }
4379
4997
  }
4380
4998
 
4999
+ /**
5000
+ * True when `sha` is a non-empty commit that is an ancestor of (or equal to)
5001
+ * HEAD in the repo at `cwd` (`git merge-base --is-ancestor`). Bounded, never
5002
+ * throws: an empty sha, an unknown sha, or any git failure is `false`, so the
5003
+ * caller's safe default is "cannot prove the work survived".
5004
+ */
5005
+ async function landedCommitIsAncestorOfHead(cwd, sha) {
5006
+ if (!sha || typeof sha !== 'string' || !cwd) return false;
5007
+ try {
5008
+ await execGitAt(resolveProjectRoot(cwd), ['merge-base', '--is-ancestor', sha, 'HEAD'], { timeout: 10_000 });
5009
+ return true;
5010
+ } catch {
5011
+ return false;
5012
+ }
5013
+ }
5014
+
5015
+ /**
5016
+ * Pure predicate: a needs_review row parked as shared_tree_reverted that
5017
+ * carries a landedCommit — the only shape reverifyNeedsReview can re-check
5018
+ * against ground truth (landedCommitIsAncestorOfHead). Deliberately
5019
+ * independent of autoFixAttempted: 1229-fo-03 was parked with
5020
+ * autoFixAttempted:true and no autoFixOutcome (isStrandedAutoFixPark shape),
5021
+ * yet isStrandedAutoFixPark could not release it — that ladder only resolves
5022
+ * once job.looksDone is set, and reverifyNeedsReview computes looksDone only
5023
+ * for isRescanCandidate / isGuardParkedWithoutAutoFix rows, neither of which
5024
+ * a shared_tree_reverted + autoFixAttempted row is.
5025
+ */
5026
+ function isStaleSharedTreeRevertedPark(job) {
5027
+ return !!job && job.status === 'needs_review'
5028
+ && job.verifierVerdict === 'shared_tree_reverted'
5029
+ && typeof job.landedCommit === 'string' && job.landedCommit.length > 0;
5030
+ }
5031
+
4381
5032
  /**
4382
5033
  * Commit exactly `paths` (must already be dirty on disk) onto a dedicated
4383
5034
  * `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
@@ -4557,7 +5208,7 @@ function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }
4557
5208
  // create it — `recursive: true` makes that race safe.
4558
5209
  function pickRunDir() {
4559
5210
  const ts = new Date().toISOString().replace(/[:.]/g, '-');
4560
- const dir = path.join(RUNS_DIR, ts);
5211
+ const dir = path.join(schedulerPaths.runsDir(), ts);
4561
5212
  return { runId: ts, dir };
4562
5213
  }
4563
5214
 
@@ -4617,7 +5268,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4617
5268
  // Sync write: this is an early-exit error path inside an async function,
4618
5269
  // so we could await, but using the sync variant keeps the error path
4619
5270
  // ordering identical to the spawn-failed branch below (also sync).
4620
- config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs: 0 });
5271
+ config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs: 0, ...SCHEDULER_META_IDENTITY });
4621
5272
  return { exitCode: -1, durationMs: 0, error: errMsg, sessionId };
4622
5273
  }
4623
5274
 
@@ -4778,7 +5429,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4778
5429
  if (!promptCheck.ok) {
4779
5430
  safeLog(`[scheduler] ${promptCheck.error}\n`);
4780
5431
  closeFd();
4781
- config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: promptCheck.error, startedAt, finishedAt: Date.now(), durationMs: 0 });
5432
+ config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: promptCheck.error, startedAt, finishedAt: Date.now(), durationMs: 0, ...SCHEDULER_META_IDENTITY });
4782
5433
  return { exitCode: -1, durationMs: 0, error: promptCheck.error, sessionId };
4783
5434
  }
4784
5435
 
@@ -4972,13 +5623,17 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4972
5623
 
4973
5624
  // ---------- spawn ----------
4974
5625
 
5626
+ // Distinct comm (`sm-claude-job`) + slug-labelled argv0 for System Monitor.
5627
+ // Both keep the word `claude`, which the /\bclaude\b/ reaper gates need.
5628
+ const jobSpawn = claudeSpawnTarget('job', job.slug, claudeBin);
5629
+
4975
5630
  const { child } = withChildAndLog({
4976
5631
  fd,
4977
5632
  logPath,
4978
5633
  safeLog,
4979
5634
  closeFd,
4980
5635
  spawn: {
4981
- command: claudeBin,
5636
+ command: jobSpawn.command,
4982
5637
  // Resume mode passes `--resume <sessionId>` (reconnect to the SAME
4983
5638
  // session) INSTEAD of `--session-id <sessionId>` (mint a new one) —
4984
5639
  // never both, see buildClaudeSpawnArgs.
@@ -4992,6 +5647,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4992
5647
  options: {
4993
5648
  cwd: spawnCwd,
4994
5649
  env: childEnv,
5650
+ ...(jobSpawn.argv0 ? { argv0: jobSpawn.argv0 } : {}),
4995
5651
  // detached:true puts the child in its own process group so we can kill
4996
5652
  // the entire descendant tree (including any stray background bashes the
4997
5653
  // agent spawned) with `process.kill(-pid)`. Without this, child.kill()
@@ -5017,7 +5673,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5017
5673
  sl(`\n[scheduler] ${errMsg}\n`);
5018
5674
  // Sync write: inside a Promise executor callback; must flush meta
5019
5675
  // before resolve() so the spawnJob mutate() that follows sees it.
5020
- config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA, originSessionId, contextDigestApplied });
5676
+ config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked, ...SCHEDULER_META_IDENTITY, originSessionId, contextDigestApplied });
5021
5677
  resolve({ exitCode: -1, durationMs, error: errMsg, leakedDescendants: leaked, sessionId });
5022
5678
  return;
5023
5679
  }
@@ -5079,7 +5735,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5079
5735
  startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
5080
5736
  agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
5081
5737
  killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
5082
- schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
5738
+ ...SCHEDULER_META_IDENTITY,
5083
5739
  originSessionId, contextDigestApplied,
5084
5740
  });
5085
5741
  resolve({
@@ -5091,6 +5747,22 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5091
5747
 
5092
5748
  if (child) {
5093
5749
  safeLog(`[scheduler] spawned pid=${child.pid} sessionId=${sessionId} (process group)\n\n`);
5750
+ // The one synchronous, authoritative dispatch record (see
5751
+ // jobSupervisorRecord.cjs). detached:true → setsid → pgid === pid, so
5752
+ // no process.getpgid. A write failure must never fail the dispatch.
5753
+ try {
5754
+ supervisorRecord.writeSupervisorRecord({
5755
+ runDir, slug: job.slug, cwd, runId: path.basename(runDir), pid: child.pid, pgid: child.pid,
5756
+ identity: procIdentityOf(child.pid), execCwd: spawnCwd,
5757
+ worktreeDir: execCwd || null, worktreeBranch: execCwd ? `sm-job/${job.slug}` : null,
5758
+ sessionId, startedAt, budgetMs: budgetExempt ? null : jobBudgetMs, maxDurationMs: null,
5759
+ idleKillMs: IDLE_OUTPUT_KILL_MS, schedulerPid: process.pid, codeSha: SCHEDULER_CODE_SHA,
5760
+ });
5761
+ } catch (e) {
5762
+ const message = e?.message ?? String(e);
5763
+ console.error(`[scheduler] FAILED to write supervisor record for ${job.slug} pid=${child.pid}: ${message}`);
5764
+ appendAuditEvent('supervisor_record_write_failed', { slug: job.slug, cwd, pid: child.pid, error: message });
5765
+ }
5094
5766
  // Make this job the OOM killer's preferred victim over Electron.
5095
5767
  biasJobOomScore(child.pid);
5096
5768
  // Persist runtime.pid with one retry — still fire-and-forget (must
@@ -5380,7 +6052,8 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
5380
6052
  return { deferred: false };
5381
6053
  }
5382
6054
  }
5383
- if (investigationsInFlight >= MAX_CONCURRENT_INVESTIGATIONS) {
6055
+ // A drain (lib/upgradeDrain.cjs) admits no NEW work: queue instead of spawning.
6056
+ if (drainActive || runtimeState.investigationCount() >= MAX_CONCURRENT_INVESTIGATIONS) {
5384
6057
  // Queue for retry when a slot frees rather than dropping — otherwise a failed
5385
6058
  // job (never 'needs_review', so reverifyNeedsReview won't retry it) would
5386
6059
  // silently never get an auto-authored fix-plan.
@@ -5394,12 +6067,12 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
5394
6067
  // both pass the cap check. Released in onExit, on any pre-spawn early return, or
5395
6068
  // on a synchronous throw (try/catch below) — and releasing hands the slot to a
5396
6069
  // queued investigation so none are stranded.
5397
- investigationsInFlight++;
6070
+ runtimeState.reserveInvestigation(failedJob.slug);
5398
6071
  let slotReleased = false;
5399
6072
  const releaseSlot = () => {
5400
6073
  if (slotReleased) return;
5401
6074
  slotReleased = true;
5402
- investigationsInFlight--;
6075
+ runtimeState.releaseInvestigation(failedJob.slug);
5403
6076
  drainDeferredInvestigation();
5404
6077
  };
5405
6078
  try {
@@ -5504,13 +6177,14 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
5504
6177
  };
5505
6178
 
5506
6179
  // Phase 2: spawn with lifecycle managed by withChildAndLog.
6180
+ const probeSpawn = claudeSpawnTarget('aux', 'investigate', claudeBin);
5507
6181
  const { child } = withChildAndLog({
5508
6182
  fd,
5509
6183
  logPath: investigationLogPath,
5510
6184
  safeLog,
5511
6185
  closeFd,
5512
6186
  spawn: {
5513
- command: claudeBin,
6187
+ command: probeSpawn.command,
5514
6188
  args: [
5515
6189
  '-p', prompt,
5516
6190
  '--model', 'opus',
@@ -5519,7 +6193,7 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
5519
6193
  '--verbose',
5520
6194
  '--session-id', sessionId,
5521
6195
  ],
5522
- options: { cwd, env: childEnv },
6196
+ options: { cwd, env: childEnv, ...(probeSpawn.argv0 ? { argv0: probeSpawn.argv0 } : {}) },
5523
6197
  },
5524
6198
  watchdogs: [deadmanWatchdog],
5525
6199
  onExit({ exitCode, error, spawnFailed, safeLog: sl }) {
@@ -5621,6 +6295,20 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
5621
6295
 
5622
6296
  if (child) {
5623
6297
  safeLog(`[scheduler] investigation pid=${child.pid}\n\n`);
6298
+ try {
6299
+ supervisorRecord.writeSupervisorRecord({
6300
+ runDir, slug: `${failedJob.slug}.investigation`, kind: 'investigation', cwd: failedJob.cwd ?? null,
6301
+ runId: path.basename(runDir), pid: child.pid, pgid: child.pid, identity: procIdentityOf(child.pid),
6302
+ execCwd: failedJob.cwd ?? null, worktreeDir: null, worktreeBranch: null, sessionId: null,
6303
+ startedAt: Date.now(), budgetMs: null, maxDurationMs: MAX_INVESTIGATION_DURATION_MS, idleKillMs: null,
6304
+ schedulerPid: process.pid, codeSha: SCHEDULER_CODE_SHA,
6305
+ });
6306
+ } catch (e) {
6307
+ const message = e?.message ?? String(e);
6308
+ console.error(`[scheduler] FAILED to write supervisor record for investigation ${failedJob.slug} pid=${child.pid}: ${message}`);
6309
+ appendAuditEvent('supervisor_record_write_failed', { slug: failedJob.slug, cwd: failedJob.cwd, pid: child.pid, error: message, kind: 'investigation' });
6310
+ }
6311
+ runtimeState.stampInvestigationPid(failedJob.slug, child.pid);
5624
6312
  // Recorded so findStrandedInvestigations (a post-restart maintenance
5625
6313
  // sweep — the live process has no other way to know a probe is still
5626
6314
  // running) can tell a live probe apart from one whose owning process is
@@ -5722,9 +6410,22 @@ async function computeDepHistorySatisfaction(state) {
5722
6410
  for (const slug of await queueHistory.completedSlugsForCwd(cwd)) satisfied.add(slug);
5723
6411
  for (const dir of listArchivedPrdDirs(cwd)) {
5724
6412
  let entries;
5725
- try { entries = await fsp.readdir(dir); } catch { continue; }
5726
- for (const name of entries) {
5727
- if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
6413
+ try { entries = await fsp.readdir(dir, { withFileTypes: true }); } catch { continue; }
6414
+ for (const ent of entries) {
6415
+ if (ent.isFile() && ent.name.endsWith('.md')) { satisfied.add(ent.name.slice(0, -3)); continue; }
6416
+ // Option (b) of PRD 1286: a manual archive (queueOps.archiveOne, the
6417
+ // schedule:archive-prd route and scheduler_archive_prd MCP tool) files the PRD
6418
+ // under prds-archived/<ISO-ts>/<slug>.md — one level DEEPER than the auto-archive
6419
+ // layout. Reading only the top level made every manually-archived slug invisible
6420
+ // here, so once its row aged out its dependents held forever as 'unresolved'.
6421
+ // A dep whose PRD file was archived is satisfied; a dep with NO file, row or
6422
+ // history record (never ran, or a typo) still holds — nothing is dropped.
6423
+ if (!ent.isDirectory()) continue;
6424
+ let inner;
6425
+ try { inner = await fsp.readdir(path.join(dir, ent.name)); } catch { continue; }
6426
+ for (const name of inner) {
6427
+ if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
6428
+ }
5728
6429
  }
5729
6430
  }
5730
6431
  } catch (e) {
@@ -5821,7 +6522,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5821
6522
  // Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
5822
6523
  // — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
5823
6524
  // leaves the job pending; the next tick retries when a slot frees up.
5824
- const slotToken = sessionSlots.acquire(`scheduler:${job.slug}`);
6525
+ const slotToken = sessionSlots.acquire(`scheduler:${job.slug}`, { claimedAt: Date.now() });
5825
6526
  if (!slotToken) {
5826
6527
  console.log(`[scheduler] no session slot free for ${job.slug} — deferring (${JSON.stringify(sessionSlots.snapshot().holders.map((h) => h.owner))})`);
5827
6528
  return;
@@ -5947,7 +6648,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5947
6648
  // targets a specific prior session on purpose) and for anything not
5948
6649
  // currently 'pending' (e.g. a needs_review->running recovery row).
5949
6650
  if (!resumeTarget && s.jobs[idx].status === 'pending') {
5950
- const outcome = latestTerminalOutcomeForSlug(job.slug, { runsDir: RUNS_DIR });
6651
+ const outcome = latestTerminalOutcomeForSlug(job.slug, { runsDir: schedulerPaths.runsDir() });
5951
6652
  const reconcileDecision = evaluateDispatchSidecarReconcile({
5952
6653
  rowStatus: s.jobs[idx].status,
5953
6654
  rowRunId: s.jobs[idx].runId ?? null,
@@ -5956,7 +6657,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5956
6657
  outcome,
5957
6658
  });
5958
6659
  if (reconcileDecision.skip) {
5959
- const sidecar = readRunOutcomeSidecars(path.join(RUNS_DIR, reconcileDecision.runId), job.slug);
6660
+ const sidecar = readRunOutcomeSidecars(path.join(schedulerPaths.runsDir(), reconcileDecision.runId), job.slug);
5960
6661
  transitionJob(s.jobs[idx], 'completed', {
5961
6662
  reason: `prior run ${reconcileDecision.runId} already completed this slug (sidecar-reconciled)`,
5962
6663
  source: 'spawnJob:dispatch-sidecar-reconcile',
@@ -5986,7 +6687,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5986
6687
  // pass_no_commit_prior_run_verified exemption can fire on this
5987
6688
  // run if it turns out to be another no-op re-verification.
5988
6689
  if (!s.jobs[idx].landedCommit && outcome?.runId) {
5989
- const sidecar = readRunOutcomeSidecars(path.join(RUNS_DIR, outcome.runId), job.slug);
6690
+ const sidecar = readRunOutcomeSidecars(path.join(schedulerPaths.runsDir(), outcome.runId), job.slug);
5990
6691
  if (sidecar.outcome?.landedCommit) {
5991
6692
  s.jobs[idx].landedCommit = sidecar.outcome.landedCommit;
5992
6693
  }
@@ -6169,6 +6870,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6169
6870
  const foreignWip = worktree.ok ? { carriedPaths } : { preRunDirtyPaths };
6170
6871
  try {
6171
6872
  res = await executeJob(job, runDir, defaultCwd, async (pid, sessionId, cwd) => {
6873
+ sessionSlots.stampPid(slotToken, pid);
6172
6874
  await mutate((s) => {
6173
6875
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
6174
6876
  if (idx >= 0) {
@@ -6312,8 +7014,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6312
7014
  }
6313
7015
 
6314
7016
  if (res.rateLimited) {
7017
+ // The executor itself observed a 429 — open the shared circuit even if
7018
+ // the billing poller has been reporting 'ok' all along (AC3: a
7019
+ // different window can 429 the executor than the one binding the
7020
+ // poller's own reads).
7021
+ billing.usageCircuit.recordFailure('executor_429');
6315
7022
  const logPath = path.join(runDir, `${job.slug}.log`);
6316
- const billingResetIso = await refreshNextReset().catch(() => cachedNextReset);
7023
+ const billingResetIso = await billingResetForPause();
6317
7024
  const resetIso = resolveRateLimitPauseReset(logPath, billingResetIso);
6318
7025
  const observedAt = dispatchStartedAtMs;
6319
7026
  const prevCount = consecutiveRapidRateLimitsBySlug.get(job.slug) || 0;
@@ -6401,6 +7108,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6401
7108
  allJobs: stateForDeps.jobs,
6402
7109
  committedDuringRun,
6403
7110
  priorLandedCommit,
7111
+ jobLandedCommitThisRun,
7112
+ exitCode: res.exitCode,
6404
7113
  }).catch((e) => ({
6405
7114
  verdict: 'verify_unavailable',
6406
7115
  reason: `verifier threw: ${e?.message ?? String(e)}`,
@@ -6506,6 +7215,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6506
7215
  dirtyBaseline: guardBaselineEntries,
6507
7216
  headBefore: guardHeadBefore,
6508
7217
  slug: job.slug,
7218
+ landedCommit: jobLandedCommitThisRun,
6509
7219
  });
6510
7220
  // A restored stash alone isn't silence — it's logged loudly above and
6511
7221
  // surfaced on the job row below — but a path that's still missing
@@ -7251,36 +7961,122 @@ async function spawnResumeRecovery(job, resumeTarget) {
7251
7961
  // is synchronous and spawnJob is fire-and-forget.
7252
7962
  let tickTail = Promise.resolve();
7253
7963
 
7964
+ // Tick watchdog. The tick BODY (not enqueue-to-settle) is bounded: a body
7965
+ // that never settles (hung reconcile / git walk) is declared wedged, the
7966
+ // chain is reset, and `tickGeneration` is bumped so the abandoned body —
7967
+ // which may resume much later — fails its generation re-check after every
7968
+ // await and returns without spawning or mutating. mutateTail is NOT touched.
7969
+ let tickGeneration = 0;
7970
+ let tickWedgeLatched = false;
7971
+ function tickWatchdogMs() {
7972
+ const raw = process.env.SM_TICK_WATCHDOG_MS;
7973
+ if (raw === undefined || raw === '') return 120_000;
7974
+ const n = Number(raw);
7975
+ return Number.isFinite(n) && n >= 0 ? n : 120_000;
7976
+ }
7977
+
7254
7978
  // `bypassLoadGate` is set only by the explicit human run-now / force-tick
7255
7979
  // paths (via runDueJobs): the human is asking, so the CPU-load gate yields
7256
7980
  // and logs that it did. Every automatic caller leaves it false.
7981
+ /**
7982
+ * Fail-CLOSED expiry of runtime reservations (slot tokens, quiet-machine
7983
+ * lease, investigation set, worktree accounting) whose owner is provably
7984
+ * dead. `now` is the pass start: reservations claimed after it are never
7985
+ * expired. Accounting only — never signals or removes anything.
7986
+ */
7987
+ function runReservationExpiryPass(jobs, now = Date.now()) {
7988
+ try {
7989
+ const liveSlugs = new Set(runningSet);
7990
+ const terminal = new Set();
7991
+ for (const j of jobs || []) {
7992
+ if (j.status === 'running' || j.status === 'investigating') liveSlugs.add(j.slug);
7993
+ else if (j.status === 'completed' || j.status === 'failed' || j.status === 'skipped') terminal.add(j.slug);
7994
+ }
7995
+ const pidAlive = (pid) => claudePidAlive(pid);
7996
+ sessionSlots.expireDead({ liveSlugs, pidAlive, now });
7997
+ quietMachineLease.expireDead({ liveSlugs, now });
7998
+ runtimeState.expireDeadInvestigations({ liveSlugs, pidAlive, now });
7999
+ gitWorktree.expireDeadWorktreeRegistrations({
8000
+ isTerminalBranch: (branch) => {
8001
+ const key = gitWorktree.keyFromBranch('job', branch);
8002
+ return !!key && !liveSlugs.has(key) && terminal.has(key);
8003
+ },
8004
+ });
8005
+ } catch (e) {
8006
+ console.warn('[scheduler] reservation expiry pass failed', e?.message);
8007
+ }
8008
+ }
8009
+
7257
8010
  function tickQueue({ bypassLoadGate = false } = {}) {
7258
- const next = tickTail.then(async () => {
8011
+ const budgetMs = tickWatchdogMs();
8012
+ const next = tickTail.then(() => {
8013
+ const gen = tickGeneration;
8014
+ return withTimeout(() => tickBody(gen, { bypassLoadGate }), budgetMs, () => {
8015
+ tickGeneration++; // fence the abandoned body
8016
+ console.warn(`[scheduler] TICK WEDGED: tick body exceeded ${budgetMs}ms — resetting tick chain`);
8017
+ if (!tickWedgeLatched) {
8018
+ tickWedgeLatched = true;
8019
+ appendAuditEvent('tick_wedged', { budgetMs });
8020
+ }
8021
+ // CAS: only reset if nothing has queued behind this wedged link.
8022
+ if (tickTail === tail) tickTail = Promise.resolve();
8023
+ return recordTick({ fired: false, reason: 'wedged' }, { detail: `tick body exceeded ${budgetMs}ms` });
8024
+ }).then((r) => {
8025
+ if (r?.reason !== 'wedged') tickWedgeLatched = false;
8026
+ return r;
8027
+ });
8028
+ });
8029
+ const tail = next.catch(() => {});
8030
+ tickTail = tail;
8031
+ return next;
8032
+ }
8033
+
8034
+ // The stale sentinel a fenced body returns: never recorded, never acted on.
8035
+ const STALE_TICK = Object.freeze({ fired: false, reason: 'stale-generation' });
8036
+
8037
+ async function tickBody(gen, { bypassLoadGate }) {
8038
+ const tickStartedAt = Date.now();
8039
+ {
8040
+ const stale = () => gen !== tickGeneration;
7259
8041
  const state = await readQueue();
8042
+ if (stale()) return STALE_TICK;
7260
8043
  // Never reconcile against an unreadable queue: reconcile() would see zero
7261
8044
  // job rows for every PRD on disk and resurrect the lot as 'pending'.
7262
8045
  if (state.unreadable) {
7263
8046
  console.error('[scheduler] tickQueue skipped: queue.json unreadable');
7264
8047
  return { fired: false, reason: 'unreadable' };
7265
8048
  }
8049
+ // Supervision of an adopted executor is independent of dispatch: it runs
8050
+ // even while paused, before any early return below.
8051
+ await superviseAdoptedRunsPass(state.jobs);
8052
+ if (stale()) return STALE_TICK;
7266
8053
  if (state.paused) {
7267
8054
  console.log('[scheduler] tickQueue skipped: paused');
7268
8055
  return recordTick({ fired: false, reason: 'paused' }, { detail: 'scheduler paused' });
7269
8056
  }
8057
+ // Upgrade drain (lib/upgradeDrain.cjs): a separate field from `paused`, so a
8058
+ // rate-limit pause can't overwrite it. Nothing new dispatches; running and
8059
+ // investigating rows finish.
8060
+ if (state.drain?.active) {
8061
+ return recordTick({ fired: false, reason: 'draining' }, { detail: 'draining for restart' });
8062
+ }
7270
8063
  if (cancelToken.cancelled) return { fired: false, reason: 'cancelled' };
7271
8064
 
7272
8065
  // Stamped here — the moment tickQueue actually reaches the picker,
7273
8066
  // regardless of whether this pass ends in a launch — so
7274
8067
  // classifyQueueStarvation can tell "the engine keeps evaluating the
7275
8068
  // queue" apart from "nothing has invoked tickQueue in a long time".
7276
- // Distinct from `lastRunAt` below, which stays true to its existing
7277
- // meaning (a batch actually launched) since other readers depend on that.
8069
+ // Distinct from `lastRunAt` below (a batch actually launched): this one
8070
+ // means only "the loop is alive" (heartbeat/health) and must NEVER feed
8071
+ // the idle clock — see dispatchIdleMs.
7278
8072
  await mutate((s) => { s.lastDispatchAttemptAt = new Date().toISOString(); });
8073
+ if (stale()) return STALE_TICK;
7279
8074
 
7280
8075
  // The retired-flat-dir sweep now lives inside reconcile() itself (see its
7281
8076
  // own comment) so every caller of reconcile — not just this tick — gets
7282
8077
  // the guarantee.
7283
8078
  await reconcile(state);
8079
+ if (stale()) return STALE_TICK;
7284
8080
  // Reclaim any job-kind worktree whose owning row already resolved
7285
8081
  // (completed/failed/skipped) without the run ever reaching
7286
8082
  // cleanupWorktree — a leaked checkout that would otherwise sit until the
@@ -7297,9 +8093,19 @@ function tickQueue({ bypassLoadGate = false } = {}) {
7297
8093
  // to also carry a private `concurrencyCap` of 3 — the exact per-consumer
7298
8094
  // cap that sessionSlots.cjs was written to replace — which silently
7299
8095
  // ceilinged the queue at 3 while the pool the user configured said 5.
7300
- const freeSlots = sessionSlots.available();
8096
+ // While the usage meter is degraded (degradedConcurrencyCapValue set by
8097
+ // pollLoop — circuit open, or a poll otherwise failed), a picker-side
8098
+ // hold narrows this SAME freeSlots figure instead of standing up a
8099
+ // second pool: the row count admitted this tick simply can't exceed the
8100
+ // degraded cap minus what's already running.
8101
+ runReservationExpiryPass(state.jobs, tickStartedAt);
8102
+ const freeSlots = degradedConcurrencyCapValue != null
8103
+ ? Math.max(0, Math.min(sessionSlots.available(), degradedConcurrencyCapValue - runningSet.size))
8104
+ : sessionSlots.available();
7301
8105
  const heldSlugs = await computeLaunchHolds(state);
8106
+ if (stale()) return STALE_TICK;
7302
8107
  const satisfiedSlugsByCwd = await computeDepHistorySatisfaction(state);
8108
+ if (stale()) return STALE_TICK;
7303
8109
  const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
7304
8110
  leaseHeld: quietMachineLease.isHeld(),
7305
8111
  machineInUse: sessionSlots.inUse(),
@@ -7399,18 +8205,18 @@ function tickQueue({ bypassLoadGate = false } = {}) {
7399
8205
  }
7400
8206
 
7401
8207
  await mutate((s) => { s.lastRunAt = new Date().toISOString(); });
8208
+ if (stale()) return STALE_TICK;
7402
8209
  await broadcast();
8210
+ if (stale()) return STALE_TICK;
7403
8211
 
7404
8212
  const { runId, dir: runDir } = pickRunDir();
7405
8213
  for (const job of gatedBatch) {
7406
- if (cancelToken.cancelled) break;
8214
+ if (cancelToken.cancelled || stale()) break;
7407
8215
  // spawnJob is fire-and-forget; it calls tickQueue() on completion.
7408
8216
  spawnJob(job, runId, runDir, state.config.defaultCwd).catch(() => {});
7409
8217
  }
7410
8218
  return recordTick({ fired: true, count: gatedBatch.length, group: gatedBatch[0]?.parallelGroup }, { holds });
7411
- });
7412
- tickTail = next.catch(() => {});
7413
- return next;
8219
+ }
7414
8220
  }
7415
8221
 
7416
8222
  // Translates a tickQueue()/runDueJobs() outcome descriptor into a renderer-facing
@@ -7486,6 +8292,40 @@ async function maybeLaunchWhenAvailable(state) {
7486
8292
  * second scheduler. */
7487
8293
  const QUEUE_STARVATION_MS = 10 * 60_000;
7488
8294
 
8295
+ /**
8296
+ * dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now }) → ms
8297
+ *
8298
+ * Pure. The ONE dispatch-idleness clock: `now - max(lastRunAt, lastPauseClearedAt,
8299
+ * schedulerBootedAt)`. `lastRunAt` has exactly one writer (tickQueue, right
8300
+ * before the spawn loop) so it already means "a batch actually launched";
8301
+ * a pause clear and a scheduler boot are the other two moments the queue
8302
+ * legitimately (re)starts. Deliberately NOT `lastDispatchAttemptAt`, which
8303
+ * tickQueue stamps before every gate — a queue that ticks every 30 s and
8304
+ * launches nothing (leaked slot, stuck hold) would refresh that stamp
8305
+ * forever and never look idle (the structural repeat of the f18e161 bug
8306
+ * where lastRunAt was refreshed every poll). No finite input → Infinity.
8307
+ */
8308
+ function dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now } = {}) {
8309
+ const stamps = [lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs].filter(Number.isFinite);
8310
+ return stamps.length ? now - Math.max(...stamps) : Infinity;
8311
+ }
8312
+
8313
+ /**
8314
+ * launchBlockedSlugs(jobs, launchBlocks) → Set<slug>
8315
+ *
8316
+ * Pure, sync (health.cjs runs as a cold process). Pending rows whose persona
8317
+ * has ANY launch-breaker entry — a superset of computeLaunchHolds, which
8318
+ * additionally lets one half-open probe row through per persona.
8319
+ */
8320
+ function launchBlockedSlugs(jobs, launchBlocks) {
8321
+ const out = new Set();
8322
+ if (!launchBlocks || !Object.keys(launchBlocks).length) return out;
8323
+ for (const j of Array.isArray(jobs) ? jobs : []) {
8324
+ if (j && j.status === 'pending' && launchBlocks[launchFailure.launchBlockKeyFor(j)]) out.add(j.slug);
8325
+ }
8326
+ return out;
8327
+ }
8328
+
7489
8329
  /**
7490
8330
  * classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs })
7491
8331
  * → null | { kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }
@@ -7513,22 +8353,28 @@ const QUEUE_STARVATION_MS = 10 * 60_000;
7513
8353
  * Returns null when the queue is healthy (work running, nothing pending,
7514
8354
  * paused on purpose, or simply not idle long enough yet).
7515
8355
  */
7516
- function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
8356
+ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, heldSlugs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
7517
8357
  if (paused) return null; // paused is a DECISION, not a stall
7518
8358
  if (runningCount > 0) return null; // work is flowing
7519
8359
  const rows = Array.isArray(jobs) ? jobs : [];
7520
8360
  const pending = rows.filter((j) => j && j.status === 'pending');
7521
8361
  if (pending.length === 0) return null; // nothing to run — not a stall
7522
8362
 
7523
- const idleMs = Number.isFinite(lastRunAtMs) ? now - lastRunAtMs : Infinity;
8363
+ const idleMs = dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now });
7524
8364
  if (idleMs < thresholdMs) return null; // give the normal path its chance first
7525
8365
 
7526
8366
  // Which pending rows could actually dispatch? Anything NOT named by a
7527
- // blocked chain. computeBlockedChains already walks dependsOn with the
7528
- // picker's own resolution, so the two can never disagree.
8367
+ // blocked chain (computeBlockedChains walks dependsOn with the picker's own
8368
+ // resolution, so the two can never disagree) and NOT held by an open launch
8369
+ // breaker / the quietMachine lease (`heldSlugs` — a separate input, never
8370
+ // folded into the dependsOn walker). Held rows can't launch no matter how
8371
+ // often we tick, so they read as 'blocked' (needs a human), not 'starved'.
7529
8372
  const blockedChains = computeBlockedChains(rows);
7530
- const blockedTotal = blockedChains.reduce((n, c) => n + c.blocked, 0);
7531
- const dispatchable = pending.length - blockedTotal;
8373
+ const held = heldSlugs?.has ? heldSlugs : new Set(heldSlugs ?? []);
8374
+ const open = held.size > 0 ? rows.filter((j) => !(j.status === 'pending' && held.has(j.slug))) : rows;
8375
+ const openBlocked = held.size > 0 ? computeBlockedChains(open) : blockedChains;
8376
+ const openPending = open.filter((j) => j.status === 'pending').length;
8377
+ const dispatchable = openPending - openBlocked.reduce((n, c) => n + c.blocked, 0);
7532
8378
 
7533
8379
  return {
7534
8380
  kind: dispatchable > 0 ? 'starved' : 'blocked',
@@ -7551,14 +8397,14 @@ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now,
7551
8397
  * projects sat starved/blocked for hours, and the watchdog never fired once
7552
8398
  * because "work is flowing" was true somewhere else. Partitioning by cwd
7553
8399
  * (the same grouping computeBlockedChains already does) fixes DETECTION only
7554
- * — the idle clock (`lastRunAtMs`) stays machine-wide, since
7555
- * `lastDispatchAttemptAt` is machine-level state, and only one tick is ever
8400
+ * — the idle clock (see dispatchIdleMs) stays machine-wide, since
8401
+ * `lastRunAt` is machine-level state, and only one tick is ever
7556
8402
  * forced per watchdog pass regardless of how many cwds are starved.
7557
8403
  *
7558
8404
  * Pure, no IO. Returns [] when paused (a DECISION, not a stall) or when no
7559
8405
  * project has a verdict.
7560
8406
  */
7561
- function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
8407
+ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, heldSlugs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
7562
8408
  if (paused) return [];
7563
8409
  const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
7564
8410
  const byCwd = new Map();
@@ -7580,6 +8426,9 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
7580
8426
  paused: false,
7581
8427
  runningCount: projRunningCount,
7582
8428
  lastRunAtMs,
8429
+ lastPauseClearedAtMs,
8430
+ schedulerBootedAtMs,
8431
+ heldSlugs,
7583
8432
  now,
7584
8433
  thresholdMs,
7585
8434
  });
@@ -7590,7 +8439,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
7590
8439
 
7591
8440
  /**
7592
8441
  * classifyQueueHealth({ jobs, paused, launchBlocks, runningSet, freeSlots,
7593
- * totalSlots, lastDispatchAttemptAtMs, now, cwd, thresholdMs })
8442
+ * totalSlots, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, cwd, thresholdMs })
7594
8443
  * → { kind, cwd, pending, dispatchable, blockedChains, needsReviewCount, runningCount, ... }
7595
8444
  *
7596
8445
  * Single source of truth for the Scheduler page's queue-health header: the
@@ -7601,7 +8450,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
7601
8450
  *
7602
8451
  * Reuses classifyQueueStarvation for the blocked/stalled read so the header
7603
8452
  * can never disagree with runQueueStarvationWatchdog's own decision to force
7604
- * a tick: both are handed the same lastDispatchAttemptAt-based idle clock and
8453
+ * a tick: both are handed the same launch-keyed idle clock (dispatchIdleMs) and
7605
8454
  * the same computeBlockedChains walk under the hood. Called here with
7606
8455
  * `thresholdMs: 0` first (a live header must say "blocked" the instant every
7607
8456
  * pending row is dependency-stuck, not wait out the watchdog's own 10-minute
@@ -7638,7 +8487,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
7638
8487
  */
7639
8488
  function classifyQueueHealth({
7640
8489
  jobs, paused, launchBlocks, runningSet: runningSlugs, freeSlots, totalSlots,
7641
- lastDispatchAttemptAtMs, now, cwd = null, thresholdMs = QUEUE_STARVATION_MS,
8490
+ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, cwd = null, thresholdMs = QUEUE_STARVATION_MS,
7642
8491
  } = {}) {
7643
8492
  const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
7644
8493
  const projectJobs = cwd ? rows.filter((j) => j.cwd === cwd) : rows;
@@ -7688,19 +8537,22 @@ function classifyQueueHealth({
7688
8537
  // over the same rows), so `base` already carries them.
7689
8538
  const immediate = classifyQueueStarvation({
7690
8539
  jobs: projectJobs, paused: false, runningCount: 0,
7691
- lastRunAtMs: lastDispatchAttemptAtMs, now, thresholdMs: 0,
8540
+ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, thresholdMs: 0,
7692
8541
  });
7693
8542
  // pending.length is already > 0 above, so `immediate` can only be null when
7694
- // lastDispatchAttemptAtMs is itself in the future (clock skew) — fall back
7695
- // to computing idleMs the same way rather than asserting a kind we can't
8543
+ // the clock stamp is itself in the future (clock skew) — fall back to
8544
+ // computing idleMs the same way rather than asserting a kind we can't
7696
8545
  // back up with a real number.
7697
8546
  const idleMs = immediate ? immediate.idleMs
7698
- : (Number.isFinite(lastDispatchAttemptAtMs) ? now - lastDispatchAttemptAtMs : Infinity);
8547
+ : dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now });
7699
8548
  if (dispatchable === 0) return { ...base, kind: 'blocked', idleMs };
7700
8549
  const kind = idleMs >= thresholdMs ? 'stalled' : 'running';
7701
8550
  return { ...base, kind, idleMs };
7702
8551
  }
7703
8552
 
8553
+ // Per-cwd latch for runQueueStarvationWatchdog: cwd → { kind, at, running }.
8554
+ const starvationLatch = new Map();
8555
+
7704
8556
  /**
7705
8557
  * The watchdog half: acts on classifyQueueStarvationByProject. Called from
7706
8558
  * the heartbeat, which already runs on its own timer independent of the
@@ -7713,33 +8565,55 @@ function classifyQueueHealth({
7713
8565
  * the tick itself is machine-wide (it drives whatever the picker finds
7714
8566
  * across every project), only the DETECTION is per-project.
7715
8567
  */
7716
- async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
7717
- // lastDispatchAttemptAt, not lastRunAt: the latter only advances when a
7718
- // batch actually launches, so a poll that keeps succeeding while dispatch
7719
- // itself never gets invoked would otherwise mask a stall behind a fresh-
7720
- // looking timestamp that was never actually tracking dispatch liveness.
8568
+ async function runQueueStarvationWatchdog(state, {
8569
+ now = Date.now(), thresholdMs = QUEUE_STARVATION_MS,
8570
+ bootedAtMs = Date.parse(SCHEDULER_BOOTED_AT), pauseClearedAtMs = lastPauseClearedAt,
8571
+ } = {}) {
8572
+ // The idle clock is launch-keyed (dispatchIdleMs): NOT lastDispatchAttemptAt,
8573
+ // which tickQueue stamps before every gate — a queue whose 30 s loop ticks
8574
+ // and launches nothing would refresh it forever and the watchdog would
8575
+ // never fire. Rows held by an open launch breaker or the quietMachine lease
8576
+ // can't launch however often we tick, so they're passed in as `heldSlugs`
8577
+ // and read as 'blocked' (needs a human) rather than a false 'starved'.
8578
+ const heldSlugs = new Set((await computeLaunchHolds(state)).keys());
8579
+ if (quietMachineLease.isHeld()) {
8580
+ for (const j of state?.jobs ?? []) if (j?.status === 'pending' && j.quietMachine === true) heldSlugs.add(j.slug);
8581
+ }
7721
8582
  const verdicts = classifyQueueStarvationByProject({
7722
8583
  jobs: state?.jobs,
7723
- paused: state?.paused,
8584
+ paused: upgradeDrain.effectivePaused(state),
7724
8585
  runningSet,
7725
- lastRunAtMs: Date.parse(state?.lastDispatchAttemptAt ?? ''),
8586
+ lastRunAtMs: Date.parse(state?.lastRunAt ?? ''),
8587
+ lastPauseClearedAtMs: pauseClearedAtMs,
8588
+ schedulerBootedAtMs: bootedAtMs,
8589
+ heldSlugs,
7726
8590
  now,
7727
8591
  thresholdMs,
7728
8592
  });
8593
+ // Latch: one episode per (cwd, kind) — re-arms only once QUEUE_STARVATION_MS
8594
+ // has elapsed again or the project's running count changes; forgotten the
8595
+ // moment the cwd stops having a verdict at all.
8596
+ const activeKeys = new Set(verdicts.map((v) => v.cwd));
8597
+ for (const cwd of [...starvationLatch.keys()]) if (!activeKeys.has(cwd)) starvationLatch.delete(cwd);
7729
8598
  if (verdicts.length === 0) return null;
7730
8599
 
7731
8600
  let anyStarved = false;
7732
8601
  let primary = null;
7733
8602
  for (const verdict of verdicts) {
7734
8603
  const mins = Math.round(verdict.idleMs / 60_000);
8604
+ const running = (state?.jobs ?? []).filter((j) => j?.cwd === verdict.cwd && (j.status === 'running' || runningSet.has(j.slug))).length;
8605
+ const latched = starvationLatch.get(verdict.cwd);
8606
+ const suppressed = !!latched && latched.kind === verdict.kind && latched.running === running && now - latched.at < QUEUE_STARVATION_MS;
8607
+ if (!primary) primary = verdict;
8608
+ if (suppressed) continue;
8609
+ starvationLatch.set(verdict.cwd, { kind: verdict.kind, at: now, running });
7735
8610
  if (verdict.kind === 'blocked') {
7736
8611
  console.warn(
7737
8612
  `[scheduler] QUEUE BLOCKED (${verdict.cwd}): ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
7738
- + `terminal or parked dependency, so ticking cannot help. Blockers: `
8613
+ + `terminal or parked dependency, an open launch breaker, or the quietMachine lease, so ticking cannot help. Blockers: `
7739
8614
  + verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
7740
8615
  );
7741
8616
  appendAuditEvent('queue_blocked_stall', { cwd: verdict.cwd, pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
7742
- if (!primary) primary = verdict;
7743
8617
  continue;
7744
8618
  }
7745
8619
 
@@ -7756,9 +8630,12 @@ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs
7756
8630
 
7757
8631
  // A never-populated utilization reading is itself one of the ways the
7758
8632
  // when-available path silently never fires (maybeLaunchWhenAvailable
7759
- // returns early on null). Treat unknown as safe here, exactly as the
7760
- // billing meter's own 429 fallback already does.
7761
- if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
8633
+ // returns early on null). Absence of information, not a green light: fall
8634
+ // back to the same conservative degraded budget the poll loop itself uses
8635
+ // rather than a blind cachedUtilization=0.
8636
+ if (cachedUtilization === null || cachedUtilization === undefined) {
8637
+ applyDegradedBudget();
8638
+ }
7762
8639
  // The in-process cancelToken is only ever reset by runDueJobs() (force-tick
7763
8640
  // / run-now / resume-timer) — every other path that clears a pause
7764
8641
  // (clearPause(), the poll loop's own auto-recovery) leaves it untouched
@@ -7924,6 +8801,72 @@ async function runBranchSweep(jobs) {
7924
8801
  * skipped too (spawn may still be mid-flight) — see selectReapableJobs for
7925
8802
  * the full predicate. Exported so unit tests can invoke it directly.
7926
8803
  */
8804
+ // Adopted-run supervision (see lib/adoptedRunSupervisor.cjs). In-memory by
8805
+ // design: the supervisors die with this process, and the next boot re-arms.
8806
+ const adoptedSupervisors = new Map();
8807
+
8808
+ /** Pid of a boot-time running row via the same ladder isBootRowAlive uses:
8809
+ * supervisor record → runtime.pid → pid spawned per the run log. */
8810
+ function bootRowPid(j, logPathOf) {
8811
+ const runDir = j.runId ? path.join(schedulerPaths.runsDir(), j.runId) : null;
8812
+ const record = runDir ? supervisorRecord.readSupervisorRecord(runDir, j.slug) : null;
8813
+ return record?.pid || j.runtime?.pid || readSpawnedPidFromLog(logPathOf(j)) || null;
8814
+ }
8815
+
8816
+ function signalAdoptedGroup(pgid, signal, pid) {
8817
+ try { process.kill(-pgid, signal); } catch {
8818
+ try { process.kill(pid, signal); } catch { /* already dead */ }
8819
+ }
8820
+ }
8821
+
8822
+ async function superviseAdoptedRunsPass(jobs) {
8823
+ try {
8824
+ const rows = jobs || (await readQueue()).jobs;
8825
+ const runDirOf = (j) => (j.runId ? path.join(schedulerPaths.runsDir(), j.runId) : null);
8826
+ return await adoptedRunSupervisor.superviseAdoptedRuns(rows, {
8827
+ registry: adoptedSupervisors,
8828
+ runDir: runDirOf,
8829
+ readRecord: supervisorRecord.readSupervisorRecord,
8830
+ lease: quietMachineLease,
8831
+ markSupervised: async (row) => {
8832
+ await mutate((s) => {
8833
+ const j = s.jobs.find((x) => x.slug === row.slug);
8834
+ if (j && j.status === 'running' && (j.runId ?? null) === (row.runId ?? null)) j.supervisedAt = new Date().toISOString();
8835
+ });
8836
+ },
8837
+ makeDeps: (row, record) => {
8838
+ const logPath = path.join(runDirOf(row), `${row.slug}.log`);
8839
+ return {
8840
+ logPath,
8841
+ statLogMtimeMs: readLogMtimeMs,
8842
+ pidAlive: claudePidAlive,
8843
+ identityOf: procIdentityOf,
8844
+ isDifferentProcess,
8845
+ killGroup: signalAdoptedGroup,
8846
+ // Stamped BEFORE the signal so reapDeadRunningJobs, which finalizes
8847
+ // the row once the process is gone, always sees why it died.
8848
+ stampKill: async (kind, reason) => {
8849
+ await mutate((s) => {
8850
+ const j = s.jobs.find((x) => x.slug === row.slug);
8851
+ if (j && j.status === 'running' && (j.runId ?? null) === (row.runId ?? null)) {
8852
+ j.adoptedKill = { watchdog: kind, reason, at: new Date().toISOString() };
8853
+ }
8854
+ });
8855
+ try { fs.appendFileSync(logPath, `\n[scheduler] adopted-run ${kind} watchdog: ${reason}\n`); } catch { /* best-effort */ }
8856
+ },
8857
+ log: (msg) => console.log(`[scheduler] ${row.slug}: ${msg}`),
8858
+ checkIntervalMs: IDLE_CHECK_INTERVAL_MS,
8859
+ sigkillAfterMs: POST_RESULT_KILL_MS,
8860
+ defaultMaxDurationMs: MAX_JOB_DURATION_MS,
8861
+ };
8862
+ },
8863
+ });
8864
+ } catch (e) {
8865
+ console.warn('[scheduler] adopted-run supervision pass failed', e?.message);
8866
+ return [];
8867
+ }
8868
+ }
8869
+
7927
8870
  async function reapDeadRunningJobs() {
7928
8871
  try {
7929
8872
  // Do NOT gate on runningSet: spawnJob()'s finally block unconditionally
@@ -7932,9 +8875,13 @@ async function reapDeadRunningJobs() {
7932
8875
  // status:"running" with no slug left in runningSet to trigger reconciliation.
7933
8876
  // queue.json is the source of truth for which jobs are actually running.
7934
8877
  const state = await readQueue();
8878
+ // A quarantined shard's rows never loaded; the filter is defence in depth
8879
+ // so a reap can never terminalize a row of a project we cannot persist.
8880
+ const reapSkip = quarantinedCwdSet(state);
8881
+ if (reapSkip.size > 0) state.jobs = state.jobs.filter((j) => !reapSkip.has(j.cwd));
7935
8882
  // Shared by the log-evidence injections below and the reapable-processing
7936
8883
  // loop further down — same `j.runId` → run log path formula either way.
7937
- const logPathForJob = (j) => (j?.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null);
8884
+ const logPathForJob = (j) => (j?.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null);
7938
8885
  const { reapable, warnings, recovered } = selectReapableJobs(state.jobs, Date.now(), {
7939
8886
  pidAlive: claudePidAlive,
7940
8887
  grace: PIDLESS_SPAWN_GRACE_MS,
@@ -8008,8 +8955,12 @@ async function reapDeadRunningJobs() {
8008
8955
  // the same still-active rate limit — the spin loop this PRD exists to
8009
8956
  // stop. Done once, outside mutate(), before finalizing any row below.
8010
8957
  if (dead.some((d) => d.outcome === 'rate_limited')) {
8958
+ // Same rationale as spawnJob's own rateLimited branch: a dead-process
8959
+ // reap that classifies as rate-limited is just as much an executor-
8960
+ // observed 429 as a live one, and must open the same shared circuit.
8961
+ billing.usageCircuit.recordFailure('executor_429');
8011
8962
  const triggering = dead.find((d) => d.outcome === 'rate_limited');
8012
- const billingResetIso = await refreshNextReset().catch(() => cachedNextReset);
8963
+ const billingResetIso = await billingResetForPause();
8013
8964
  const resetIso = resolveRateLimitPauseReset(triggering.logPath, billingResetIso);
8014
8965
  const triggeringRow = triggering ? state.jobs.find((x) => x.slug === triggering.slug) : null;
8015
8966
  const observedAtMs = triggeringRow?.startedAt ? Date.parse(triggeringRow.startedAt) : null;
@@ -8115,9 +9066,10 @@ async function reapDeadRunningJobs() {
8115
9066
  // this runs the whole dead-job batch concurrently rather than one
8116
9067
  // dispatch's git-spawn latency at a time.
8117
9068
  await Promise.all(dead.map(async (d) => {
8118
- if (d.outcome === 'rate_limited' || d.outcome === 'success') return;
8119
- if (d.pidless && d.failureOverride) return;
8120
9069
  const row = state.jobs.find((x) => x.slug === d.slug);
9070
+ const adoptedBudgetKill = row?.adoptedKill?.watchdog === 'budget';
9071
+ if (d.outcome === 'rate_limited' || (d.outcome === 'success' && !adoptedBudgetKill)) return;
9072
+ if (d.pidless && d.failureOverride) return;
8121
9073
  if (!row?.landedCommit) return;
8122
9074
  const rowCwd = row.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD;
8123
9075
  const resolved = await resolveLandedCommitEvidence(rowCwd, row.landedCommit, row.startedAt);
@@ -8150,7 +9102,7 @@ async function reapDeadRunningJobs() {
8150
9102
  const baseSet = new Set(s.jobs[idx].guardBaseline);
8151
9103
  deltaPaths = after.filter((p) => !baseSet.has(p));
8152
9104
  if (deltaPaths.length) {
8153
- const salvagePath = path.join(RUNS_DIR, s.jobs[idx].runId, `${slug}.uncommitted.patch`);
9105
+ const salvagePath = path.join(schedulerPaths.runsDir(), s.jobs[idx].runId, `${slug}.uncommitted.patch`);
8154
9106
  const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: rowCwd, paths: deltaPaths, outFile: salvagePath });
8155
9107
  if (salvage && salvage.ok) {
8156
9108
  s.jobs[idx].salvagePatch = salvagePath;
@@ -8198,6 +9150,17 @@ async function reapDeadRunningJobs() {
8198
9150
  if (pidless && !effectiveSuccess && !rateLimited && !notLandedInfo && failureOverride) {
8199
9151
  notLandedInfo = { verdict: failureOverride.verdict, reason: failureOverride.reason };
8200
9152
  }
9153
+ // An adopted run the budget watchdog killed (adoptedKill, stamped
9154
+ // before the signal) parks exactly like a native budget kill: the
9155
+ // shared classifyBudgetKill decides, and it wins over success/failure.
9156
+ const adoptedBudgetKill = rateLimited ? null : classifyBudgetKill({
9157
+ killedByWatchdog: s.jobs[idx].adoptedKill?.watchdog,
9158
+ budgetKillReason: s.jobs[idx].adoptedKill?.reason,
9159
+ }, landedCommitEvidence.get(slug) || null);
9160
+ if (adoptedBudgetKill) {
9161
+ effectiveSuccess = false;
9162
+ notLandedInfo = { verdict: 'budget_exceeded', reason: adoptedBudgetKill.reason };
9163
+ }
8201
9164
 
8202
9165
  const leftoverSuffix = deltaPaths && deltaPaths.length
8203
9166
  ? ` — left ${deltaPaths.length} files uncommitted`
@@ -8249,6 +9212,7 @@ async function reapDeadRunningJobs() {
8249
9212
  s.jobs[idx].gateOutcome = gateOutcome;
8250
9213
  if (confirmedLandedCommit) s.jobs[idx].landedCommit = confirmedLandedCommit;
8251
9214
  if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
9215
+ if (adoptedBudgetKill?.landedCommit) s.jobs[idx].landedCommit = adoptedBudgetKill.landedCommit;
8252
9216
  }
8253
9217
  // A pidless spawn that never wrote its own '<slug>.log' into the
8254
9218
  // batch runId dir it was stamped with must not keep that runId —
@@ -8263,6 +9227,7 @@ async function reapDeadRunningJobs() {
8263
9227
  s.jobs[idx].runId = null;
8264
9228
  }
8265
9229
  delete s.jobs[idx].runtime;
9230
+ delete s.jobs[idx].adoptedKill;
8266
9231
  delete s.jobs[idx].dispatchPhase;
8267
9232
  delete s.jobs[idx].dispatchPhaseAt;
8268
9233
  delete s.jobs[idx].overrun;
@@ -8327,14 +9292,21 @@ async function pollLoop() {
8327
9292
  // 404/time-out and eventually pause the queue on 'network' — treat usage as
8328
9293
  // wide-open and fire on pending + memory alone. (Blackrock-style machines.)
8329
9294
  if (!billing.usageMeterApplicable()) {
8330
- cachedUtilization = 0;
9295
+ // Close the shared circuit if a PRIOR consumer-auth session left it
9296
+ // open/half_open — this process has stopped polling the meter
9297
+ // entirely, so nothing else will ever call recordSuccess() to clear
9298
+ // it, and health.cjs would otherwise read a stale open circuit as
9299
+ // YELLOW/RED forever even though nothing is actually degraded.
9300
+ if (billing.usageCircuit.state() !== 'closed') billing.usageCircuit.recordSuccess({});
9301
+ cachedUtilization = NO_METER_UTILIZATION;
9302
+ degradedConcurrencyCapValue = null;
8331
9303
  consecutiveFailures = 0;
8332
9304
  backoffMs = 0;
8333
9305
  backoffNextAt = null;
8334
9306
  firstFailureAt = null;
8335
9307
  firstNon429FailureAt = null;
8336
9308
  lastFailureKind = null;
8337
- failureStreakWarned = false;
9309
+ resetFailureStreak();
8338
9310
  lastPollAt = Date.now();
8339
9311
  lastPollOk = true;
8340
9312
  persistSchedulerState();
@@ -8350,18 +9322,40 @@ async function pollLoop() {
8350
9322
  return; // finally re-arms the timer
8351
9323
  }
8352
9324
 
9325
+ // Shared breaker over the meter (AC1): while it is OPEN, no request is
9326
+ // made except the half-open probe below (billing.fetchUsage() is only
9327
+ // ever reached, further down, from the closed/half_open paths). "Meter
9328
+ // down" reads as absence of information, not a green light — the
9329
+ // conservative degraded budget stands in for both the utilization-
9330
+ // threshold gate (maybeLaunchWhenAvailable) and the concurrency cap
9331
+ // (tickQueue's freeSlots), never a blind cachedUtilization=0.
9332
+ if (billing.usageCircuit.state() === 'open') {
9333
+ applyDegradedBudget();
9334
+ lastPollAt = Date.now();
9335
+ lastPollOk = false;
9336
+ warnFailureStreakIfNeeded();
9337
+ persistSchedulerState();
9338
+ const cur = await readQueue();
9339
+ await maybeLaunchWhenAvailable(cur);
9340
+ await broadcast();
9341
+ return;
9342
+ }
9343
+
8353
9344
  const r = await billing.fetchUsage();
8354
9345
 
8355
9346
  if (r.kind === 'ok') {
8356
- cachedNextReset = r.data?.usage?.five_hour?.resets_at ?? cachedNextReset;
8357
- cachedUtilization = r.data?.usage?.five_hour?.utilization ?? cachedUtilization;
9347
+ const window = bindingWindow(r.data?.usage);
9348
+ recordObservedReset(window.resets_at ?? null);
9349
+ cachedUtilization = Number.isFinite(window.utilization) ? window.utilization : cachedUtilization;
9350
+ lastGoodUsagePayload = r.data?.usage ?? lastGoodUsagePayload;
9351
+ degradedConcurrencyCapValue = null;
8358
9352
  consecutiveFailures = 0;
8359
9353
  backoffMs = 0;
8360
9354
  backoffNextAt = null;
8361
9355
  firstFailureAt = null;
8362
9356
  firstNon429FailureAt = null;
8363
9357
  lastFailureKind = null;
8364
- failureStreakWarned = false;
9358
+ resetFailureStreak();
8365
9359
  lastPollAt = Date.now();
8366
9360
  lastPollOk = true;
8367
9361
  persistSchedulerState();
@@ -8377,14 +9371,14 @@ async function pollLoop() {
8377
9371
  await maybeLaunchWhenAvailable(cur);
8378
9372
  await broadcast();
8379
9373
  } else if (r.kind === 'meter_rate_limited') {
8380
- // Billing meter is itself being rate-limited. Treat as "utilization unknown but safe":
8381
- // fire available jobs anyway at utilization=0 rather than pausing the queue.
8382
- // Still back off the POLL cadence itself (same curve/cap as the transient
8383
- // branch) and persist state every cycle — without this, a sustained 429
8384
- // streak hammered the already-rate-limited endpoint every POLL_INTERVAL_MS
8385
- // forever AND never wrote lastPollAt/consecutiveFailures back to
8386
- // scheduler-state.json, so the sidecar froze stale while the loop kept
8387
- // failing silently underneath it (the 57-consecutive-failure incident).
9374
+ // Billing meter is itself being rate-limited — absence of information,
9375
+ // not a green light. Still back off the POLL cadence itself (same
9376
+ // curve/cap as the transient branch) and persist state every cycle —
9377
+ // without this, a sustained 429 streak hammered the already-rate-
9378
+ // limited endpoint every POLL_INTERVAL_MS forever AND never wrote
9379
+ // lastPollAt/consecutiveFailures back to scheduler-state.json, so the
9380
+ // sidecar froze stale while the loop kept failing silently underneath
9381
+ // it (the 57-consecutive-failure incident).
8388
9382
  lastPollAt = Date.now();
8389
9383
  lastPollOk = false;
8390
9384
  consecutiveFailures++;
@@ -8392,8 +9386,8 @@ async function pollLoop() {
8392
9386
  // Don't update firstNon429FailureAt — 429s don't count toward the 30-min network-pause threshold.
8393
9387
  backoffMs = nextBackoffMs(backoffMs);
8394
9388
  backoffNextAt = Date.now() + backoffMs;
8395
- cachedUtilization = 0; // assume safe; fire any pending work
8396
- console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on heuristic (failure #${consecutiveFailures}); retry in ${backoffMs / 1000}s`);
9389
+ applyDegradedBudget();
9390
+ console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on degraded budget (util=${cachedUtilization}%, cap=${degradedConcurrencyCapValue}) (failure #${consecutiveFailures}); retry in ${backoffMs / 1000}s`);
8397
9391
  warnFailureStreakIfNeeded();
8398
9392
  persistSchedulerState();
8399
9393
  const cur = await readQueue();
@@ -8433,13 +9427,12 @@ async function pollLoop() {
8433
9427
  // 'ok' and 'meter_rate_limited' branches used to reach
8434
9428
  // maybeLaunchWhenAvailable, so auth/transient failures left ready
8435
9429
  // pending work untouched until either the queue-starvation watchdog's
8436
- // 10-minute safety net fired or the poll itself recovered. Utilization
8437
- // is unknown during a failed poll, not unsafe — treated the same way
8438
- // the meter_rate_limited branch above already treats a 429 as safe to
8439
- // fire through. maybeLaunchWhenAvailable itself still honors an
8440
- // 'auth'/'network' pause (state.paused), so this is a no-op whenever
8441
- // setPaused() above actually engaged one.
8442
- if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
9430
+ // 10-minute safety net fired or the poll itself recovered. Absence of
9431
+ // information, not a green light: fall back to the degraded budget
9432
+ // rather than a blind cachedUtilization=0. maybeLaunchWhenAvailable
9433
+ // itself still honors an 'auth'/'network' pause (state.paused), so
9434
+ // this is a no-op whenever setPaused() above actually engaged one.
9435
+ applyDegradedBudget();
8443
9436
  await maybeLaunchWhenAvailable(await readQueue());
8444
9437
  await broadcast();
8445
9438
  }
@@ -8458,7 +9451,7 @@ async function pollLoop() {
8458
9451
  // Same rationale as the auth/transient branch above: the outer catch
8459
9452
  // must not be a silent dispatch dead-end either.
8460
9453
  try {
8461
- if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
9454
+ applyDegradedBudget();
8462
9455
  await maybeLaunchWhenAvailable(await readQueue());
8463
9456
  await broadcast();
8464
9457
  } catch { /* best-effort — the poll loop must still re-arm below */ }
@@ -8521,6 +9514,34 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
8521
9514
  // anyway). For non-fix-plan jobs the exemption never applies, so rescanning
8522
9515
  // their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
8523
9516
  const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'abandoned_background_task', 'pass_no_commit', 'pass_no_commit_already_shipped']);
9517
+ // RESCANNABLE_VERDICTS is a HINT, not a gate: it names the verdicts whose
9518
+ // recovery rung is a transcript re-verification (verifyRun). Every other
9519
+ // needs_review verdict is still a heal candidate (isRescanCandidate) — it just
9520
+ // gets the evidence-only rung (computeLooksDone) instead of a transcript
9521
+ // rescan, because verifyRun cannot see a commit-guard / shared-tree verdict and
9522
+ // would return 'clean' and falsely heal it.
9523
+
9524
+ // The ONLY needs_review verdicts NOT eligible for the periodic heal ladder.
9525
+ // An allow-list here was reopened three times (2026-09-12 x2, 2026-09-18
9526
+ // shared_tree_reverted) because a new park reason was born invisible to
9527
+ // self-healing. Add a verdict here only with a one-line proof that no
9528
+ // re-verification or evidence scan can ever change it.
9529
+ const RESCAN_EXCLUDED_VERDICTS = new Set([
9530
+ // Commit-guard verdict verifyRun never inspects: a rescan returns 'clean' and would heal genuinely unfinished work.
9531
+ 'uncommitted_changes',
9532
+ // Its damage IS a commit stranded on an unmerged sm-job branch — a landedCommit restates it; selectMechanicalRecoveryTarget owns the real re-merge.
9533
+ 'worktree_integration_failed',
9534
+ // The run overran its own time/cost estimate; no transcript or git evidence can un-overrun it (selectAutoFixTargets excludes it too).
9535
+ 'budget_exceeded',
9536
+ ]);
9537
+
9538
+ // Per-pass / per-row bounds on the evidence-only rung (the widened candidate
9539
+ // set). Each scan costs one computeLooksDone: a per-cwd-deduped `git fetch`
9540
+ // (<=~20s) + a git log. Unbounded, a backlog of N parked rows would pay N of
9541
+ // those every 10 minutes forever.
9542
+ const REVERIFY_INTERVAL_MS = 10 * 60_000;
9543
+ const EVIDENCE_SCAN_MAX_PER_PASS = 20;
9544
+ const EVIDENCE_SCAN_MIN_INTERVAL_MS = 6 * REVERIFY_INTERVAL_MS;
8524
9545
 
8525
9546
  // Bounds fix-plan recursion: cap N permits at most N+1 fix jobs per original
8526
9547
  // slug (depth 1 = the original job, depth 2 = its `-fix`, depth 3+ is
@@ -8563,7 +9584,7 @@ function isFixPlanBeyondDepthCap(slug, investigationDepth, isFixPlan) {
8563
9584
  * check, no nested loop over user-scaled data. Dir names are ISO timestamps,
8564
9585
  * so lexical-descending sort picks the newest match. Exported for tests.
8565
9586
  */
8566
- function resolveRunId(job, { runsDir = RUNS_DIR } = {}) {
9587
+ function resolveRunId(job, { runsDir = schedulerPaths.runsDir() } = {}) {
8567
9588
  if (!job || job.runId) return job?.runId || null;
8568
9589
  if (!job.slug) return null;
8569
9590
  let dirs;
@@ -8665,18 +9686,71 @@ function isGuardParkedWithoutAutoFix(job) {
8665
9686
  return GUARD_VERDICT_EVIDENCE_ELIGIBLE.has(job.verifierVerdict);
8666
9687
  }
8667
9688
 
9689
+ /**
9690
+ * Pure predicate, no I/O: a needs_review row whose auto-fix investigation
9691
+ * genuinely ran (autoFixAttempted === true) but whose outcome was NEVER
9692
+ * durably stamped at all — and that has nothing left in flight to wait on:
9693
+ * no live/queued fix-plan row at fixSlugFor(job).
9694
+ *
9695
+ * Distinct from isExhaustedAutoFix, which requires autoFixRetries >= 1 to
9696
+ * have already accumulated. spawnInvestigation's onExit handler restores the
9697
+ * job's status from 'investigating' back to needs_review in ONE mutate()
9698
+ * call (scheduler.cjs's spawnInvestigation, source
9699
+ * 'spawnInvestigation:onExit') and stamps autoFixOutcome ('plan' / 'no-plan'
9700
+ * / 'error') in a SEPARATE, later mutate() call — an app restart or process
9701
+ * death between the two leaves autoFixOutcome permanently unset, with
9702
+ * autoFixRetries never incremented either, so isExhaustedAutoFix never fires
9703
+ * and the row falls through every existing resolving door forever, re-scanned
9704
+ * by the periodic reverify pass against the same frozen transcript with no
9705
+ * new outcome to observe.
9706
+ *
9707
+ * Job 1218-fo-01 (2026-09-13, findings filed at
9708
+ * session-manager-operations/reviews/2026-09-13-scheduler-stability-investigation.md,
9709
+ * "post-run adjudication" section) sat exactly in this state: needs_review,
9710
+ * verifierVerdict transcript_errors, autoFixAttempted: true, autoFixOutcome:
9711
+ * undefined, autoFixRetries: undefined, statusHistory ending in
9712
+ * "investigation probe exited — restoring prior status" — with a landed
9713
+ * commit no existing ladder rung would credit.
9714
+ *
9715
+ * Deliberately narrower than "unset, 'error', or 'no-plan'": a row that DID
9716
+ * get a durably-stamped 'error'/'no-plan' outcome with its one bounded retry
9717
+ * still unspent (autoFixRetries < 1) is exactly the row
9718
+ * selectAutoFixTargets's own retryEligible check still owns and will retry
9719
+ * on its own — pulling it into THIS ladder instead would race it away from
9720
+ * that retry (scheduler-needs-review-autoresolve.test.cjs's "a non-exhausted
9721
+ * needs_review row … is left alone" guards exactly this). Only the
9722
+ * outcome-truly-never-stamped case is structurally unrecoverable by any
9723
+ * OTHER existing door, because nothing ever wrote a value selectAutoFixTargets
9724
+ * or isExhaustedAutoFix could act on.
9725
+ * Exported for tests.
9726
+ */
9727
+ function isStrandedAutoFixPark(job, jobsInProject) {
9728
+ if (!job || job.status !== 'needs_review') return false;
9729
+ if (job.autoFixAttempted !== true) return false;
9730
+ if (job.autoFixOutcome != null) return false;
9731
+ const fixSlug = fixSlugFor(job);
9732
+ const liveOrQueuedChild = (jobsInProject || []).some(
9733
+ (j) => j.slug === fixSlug && j.status !== 'completed' && !DEAD_FIX_CHILD_STATUSES.has(j.status),
9734
+ );
9735
+ return !liveOrQueuedChild;
9736
+ }
9737
+
8668
9738
  /**
8669
9739
  * Pure predicate, no I/O: is this needs_review row eligible for the bounded
8670
9740
  * auto-resolve ladder at all — either because its auto-fix path is genuinely
8671
- * spent (isExhaustedAutoFix), or because it was parked by a GUARD verdict
8672
- * that never entered auto-fix in the first place (isGuardParkedWithoutAutoFix).
8673
- * Both classes share ONE ladder (applyNeedsReviewAutoResolve) rather than a
8674
- * duplicated one — the ladder itself doesn't care which door a row came
8675
- * through, only whether it now carries completion evidence (job.looksDone).
9741
+ * spent (isExhaustedAutoFix), because it was parked by a GUARD verdict that
9742
+ * never entered auto-fix in the first place (isGuardParkedWithoutAutoFix),
9743
+ * or because its auto-fix investigation ran but was stranded before
9744
+ * recording any outcome (isStrandedAutoFixPark). All three classes share ONE
9745
+ * ladder (applyNeedsReviewAutoResolve) rather than a duplicated one — the
9746
+ * ladder itself doesn't care which door a row came through, only whether it
9747
+ * now carries completion evidence (job.looksDone). `jobsInProject` is only
9748
+ * consulted by isStrandedAutoFixPark (to check for a live/queued fix-plan
9749
+ * child) and defaults to empty so existing single-arg callers are unaffected.
8676
9750
  * Exported for tests.
8677
9751
  */
8678
- function isEligibleForNeedsReviewAutoResolve(job) {
8679
- return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job);
9752
+ function isEligibleForNeedsReviewAutoResolve(job, jobsInProject = []) {
9753
+ return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job) || isStrandedAutoFixPark(job, jobsInProject);
8680
9754
  }
8681
9755
 
8682
9756
  /**
@@ -8789,18 +9863,55 @@ function isFailedUnverifiedShaped(job) {
8789
9863
  if (job.verifierVerdict && RESCANNABLE_VERDICTS.has(job.verifierVerdict)) return true;
8790
9864
  const runId = job.runId || resolveRunId(job);
8791
9865
  if (!runId) return false;
8792
- const logPath = path.join(RUNS_DIR, runId, `${job.slug}.log`);
9866
+ const logPath = path.join(schedulerPaths.runsDir(), runId, `${job.slug}.log`);
8793
9867
  return classifyRunOutcome(logPath) === 'no_result';
8794
9868
  }
8795
9869
 
8796
9870
  function isRescanCandidate(job) {
8797
9871
  if (!job) return false;
9872
+ // Default-ELIGIBLE: every needs_review row is a heal candidate unless its
9873
+ // verdict is in RESCAN_EXCLUDED_VERDICTS. No runId requirement here — a row
9874
+ // without one still gets the evidence rung and the unresolvable annotation.
9875
+ if (job.status === 'needs_review') return !RESCAN_EXCLUDED_VERDICTS.has(job.verifierVerdict);
8798
9876
  if (!(job.runId || resolveRunId(job))) return false;
8799
- if (job.status === 'needs_review') return RESCANNABLE_VERDICTS.has(job.verifierVerdict);
8800
9877
  if (job.status === 'failed') return isFailedUnverifiedShaped(job);
8801
9878
  return false;
8802
9879
  }
8803
9880
 
9881
+ /**
9882
+ * Which rung a needs_review candidate gets (RESCANNABLE_VERDICTS as a hint):
9883
+ * true = transcript re-verification (needs a run dir to read); false = the
9884
+ * evidence-only rung. I/O only when a rescannable-verdict row lacks a runId.
9885
+ */
9886
+ function isTranscriptRescannable(job) {
9887
+ return !!job && RESCANNABLE_VERDICTS.has(job.verifierVerdict) && !!(job.runId || resolveRunId(job));
9888
+ }
9889
+
9890
+ /**
9891
+ * Pure, no I/O: the bounded subset of evidence-only needs_review candidates
9892
+ * reverifyNeedsReview scans this pass. Skips rows already carrying looksDone,
9893
+ * rows scanned within EVIDENCE_SCAN_MIN_INTERVAL_MS (evidenceScannedAt), and —
9894
+ * PRD 1136 — rows with a live auto-fix history unless they are an
9895
+ * auto-resolve door (isEligibleForNeedsReviewAutoResolve, which is what
9896
+ * consumes looksDone). Never-scanned rows go first, then least-recently
9897
+ * scanned; capped at EVIDENCE_SCAN_MAX_PER_PASS. O(n log n) in needs_review rows.
9898
+ * Per-pass cost ceiling: EVIDENCE_SCAN_MAX_PER_PASS computeLooksDone calls.
9899
+ */
9900
+ function selectEvidenceScanTargets(jobs, now = Date.now()) {
9901
+ const due = [];
9902
+ for (const j of jobs ?? []) {
9903
+ if (j.status !== 'needs_review' || !isRescanCandidate(j)) continue;
9904
+ if (isTranscriptRescannable(j)) continue;
9905
+ if (j.looksDone) continue;
9906
+ if (j.autoFixAttempted === true && !isEligibleForNeedsReviewAutoResolve(j, jobs)) continue;
9907
+ const last = Date.parse(j.evidenceScannedAt ?? '');
9908
+ if (!Number.isNaN(last) && now - last < EVIDENCE_SCAN_MIN_INTERVAL_MS) continue;
9909
+ due.push({ j, last: Number.isNaN(last) ? 0 : last });
9910
+ }
9911
+ due.sort((a, b) => a.last - b.last);
9912
+ return due.slice(0, EVIDENCE_SCAN_MAX_PER_PASS).map((d) => d.j);
9913
+ }
9914
+
8804
9915
  /**
8805
9916
  * Cheap-guard for the 10-minute periodic reverify tick. MUST be expressed in
8806
9917
  * terms of isRescanCandidate — not a hand-written status test — because the
@@ -8834,6 +9945,14 @@ function isRescanCandidate(job) {
8834
9945
  * that function). Same rule as always: never let this guard be narrower than
8835
9946
  * the work reverifyNeedsReview actually performs.
8836
9947
  *
9948
+ * Reopened a THIRD time 2026-09-18 (shared_tree_reverted parked 1229-fo-03
9949
+ * falsely, 19 of 20 pending rows held): the fix was not another OR-clause but
9950
+ * inverting the default — isRescanCandidate is now default-ELIGIBLE for every
9951
+ * needs_review row (RESCAN_EXCLUDED_VERDICTS names the few exceptions), so a
9952
+ * new park reason can never again be born unhealable. The OR-clauses below
9953
+ * are now redundant for needs_review rows and kept only for their
9954
+ * non-needs_review inputs.
9955
+ *
8837
9956
  * Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget and
8838
9957
  * isGuardParkedWithoutAutoFix are pure (no I/O). selectAutoFixTargets is
8839
9958
  * called with an injected fixSlugExists that always returns false — cheap
@@ -8845,7 +9964,7 @@ function isRescanCandidate(job) {
8845
9964
  */
8846
9965
  function shouldRunPeriodicReverify(jobs) {
8847
9966
  if (!Array.isArray(jobs)) return false;
8848
- if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j))) return true;
9967
+ if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j) || isStaleSharedTreeRevertedPark(j))) return true;
8849
9968
  if (jobs.some((j) => selectMechanicalRecoveryTarget(j) || selectResumeRecoveryTarget(j))) return true;
8850
9969
  return selectAutoFixTargets(jobs, { fixSlugExists: () => false }).length > 0;
8851
9970
  }
@@ -9012,11 +10131,12 @@ function needsReviewAutoResolveDisabled() {
9012
10131
  * [{ slug, cwd, ageMs, attempts }]
9013
10132
  *
9014
10133
  * Pure selector — no IO. Selects `needs_review` rows eligible for the
9015
- * bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve — either
9016
- * auto-fix genuinely spent, or parked by a GUARD verdict that never entered
9017
- * auto-fix at all), whose newest statusHistory entry with `to ===
9018
- * 'needs_review'` is older than `thresholdMs`, and whose
9019
- * exhaustedResolveAttempts counter has not yet spent its cap.
10134
+ * bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve — auto-fix
10135
+ * genuinely spent, parked by a GUARD verdict that never entered auto-fix at
10136
+ * all, or a stranded auto-fix park with no outcome ever recorded), whose
10137
+ * newest statusHistory entry with `to === 'needs_review'` is older than
10138
+ * `thresholdMs`, and whose exhaustedResolveAttempts counter has not yet
10139
+ * spent its cap.
9020
10140
  *
9021
10141
  * The inclusion bound is inclusive of the cap itself (`<= CAP`, not `<
9022
10142
  * CAP`): NEEDS_REVIEW_RESOLVE_CAP counts REQUEUE attempts already spent, and
@@ -9029,7 +10149,7 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
9029
10149
  const targets = [];
9030
10150
  for (const j of jobs ?? []) {
9031
10151
  if (j.status !== 'needs_review') continue;
9032
- if (!isEligibleForNeedsReviewAutoResolve(j)) continue;
10152
+ if (!isEligibleForNeedsReviewAutoResolve(j, jobs)) continue;
9033
10153
  if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) continue;
9034
10154
  const history = j.statusHistory || [];
9035
10155
  let entry = null;
@@ -9073,8 +10193,8 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
9073
10193
  * reason text (and the Queue UI's job.error) name the RIGHT evidence — a
9074
10194
  * guard-parked row was never "exhausted auto-fix" and must never claim to be.
9075
10195
  */
9076
- function applyNeedsReviewAutoResolve(j) {
9077
- if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j)) return null;
10196
+ function applyNeedsReviewAutoResolve(j, jobsInProject = []) {
10197
+ if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j, jobsInProject)) return null;
9078
10198
  if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) return null;
9079
10199
  const originIsGuardParked = !isExhaustedAutoFix(j) && isGuardParkedWithoutAutoFix(j);
9080
10200
 
@@ -9356,6 +10476,55 @@ async function computeLooksDone(job, fetchedCwds) {
9356
10476
  return { commits: attributed.commits, paths, detectedAt: new Date().toISOString(), rule: attributed.rule };
9357
10477
  }
9358
10478
 
10479
+ /**
10480
+ * Shadow gate (observation only): run a needs_review row's authored gate at
10481
+ * the project's current HEAD and record what it WOULD have decided as
10482
+ * `gateShadow` on the verdicts sidecar and the row. Changes NO status, takes
10483
+ * no slot (not a claude -p run — runGateSequence keeps one shadow gate in
10484
+ * flight machine-wide). Never called from finalize: only the reverify pass.
10485
+ * Returns the recorded gateShadow, or null when nothing was recorded (already
10486
+ * recorded at this HEAD, PRD unreadable, or another shadow gate is running).
10487
+ */
10488
+ async function runGateShadow(job) {
10489
+ if (!job || !job.slug || !job.cwd) return null;
10490
+ const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
10491
+ let prdText;
10492
+ try { prdText = fs.readFileSync(prdPath, 'utf8'); } catch { return null; }
10493
+ const head = await gitHead(job.cwd);
10494
+ if (job.gateShadow && job.gateShadow.head === head) return null;
10495
+ const gate = resolveGate(prdText);
10496
+ let outcome;
10497
+ if (gate.source === 'none') outcome = { status: 'unavailable', reason: 'gate-opt-out', results: [] };
10498
+ else if (!gate.sequence.length) outcome = { status: 'unavailable', reason: 'no-parseable-gate', results: [] };
10499
+ else {
10500
+ const r = await runGateSequence(gate.sequence, { cwd: job.cwd });
10501
+ if (r.status === 'busy') return null;
10502
+ outcome = r;
10503
+ }
10504
+ const gateShadow = { ...outcome, head, source: gate.source, ranAt: new Date().toISOString() };
10505
+ const runId = job.runId || resolveRunId(job);
10506
+ if (runId) {
10507
+ const verdictsPath = path.join(schedulerPaths.runsDir(), runId, `${job.slug}.verdicts.json`);
10508
+ // Read-merge (single-writer law: runVerify owns the sidecar's other keys).
10509
+ // Only merge into an existing run dir — never conjure one.
10510
+ if (fs.existsSync(path.dirname(verdictsPath))) {
10511
+ let existing = {};
10512
+ try { existing = JSON.parse(fs.readFileSync(verdictsPath, 'utf8')) || {}; } catch { /* absent/unparseable → fresh */ }
10513
+ try { atomicWriteJsonSync(verdictsPath, { ...existing, gateShadow }); } catch { /* best-effort */ }
10514
+ }
10515
+ }
10516
+ await mutate((s) => {
10517
+ for (const j of s.jobs) {
10518
+ if (j.slug === job.slug && j.status === 'needs_review') j.gateShadow = gateShadow;
10519
+ }
10520
+ });
10521
+ await broadcast();
10522
+ return gateShadow;
10523
+ }
10524
+
10525
+ // Tail of the last background shadow gate — lets tests (and only tests) await it.
10526
+ let gateShadowPending = null;
10527
+
9359
10528
  async function reverifyNeedsReview() {
9360
10529
  const snap = await readQueue();
9361
10530
  // isGuardParkedWithoutAutoFix rows are NOT isRescanCandidate (their
@@ -9365,7 +10534,7 @@ async function reverifyNeedsReview() {
9365
10534
  // guard-verdict auto-resolve gap this PRD closes. Handled in its own
9366
10535
  // branch below (no transcript rescan — there is no transcript verdict to
9367
10536
  // rescan) rather than through the isRescanCandidate machinery.
9368
- const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j));
10537
+ const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j) || isStaleSharedTreeRevertedPark(j));
9369
10538
  const healed = [];
9370
10539
  const leftForReview = [];
9371
10540
  const looksDoneUpdates = [];
@@ -9373,12 +10542,33 @@ async function reverifyNeedsReview() {
9373
10542
  // the `git fetch --all --prune` per distinct cwd (see computeLooksDone's
9374
10543
  // header) rather than re-fetching the same repo once per candidate row.
9375
10544
  const fetchedCwds = new Set();
10545
+ const evidenceSlugs = new Set(selectEvidenceScanTargets(snap.jobs).map((j) => j.slug));
10546
+ const evidenceScanned = [];
9376
10547
  for (const job of candidates) {
9377
- if (!isRescanCandidate(job) && isGuardParkedWithoutAutoFix(job)) {
9378
- // Guard-verdict park, never auto-fixed: only evidence gathering, never
9379
- // a transcript rescan (there was never a transcript-verifier verdict
9380
- // here) and never a direct heal — applyNeedsReviewAutoResolve is the
9381
- // sole place that turns this annotation into a status change.
10548
+ if (isStaleSharedTreeRevertedPark(job)) {
10549
+ // Re-apply the corrected shared-tree check: the row's own landedCommit
10550
+ // (this dispatch's, per resolveLandedCommitEvidence) still being an
10551
+ // ancestor of HEAD means the park was a false positive — heal it.
10552
+ const cwd = job.cwd || DEFAULT_PROJECT_CWD;
10553
+ if (await resolveLandedCommitEvidence(cwd, job.landedCommit, job.startedAt)
10554
+ && await module.exports.landedCommitIsAncestorOfHead(cwd, job.landedCommit)) {
10555
+ healed.push(job.slug);
10556
+ continue;
10557
+ }
10558
+ if (!isRescanCandidate(job) && !isGuardParkedWithoutAutoFix(job)) {
10559
+ leftForReview.push({ slug: job.slug, reason: 'shared_tree_reverted: landed commit not an ancestor of HEAD' });
10560
+ continue;
10561
+ }
10562
+ }
10563
+ if (job.status === 'needs_review' && !isTranscriptRescannable(job)) {
10564
+ // Any needs_review row whose verdict is not a transcript-verifier one
10565
+ // (a guard verdict, a not-yet-invented verdict, a stranded auto-fix
10566
+ // park): only evidence gathering, never a transcript rescan (verifyRun
10567
+ // would call it clean) and never a direct heal —
10568
+ // applyNeedsReviewAutoResolve is the sole place that turns this
10569
+ // annotation into a status change. Bounded by selectEvidenceScanTargets.
10570
+ if (!evidenceSlugs.has(job.slug)) continue;
10571
+ evidenceScanned.push(job.slug);
9382
10572
  const looksDone = await computeLooksDone(job, fetchedCwds);
9383
10573
  if (looksDone) {
9384
10574
  looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
@@ -9401,7 +10591,7 @@ async function reverifyNeedsReview() {
9401
10591
  }
9402
10592
  continue;
9403
10593
  }
9404
- const runDir = path.join(RUNS_DIR, job.runId || resolveRunId(job));
10594
+ const runDir = path.join(schedulerPaths.runsDir(), job.runId || resolveRunId(job));
9405
10595
  const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
9406
10596
  // Derive committedDuringRun from the recorded run window. The live
9407
10597
  // commit-guard uses gitHead() (before/after HEAD diff); here the run is
@@ -9424,6 +10614,13 @@ async function reverifyNeedsReview() {
9424
10614
  committedDuringRun,
9425
10615
  allowPreSentinelHeal: true,
9426
10616
  priorLandedCommit,
10617
+ // job.landedCommit is THIS row's own last-run attribution (stamped by
10618
+ // spawnJob's finalize, survives resetJobFields) — the same
10619
+ // ground-truth-outranks-heuristics evidence spawnJob passes live,
10620
+ // just read back post-hoc since there is no in-flight guardHeadBefore/
10621
+ // headAtExit pair to recompute for an already-terminal row.
10622
+ jobLandedCommitThisRun: job.landedCommit ?? null,
10623
+ exitCode: job.exitCode ?? null,
9427
10624
  });
9428
10625
  } catch { leftForReview.push({ slug: job.slug, reason: 'verifyRun threw' }); continue; }
9429
10626
  const refusal = healRefusalReason(job, v, committedDuringRun);
@@ -9454,6 +10651,23 @@ async function reverifyNeedsReview() {
9454
10651
  }
9455
10652
  }
9456
10653
  }
10654
+ // Shadow gate (observation only): at most ONE needs_review row per pass,
10655
+ // fired in the background so a 15-minute gate never stalls this pass.
10656
+ if (!gateShadowPending && process.env.SM_GATE_SHADOW_DISABLE !== '1') {
10657
+ const gateTarget = snap.jobs.find((j) => j.status === 'needs_review' && !j.gateShadow);
10658
+ if (gateTarget) {
10659
+ gateShadowPending = runGateShadow(gateTarget)
10660
+ .catch((e) => { console.error('[scheduler] gate shadow error', gateTarget.slug, e); })
10661
+ .finally(() => { gateShadowPending = null; });
10662
+ }
10663
+ }
10664
+ if (evidenceScanned.length) {
10665
+ const scannedSet = new Set(evidenceScanned);
10666
+ const stamp = new Date().toISOString();
10667
+ await mutate((s) => {
10668
+ for (const j of s.jobs) if (scannedSet.has(j.slug) && j.status === 'needs_review') j.evidenceScannedAt = stamp;
10669
+ });
10670
+ }
9457
10671
  if (looksDoneUpdates.length) {
9458
10672
  const bySlug = new Map(looksDoneUpdates.map((u) => [u.slug, u]));
9459
10673
  await mutate((s) => {
@@ -9666,7 +10880,7 @@ async function reverifyNeedsReview() {
9666
10880
  });
9667
10881
  for (const job of targets) {
9668
10882
  const runId = job.runId || resolveRunId(job);
9669
- const runDir = path.join(RUNS_DIR, runId);
10883
+ const runDir = path.join(schedulerPaths.runsDir(), runId);
9670
10884
  const isRetryAttempt = job.autoFixAttempted === true;
9671
10885
  const isDeadFixPlanReopen = isFixPlanDead(job, queueForResumeAndAutofix.jobs);
9672
10886
  const deadChild = isDeadFixPlanReopen
@@ -9813,12 +11027,14 @@ function registerScheduleHandlers() {
9813
11027
  const freeSlots = Math.max(0, slotSnapshot.total - slotSnapshot.inUse);
9814
11028
  const verdict = classifyQueueHealth({
9815
11029
  jobs: state.jobs,
9816
- paused: state.paused,
11030
+ paused: upgradeDrain.effectivePaused(state),
9817
11031
  launchBlocks: state.launchBlocks,
9818
11032
  runningSet,
9819
11033
  freeSlots,
9820
11034
  totalSlots: slotSnapshot.total,
9821
- lastDispatchAttemptAtMs: Date.parse(state.lastDispatchAttemptAt ?? ''),
11035
+ lastRunAtMs: Date.parse(state.lastRunAt ?? ''),
11036
+ lastPauseClearedAtMs: lastPauseClearedAt,
11037
+ schedulerBootedAtMs: Date.parse(SCHEDULER_BOOTED_AT),
9822
11038
  now,
9823
11039
  cwd,
9824
11040
  });
@@ -9949,6 +11165,11 @@ function registerScheduleHandlers() {
9949
11165
  return { ok: true };
9950
11166
  });
9951
11167
 
11168
+ ipcMain.handle('schedule:pause', async () => {
11169
+ await setPaused('manual', null);
11170
+ return { ok: true };
11171
+ });
11172
+
9952
11173
  ipcMain.handle('schedule:resume', async () => {
9953
11174
  await clearPause('manual');
9954
11175
  return { ok: true };
@@ -9980,7 +11201,7 @@ function registerScheduleHandlers() {
9980
11201
  ipcMain.handle('schedule:clear-queue', async () => {
9981
11202
  ensureDirs();
9982
11203
  const ts = new Date().toISOString().replace(/[:.]/g, '-');
9983
- const archiveDir = path.join(PRDS_ARCHIVE_DIR, ts);
11204
+ const archiveDir = path.join(schedulerPaths.scheduledPlansRoot(), 'prds-archived', ts);
9984
11205
  const state = await readQueue();
9985
11206
  const victims = state.jobs.filter((j) => j.status !== 'running');
9986
11207
  if (victims.length === 0) {
@@ -10027,7 +11248,7 @@ function registerScheduleHandlers() {
10027
11248
 
10028
11249
  ipcMain.handle('schedule:open-folder', async () => {
10029
11250
  const { shell } = require('electron');
10030
- await shell.openPath(ROOT);
11251
+ await shell.openPath(schedulerPaths.scheduledPlansRoot());
10031
11252
  return { ok: true };
10032
11253
  });
10033
11254
 
@@ -10045,8 +11266,8 @@ function registerScheduleHandlers() {
10045
11266
  ipcMain.handle('schedule:read-log', validated(schemas.scheduleReadLog, async ({ slug, runId }) => {
10046
11267
  // Defense-in-depth: re-check containment after path.resolve even though
10047
11268
  // SLUG_RE / RUN_ID_RE already forbid path separators.
10048
- const logPath = path.resolve(path.join(RUNS_DIR, runId, `${slug}.log`));
10049
- if (!logPath.startsWith(RUNS_DIR + path.sep)) {
11269
+ const logPath = path.resolve(path.join(schedulerPaths.runsDir(), runId, `${slug}.log`));
11270
+ if (!logPath.startsWith(schedulerPaths.runsDir() + path.sep)) {
10050
11271
  return { ok: false, error: 'invalid slug or runId' };
10051
11272
  }
10052
11273
  try {
@@ -10063,8 +11284,8 @@ function registerScheduleHandlers() {
10063
11284
  // template, authored before the user fills in `cwd`) falls back to the
10064
11285
  // legacy global dir until it's re-saved with a real cwd and migrated by
10065
11286
  // the next reconcile-driven scan.
10066
- const dir = (await findPrdDir(data.slug)) ?? PRDS_DIR;
10067
- if (dir === PRDS_DIR) ensureDirs();
11287
+ const dir = (await findPrdDir(data.slug)) ?? schedulerPaths.prdsRoot();
11288
+ if (dir === schedulerPaths.prdsRoot()) ensureDirs();
10068
11289
  const resolved = safeSlugPathIn(dir, data.slug);
10069
11290
  if (!resolved) return { ok: false, error: 'invalid slug' };
10070
11291
  try {
@@ -10096,6 +11317,15 @@ function registerScheduleHandlers() {
10096
11317
  });
10097
11318
  }
10098
11319
 
11320
+ function stopDispatchLoop() {
11321
+ if (dispatchLoopHandle) { dispatchLoopHandle.stop(); dispatchLoopHandle = null; }
11322
+ }
11323
+
11324
+ /** Shutdown path: stop the timers this module owns. */
11325
+ function stop() {
11326
+ stopDispatchLoop();
11327
+ }
11328
+
10099
11329
  async function init() {
10100
11330
  ensureDirs();
10101
11331
  // Boot phase — reconciliation, migrations, self-heal, first reset probe.
@@ -10111,6 +11341,9 @@ async function init() {
10111
11341
  // A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
10112
11342
  // batch — advance the queue without waiting for the next 60s poll.
10113
11343
  sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
11344
+ // Boot-time expiry pass (process-local state is empty after a restart, so
11345
+ // this is a cheap belt-and-braces run against the freshly read queue).
11346
+ try { runReservationExpiryPass((await readQueue()).jobs); } catch { /* best-effort */ }
10114
11347
  // Retire the global queue.json: split its rows into per-project shards
10115
11348
  // BEFORE the first read below, so boot reconciliation sees the shards.
10116
11349
  try {
@@ -10131,17 +11364,17 @@ async function init() {
10131
11364
  // Boot reconciliation: finalize any job that was 'running' when the app died.
10132
11365
  // Check the run log first — a job that emitted result/success before the crash
10133
11366
  // should be marked 'completed', not 'failed', so it doesn't wedge the queue
10134
- // via the failure-gate. Also kill any still-live orphan claude child to prevent
10135
- // it from continuing to write to the project unsupervised (2026-05-21 incident).
11367
+ // via the failure-gate. A still-live executor is spared, not killed.
10136
11368
  //
10137
11369
  // classifyRunOutcome calls readTail → fs.readFileSync (up to 64 KB per job).
10138
11370
  // Pre-compute all outcomes BEFORE entering the mutate lock so the blocking I/O
10139
11371
  // does not stall the event loop or hold the mutateTail chain during startup.
10140
11372
  //
10141
- // Jobs whose recorded pid is still alive are deferred (not classified here) —
10142
- // see partitionBootOrphans. Everything else (dead pid or no pid) is safe to
11373
+ // Rows proven alive are adopted (left running, never killed) — see
11374
+ // partitionBootOrphans. Everything else is proven dead/exited and is safe to
10143
11375
  // classify immediately below.
10144
11376
  const bootSnap = readQueueSync();
11377
+ const bootLogPath = (j) => (j?.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null);
10145
11378
 
10146
11379
  // Worktree boot reconciliation (PRD 994): a job worktree that survives an
10147
11380
  // app crash/host reboot must not leak disk or a dangling branch forever —
@@ -10156,11 +11389,15 @@ async function init() {
10156
11389
  // itself, proof its run already died. isLive checks the already-read
10157
11390
  // bootSnap (no extra queue read) for a live running-row pid, OR a live
10158
11391
  // /proc cwd holder under the checkout itself. See jobWorktreeBootLive.cjs.
11392
+ // rowPid walks the SAME record → runtime.pid → log-pid ladder
11393
+ // partitionBootOrphans uses, so the sweep and the partition can never
11394
+ // disagree about which executor is alive.
10159
11395
  const isLive = buildJobWorktreeIsLive({
10160
11396
  bootJobs: bootSnap.jobs,
10161
11397
  claudePidAlive,
10162
11398
  hasLiveHolder: gitWorktree.hasLiveHolder,
10163
11399
  cwdHolders: gitWorktree.listCwdHolders(),
11400
+ rowPid: (j) => bootRowPid(j, bootLogPath),
10164
11401
  });
10165
11402
  await jobWorktree.reconcileWorktreesOnBoot([...worktreeCwds], { isLive });
10166
11403
  } catch (e) {
@@ -10180,12 +11417,30 @@ async function init() {
10180
11417
  console.error('[scheduler] boot epic-worktree reconciliation failed', e?.message);
10181
11418
  }
10182
11419
 
10183
- const { immediate: immediateSlugs, deferred: deferredSlugs } = partitionBootOrphans(bootSnap.jobs);
11420
+ const { immediate: immediateSlugs, adopted: adoptedSlugs } = partitionBootOrphans(bootSnap.jobs, {
11421
+ pidAlive: claudePidAlive,
11422
+ getLogPid: (j) => readSpawnedPidFromLog(bootLogPath(j)),
11423
+ getLogMtimeMs: (j) => readLogMtimeMs(bootLogPath(j)),
11424
+ logFreshWindowMs: IDLE_OUTPUT_KILL_MS,
11425
+ findLiveProcess: (j) => findLiveProcessForJob(j, {
11426
+ worktreeDir: jobWorktree.worktreeDirFor(j.cwd || DEFAULT_PROJECT_CWD, j.slug),
11427
+ runCwd: j.runtime?.cwd || j.cwd,
11428
+ }),
11429
+ readRecord: supervisorRecord.readSupervisorRecord,
11430
+ });
10184
11431
  const bootOutcomes = new Map();
10185
11432
  for (const j of bootSnap.jobs) {
10186
11433
  if (!immediateSlugs.includes(j.slug)) continue;
10187
- const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
10188
- bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
11434
+ const logPath = j.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null;
11435
+ let outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
11436
+ // A row whose run already wrote its exit marker (meta.json) is finalized
11437
+ // from that meta when the log tail alone can't say (killed/torn tail).
11438
+ if (outcome === 'unknown' || outcome === 'no_result') {
11439
+ let meta = null;
11440
+ try { meta = j.runId ? JSON.parse(fs.readFileSync(path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.meta.json`), 'utf8')) : null; } catch { /* no/torn meta — keep the log outcome */ }
11441
+ if (meta && typeof meta.exitCode === 'number') outcome = meta.exitCode === 0 ? 'success' : 'failed';
11442
+ }
11443
+ bootOutcomes.set(j.slug, outcome);
10189
11444
  }
10190
11445
  // Same evidence-before-failure gate reapDeadRunningJobs applies, resolved
10191
11446
  // BEFORE mutate() for the same reason (git spawn work must never run
@@ -10215,53 +11470,32 @@ async function init() {
10215
11470
  await archiveCompletedPrd(slug, cwd);
10216
11471
  }
10217
11472
 
10218
- // Still-alive orphans: SIGTERM (+ killOrphanClaudePid's own deferred SIGKILL
10219
- // follow-up) now, but classification waits until BOOT_ORPHAN_KILL_GRACE_MS
10220
- // later — reading the log while the orphan might still be writing to it
10221
- // could misclassify an about-to-succeed run as no_result and double-run the
10222
- // same PRD (2026-05-21 incident this guard exists for).
10223
- for (const slug of deferredSlugs) {
10224
- const j = bootSnap.jobs.find((x) => x.slug === slug);
10225
- const pid = j?.runtime?.pid;
10226
- const bootRunId = j?.runId ?? null; // captured now — guards against reconciling a DIFFERENT later run of the same slug
10227
- if (!pid) continue;
10228
- const result = killOrphanClaudePid(pid);
10229
- const killNote = ` (orphan pid=${pid}: ${result})`;
10230
- if (result === 'killed') {
10231
- console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
10232
- }
10233
- setTimeout(async () => {
10234
- const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
10235
- const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
10236
- // Same evidence-before-failure gate as the immediate-orphan path
10237
- // above, resolved before mutate() for the same reason (git spawn
10238
- // work must never run inside mutate()'s serialization chain). Uses
10239
- // the captured pre-kill snapshot's landedCommit/cwd/startedAt — the
10240
- // race guard below already confirms `cur` is still this same run
10241
- // (runId === bootRunId) before this evidence is applied.
10242
- const confirmedLandedCommit = (outcome !== 'success' && j.landedCommit)
10243
- ? (await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt) ? j.landedCommit : null)
10244
- : null;
10245
- let deferredCompletedCwd;
10246
- mutate((state) => {
10247
- const cur = state.jobs.find((x) => x.slug === slug);
10248
- // Race guard: bail if the job already resolved, OR if it's already been
10249
- // re-picked into a NEW run (different runId) within the grace window —
10250
- // that new run is not the boot orphan we SIGTERM'd and must not be
10251
- // touched by this stale classification.
10252
- if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
10253
- applyOrphanOutcome(cur, outcome, killNote, confirmedLandedCommit);
10254
- console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
10255
- deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
10256
- }).then(() => {
10257
- if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
10258
- }).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
10259
- }, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
11473
+ // Proven-alive rows stay `running` and are never signalled (the boot worktree
11474
+ // sweep above already spares their checkout). No sessionSlots token is
11475
+ // acquired: pickNextBatch's untrackedRunning correction counts the row
11476
+ // against the pool, and reapDeadRunningJobs finalizes it on exit.
11477
+ if (adoptedSlugs.length) {
11478
+ const adoptedAtBoot = new Date().toISOString();
11479
+ const adoptedRunIds = new Map(bootSnap.jobs.filter((j) => adoptedSlugs.includes(j.slug)).map((j) => [j.slug, j.runId ?? null]));
11480
+ await mutate((state) => {
11481
+ for (const j of state.jobs) {
11482
+ // runId guard: never stamp a DIFFERENT later run of the same slug.
11483
+ if (j.status !== 'running' || !adoptedRunIds.has(j.slug) || (j.runId ?? null) !== adoptedRunIds.get(j.slug)) continue;
11484
+ j.adoptedAtBoot = adoptedAtBoot;
11485
+ delete j.supervisedAt; // a prior process's supervisor died with it
11486
+ console.log(`[scheduler] boot: adopted live executor for ${j.slug} (pid=${j.runtime?.pid ?? 'unknown'}) — left running, no signal`);
11487
+ }
11488
+ });
10260
11489
  }
10261
11490
 
11491
+ // Re-arm budget/idle/deadman + the quietMachine lease for the adopted rows
11492
+ // (a dispatch-loop pass repeats this for any row left without a supervisor).
11493
+ await superviseAdoptedRunsPass();
11494
+
10262
11495
  // If we boot up while paused with a resumeAt in the past, clear it. This
10263
11496
  // happens when the app was closed across the reset window.
10264
11497
  const boot = await readQueue();
11498
+ await clearStaleDrainAtBoot(boot);
10265
11499
  if (boot.paused && boot.paused.resumeAt && new Date(boot.paused.resumeAt).getTime() <= Date.now()) {
10266
11500
  await clearPause('boot-elapsed');
10267
11501
  } else if (boot.paused && boot.paused.resumeAt) {
@@ -10483,7 +11717,7 @@ async function init() {
10483
11717
  }
10484
11718
  for (const target of exhaustedNeedsReviewTargets) {
10485
11719
  const j = ms.jobs.find((x) => x.slug === target.slug);
10486
- const outcome = applyNeedsReviewAutoResolve(j);
11720
+ const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
10487
11721
  if (outcome) {
10488
11722
  console.warn(
10489
11723
  `[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
@@ -10504,7 +11738,7 @@ async function init() {
10504
11738
  }
10505
11739
  }).catch(() => {});
10506
11740
  }
10507
- }, 10 * 60_000);
11741
+ }, REVERIFY_INTERVAL_MS);
10508
11742
 
10509
11743
  // Self-rescheduling poll loop with exponential backoff. Replaces the
10510
11744
  // old fixed-interval pollTimer + initialPollTimeout.
@@ -10525,87 +11759,18 @@ async function init() {
10525
11759
  // setInterval callback is sync; readQueueSync stays sync to avoid awaiting
10526
11760
  // inside the timer body (and the 60s cadence makes the cost moot).
10527
11761
  if (heartbeatInterval) clearInterval(heartbeatInterval);
10528
- heartbeatInterval = setInterval(() => {
10529
- const s = readQueueSync();
10530
- // NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
10531
- // running, something must drive it. This is the only driver that does
10532
- // not depend on the billing poll loop, a pause timer, or a completing
10533
- // job to schedule the next tick — every one of which has failed at
10534
- // least once. See classifyQueueStarvation.
10535
- if (!s.unreadable) {
10536
- runQueueStarvationWatchdog(s).catch((e) => console.error('[scheduler] starvation watchdog error', e));
10537
- }
10538
- // Initialise from the real status union (scheduleJobSchema.cjs) rather
10539
- // than a hand-maintained subset — the old `{ pending, running, completed,
10540
- // failed }` literal silently minted a NEW key for any other value
10541
- // (`counts[j.status] = (counts[j.status]||0)+1`), which is exactly how a
10542
- // heartbeat with a `queued: 2` bucket looked like "normal" 24h
10543
- // visibility instead of the alarm it should have been. Any row whose
10544
- // status isn't in JOB_STATUSES (shouldn't happen post-quarantine, but
10545
- // this is the last line of defence) routes into `unknown`, never a
10546
- // freshly-minted key.
10547
- const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
10548
- counts.unknown = 0;
10549
- for (const j of s.jobs) {
10550
- if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
10551
- counts[j.status] += 1;
10552
- } else {
10553
- counts.unknown += 1;
10554
- }
10555
- }
10556
-
10557
- const stall = computeStallSummary(s);
10558
- // Per-project alerting (see computeStallSummary's header): a project
10559
- // stalled while others are busy must still fire, and one project
10560
- // recovering must not clear or suppress another's still-open episode —
10561
- // that is exactly what a single module-level stallSince/stallToasted
10562
- // flag masked before (the burrow-vs-others incident this PRD fixes).
10563
- const now = Date.now();
10564
- const stalledCwds = Object.keys(stall.byProject).filter((cwd) => stall.byProject[cwd].stalled);
10565
- for (const cwd of [...stallSince.keys()]) {
10566
- if (!stalledCwds.includes(cwd)) {
10567
- stallSince.delete(cwd);
10568
- stallToasted.delete(cwd);
10569
- }
10570
- }
10571
- const toAlert = [];
10572
- for (const cwd of stalledCwds) {
10573
- if (!stallSince.has(cwd)) stallSince.set(cwd, now);
10574
- if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
10575
- stallToasted.set(cwd, true);
10576
- toAlert.push(cwd);
10577
- }
10578
- }
10579
- if (toAlert.length > 0) {
10580
- console.error(
10581
- `[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
10582
- + `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
10583
- stall.byProject,
10584
- );
10585
- appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: stall.total, byProject: stall.byProject });
10586
- if (mainWindow && !mainWindow.isDestroyed()) {
10587
- sendIfAlive(mainWindow, 'schedule:stall', {
10588
- message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
10589
- projects: toAlert,
10590
- total: stall.total,
10591
- byProject: stall.byProject,
10592
- });
10593
- }
10594
- }
10595
-
10596
- appendHeartbeat({
10597
- ts: Date.now(),
10598
- pid: process.pid,
10599
- counts,
10600
- stall: { stalled: stall.stalled, total: stall.total },
10601
- paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
10602
- nextReset: cachedNextReset,
10603
- utilization: cachedUtilization,
10604
- consecutiveFailures,
10605
- });
10606
- }, 60_000);
11762
+ heartbeatInterval = setInterval(() => heartbeatTick(), 60_000);
10607
11763
  if (heartbeatInterval.unref) heartbeatInterval.unref();
10608
11764
 
11765
+ // Dispatch's own periodic driver: cadence is independent of pollLoop's billing
11766
+ // backoff. A loop tick meeting a cancelled cancelToken returns 'cancelled' from
11767
+ // tickBody; clearing stays with the starvation watchdog's existing force-clear.
11768
+ stopDispatchLoop();
11769
+ dispatchLoopHandle = startDispatchLoop({
11770
+ tick: () => tickQueue(),
11771
+ onError: (e) => console.warn('[scheduler] dispatch loop tick failed', e?.message),
11772
+ });
11773
+
10609
11774
  // Wake-from-sleep: immediately re-poll and re-evaluate the queue.
10610
11775
  try {
10611
11776
  const { powerMonitor } = require('electron');
@@ -10828,8 +11993,8 @@ const remote = {
10828
11993
  }
10829
11994
  await fsp.mkdir(dir, { recursive: true });
10830
11995
  } else {
10831
- dir = (await findPrdDir(slug)) ?? PRDS_DIR;
10832
- if (dir === PRDS_DIR) ensureDirs();
11996
+ dir = (await findPrdDir(slug)) ?? schedulerPaths.prdsRoot();
11997
+ if (dir === schedulerPaths.prdsRoot()) ensureDirs();
10833
11998
  }
10834
11999
 
10835
12000
  // writePrd only ever JOINS an existing Epic now (no mintAuthority
@@ -10866,6 +12031,19 @@ const remote = {
10866
12031
  }
10867
12032
  },
10868
12033
 
12034
+ // User-initiated pause/resume — the admin-route/MCP twins of the
12035
+ // schedule:pause / schedule:resume IPC handlers, through the same setPaused /
12036
+ // clearPause. Pause stops NEW dispatch only; running jobs are never touched.
12037
+ async pause() {
12038
+ await setPaused('manual', null);
12039
+ return { ok: true };
12040
+ },
12041
+
12042
+ async resume() {
12043
+ await clearPause('manual');
12044
+ return { ok: true };
12045
+ },
12046
+
10869
12047
  async resetJob(slug, opts = {}) {
10870
12048
  const resolved = await resolveSlugOrReason(slug, opts.cwd);
10871
12049
  if (!resolved.ok) {
@@ -11193,6 +12371,14 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
11193
12371
  sendJson(res, 200, jobs);
11194
12372
  });
11195
12373
 
12374
+ adminHttp.registerRoute('POST', '/admin/scheduler/pause', async (req, res) => {
12375
+ sendJson(res, 200, await remoteObj.pause());
12376
+ });
12377
+
12378
+ adminHttp.registerRoute('POST', '/admin/scheduler/resume', async (req, res) => {
12379
+ sendJson(res, 200, await remoteObj.resume());
12380
+ });
12381
+
11196
12382
  adminHttp.registerRoute('POST', '/admin/scheduler/reset-job', async (req, res) => {
11197
12383
  const raw = await readBody(req);
11198
12384
  let parsed;
@@ -11217,6 +12403,8 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
11217
12403
  module.exports = {
11218
12404
  classifyQueueStarvation,
11219
12405
  classifyQueueStarvationByProject,
12406
+ dispatchIdleMs,
12407
+ launchBlockedSlugs,
11220
12408
  classifyQueueHealth,
11221
12409
  runQueueStarvationWatchdog,
11222
12410
  QUEUE_STARVATION_MS,
@@ -11240,13 +12428,13 @@ module.exports = {
11240
12428
  registerScheduleHandlers,
11241
12429
  attachWindow,
11242
12430
  init,
11243
- ROOT,
11244
- PRDS_DIR,
11245
- SCHEDULER_STATE_PATH,
11246
12431
  BACKOFF_MAX_MS,
11247
12432
  FAILURE_STREAK_WARN_THRESHOLD,
12433
+ FAILURE_STREAK_ESCALATION_MS,
11248
12434
  nextBackoffMs,
11249
12435
  shouldWarnFailureStreak,
12436
+ shouldEscalateFailureStreak,
12437
+ computeDegradedBudget,
11250
12438
  healRefusalReason,
11251
12439
  writeQueue,
11252
12440
  reconcile,
@@ -11269,6 +12457,8 @@ module.exports = {
11269
12457
  memoryLimitedBatchSize,
11270
12458
  availableForJobs,
11271
12459
  reverifyNeedsReview,
12460
+ runGateShadow,
12461
+ awaitGateShadowIdle: async () => { while (gateShadowPending) await gateShadowPending; },
11272
12462
  shouldRunPeriodicReverify,
11273
12463
  findStuckFailedJobs,
11274
12464
  STUCK_FAILED_ESCALATE_MS,
@@ -11283,6 +12473,13 @@ module.exports = {
11283
12473
  NEEDS_REVIEW_RESOLVE_MS,
11284
12474
  needsReviewAutoResolveDisabled,
11285
12475
  isRescanCandidate,
12476
+ isTranscriptRescannable,
12477
+ selectEvidenceScanTargets,
12478
+ RESCAN_EXCLUDED_VERDICTS,
12479
+ RESCANNABLE_VERDICTS,
12480
+ EVIDENCE_SCAN_MAX_PER_PASS,
12481
+ EVIDENCE_SCAN_MIN_INTERVAL_MS,
12482
+ REVERIFY_INTERVAL_MS,
11286
12483
  isFailedUnverifiedShaped,
11287
12484
  computeLooksDone,
11288
12485
  attributeLandedCommits,
@@ -11295,6 +12492,9 @@ module.exports = {
11295
12492
  isExhaustedAutoFix,
11296
12493
  GUARD_VERDICT_EVIDENCE_ELIGIBLE,
11297
12494
  isGuardParkedWithoutAutoFix,
12495
+ isStrandedAutoFixPark,
12496
+ isStaleSharedTreeRevertedPark,
12497
+ landedCommitIsAncestorOfHead,
11298
12498
  isEligibleForNeedsReviewAutoResolve,
11299
12499
  isPlanUnqueued,
11300
12500
  isFixPlanDead,
@@ -11334,7 +12534,6 @@ module.exports = {
11334
12534
  buildScheduleStatePayload,
11335
12535
  partitionBootOrphans,
11336
12536
  applyOrphanOutcome,
11337
- BOOT_ORPHAN_KILL_GRACE_MS,
11338
12537
  registerAdminRoutes,
11339
12538
  notifyOriginatingTab,
11340
12539
  notifyNeedsReview,
@@ -11360,10 +12559,13 @@ module.exports = {
11360
12559
  SCHEDULER_CODE_SHA,
11361
12560
  resetJobFields,
11362
12561
  executeJob,
12562
+ killOrphanClaudePid,
11363
12563
  prdArchivedSkipResult,
11364
12564
  spawnJob,
11365
12565
  listPrdsInternal,
11366
12566
  computeStallSummary,
12567
+ heartbeatTick,
12568
+ appendHeartbeat,
11367
12569
  findStaleQuarantinedJobs,
11368
12570
  QUARANTINE_ESCALATE_MS,
11369
12571
  selectQuarantineAutoResolveTargets,
@@ -11402,6 +12604,10 @@ module.exports = {
11402
12604
  setPaused,
11403
12605
  clearPause,
11404
12606
  tickQueue,
12607
+ setRestartHandler,
12608
+ driveUpgradeDrain,
12609
+ clearStaleDrainAtBoot,
12610
+ stop,
11405
12611
  runDueJobs,
11406
12612
  pollLoop,
11407
12613
  maybeLaunchWhenAvailable,
@@ -11410,12 +12616,23 @@ module.exports = {
11410
12616
  CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
11411
12617
  RAPID_RATE_LIMIT_WINDOW_MS,
11412
12618
  MANUAL_PAUSE_COOLDOWN_MS,
11413
- RUNS_DIR,
11414
12619
  pickRunDir,
11415
12620
  resolveRateLimitPauseReset,
12621
+ billingResetForPause,
11416
12622
  computeEffectiveResumeAt,
11417
12623
  computeResumeDelay,
11418
12624
  FOREIGN_WIP_BLOCK_STREAK_LIMIT,
11419
12625
  validateForeignWipBlockClaim,
11420
12626
  requeueForeignWipBlockedJobs,
11421
12627
  };
12628
+
12629
+ // Lazy path getters: resolved from SM_SCHEDULER_HOME at each read, never frozen
12630
+ // at require time (see lib/schedulerPaths.cjs).
12631
+ for (const [name, resolve] of [
12632
+ ['ROOT', schedulerPaths.scheduledPlansRoot],
12633
+ ['PRDS_DIR', schedulerPaths.prdsRoot],
12634
+ ['RUNS_DIR', schedulerPaths.runsDir],
12635
+ ['SCHEDULER_STATE_PATH', schedulerPaths.schedulerStatePath],
12636
+ ]) {
12637
+ Object.defineProperty(module.exports, name, { get: resolve, enumerable: true });
12638
+ }