claude-code-session-manager 0.97.0 → 0.100.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (430) hide show
  1. package/README.md +3 -3
  2. package/bin/cli.cjs +16 -4
  3. package/bin/node-floor.cjs +41 -0
  4. package/dist/assets/{DataModel-D93FoaOm.js → DataModel-HdR0AxFq.js} +1 -1
  5. package/dist/assets/{History-CXxxqNP1.js → History-Dr4Jr_5z.js} +2 -2
  6. package/dist/assets/{Hooks-CzeYFQjq.js → Hooks-NhrvH3JP.js} +3 -3
  7. package/dist/assets/{Library-D2zPkwOC.js → Library-Dv_EtIMr.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-DbinP78A.js → MarkdownEditor-DTKDWV_n.js} +1 -1
  9. package/dist/assets/{McpServers-Bu_sVGLr.js → McpServers-M034G9B3.js} +2 -2
  10. package/dist/assets/{Memory-BgiI6LZ_.js → Memory-DzF1B4gx.js} +6 -6
  11. package/dist/assets/{Permissions-C1MMzZRm.js → Permissions-DrgbSGWl.js} +3 -3
  12. package/dist/assets/{Plugins-Dhu6wAZi.js → Plugins-Bd3ef35w.js} +2 -2
  13. package/dist/assets/{ProvenanceBadge-C3AcWKMG.js → ProvenanceBadge-DcBHT1Iy.js} +1 -1
  14. package/dist/assets/{SaveBar-plwYm0iO.js → SaveBar-rimJem-Y.js} +1 -1
  15. package/dist/assets/Scheduler-DjgFJl2w.js +16 -0
  16. package/dist/assets/{ScopeSwitcher-DFy0vv_o.js → ScopeSwitcher-CnAzTBrl.js} +1 -1
  17. package/dist/assets/Settings-Dc1F_sSg.js +3 -0
  18. package/dist/assets/{SkillReferenceGraph-DsI7hRPB.js → SkillReferenceGraph-B9aZDecM.js} +1 -1
  19. package/dist/assets/{Skills-3E748UfB.js → Skills-DRXd18gu.js} +2 -2
  20. package/dist/assets/{SystemPrompt-BqmfUJOd.js → SystemPrompt-ahy5-Fwe.js} +1 -1
  21. package/dist/assets/{TagLibrary-CPz6fO8Q.js → TagLibrary-C_tToDeL.js} +1 -1
  22. package/dist/assets/{TiptapBody-D9EoBQ1P.js → TiptapBody-DvdefS6K.js} +1 -1
  23. package/dist/assets/{Toggle-BHAmZNDn.js → Toggle-6Jl5Njc6.js} +1 -1
  24. package/dist/assets/{index-CKH5Uxik.css → index-Bta-hwud.css} +1 -1
  25. package/dist/assets/{index-tn5JIVPj.js → index-vx8O73l8.js} +497 -502
  26. package/dist/assets/{settingsSchema-Crqy_aNz.js → settingsSchema-D6iRguNV.js} +1 -1
  27. package/dist/index.html +2 -2
  28. package/package.json +26 -29
  29. package/plugins/CLAUDE.md +6 -6
  30. package/plugins/session-manager-dev/skills/builder/4-manual/SKILL.md +23 -19
  31. package/plugins/session-manager-dev/skills/builder/SKILL.md +3 -3
  32. package/plugins/session-manager-dev/skills/develop/SKILL.md +254 -510
  33. package/plugins/session-manager-dev/skills/develop/standards.md +17 -22
  34. package/scripts/README.md +3 -11
  35. package/scripts/audit-ops-hygiene.cjs +3 -3
  36. package/scripts/hooks/lib/guard-prd-writes-policy.cjs +1 -1
  37. package/scripts/hooks/lib/guard-self-schedule-policy.cjs +1 -1
  38. package/scripts/mint-epic.cjs +2 -3
  39. package/scripts/scheduler-mcp-server.cjs +36 -7
  40. package/src/main/agentLibrary.cjs +22 -2
  41. package/src/main/build-info.json +4 -4
  42. package/src/main/config.cjs +41 -38
  43. package/src/main/docEdit.cjs +4 -1
  44. package/src/main/files.cjs +2 -5
  45. package/src/main/health.cjs +1 -1
  46. package/src/main/index.cjs +23 -10
  47. package/src/main/ipcSchemas.cjs +28 -35
  48. package/src/main/lib/agentPersonaSchema.cjs +13 -2
  49. package/src/main/lib/atomicFs.cjs +116 -0
  50. package/src/main/lib/branchSweep.cjs +13 -12
  51. package/src/main/lib/buildTarget.cjs +3 -3
  52. package/src/main/lib/claudeCliCaps.cjs +129 -0
  53. package/src/main/lib/credentials.cjs +2 -4
  54. package/src/main/lib/crossProjectFeedback.cjs +3 -3
  55. package/src/main/lib/cwdClassify.cjs +15 -3
  56. package/src/main/lib/definitionOfDone.cjs +207 -125
  57. package/src/main/lib/dodDrainHook.cjs +11 -5
  58. package/src/main/lib/epicMint.cjs +2 -4
  59. package/src/main/lib/epicStatusMirror.cjs +8 -10
  60. package/src/main/lib/epicWorktreeProjectConfig.cjs +2 -4
  61. package/src/main/lib/gateAuthority.cjs +96 -0
  62. package/src/main/lib/gitExec.cjs +69 -0
  63. package/src/main/lib/gitWorktree.cjs +656 -92
  64. package/src/main/lib/instanceLock.cjs +5 -8
  65. package/src/main/lib/launchFailure.cjs +2 -3
  66. package/src/main/lib/macroLibrary.cjs +386 -0
  67. package/src/main/lib/mcpToolCatalog.cjs +26 -17
  68. package/src/main/lib/needsReviewLedger.cjs +1 -1
  69. package/src/main/lib/opsErrorLog.cjs +11 -1
  70. package/src/main/lib/opsOwnership.cjs +0 -5
  71. package/src/main/lib/pidAlive.cjs +32 -0
  72. package/src/main/lib/prdCreate.cjs +153 -19
  73. package/src/main/lib/prdDisposition.cjs +2 -2
  74. package/src/main/lib/prdGateFiles.cjs +284 -0
  75. package/src/main/lib/prdLocations.cjs +61 -0
  76. package/src/main/lib/prdMigration.cjs +2 -3
  77. package/src/main/lib/prdSizing.cjs +2 -1
  78. package/src/main/lib/promptSessionSchema.cjs +10 -2
  79. package/src/main/lib/rcaReport.cjs +4 -6
  80. package/src/main/lib/reaperHelpers.cjs +8 -1
  81. package/src/main/lib/reservationExpiry.cjs +6 -1
  82. package/src/main/lib/reviewNotice.cjs +263 -0
  83. package/src/main/lib/runClaudeP.cjs +4 -1
  84. package/src/main/lib/schedulerPaths.cjs +22 -2
  85. package/src/main/lib/sessionSlots.cjs +2 -4
  86. package/src/main/lib/shippedPersonaSeeds.cjs +179 -0
  87. package/src/main/lib/timeoutShim.cjs +123 -0
  88. package/src/main/lib/timeoutShimScript.cjs +275 -0
  89. package/src/main/lib/upgradeDrain.cjs +4 -10
  90. package/src/main/lib/watchdogHelpers.cjs +33 -30
  91. package/src/main/lib/workTypeLibrary.cjs +7 -1
  92. package/src/main/promptSessionEvents.cjs +37 -13
  93. package/src/main/queueOps.cjs +17 -6
  94. package/src/main/scheduler.cjs +975 -203
  95. package/src/main/seedAgentPersonas.cjs +278 -23
  96. package/src/main/supervisor.cjs +4 -1
  97. package/src/main/templates/PRD_AUTHORING.md +126 -406
  98. package/src/preload/__tests__/preload-surface.test.cjs +68 -0
  99. package/src/preload/api.d.ts +41 -94
  100. package/src/preload/index.cjs +14 -9
  101. package/src/seed/agents/architect.md +1 -0
  102. package/src/seed/agents/dev-lead.md +17 -18
  103. package/src/seed/agents/project-home-builder.md +1 -0
  104. package/src/seed/agents/validator.md +10 -4
  105. package/dist/assets/HostBilko-nESrTRIg.js +0 -1
  106. package/dist/assets/Scheduler-BVYTeh39.js +0 -16
  107. package/dist/assets/Settings-DIhGgdfs.js +0 -3
  108. package/src/main/__tests__/activeIndexMerge.test.cjs +0 -235
  109. package/src/main/__tests__/agentEffortResolve.test.cjs +0 -117
  110. package/src/main/__tests__/agentLibrary.test.cjs +0 -276
  111. package/src/main/__tests__/agentModelResolve.test.cjs +0 -311
  112. package/src/main/__tests__/agentOverlayWrite.test.cjs +0 -95
  113. package/src/main/__tests__/bilkoHost-deriveSlug.test.cjs +0 -26
  114. package/src/main/__tests__/bilkoHost-integration.test.cjs +0 -118
  115. package/src/main/__tests__/bilkoHostCore.test.cjs +0 -72
  116. package/src/main/__tests__/broadcastCoalescer.test.cjs +0 -122
  117. package/src/main/__tests__/chat-cancel-terminal.test.cjs +0 -120
  118. package/src/main/__tests__/chat-dead-channels.test.cjs +0 -63
  119. package/src/main/__tests__/chat-exit-close-race.test.cjs +0 -146
  120. package/src/main/__tests__/chat-mcp-consent-notice.test.cjs +0 -139
  121. package/src/main/__tests__/chat-preamble-anchors.test.cjs +0 -100
  122. package/src/main/__tests__/chat-queue.test.cjs +0 -97
  123. package/src/main/__tests__/chat-stop-signal.test.cjs +0 -89
  124. package/src/main/__tests__/chatRunner-epic-worktree-execcwd.test.cjs +0 -125
  125. package/src/main/__tests__/chatRunner-session-flag-retry.test.cjs +0 -252
  126. package/src/main/__tests__/classifyPromptTicket.test.cjs +0 -101
  127. package/src/main/__tests__/classifyTranscriptLine.test.cjs +0 -201
  128. package/src/main/__tests__/computeDepHistorySatisfaction.test.cjs +0 -66
  129. package/src/main/__tests__/config-readText-bounded.test.cjs +0 -84
  130. package/src/main/__tests__/configWriteBoundaryOwners.test.cjs +0 -59
  131. package/src/main/__tests__/crossProjectFeedback.test.cjs +0 -334
  132. package/src/main/__tests__/crossProjectFeedbackRoutes.test.cjs +0 -161
  133. package/src/main/__tests__/dep-orphan-archive-health.test.cjs +0 -77
  134. package/src/main/__tests__/develop-skill-failure-modes.test.cjs +0 -70
  135. package/src/main/__tests__/docEdit.test.cjs +0 -244
  136. package/src/main/__tests__/dod-batchkey.test.cjs +0 -183
  137. package/src/main/__tests__/dod-drain-hook.test.cjs +0 -302
  138. package/src/main/__tests__/dod-report.test.cjs +0 -304
  139. package/src/main/__tests__/dod-reverify.test.cjs +0 -285
  140. package/src/main/__tests__/epicContextDigest.test.cjs +0 -174
  141. package/src/main/__tests__/epicMint.test.cjs +0 -332
  142. package/src/main/__tests__/epicMintTelemetryTap.test.cjs +0 -64
  143. package/src/main/__tests__/epicStatusMirror.test.cjs +0 -110
  144. package/src/main/__tests__/epicValidationHook.test.cjs +0 -291
  145. package/src/main/__tests__/exchanges.test.cjs +0 -122
  146. package/src/main/__tests__/exchangesPromptId.test.cjs +0 -61
  147. package/src/main/__tests__/extractJson.test.cjs +0 -51
  148. package/src/main/__tests__/files-reject-credentials.test.cjs +0 -40
  149. package/src/main/__tests__/fixtures/1218-fo-01-move-scripts-lib-into-src-main-lib.log +0 -556
  150. package/src/main/__tests__/flatPrdTickSweep.test.cjs +0 -110
  151. package/src/main/__tests__/health-build-freshness.test.cjs +0 -39
  152. package/src/main/__tests__/health-claude-md-budget.test.cjs +0 -57
  153. package/src/main/__tests__/health-credentials.test.cjs +0 -81
  154. package/src/main/__tests__/health-delegation-chain.test.cjs +0 -124
  155. package/src/main/__tests__/health-per-project-stall.test.cjs +0 -84
  156. package/src/main/__tests__/health-prd-migration.test.cjs +0 -37
  157. package/src/main/__tests__/health-queue-dispatch.test.cjs +0 -135
  158. package/src/main/__tests__/health-starve-escalation.test.cjs +0 -94
  159. package/src/main/__tests__/health-tick-liveness.test.cjs +0 -179
  160. package/src/main/__tests__/health-usage-poller.test.cjs +0 -144
  161. package/src/main/__tests__/health-worktree-cap-blocked.test.cjs +0 -65
  162. package/src/main/__tests__/heapSnapshot.test.cjs +0 -121
  163. package/src/main/__tests__/historyAggregatorIntraday.test.cjs +0 -313
  164. package/src/main/__tests__/historyDashboard.test.cjs +0 -163
  165. package/src/main/__tests__/historyRollup.test.cjs +0 -333
  166. package/src/main/__tests__/intradayRefresh.test.cjs +0 -39
  167. package/src/main/__tests__/ipcSchemas-dependsOn.test.cjs +0 -39
  168. package/src/main/__tests__/kg-augment.test.cjs +0 -195
  169. package/src/main/__tests__/loadGateDetailTick.test.cjs +0 -31
  170. package/src/main/__tests__/machineProfile.test.cjs +0 -152
  171. package/src/main/__tests__/mcpStatus.test.cjs +0 -61
  172. package/src/main/__tests__/memoryAggregate.test.cjs +0 -109
  173. package/src/main/__tests__/memoryStale.test.cjs +0 -88
  174. package/src/main/__tests__/needsReviewLedger.test.cjs +0 -162
  175. package/src/main/__tests__/openExternalApp-spawn-error.test.cjs +0 -25
  176. package/src/main/__tests__/opsErrorLog.test.cjs +0 -109
  177. package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +0 -173
  178. package/src/main/__tests__/personaMerge.test.cjs +0 -169
  179. package/src/main/__tests__/planValidator.test.cjs +0 -125
  180. package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +0 -176
  181. package/src/main/__tests__/prd-group-allocator.test.cjs +0 -119
  182. package/src/main/__tests__/prdAdminRouteParity.test.cjs +0 -70
  183. package/src/main/__tests__/prdAdminRoutes.test.cjs +0 -718
  184. package/src/main/__tests__/prdAgentType.test.cjs +0 -103
  185. package/src/main/__tests__/prdAuthoringSeed.test.cjs +0 -39
  186. package/src/main/__tests__/prdCreate.test.cjs +0 -979
  187. package/src/main/__tests__/prdCreateDisposition.test.cjs +0 -201
  188. package/src/main/__tests__/prdCreatePlanId.test.cjs +0 -132
  189. package/src/main/__tests__/prdFrontmatterAgentType.test.cjs +0 -117
  190. package/src/main/__tests__/prdFrontmatterDependsOn.test.cjs +0 -136
  191. package/src/main/__tests__/prdFrontmatterDisposition.test.cjs +0 -125
  192. package/src/main/__tests__/prdFrontmatterQuietMachine.test.cjs +0 -108
  193. package/src/main/__tests__/prdLocations.test.cjs +0 -195
  194. package/src/main/__tests__/prdLocationsArchived.test.cjs +0 -201
  195. package/src/main/__tests__/prdMigration.test.cjs +0 -349
  196. package/src/main/__tests__/prdMigrationLegacyAdopt.test.cjs +0 -91
  197. package/src/main/__tests__/prdParserHighWater.test.cjs +0 -74
  198. package/src/main/__tests__/prdParserSourcePromptId.test.cjs +0 -65
  199. package/src/main/__tests__/prdSetDisposition.test.cjs +0 -222
  200. package/src/main/__tests__/prdSizing.test.cjs +0 -106
  201. package/src/main/__tests__/prdSourcePromptIdBackfill.test.cjs +0 -118
  202. package/src/main/__tests__/prdUpdateDependsOn.test.cjs +0 -160
  203. package/src/main/__tests__/proc-role-env.test.cjs +0 -125
  204. package/src/main/__tests__/procname-claude-spawn-sites.test.cjs +0 -304
  205. package/src/main/__tests__/procname-sm-processes.test.cjs +0 -127
  206. package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +0 -177
  207. package/src/main/__tests__/projectPages.test.cjs +0 -137
  208. package/src/main/__tests__/promptSessionEvents.test.cjs +0 -234
  209. package/src/main/__tests__/promptSessionSchema.test.cjs +0 -101
  210. package/src/main/__tests__/promptSessionTranscript.test.cjs +0 -0
  211. package/src/main/__tests__/promptSessionsCreateEpicHandler.test.cjs +0 -159
  212. package/src/main/__tests__/pty-epic-worktree-spawn-cwd.test.cjs +0 -282
  213. package/src/main/__tests__/pty-session-open-telemetry.test.cjs +0 -96
  214. package/src/main/__tests__/pty-write-result.test.cjs +0 -46
  215. package/src/main/__tests__/queue-health-verdict.test.cjs +0 -170
  216. package/src/main/__tests__/queue-starvation-dispatch-driver.test.cjs +0 -286
  217. package/src/main/__tests__/queue-starvation-per-project.test.cjs +0 -147
  218. package/src/main/__tests__/queueHistory.test.cjs +0 -355
  219. package/src/main/__tests__/queueOps-interactive-ac-lint.test.cjs +0 -65
  220. package/src/main/__tests__/queueOpsArchiveDestination.test.cjs +0 -65
  221. package/src/main/__tests__/queueOpsAutoArchive.test.cjs +0 -154
  222. package/src/main/__tests__/rateLimitPollerStreak.test.cjs +0 -128
  223. package/src/main/__tests__/rcaReport.test.cjs +0 -266
  224. package/src/main/__tests__/reconcileFlatPrdSweep.test.cjs +0 -119
  225. package/src/main/__tests__/reconcileTiming.test.cjs +0 -135
  226. package/src/main/__tests__/runLogRetention.test.cjs +0 -489
  227. package/src/main/__tests__/runVerify-atomic-verdicts.test.cjs +0 -26
  228. package/src/main/__tests__/runVerify-blocked-by-foreign-wip.test.cjs +0 -58
  229. package/src/main/__tests__/runVerify-landed-commit-outranks.test.cjs +0 -191
  230. package/src/main/__tests__/runVerify-policy-denial.test.cjs +0 -89
  231. package/src/main/__tests__/runVerify-transcript-commit-evidence.test.cjs +0 -225
  232. package/src/main/__tests__/runVerify.test.cjs +0 -1784
  233. package/src/main/__tests__/scheduleJobSchema.test.cjs +0 -127
  234. package/src/main/__tests__/scheduleJobStatusDrift.test.cjs +0 -65
  235. package/src/main/__tests__/scheduleJobTransitions.test.cjs +0 -277
  236. package/src/main/__tests__/scheduleJobTransitionsGrep.test.cjs +0 -59
  237. package/src/main/__tests__/scheduleJobTransitionsTelemetryTap.test.cjs +0 -72
  238. package/src/main/__tests__/scheduler-admin-routes.test.cjs +0 -199
  239. package/src/main/__tests__/scheduler-adopted-run-supervision.test.cjs +0 -143
  240. package/src/main/__tests__/scheduler-already-satisfied-on-main.test.cjs +0 -105
  241. package/src/main/__tests__/scheduler-archive-completed-prd.test.cjs +0 -102
  242. package/src/main/__tests__/scheduler-archived-twin-guard.test.cjs +0 -155
  243. package/src/main/__tests__/scheduler-autofix-outcome.test.cjs +0 -188
  244. package/src/main/__tests__/scheduler-autofix-select.test.cjs +0 -439
  245. package/src/main/__tests__/scheduler-autopromote.test.cjs +0 -51
  246. package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +0 -101
  247. package/src/main/__tests__/scheduler-blocked-by-foreign-wip.test.cjs +0 -107
  248. package/src/main/__tests__/scheduler-boot-orphans.test.cjs +0 -153
  249. package/src/main/__tests__/scheduler-broadcast-reconcile.test.cjs +0 -121
  250. package/src/main/__tests__/scheduler-clear-queue-history.test.cjs +0 -134
  251. package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +0 -244
  252. package/src/main/__tests__/scheduler-committed-in-window.test.cjs +0 -182
  253. package/src/main/__tests__/scheduler-cross-project-batch.test.cjs +0 -43
  254. package/src/main/__tests__/scheduler-default-eligible-heal.test.cjs +0 -168
  255. package/src/main/__tests__/scheduler-dispatch-loop.test.cjs +0 -58
  256. package/src/main/__tests__/scheduler-effective-concurrency.test.cjs +0 -81
  257. package/src/main/__tests__/scheduler-epic-digest.test.cjs +0 -227
  258. package/src/main/__tests__/scheduler-failed-autoreset.test.cjs +0 -121
  259. package/src/main/__tests__/scheduler-finalize-dispatch-guards.test.cjs +0 -229
  260. package/src/main/__tests__/scheduler-find-prd-dir.test.cjs +0 -75
  261. package/src/main/__tests__/scheduler-fix-plan-path.test.cjs +0 -119
  262. package/src/main/__tests__/scheduler-force-tick-outcome.test.cjs +0 -46
  263. package/src/main/__tests__/scheduler-foreign-wip-manifest.test.cjs +0 -78
  264. package/src/main/__tests__/scheduler-gate-shadow.test.cjs +0 -119
  265. package/src/main/__tests__/scheduler-guard-verdict-autoresolve.test.cjs +0 -390
  266. package/src/main/__tests__/scheduler-heal-refusal.test.cjs +0 -61
  267. package/src/main/__tests__/scheduler-heartbeat-payload.test.cjs +0 -80
  268. package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +0 -323
  269. package/src/main/__tests__/scheduler-integration-failure-stamp.test.cjs +0 -41
  270. package/src/main/__tests__/scheduler-investigation-clean-skip.test.cjs +0 -63
  271. package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +0 -123
  272. package/src/main/__tests__/scheduler-job-budget.test.cjs +0 -172
  273. package/src/main/__tests__/scheduler-job-overrun.test.cjs +0 -175
  274. package/src/main/__tests__/scheduler-launch-failure.test.cjs +0 -199
  275. package/src/main/__tests__/scheduler-leftover-fields.test.cjs +0 -52
  276. package/src/main/__tests__/scheduler-leftover-quarantine.test.cjs +0 -199
  277. package/src/main/__tests__/scheduler-looks-done.test.cjs +0 -537
  278. package/src/main/__tests__/scheduler-manual-pause.test.cjs +0 -118
  279. package/src/main/__tests__/scheduler-mechanical-recovery.test.cjs +0 -245
  280. package/src/main/__tests__/scheduler-meta-code-sha.test.cjs +0 -46
  281. package/src/main/__tests__/scheduler-needs-review-autoresolve.test.cjs +0 -197
  282. package/src/main/__tests__/scheduler-never-stop.test.cjs +0 -157
  283. package/src/main/__tests__/scheduler-no-dead-end-status.test.cjs +0 -152
  284. package/src/main/__tests__/scheduler-no-orphan-run-dir.test.cjs +0 -81
  285. package/src/main/__tests__/scheduler-notify-originating-tab-transcript.test.cjs +0 -87
  286. package/src/main/__tests__/scheduler-notify-originating-tab.test.cjs +0 -343
  287. package/src/main/__tests__/scheduler-periodic-reverify-guard.test.cjs +0 -217
  288. package/src/main/__tests__/scheduler-porcelain-rename.test.cjs +0 -164
  289. package/src/main/__tests__/scheduler-prd-missing-skip.test.cjs +0 -161
  290. package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +0 -178
  291. package/src/main/__tests__/scheduler-quarantine-autoresolve.test.cjs +0 -165
  292. package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +0 -257
  293. package/src/main/__tests__/scheduler-rate-limit-cooldown-freshness.test.cjs +0 -123
  294. package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +0 -152
  295. package/src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs +0 -156
  296. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +0 -754
  297. package/src/main/__tests__/scheduler-reaper-helpers-basics.test.cjs +0 -87
  298. package/src/main/__tests__/scheduler-reconcile-cwd-preserve.test.cjs +0 -100
  299. package/src/main/__tests__/scheduler-reconcile-history-backfill.test.cjs +0 -105
  300. package/src/main/__tests__/scheduler-reconcile-invalid-repair.test.cjs +0 -203
  301. package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +0 -247
  302. package/src/main/__tests__/scheduler-reset-job-fields-guard.test.cjs +0 -77
  303. package/src/main/__tests__/scheduler-resume-recovery.test.cjs +0 -254
  304. package/src/main/__tests__/scheduler-shard-quarantine.test.cjs +0 -115
  305. package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +0 -380
  306. package/src/main/__tests__/scheduler-sigterm-commit.test.cjs +0 -43
  307. package/src/main/__tests__/scheduler-stall-per-project.test.cjs +0 -126
  308. package/src/main/__tests__/scheduler-starve-escalation.test.cjs +0 -154
  309. package/src/main/__tests__/scheduler-stranded-autofix-park.test.cjs +0 -252
  310. package/src/main/__tests__/scheduler-stranded-investigation.test.cjs +0 -185
  311. package/src/main/__tests__/scheduler-stuck-failed-escalation.test.cjs +0 -136
  312. package/src/main/__tests__/scheduler-supervisor-record.test.cjs +0 -81
  313. package/src/main/__tests__/scheduler-tick-cancel-token.test.cjs +0 -54
  314. package/src/main/__tests__/scheduler-tick-wedge.test.cjs +0 -172
  315. package/src/main/__tests__/scheduler-transient-failure.test.cjs +0 -141
  316. package/src/main/__tests__/scheduler-unreadable-queue-guard.test.cjs +0 -62
  317. package/src/main/__tests__/scheduler-utilization-hold.test.cjs +0 -89
  318. package/src/main/__tests__/scheduler-verify-prd-path.test.cjs +0 -109
  319. package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +0 -234
  320. package/src/main/__tests__/scheduler-worktree-exec-cwd.test.cjs +0 -120
  321. package/src/main/__tests__/scheduler-writeprd-epic-rollback.test.cjs +0 -105
  322. package/src/main/__tests__/schedulerBatchRootBlocker.test.cjs +0 -117
  323. package/src/main/__tests__/schedulerStateSidecarRestore.test.cjs +0 -110
  324. package/src/main/__tests__/seedAgentPersonas.test.cjs +0 -184
  325. package/src/main/__tests__/seedSchedulerMcp.test.cjs +0 -211
  326. package/src/main/__tests__/seedStatus.test.cjs +0 -100
  327. package/src/main/__tests__/seedValidatorPersona.test.cjs +0 -116
  328. package/src/main/__tests__/stop-signal-anchor.test.cjs +0 -75
  329. package/src/main/__tests__/telemetryClient.test.cjs +0 -1055
  330. package/src/main/__tests__/telemetryContract.test.cjs +0 -930
  331. package/src/main/__tests__/telemetrySettings.test.cjs +0 -210
  332. package/src/main/__tests__/transcripts-batch-flush.test.cjs +0 -249
  333. package/src/main/__tests__/transcripts-doFlush-array.test.cjs +0 -124
  334. package/src/main/__tests__/transcripts-paged-reads.test.cjs +0 -241
  335. package/src/main/__tests__/transcripts-worktree-epic-path.test.cjs +0 -154
  336. package/src/main/__tests__/transcriptsUsageFor.test.cjs +0 -206
  337. package/src/main/__tests__/uniquePrdNumbers.test.cjs +0 -153
  338. package/src/main/__tests__/usageSingleFlight.test.cjs +0 -169
  339. package/src/main/__tests__/validationSentinels.test.cjs +0 -84
  340. package/src/main/__tests__/workTypeLibrary.test.cjs +0 -89
  341. package/src/main/bilkoHost.cjs +0 -314
  342. package/src/main/bilkoHostCore.cjs +0 -89
  343. package/src/main/lib/__tests__/active-sessions.test.cjs +0 -251
  344. package/src/main/lib/__tests__/activeIndexRebuild.test.cjs +0 -179
  345. package/src/main/lib/__tests__/agentPersonaSchema.test.cjs +0 -67
  346. package/src/main/lib/__tests__/auditLog.test.cjs +0 -38
  347. package/src/main/lib/__tests__/bootSelfHeal.test.cjs +0 -107
  348. package/src/main/lib/__tests__/branchSweep.test.cjs +0 -164
  349. package/src/main/lib/__tests__/buildIdentity.test.cjs +0 -121
  350. package/src/main/lib/__tests__/buildTarget.test.cjs +0 -52
  351. package/src/main/lib/__tests__/childWithLog.test.cjs +0 -321
  352. package/src/main/lib/__tests__/coldBootPromptSessionsWrite.test.cjs +0 -87
  353. package/src/main/lib/__tests__/crashTelemetry.test.cjs +0 -103
  354. package/src/main/lib/__tests__/credentials-futile-refresh.test.cjs +0 -115
  355. package/src/main/lib/__tests__/cwdClassify.test.cjs +0 -111
  356. package/src/main/lib/__tests__/definitionOfDoneSequence.test.cjs +0 -95
  357. package/src/main/lib/__tests__/delegationReadiness.test.cjs +0 -1175
  358. package/src/main/lib/__tests__/dispatchLoop.test.cjs +0 -63
  359. package/src/main/lib/__tests__/effectiveModelInfo.test.cjs +0 -244
  360. package/src/main/lib/__tests__/ephemeralCwd.test.cjs +0 -91
  361. package/src/main/lib/__tests__/epicDelegationStats.test.cjs +0 -137
  362. package/src/main/lib/__tests__/epicSpawnCwd.test.cjs +0 -283
  363. package/src/main/lib/__tests__/epicSpawnPlan.test.cjs +0 -196
  364. package/src/main/lib/__tests__/epicTranscriptPath.test.cjs +0 -163
  365. package/src/main/lib/__tests__/epicWorktreeBoot.test.cjs +0 -136
  366. package/src/main/lib/__tests__/epicWorktreeMerge.test.cjs +0 -130
  367. package/src/main/lib/__tests__/epicWorktreeMint.test.cjs +0 -117
  368. package/src/main/lib/__tests__/epicWorktreeProjectConfig.test.cjs +0 -137
  369. package/src/main/lib/__tests__/fixChainDepth.test.cjs +0 -40
  370. package/src/main/lib/__tests__/fixtures/204-mercury-steam-horse.log.txt +0 -13
  371. package/src/main/lib/__tests__/fixtures/scheduler-machine.json.corrupt-1789147548 +0 -34
  372. package/src/main/lib/__tests__/gateFixtures.json +0 -20
  373. package/src/main/lib/__tests__/gitCacheBound.test.cjs +0 -69
  374. package/src/main/lib/__tests__/gitWorktree.test.cjs +0 -1478
  375. package/src/main/lib/__tests__/gitWorktreeSalvage.test.cjs +0 -107
  376. package/src/main/lib/__tests__/gitWorktreeSalvageDelta.test.cjs +0 -153
  377. package/src/main/lib/__tests__/guardShims.test.cjs +0 -158
  378. package/src/main/lib/__tests__/importReferences.spec.cjs +0 -56
  379. package/src/main/lib/__tests__/instanceLock.test.cjs +0 -173
  380. package/src/main/lib/__tests__/jobSupervisorRecord.test.cjs +0 -78
  381. package/src/main/lib/__tests__/jobWorktree.test.cjs +0 -199
  382. package/src/main/lib/__tests__/jobWorktreeBootLive.test.cjs +0 -82
  383. package/src/main/lib/__tests__/landedSinceRun.test.cjs +0 -133
  384. package/src/main/lib/__tests__/launchFailure.test.cjs +0 -220
  385. package/src/main/lib/__tests__/loadGate.test.cjs +0 -303
  386. package/src/main/lib/__tests__/localAdminHttp.test.cjs +0 -214
  387. package/src/main/lib/__tests__/loopDelay.test.cjs +0 -68
  388. package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +0 -107
  389. package/src/main/lib/__tests__/modelCatalog.test.cjs +0 -202
  390. package/src/main/lib/__tests__/opsOwnership.test.cjs +0 -113
  391. package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +0 -328
  392. package/src/main/lib/__tests__/opsRootNestedWrite.test.cjs +0 -51
  393. package/src/main/lib/__tests__/opsRootResolve.test.cjs +0 -149
  394. package/src/main/lib/__tests__/prdDeclaredPaths.test.cjs +0 -82
  395. package/src/main/lib/__tests__/prdDisposition.test.cjs +0 -224
  396. package/src/main/lib/__tests__/procIdentity.test.cjs +0 -119
  397. package/src/main/lib/__tests__/procName.test.cjs +0 -92
  398. package/src/main/lib/__tests__/projectBriefCore.test.cjs +0 -216
  399. package/src/main/lib/__tests__/projectRootResolve.test.cjs +0 -148
  400. package/src/main/lib/__tests__/queueHealth.test.cjs +0 -58
  401. package/src/main/lib/__tests__/queueStoreAtomicWrite.test.cjs +0 -88
  402. package/src/main/lib/__tests__/queueStoreMachineStateRecovery.test.cjs +0 -190
  403. package/src/main/lib/__tests__/quietMachineLease.test.cjs +0 -39
  404. package/src/main/lib/__tests__/rateLimitWindow.test.cjs +0 -88
  405. package/src/main/lib/__tests__/reaperHelpers.test.cjs +0 -577
  406. package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +0 -312
  407. package/src/main/lib/__tests__/schedulerBatchFairness.test.cjs +0 -213
  408. package/src/main/lib/__tests__/schedulerBatchLaunchHold.test.cjs +0 -125
  409. package/src/main/lib/__tests__/schedulerBatchProjectCap.test.cjs +0 -127
  410. package/src/main/lib/__tests__/schedulerBatchQuietMachine.test.cjs +0 -109
  411. package/src/main/lib/__tests__/schedulerMcpServerHeadlessRefusal.test.cjs +0 -71
  412. package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +0 -216
  413. package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +0 -183
  414. package/src/main/lib/__tests__/schedulerPaths.test.cjs +0 -226
  415. package/src/main/lib/__tests__/schedulerPathsWorktree.test.cjs +0 -133
  416. package/src/main/lib/__tests__/schedulerRuntimeState.test.cjs +0 -56
  417. package/src/main/lib/__tests__/sessionSlots.test.cjs +0 -144
  418. package/src/main/lib/__tests__/telemetryBacklog.test.cjs +0 -626
  419. package/src/main/lib/__tests__/telemetryBoot.test.cjs +0 -134
  420. package/src/main/lib/__tests__/telemetryConsent.test.cjs +0 -136
  421. package/src/main/lib/__tests__/telemetryCounters.test.cjs +0 -57
  422. package/src/main/lib/__tests__/telemetryCountersMetadataColumn.test.cjs +0 -98
  423. package/src/main/lib/__tests__/terminalRunOutcome.test.cjs +0 -200
  424. package/src/main/lib/__tests__/toolUseClassify.test.cjs +0 -53
  425. package/src/main/lib/__tests__/updateCheck.test.cjs +0 -63
  426. package/src/main/lib/__tests__/upgradeDrain.test.cjs +0 -130
  427. package/src/main/lib/__tests__/usageCircuit.test.cjs +0 -354
  428. package/src/main/lib/__tests__/watchdog-helpers.test.cjs +0 -375
  429. package/src/main/lib/__tests__/watchdog-relaunch.test.cjs +0 -266
  430. package/src/main/lib/kgExchangePairing.cjs +0 -75
@@ -47,6 +47,7 @@ const fs = require('node:fs');
47
47
  const fsp = require('node:fs/promises');
48
48
  const path = require('node:path');
49
49
  const os = require('node:os');
50
+ const { AsyncLocalStorage } = require('node:async_hooks');
50
51
  const { startDispatchLoop } = require('./lib/dispatchLoop.cjs');
51
52
  const schedulerPaths = require('./lib/schedulerPaths.cjs');
52
53
  const { randomUUID } = require('node:crypto');
@@ -54,8 +55,15 @@ const { execFile, execFileSync } = require('node:child_process');
54
55
  const { ipcMain } = require('electron');
55
56
  const billing = require('./usage.cjs');
56
57
  const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
58
+ const { ensureTimeoutShim, withTimeoutShimOnPath } = require('./lib/timeoutShim.cjs');
57
59
  const supervisor = require('./supervisor.cjs');
58
60
  const { resolveClaudeBin, claudeSpawnTarget, probeClaudeVersion } = require('./lib/claudeBin.cjs');
61
+ const { headlessPermissionArgs, ensureCliCapsProbed } = require('./lib/claudeCliCaps.cjs');
62
+ // Fire-and-forget priming — by the time the first job dispatches, the cached
63
+ // probe result is almost always already warm for buildClaudeSpawnArgs's sync
64
+ // callers; async spawn sites still `await ensureCliCapsProbed()` themselves
65
+ // as the authoritative guarantee.
66
+ ensureCliCapsProbed().catch(() => {});
59
67
  const launchFailure = require('./lib/launchFailure.cjs');
60
68
  const { appendError } = require('./lib/opsErrorLog.cjs');
61
69
  const { readTail } = require('./lib/fileTail.cjs');
@@ -68,6 +76,7 @@ const { resolveProjectRoot } = require('./lib/opsOwnership.cjs');
68
76
  const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
69
77
  const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
70
78
  const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
79
+ const { GATE_AUTHORITY_VERDICTS, decideGateAuthority } = require('./lib/gateAuthority.cjs');
71
80
  const { resolveBindingRateLimitReset } = require('./lib/rateLimitWindow.cjs');
72
81
  const { isResetFresh, bindingWindow, degradedBudget } = require('./lib/usageCircuit.cjs');
73
82
  const { computeQueueHealth } = require('./lib/queueHealth.cjs');
@@ -78,7 +87,8 @@ const { createBroadcastCoalescer } = require('./lib/broadcastCoalescer.cjs');
78
87
  const prdParser = require('./scheduler/prdParser.cjs');
79
88
  const sessionsStore = require('./sessionsStore.cjs');
80
89
  const { enqueueExternalPrompt } = require('./chatRunner.cjs');
81
- const { appendResponseEventIfKnown } = require('./promptSessionEvents.cjs');
90
+ const { appendResponseEventIfKnown, appendResponseEventWithReason } = require('./promptSessionEvents.cjs');
91
+ const { buildReviewNotice, selectDueReviewNotices, formatReviewNotice, holdMsFromEnv } = require('./lib/reviewNotice.cjs');
82
92
  const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs');
83
93
  const { parseValidationSentinels } = require('./lib/validationSentinels.cjs');
84
94
  const { hasDownstreamValidator } = require('./lib/planValidator.cjs');
@@ -144,7 +154,7 @@ const queueOps = require('./queueOps.cjs');
144
154
  // Plain Node module, no Electron dependency; queuePath/prdsDir defaults already
145
155
  // match ROOT/QUEUE_PATH below since both resolve the same ~/.claude/session-manager
146
156
  // home-dir layout.
147
- const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs, deriveProjectCwdFromPrdPath } = require('./lib/prdLocations.cjs');
157
+ const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs, deriveProjectCwdFromPrdPath, ancestorEpicPrdDirs, ancestorEpicArchivedPrdDirs } = require('./lib/prdLocations.cjs');
148
158
  const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
149
159
  const agentModelResolve = require('./lib/agentModelResolve.cjs');
150
160
  const { resolveEpicEffort, effortArgs } = require('./lib/agentEffortResolve.cjs');
@@ -176,6 +186,7 @@ const quietMachineLease = require('./lib/quietMachineLease.cjs');
176
186
  const runtimeState = require('./lib/schedulerRuntimeState.cjs');
177
187
  const jobWorktree = require('./lib/jobWorktree.cjs');
178
188
  const gitWorktree = require('./lib/gitWorktree.cjs');
189
+ const sharedGitExec = require('./lib/gitExec.cjs');
179
190
  const { buildJobWorktreeIsLive } = require('./lib/jobWorktreeBootLive.cjs');
180
191
  const { buildTerminalOrphanIsLive } = require('./lib/jobWorktreeTerminalOrphanLive.cjs');
181
192
  const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
@@ -271,113 +282,118 @@ const BASH_MAX_TIMEOUT_MS = 900_000; // 15 min — must stay below IDLE_OUTPUT_K
271
282
  // Value lives in reaperHelpers.cjs (imported above) so the external watchdog
272
283
  // shares the exact same budget — both increment the same j.orphanRetries field.
273
284
 
274
- // Appended to every scheduled job prompt so the queue can be RELIED ON to finish
275
- // work to a consistent bar: review → security-review → verify → commit. Enforced
276
- // centrally here (not per-PRD) so it applies to every current and future PRD.
277
- // The commit step is also backstopped by the post-run commit guard below: a
278
- // clean exit that leaves uncommitted changes is downgraded to needs_review.
285
+ // Appended to every scheduled job prompt so every PRD finishes the same way:
286
+ // review → security review → verify → commit → verdict line. It lives here,
287
+ // not in each PRD, so it covers every current and future PRD. The post-run
288
+ // commit guard backs up the commit step: a clean exit that leaves
289
+ // uncommitted changes is downgraded to needs_review.
279
290
  //
280
291
  // PRD 1408: a plan that ends in its own trailing `validator` job (PRD
281
- // 1405/1407) already re-reviews every work-item's diff once, so asking EVERY
282
- // work-item to also run /code-review + /security-review inline is duplicate
283
- // review work — and the dominant cost tail on multi-file PRDs. buildFinishProtocol
284
- // derives a `reviewInRun: false` variant that collapses those two steps into
285
- // one deferral note and renumbers the rest, sharing every other word
286
- // byte-for-byte with the default (`reviewInRun: true`, which is exactly
287
- // FINISH_PROTOCOL) via the head/review-steps/tail pieces below.
292
+ // 1405/1407) already reviews every work item's diff once, so asking EVERY
293
+ // work item to also run /code-review + /security-review is duplicate work
294
+ // and the main cost tail on multi-file PRDs. buildFinishProtocol builds a
295
+ // `reviewInRun: false` variant that folds those two steps into one deferral
296
+ // note and renumbers the rest. Every other word is shared byte-for-byte with
297
+ // the default (`reviewInRun: true`, which is exactly FINISH_PROTOCOL) through
298
+ // the head / review-steps / tail pieces below.
299
+ //
300
+ // Plain words on purpose: short sentences, one rule each, rule first, then
301
+ // "Why:". Tests pin the load-bearing tokens (SYNCHRONOUSLY, background Bash,
302
+ // Monitor, TaskOutput, ScheduleWakeup, "no later turn", `timeout <n>`,
303
+ // `git add <path>`, the SCHEDULER_VERDICT lines) — keep them exact. No line
304
+ // of this text may START with SCHEDULER_VERDICT: or FOREIGN_WIP_PATHS: —
305
+ // runVerify's scanners are line-anchored.
288
306
  function buildFinishHead(verifyStepNum) {
289
307
  return `
290
308
 
291
309
  ---
292
- # SCHEDULER FINISH PROTOCOL (mandatory — runs AFTER the work above)
293
-
294
- Once every acceptance-criteria line above is satisfied, finish in this EXACT
295
- sequence. Do not stop before the commit lands; committing is part of the job.
296
-
297
- RUN VERIFICATION IN THE FOREGROUND — this applies to the whole run, not just
298
- step ${verifyStepNum} below: every test/typecheck/lint/build command you run, whether while
299
- implementing the AC or during VERIFY, must run SYNCHRONOUSLY and you must wait
300
- for it to return. Never start a verification command as a background task
301
- (no background Bash) and then call Monitor, TaskOutput, or ScheduleWakeup to
302
- pick up its result later — a headless \`claude -p\` run has no later turn, so
303
- nothing ever delivers that notification and the run dies mid-verification with
304
- no commit and no verdict. Your foreground Bash budget for this run is
305
- ${BASH_DEFAULT_TIMEOUT_MS / 1000}s by default, up to ${BASH_MAX_TIMEOUT_MS / 1000}s max
306
- — size your own \`timeout <n>\` wrapper (e.g. \`timeout ${Math.floor(BASH_MAX_TIMEOUT_MS / 1000)} npm test\`)
307
- to fit inside that ceiling; if a gate command still cannot finish inside
308
- budget, stop and emit SCHEDULER_VERDICT: FAIL with the reason instead of
309
- deferring it.
310
+ # SCHEDULER FINISH PROTOCOL (required — do this after the work above)
311
+
312
+ When every acceptance criterion above is met, do the steps below in order.
313
+ Do not stop before your commit lands. The commit is part of the job.
314
+
315
+ Do not stop to ask a question. No one can answer it: this run has no later
316
+ turn. If something is unclear, make the safest reasonable choice, say what you
317
+ chose in your final report, and finish. If you truly cannot go on, end with the
318
+ verdict line SCHEDULER_VERDICT: FAIL and the reason.
319
+
320
+ Run every check in the foreground. This rule covers the whole run, not only step ${verifyStepNum}.
321
+ It covers every test, typecheck, lint and build command, while you build and while you verify.
322
+ - Run each command SYNCHRONOUSLY and wait for it to return.
323
+ - Never start a check as a background task (no background Bash).
324
+ - Never call Monitor, TaskOutput or ScheduleWakeup to collect a result later.
325
+ Why: a headless \`claude -p\` run has no later turn. Nothing will deliver
326
+ that result, and the run ends with no commit and no verdict.
327
+ - Your foreground Bash limit is ${BASH_DEFAULT_TIMEOUT_MS / 1000}s by default and ${BASH_MAX_TIMEOUT_MS / 1000}s at most.
328
+ Wrap long commands to fit inside it, for example \`timeout ${Math.floor(BASH_MAX_TIMEOUT_MS / 1000)} npm test\`.
329
+ - If a check still cannot finish in time, stop and end with the verdict line
330
+ SCHEDULER_VERDICT: FAIL and the reason. Do not put it off.
310
331
 
311
332
  `;
312
333
  }
313
334
 
314
- const FINISH_REVIEW_STEPS_IN_RUN = `1. CODE REVIEW — run \`/code-review --fix\` on your changes and apply the fixes it
315
- surfaces (correctness first). For any finding you judge a false positive, say
316
- why in your result; do not silently skip it. If \`/code-review\` is not
317
- available in this environment, do an equivalent careful self-review instead.
318
- 2. SECURITY REVIEW — run \`/security-review\` and address every finding (or
319
- justify it). If unavailable, self-review the diff for injection, secrets,
320
- path traversal, and unsafe input handling.
335
+ const FINISH_REVIEW_STEPS_IN_RUN = `1. CODE REVIEW — run \`/code-review --fix\` on your changes. Apply the fixes it finds,
336
+ correctness first. If you think a finding is wrong, say why in your final report.
337
+ Do not skip a finding silently. If \`/code-review\` is not available here,
338
+ review your own diff with the same care.
339
+ 2. SECURITY REVIEW — run \`/security-review\` and fix every finding, or explain
340
+ why it is safe. If it is not available, check your diff yourself for
341
+ injection, secrets, path traversal and unsafe input handling.
321
342
  `;
322
343
 
323
- const FINISH_REVIEW_STEP_DEFERRED = `1. REVIEW — code and security review for this plan run once, in its trailing validator job. Do NOT run /code-review or /security-review here.
344
+ const FINISH_REVIEW_STEP_DEFERRED = `1. REVIEW — skip it in this run. The plan's trailing validator job reviews code and security once, for the whole plan. Do NOT run /code-review or /security-review here.
324
345
  `;
325
346
 
326
347
  function buildFinishTail({ verifyStepNum, commitStepNum, verdictStepNum }) {
327
- return `${verifyStepNum}. VERIFY — run the project's OWN check commands (typecheck / lint / tests — the
328
- project's CLAUDE.md names them; infer from the repo if not) and make them
329
- pass. Do not assume npm; use whatever the target project uses.
330
- ${commitStepNum}. COMMIT — the queue can run several jobs against this SAME working tree at
331
- once. Do NOT stage the whole working tree in one blanket/wildcard git-add
332
- sweep — that captures whatever a concurrent sibling job is mid-writing and
333
- mis-attributes its work to this commit, corrupting both jobs' verdicts.
334
- Stage only the exact paths YOU created or modified for this PRD, then
335
- commit: \`git add <path> [<path>...] && git commit -m "<type>(<scope>): <summary>"\`.
336
- Your own work must still never be left uncommitted — this only changes
337
- which paths get staged, never whether you commit.
338
- ${verdictStepNum}. VERDICT SENTINEL — as the LAST LINE of your final result text, emit exactly
339
- one of these lines (no trailing text after it):
348
+ return `${verifyStepNum}. VERIFY — run every command in the PRD's \`# Gate\` section, in order, in the
349
+ foreground. Each must exit 0. If the PRD has no \`# Gate\` section, run the
350
+ project's own typecheck, lint and test commands instead. The project's
351
+ CLAUDE.md names them; if it does not, find them in the repo. Do not assume npm.
352
+ ${commitStepNum}. COMMIT — stage only the exact paths you created or changed for this PRD,
353
+ then commit:
354
+ \`git add <path> [<path>...] && git commit -m "<type>(<scope>): <summary>"\`
355
+ Never stage the whole tree with a wildcard or blanket git add.
356
+ Why: other jobs may be writing files in this same working tree right now.
357
+ A blanket add puts their half-done work in your commit and breaks both
358
+ jobs' verdicts. Always commit your own work: this rule changes what you
359
+ stage, never whether you commit.
360
+ ${verdictStepNum}. VERDICT LINE — end your final report with exactly one of these lines.
361
+ Put nothing after it, except the FOREIGN_WIP_PATHS line described below.
340
362
  SCHEDULER_VERDICT: PASS
341
363
  SCHEDULER_VERDICT: FAIL <one-line reason>
342
364
  SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP
343
- Print PASS only when the AC gate is green AND the commit from step ${commitStepNum} landed.
344
- Print FAIL (and exit 1) if the AC gate was red or the commit could not land.
345
- NEVER print PASS on a red AC gate — a lying PASS turns the verifier from a
346
- false-failure catcher into a silent-failure shipper. A truthful PASS + a
347
- landed commit lets the verifier override incidental transcript noise (grep
348
- results containing "Error", a TDD red-test run early in the session, debug
349
- Tracebacks) so those do not false-trip a needs_review downgrade.
350
- Print BLOCKED_BY_FOREIGN_WIP (and exit 1) ONLY when your own AC gate failed
351
- because it ran against a SIBLING job's in-flight, uncommitted file — never
352
- because of your own regression — AND every failing path is one this prompt
353
- already disclosed to you as foreign (see the "FOREIGN WORKING-TREE STATE"
354
- section above, if present). It MUST be accompanied by a second line naming
355
- every such path:
356
- FOREIGN_WIP_PATHS: <path1>, <path2>, ...
357
- The scheduler independently validates every listed path against the exact
358
- foreign-WIP manifest it disclosed to you. Only list a path that (a) this
359
- prompt already told you is foreign WIP, not yours, AND (b) is why your own
360
- AC gate failed — never a path you own, and never a path that failed for
361
- some other reason of your own making. Any listed path NOT in that manifest
362
- downgrades this whole verdict back to FAIL, with your claim rejected. This
363
- is not an escape hatch for your own broken code: claiming it for a
364
- regression you introduced, or for a path you never received as foreign
365
- WIP, is a lying verdict exactly like a false PASS.
366
-
367
- A job that exits with uncommitted changes is treated as INCOMPLETE and flagged
368
- for review. Do NOT add work beyond the acceptance criteria — this protocol is the
369
- only post-AC work. If a review finding can't be fixed within scope, commit what
370
- you have, describe the finding in the commit body, and note the follow-up in your
371
- final result.`;
365
+ - Print PASS only when the gate is green AND the commit from step ${commitStepNum} landed.
366
+ - Print FAIL when the gate is red or the commit could not land.
367
+ - Never print PASS on a red gate. Why: a false PASS ships broken work that
368
+ no one notices. A true PASS plus a landed commit lets the scheduler ignore
369
+ harmless noise in your transcript, such as grep hits for "Error", an early
370
+ red test run, or debug tracebacks.
371
+ - Print BLOCKED_BY_FOREIGN_WIP ONLY when your gate failed because of a
372
+ sibling job's unfinished, uncommitted file — never because of your own
373
+ change — AND every failing path is one this prompt already listed as
374
+ foreign (see the "FOREIGN WORKING-TREE STATE" section above, if present).
375
+ Add a second line that names every such path:
376
+ FOREIGN_WIP_PATHS: <path1>, <path2>, ...
377
+ The scheduler checks each listed path against the list it gave you. List
378
+ a path only if (a) this prompt told you it is foreign work, not yours,
379
+ and (b) it is why your gate failed. One path that is not on that list
380
+ turns the whole verdict back into FAIL. Claiming this for your own bug is
381
+ a false verdict, just like a false PASS.
382
+
383
+ A job that ends with uncommitted changes counts as INCOMPLETE and is flagged
384
+ for review. Do not add work beyond the acceptance criteria: these steps are the
385
+ only work after them. If you cannot fix a review finding within scope, commit
386
+ what you have, describe the finding in the commit body, and note the follow-up
387
+ in your final report.`;
372
388
  }
373
389
 
374
390
  /**
375
391
  * buildFinishProtocol({ reviewInRun }) → string
376
392
  *
377
393
  * `reviewInRun: true` (default) is byte-for-byte FINISH_PROTOCOL: steps 1-5
378
- * are CODE REVIEW, SECURITY REVIEW, VERIFY, COMMIT, VERDICT SENTINEL.
394
+ * are CODE REVIEW, SECURITY REVIEW, VERIFY, COMMIT, VERDICT LINE.
379
395
  * `reviewInRun: false` collapses the two review steps into one deferral note
380
- * (step 1) and renumbers VERIFY/COMMIT/VERDICT SENTINEL to 2-4 — used when
396
+ * (step 1) and renumbers VERIFY/COMMIT/VERDICT LINE to 2-4 — used when
381
397
  * `hasDownstreamValidator` (lib/planValidator.cjs) finds this job's diff will
382
398
  * be re-reviewed once by the plan's own trailing validator job, so asking
383
399
  * every work-item to also run /code-review + /security-review inline is
@@ -1175,7 +1191,15 @@ function archivedPrdPathForJob(job) {
1175
1191
  */
1176
1192
  async function archivedTwinExists(job) {
1177
1193
  const slug = job && job.slug;
1178
- for (const dir of listArchivedPrdDirs((job && job.cwd) || DEFAULT_PROJECT_CWD)) {
1194
+ const dirs = [...listArchivedPrdDirs((job && job.cwd) || DEFAULT_PROJECT_CWD)];
1195
+ // Epic-in-non-git-parent case (see prdLocations.ancestorEpicDirs' header
1196
+ // comment): job.cwd's own project never sees the ancestor Epic's archive
1197
+ // dir via listArchivedPrdDirs, so also walk up from job.cwd for this
1198
+ // job's EXACT epicId.
1199
+ if (job && job.cwd && job.epicId) {
1200
+ dirs.push(...ancestorEpicArchivedPrdDirs(job.cwd, job.epicId));
1201
+ }
1202
+ for (const dir of dirs) {
1179
1203
  const candidate = safeSlugPathIn(dir, slug);
1180
1204
  if (!candidate) continue;
1181
1205
  try {
@@ -1232,6 +1256,41 @@ async function findPrdDir(slug) {
1232
1256
  return null;
1233
1257
  }
1234
1258
 
1259
+ /**
1260
+ * Job-aware widening of findPrdDir: tries the ordinary global candidate
1261
+ * search first (unchanged), then a stored absolute PRD path on the row (if
1262
+ * the queue schema ever carries one — checked defensively), then walks up
1263
+ * to 3 ancestor directories above `job.cwd` for that EXACT `job.epicId`'s
1264
+ * `prds` dir (prdLocations.ancestorEpicPrdDirs).
1265
+ *
1266
+ * Fixes the incident where an Epic lives in a non-git PARENT folder P
1267
+ * (P/session-manager-operations/scheduler/epics/<epicId>/prds/*.md) but its
1268
+ * PRDs' frontmatter `cwd` is the sub-repo P/repo — P is never itself a
1269
+ * tracked project cwd, so findPrdDir's candidatePrdsDirs() never visits it
1270
+ * and the job was wrongly retired as `prd-missing`. epicId is matched
1271
+ * exactly (never a glob); every path is built with path.join, never string
1272
+ * concatenation with user input.
1273
+ */
1274
+ async function findPrdDirForJob(job) {
1275
+ const viaGlobalSearch = await findPrdDir(job && job.slug);
1276
+ if (viaGlobalSearch) return viaGlobalSearch;
1277
+ if (job && typeof job.prdPath === 'string' && job.prdPath) {
1278
+ try {
1279
+ await fsp.access(job.prdPath);
1280
+ return path.dirname(job.prdPath);
1281
+ } catch { /* stored path stale — fall through to the ancestor walk */ }
1282
+ }
1283
+ if (job && job.cwd && job.epicId) {
1284
+ for (const dir of ancestorEpicPrdDirs(job.cwd, job.epicId)) {
1285
+ try {
1286
+ await fsp.access(path.join(dir, `${job.slug}.md`));
1287
+ return dir;
1288
+ } catch { /* not here — try the next ancestor */ }
1289
+ }
1290
+ }
1291
+ return null;
1292
+ }
1293
+
1235
1294
  /**
1236
1295
  * Resolve `<dir>/<slug>.md` for a directory already known to contain (or be
1237
1296
  * about to receive) the slug, and enforce path containment. Returns the
@@ -1246,6 +1305,26 @@ function safeSlugPathIn(dir, slug) {
1246
1305
  return resolved;
1247
1306
  }
1248
1307
 
1308
+ /**
1309
+ * Realpath of a PRDs dir for the symlink-containment re-checks below, or null
1310
+ * when `dir` or any component between it and its project cwd (the parent of
1311
+ * `session-manager-operations`) is a symlink. Why: comparing a file's realpath
1312
+ * against the RAW dir rejected every PRD under a symlinked path (macOS tmpdir
1313
+ * /var → /private/var), but trusting realpath(dir) alone lets a rogue job swap
1314
+ * `epics/<id>/prds` for a symlink to e.g. ~/.claude and write there. Symlinks
1315
+ * ABOVE the project cwd (a symlinked home or project folder) stay allowed.
1316
+ */
1317
+ async function realPrdsDir(dir) {
1318
+ const marker = `${path.sep}session-manager-operations${path.sep}`;
1319
+ const idx = (dir + path.sep).lastIndexOf(marker);
1320
+ const stop = idx === -1 ? path.dirname(dir) : dir.slice(0, idx);
1321
+ for (let p = dir; p.length > stop.length; p = path.dirname(p)) {
1322
+ const st = await fsp.lstat(p).catch(() => null);
1323
+ if (st && st.isSymbolicLink()) return null;
1324
+ }
1325
+ try { return await fsp.realpath(dir); } catch { return null; }
1326
+ }
1327
+
1249
1328
  /**
1250
1329
  * Locate an EXISTING `<slug>.md` across every candidate PRD dir and return
1251
1330
  * its safe, containment-checked absolute path — or null if the slug isn't
@@ -1289,6 +1368,18 @@ async function resolveSlugOrReason(slug, cwd) {
1289
1368
  return { ok: true, path: p };
1290
1369
  }
1291
1370
 
1371
+ /**
1372
+ * Shared row-match predicate for a reset-job request: slugs carry no cwd
1373
+ * salt, so two different projects can independently produce the identical
1374
+ * slug. Both the renderer-facing `schedule:reset-job` IPC handler and
1375
+ * remote.resetJob (admin/MCP) go through this one predicate so a cwd-bearing
1376
+ * caller always resets THAT project's job, never just any row matching the
1377
+ * string.
1378
+ */
1379
+ function jobMatchesSlugAndCwd(j, slug, cwd) {
1380
+ return j.slug === slug && (!cwd || j.cwd === cwd);
1381
+ }
1382
+
1292
1383
  /** Actionable message for `resolveSlugOrReason`'s 'not-found' reason. */
1293
1384
  function unknownSlugMessage(slug) {
1294
1385
  return `unknown slug "${slug}": no PRD file with that name in any known project — call scheduler_list_prds (optionally with cwd) to see what exists`;
@@ -1328,9 +1419,17 @@ async function archiveCompletedPrd(slug, cwd) {
1328
1419
  * `schedule:archive-prd`) so a stale queue entry can never survive to fire
1329
1420
  * against a PRD that no longer exists in the live prds/ dir — the same
1330
1421
  * ENOENT-avoidance archivedTwinExists provides in executeJob, applied at the
1331
- * archiving source instead of at fire-time. auto-archived slugs never need
1332
- * this (selectAutoArchivable in queueOps.cjs only selects already-completed
1333
- * jobs), so this is exercised only by the manual archive path.
1422
+ * archiving source instead of at fire-time.
1423
+ *
1424
+ * MUST NEVER be called from inside a mutate() body (directly, or transitively
1425
+ * via reconcile()/tickBody, which always run inside one) — this function
1426
+ * itself calls mutate() below, and a mutate queued from inside a mutate body
1427
+ * queues behind the very mutate awaiting it, so mutateTail never settles:
1428
+ * every tick wedges machine-wide until the app restarts (proven live
1429
+ * 2026-09-25). That is exactly why queueOps.cjs's autoArchiveCompleted (which
1430
+ * DOES run inside reconcile()) calls archiveMany with { retire: false } and
1431
+ * skips this function entirely — it is exercised only by the manual archive
1432
+ * path (queueOps.cjs's `schedule:archive-prd`, called from outside mutate()).
1334
1433
  */
1335
1434
  async function retireCompletedSlugs(slugs) {
1336
1435
  const list = Array.isArray(slugs) ? slugs.filter(Boolean) : [];
@@ -2469,6 +2568,15 @@ async function writeQueue(state) {
2469
2568
  // preceding mutate threw, so the chain never deadlocks.
2470
2569
  let mutateTail = Promise.resolve();
2471
2570
 
2571
+ // Tracks whether the current async continuation is running inside a live
2572
+ // mutate() body. A mutate() called (and awaited) from inside another mutate's
2573
+ // `fn` would otherwise queue behind its own caller on mutateTail and deadlock
2574
+ // forever — this turns that class of bug into an immediate, audited rejection.
2575
+ // A callback (setTimeout/promise) scheduled from inside a mutate body but
2576
+ // firing AFTER that body has finished sees ctx.active === false and proceeds
2577
+ // normally, since AsyncLocalStorage propagates the store into it.
2578
+ const mutateCtx = new AsyncLocalStorage();
2579
+
2472
2580
  // Observe-only watchdog: a mutate body over MUTATE_WATCHDOG_MS is logged and
2473
2581
  // audited once per episode (latched until a mutate completes). mutateTail is
2474
2582
  // deliberately NEVER reset — it is what enforces the single-writer law, and
@@ -2477,6 +2585,12 @@ const MUTATE_WATCHDOG_MS = 60_000;
2477
2585
  let mutateWedgeLatched = false;
2478
2586
 
2479
2587
  function mutate(fn) {
2588
+ if (mutateCtx.getStore()?.active === true) {
2589
+ const err = new Error('mutate() re-entered from inside a running mutate body — would deadlock');
2590
+ console.error('[scheduler] MUTATE RE-ENTRANT', err.stack);
2591
+ appendAuditEvent('mutate_reentrant', { stack: String(err.stack).split('\n').slice(0, 8).join('\n') });
2592
+ return Promise.reject(err);
2593
+ }
2480
2594
  const next = mutateTail.then(async () => {
2481
2595
  const wedgeTimer = setTimeout(() => {
2482
2596
  if (mutateWedgeLatched) return;
@@ -2485,9 +2599,11 @@ function mutate(fn) {
2485
2599
  appendAuditEvent('mutate_wedged', { budgetMs: MUTATE_WATCHDOG_MS });
2486
2600
  }, MUTATE_WATCHDOG_MS);
2487
2601
  if (typeof wedgeTimer.unref === 'function') wedgeTimer.unref();
2602
+ const ctx = { active: true };
2488
2603
  try {
2489
- return await mutateBody(fn);
2604
+ return await mutateCtx.run(ctx, () => mutateBody(fn));
2490
2605
  } finally {
2606
+ ctx.active = false;
2491
2607
  clearTimeout(wedgeTimer);
2492
2608
  mutateWedgeLatched = false;
2493
2609
  }
@@ -3592,13 +3708,17 @@ let drainActive = false;
3592
3708
 
3593
3709
  function drainDeferredInvestigation() {
3594
3710
  if (drainActive || runtimeState.investigationCount() >= MAX_CONCURRENT_INVESTIGATIONS) return;
3595
- const next = deferredInvestigations.entries().next();
3596
- if (next.done) return;
3597
- const [slug, ctx] = next.value;
3598
- deferredInvestigations.delete(slug);
3599
- spawnInvestigation(ctx.failedJob, ctx.runDir).catch((e) => {
3600
- console.error('[scheduler] drained investigation error', slug, e);
3601
- });
3711
+ // Skip the slug the background gate shadow is still deciding (gateShadowSlug,
3712
+ // set near reverifyNeedsReview's pick) — leave it queued rather than drain it
3713
+ // out from under the gate. Take the first OTHER entry instead, in queue order.
3714
+ for (const [slug, ctx] of deferredInvestigations) {
3715
+ if (slug === gateShadowSlug) continue;
3716
+ deferredInvestigations.delete(slug);
3717
+ spawnInvestigation(ctx.failedJob, ctx.runDir).catch((e) => {
3718
+ console.error('[scheduler] drained investigation error', slug, e);
3719
+ });
3720
+ return;
3721
+ }
3602
3722
  }
3603
3723
  let cancelToken = { cancelled: false };
3604
3724
  // Last memory-gate observation; included in snapshot for renderer visibility.
@@ -3775,13 +3895,31 @@ function computeFireAt(state, nextResetIso) {
3775
3895
  return reset + (state.config.offsetMinutes * 60_000);
3776
3896
  }
3777
3897
 
3898
+ // PRD 1446: refreshNextReset() awaits billing.fetchUsage(), which can hang
3899
+ // or back off for minutes while the usage meter is rate-limited. Left
3900
+ // unbounded, that stalled THIS function's own mutate(reconcile) call for as
3901
+ // long as the fetch took — the 10-minute rescheduleTimer was meant to be a
3902
+ // second, independent path into reconcile() alongside the 60s pollLoop, but
3903
+ // a hung fetch silently took it out too, leaving newly-created PRDs
3904
+ // unadopted (PRDs 1420-1441, 2026-09-25). Racing against this timeout lets
3905
+ // the reconcile still run on the cached reset value; the abandoned
3906
+ // refreshNextReset() call is left to resolve on its own and update the
3907
+ // cache for next time.
3908
+ const RESCHEDULE_TIMER_BILLING_RACE_MS = 10_000;
3909
+
3778
3910
  async function rescheduleTimer() {
3779
3911
  clearFireTimer();
3780
3912
  // Wrap in try/catch — on failure use the cached value so the on-reset
3781
3913
  // timer can still be armed from the last known reset.
3782
3914
  let nextResetIso;
3783
3915
  try {
3784
- nextResetIso = await refreshNextReset();
3916
+ nextResetIso = await Promise.race([
3917
+ refreshNextReset(),
3918
+ new Promise((resolve) => {
3919
+ const t = setTimeout(() => resolve(cachedNextReset), RESCHEDULE_TIMER_BILLING_RACE_MS);
3920
+ if (typeof t.unref === 'function') t.unref();
3921
+ }),
3922
+ ]);
3785
3923
  } catch {
3786
3924
  nextResetIso = cachedNextReset;
3787
3925
  }
@@ -3978,15 +4116,21 @@ async function clearPause(source) {
3978
4116
  /**
3979
4117
  * Mutate a job in place to "pending" with cleared run metadata.
3980
4118
  *
3981
- * Refuses (no-ops, returns false) on a job already in a terminal success
3982
- * state ('completed') unless opts.force is true — resetting a completed job
3983
- * re-fires the PRD and re-executes already-shipped work (the false-failure
3984
- * class PRD 812-workbench-review-nits-cleanup demonstrated: a completed job
3985
- * was reset to pending and re-ran a correct no-op that then got flagged
3986
- * needs_review). All internal call sites operate on jobs that are still
3987
- * 'running'/'failed' at the point they call this, so the guard is a no-op
3988
- * for them; only an external reset request (IPC/admin API) can target an
3989
- * already-'completed' job, and that path is exactly what this guards.
4119
+ * Refuses (no-ops, returns false) on a job already in one of two terminal
4120
+ * statuses, unless opts.force is true:
4121
+ * - 'completed': the work already shipped. Resetting it re-fires the PRD
4122
+ * and re-executes already-shipped work (the false-failure class PRD
4123
+ * 812-workbench-review-nits-cleanup demonstrated: a completed job was
4124
+ * reset to pending and re-ran a correct no-op that then got flagged
4125
+ * needs_review).
4126
+ * - 'skipped': the scheduler or a human already chose not to run this job.
4127
+ * force:true is the deliberate override for that choice.
4128
+ * resetRefusalMessage() (right below) names the one of these two that
4129
+ * applies, for the two external callers that report a refusal to a caller.
4130
+ * All internal call sites operate on jobs that are still 'running'/'failed'
4131
+ * at the point they call this, so the guard is a no-op for them; only an
4132
+ * external reset request (IPC/admin API) can target an already-'completed'
4133
+ * or already-'skipped' job, and that path is exactly what this guards.
3990
4134
  */
3991
4135
  function resetJobFields(job, errorMsg, opts = {}) {
3992
4136
  if ((job.status === 'completed' || job.status === 'skipped') && opts.force !== true) return false;
@@ -4046,12 +4190,51 @@ function resetJobFields(job, errorMsg, opts = {}) {
4046
4190
  delete job.blockedByForeignWip;
4047
4191
  delete job.foreignWipBlockedPaths;
4048
4192
  delete job.foreignWipBlockCount;
4193
+ // A reset starts a new episode: whatever this row's last needs_review park
4194
+ // was about is over, so a later park must start its own hold clock from
4195
+ // scratch rather than inheriting a stale firstParkedAt from days earlier
4196
+ // (reviewNotice.cjs's own doc explains why that matters). The rung-6
4197
+ // ladder requeue (applyNeedsReviewAutoResolve) does NOT call
4198
+ // resetJobFields — it transitions the row directly — so that requeue
4199
+ // correctly keeps the same clock, because it is still the same episode.
4200
+ delete job.reviewNotice;
4201
+ // This run's completion evidence, not durable across a reset — an old
4202
+ // run's looksDone must never complete the NEXT run
4203
+ // (applyNeedsReviewAutoResolve completes on looksDone alone), and a stale
4204
+ // evidenceScannedAt must not block the next park's own scan.
4205
+ // computeLooksDone only counts commits since job.startedAt, so the next
4206
+ // park re-scans fresh once this run's own startedAt is set.
4207
+ delete job.looksDone;
4208
+ delete job.evidenceScannedAt;
4049
4209
  // Deliberately NOT deleting job.landedCommit: it must outlive a reset so a
4050
4210
  // re-fired run of this same slug can pass it to verifyRun as
4051
4211
  // priorLandedCommit (pass_no_commit_prior_run_verified exemption).
4052
4212
  return true;
4053
4213
  }
4054
4214
 
4215
+ /**
4216
+ * Plain-word refusal message for a reset that resetJobFields' terminal-
4217
+ * status guard blocked. One branch per status that guard can refuse, plus a
4218
+ * generic fallback for any other status a caller might pass in (defensive —
4219
+ * resetJobFields today only refuses 'completed' and 'skipped').
4220
+ *
4221
+ * `canForce` tells the message whether force:true is actually available to
4222
+ * the caller: true for the admin/MCP resetJob (force threads through), false
4223
+ * for the renderer IPC schedule:reset-job (no force option there — see that
4224
+ * handler's own comment for why).
4225
+ */
4226
+ function resetRefusalMessage(status, { canForce } = {}) {
4227
+ if (status === 'completed') {
4228
+ return `job already completed — resetting it would re-execute shipped work; archive the PRD instead${canForce ? ', or pass force:true' : ''}`;
4229
+ }
4230
+ if (status === 'skipped') {
4231
+ return canForce
4232
+ ? 'job was skipped — pass force:true to run it again'
4233
+ : 'job was skipped — only scheduler_reset_job with force:true can run it again';
4234
+ }
4235
+ return `job status is "${status}" — it cannot be reset now`;
4236
+ }
4237
+
4055
4238
  /**
4056
4239
  * partitionBootOrphans(jobs, liveness) → { immediate: string[], adopted: string[] }
4057
4240
  *
@@ -4547,6 +4730,16 @@ async function notifyOriginatingTab(job, {
4547
4730
  * non-active session and returns false. Nothing here can create an Epic — if
4548
4731
  * no open authoring Epic exists, the root-cause report in the run directory
4549
4732
  * is the whole record and the Scheduler tab is where it surfaces.
4733
+ *
4734
+ * Quiet by default (this PRD): the park path no longer calls this at park
4735
+ * time. It records a `reviewNotice` on the job row instead and lets
4736
+ * flushDueReviewNotices send one grouped message once the self-heal ladder
4737
+ * gives up or the hold time passes. This function now only runs under the
4738
+ * SM_REVIEW_NOTICE_IMMEDIATE=1 kill switch (same park-time call site, for
4739
+ * local debugging) — it is otherwise dead code outside its own tests. That
4740
+ * call site stamps `reviewNotice.sentAt` itself afterward, and only when
4741
+ * this returns true AND the row's live notice still matches the one it just
4742
+ * built — never on the strength of this function's return value alone.
4550
4743
  */
4551
4744
  async function notifyNeedsReview(job, report, {
4552
4745
  parsePrdRaw = prdParser.parsePrdRaw,
@@ -4577,6 +4770,133 @@ async function notifyNeedsReview(job, report, {
4577
4770
  }
4578
4771
  }
4579
4772
 
4773
+ // Single-flight guard for flushDueReviewNotices — same shape as
4774
+ // gateShadowPending above: a module-level flag rather than a class, so two
4775
+ // triggers landing close together (an auto-resolve skip right next to the
4776
+ // heartbeat interval) read-modify-write queue.json in series, never racing
4777
+ // each other over the same rows.
4778
+ let reviewNoticeFlushPending = null;
4779
+
4780
+ /**
4781
+ * flushDueReviewNotices() → Promise<void>
4782
+ *
4783
+ * The only path that still sends a needs_review notice (outside the
4784
+ * SM_REVIEW_NOTICE_IMMEDIATE=1 kill switch above notifyNeedsReview). Reads
4785
+ * the live queue, asks reviewNotice.cjs's selectDueReviewNotices which (cwd,
4786
+ * Epic, cause) groups have gone quiet — the ladder gave up
4787
+ * (needsReviewAutoResolvedSkip) or the hold time passed (holdMsFromEnv) —
4788
+ * and sends each group ONE message via formatReviewNotice, passing the
4789
+ * tick's own full job list so the message can name any `pending` row the
4790
+ * group still blocks.
4791
+ *
4792
+ * The default sender is appendResponseEventWithReason, which reports WHY a
4793
+ * send didn't land instead of collapsing every case to `false`. An injected
4794
+ * sender (tests) may still return a plain boolean — `true` is treated as ok,
4795
+ * `false` as a refusal with reason `'refused'` — and a sender that throws is
4796
+ * treated as reason `'error'`.
4797
+ *
4798
+ * What each outcome does to reviewNotice.sentAt, per row still at its
4799
+ * snapshotted (firstParkedAt, cause) and still needs_review/skipped live:
4800
+ * - ok, or a refusal with any reason OTHER than 'error': stamp sentAt now.
4801
+ * A refusal is logged (today it was silent) but is otherwise treated as
4802
+ * permanent — a deleted Epic or a completed session will never un-refuse
4803
+ * itself, so resending every tick forever would help no one.
4804
+ * - 'error' (an exception, e.g. a disk hiccup): do NOT stamp. Increment
4805
+ * reviewNotice.sendErrors instead, so the next pass retries. Once that
4806
+ * reaches 3, stamp sentAt anyway, set sendFailed: true, and log it —
4807
+ * retry a transient failure, but never loop on it forever.
4808
+ * A group with no resolved Epic/cwd never calls the sender at all — same
4809
+ * "report only" shape as before — and every row in it is stamped
4810
+ * unconditionally, same as a successful send.
4811
+ *
4812
+ * Single-flight via reviewNoticeFlushPending: a call while one is already in
4813
+ * flight returns the SAME promise instead of starting a second pass over the
4814
+ * same rows. Never throws — a bad read just leaves the notice for the next
4815
+ * pass to retry.
4816
+ */
4817
+ async function flushDueReviewNotices({
4818
+ appendResponseEvent = appendResponseEventWithReason,
4819
+ now = Date.now(),
4820
+ } = {}) {
4821
+ if (reviewNoticeFlushPending) return reviewNoticeFlushPending;
4822
+ reviewNoticeFlushPending = (async () => {
4823
+ try {
4824
+ const state = await readQueue();
4825
+ if (state.unreadable) return;
4826
+ const groups = selectDueReviewNotices(state.jobs || [], { now, holdMs: holdMsFromEnv() });
4827
+ for (const group of groups) {
4828
+ const slugs = new Set(group.jobs.map((j) => j.slug));
4829
+ // Snapshot each row's own (firstParkedAt, cause) as it was when THIS
4830
+ // message was built — the mutate below only touches a row whose live
4831
+ // notice still matches its own snapshot, so a row that re-parked
4832
+ // with a new episode between this read and that mutate is left
4833
+ // alone rather than wrongly marked sent for a notice that never
4834
+ // described it.
4835
+ const snapshotBySlug = new Map(group.jobs.map((j) => [
4836
+ j.slug,
4837
+ { firstParkedAt: j.reviewNotice?.firstParkedAt ?? null, cause: j.reviewNotice?.cause ?? null },
4838
+ ]));
4839
+ const sentAt = new Date(now).toISOString();
4840
+ const stampIfLiveMatches = (s, apply) => {
4841
+ for (const j of s.jobs) {
4842
+ if (!slugs.has(j.slug) || !j.reviewNotice || j.reviewNotice.sentAt) continue;
4843
+ if (j.status !== 'needs_review' && j.status !== 'skipped') continue;
4844
+ const snap = snapshotBySlug.get(j.slug);
4845
+ if (j.reviewNotice.firstParkedAt !== snap.firstParkedAt || j.reviewNotice.cause !== snap.cause) continue;
4846
+ apply(j);
4847
+ }
4848
+ };
4849
+
4850
+ if (!group.epicId || !group.cwd) {
4851
+ console.log(`[scheduler] flushDueReviewNotices: no authoring Epic for ${group.jobs.map((j) => j.slug).join(', ')}, report only`);
4852
+ await mutate((s) => {
4853
+ stampIfLiveMatches(s, (j) => { j.reviewNotice.sentAt = sentAt; });
4854
+ }).catch(() => {});
4855
+ continue;
4856
+ }
4857
+
4858
+ let outcome;
4859
+ try {
4860
+ const result = await appendResponseEvent(group.cwd, group.epicId, formatReviewNotice(group, { jobs: state.jobs }), {
4861
+ prdSlug: group.jobs[0].slug,
4862
+ outcome: 'needs_review',
4863
+ validation: 'unvalidated',
4864
+ });
4865
+ outcome = typeof result === 'boolean' ? { ok: result, reason: result ? undefined : 'refused' } : result;
4866
+ } catch (e) {
4867
+ console.error('[scheduler] flushDueReviewNotices appendResponseEvent error', group.epicId, e);
4868
+ outcome = { ok: false, reason: 'error' };
4869
+ }
4870
+
4871
+ if (!outcome.ok && outcome.reason !== 'error') {
4872
+ console.warn(`[scheduler] flushDueReviewNotices: send refused (${outcome.reason}) for ${group.jobs.map((j) => j.slug).join(', ')}`);
4873
+ }
4874
+
4875
+ await mutate((s) => {
4876
+ stampIfLiveMatches(s, (j) => {
4877
+ if (outcome.ok || outcome.reason !== 'error') {
4878
+ j.reviewNotice.sentAt = sentAt;
4879
+ return;
4880
+ }
4881
+ const sendErrors = (j.reviewNotice.sendErrors ?? 0) + 1;
4882
+ j.reviewNotice.sendErrors = sendErrors;
4883
+ if (sendErrors >= 3) {
4884
+ j.reviewNotice.sentAt = sentAt;
4885
+ j.reviewNotice.sendFailed = true;
4886
+ console.error(`[scheduler] flushDueReviewNotices: giving up on ${j.slug} after ${sendErrors} send errors`);
4887
+ }
4888
+ });
4889
+ }).catch(() => {});
4890
+ }
4891
+ } catch (e) {
4892
+ console.error('[scheduler] flushDueReviewNotices error', e);
4893
+ } finally {
4894
+ reviewNoticeFlushPending = null;
4895
+ }
4896
+ })();
4897
+ return reviewNoticeFlushPending;
4898
+ }
4899
+
4580
4900
  /** Scan the tail of a job's log for a network-outage signal: the structured
4581
4901
  * `terminal_reason":"api_error"` field alongside a network-class error
4582
4902
  * string. This is NOT a real code defect — spawning an auto-fix
@@ -4996,10 +5316,16 @@ function stampIntegrationFailure(row, integration) {
4996
5316
  else delete row.integrationConflictPaths;
4997
5317
  if (integration.baseHeadSha) row.integrationBaseHeadSha = integration.baseHeadSha;
4998
5318
  else delete row.integrationBaseHeadSha;
5319
+ if (integration.expectedBranch) row.integrationExpectedBranch = integration.expectedBranch;
5320
+ else delete row.integrationExpectedBranch;
5321
+ if (integration.actualBranch) row.integrationActualBranch = integration.actualBranch;
5322
+ else delete row.integrationActualBranch;
4999
5323
  } else {
5000
5324
  delete row.integrationFailureKind;
5001
5325
  delete row.integrationConflictPaths;
5002
5326
  delete row.integrationBaseHeadSha;
5327
+ delete row.integrationExpectedBranch;
5328
+ delete row.integrationActualBranch;
5003
5329
  }
5004
5330
  }
5005
5331
 
@@ -5077,7 +5403,13 @@ function selectMechanicalRecoveryTarget(job, currentHeadSha = null) {
5077
5403
  if (job.mechanicalRecoveryAttempted === true) return null;
5078
5404
  if (isMechanicalRecoveryFutile(job, currentHeadSha)) return null;
5079
5405
  const cwd = job.cwd || DEFAULT_PROJECT_CWD;
5080
- return { slug: job.slug, cwd, branch: jobWorktree.branchNameFor(job.slug), carriedPaths: job.carriedPaths || [] };
5406
+ return {
5407
+ slug: job.slug,
5408
+ cwd,
5409
+ branch: jobWorktree.branchNameFor(job.slug),
5410
+ carriedPaths: job.carriedPaths || [],
5411
+ baseBranch: job.worktreeBaseBranch || null,
5412
+ };
5081
5413
  }
5082
5414
 
5083
5415
  /**
@@ -5096,7 +5428,7 @@ function selectMechanicalRecoveryTarget(job, currentHeadSha = null) {
5096
5428
  */
5097
5429
  async function performMechanicalRecovery(job, target) {
5098
5430
  const integration = await jobWorktree.integrateJobBranch({
5099
- cwd: target.cwd, branch: target.branch, slug: target.slug, carriedPaths: target.carriedPaths,
5431
+ cwd: target.cwd, branch: target.branch, slug: target.slug, carriedPaths: target.carriedPaths, baseBranch: target.baseBranch, allowRefLanding: true,
5100
5432
  });
5101
5433
  if (integration.ok) {
5102
5434
  await jobWorktree.cleanupJobWorktree({ cwd: target.cwd, dir: undefined, branch: target.branch, keepBranch: false });
@@ -5107,6 +5439,7 @@ async function performMechanicalRecovery(job, target) {
5107
5439
  if (!j) return;
5108
5440
  j.mechanicalRecoveryAttempted = true;
5109
5441
  if (integration.ok) {
5442
+ if (integration.viaRef) j.landedCommit = integration.sha;
5110
5443
  if (transitionJob(j, 'completed', {
5111
5444
  reason: `mechanical recovery: ${target.branch} re-integrated successfully`,
5112
5445
  source: 'scheduler:mechanicalRecovery',
@@ -5122,6 +5455,9 @@ async function performMechanicalRecovery(job, target) {
5122
5455
  }
5123
5456
  });
5124
5457
  if (integration.ok) {
5458
+ if (integration.viaRef) {
5459
+ console.log(`[scheduler] mechanical-recovery: ${job.slug} landed onto the base ref directly; the main checkout was busy on another branch and was never touched.`);
5460
+ }
5125
5461
  console.log(`[scheduler] mechanical-recovery: ${job.slug} → completed (branch ${target.branch} re-integrated)`);
5126
5462
  if (becameCompleted) await archiveCompletedPrd(job.slug, job.cwd);
5127
5463
  } else {
@@ -5165,22 +5501,11 @@ function selectLeftoverQuarantineTarget(job) {
5165
5501
  return { slug: job.slug, cwd: job.cwd, paths };
5166
5502
  }
5167
5503
 
5504
+ // Delegates to the shared gitExec.execGit (gitExec.cjs has zero requires of
5505
+ // scheduler.cjs/gitWorktree.cjs, so no cycle) — semantics match: execFile
5506
+ // with an argv array, same timeout default, same env-merge-over-process.env.
5168
5507
  function execGitAt(cwd, args, { env, timeout = 20_000 } = {}) {
5169
- return new Promise((resolve, reject) => {
5170
- execFile(
5171
- 'git',
5172
- ['-C', cwd, ...args],
5173
- { timeout, windowsHide: true, encoding: 'utf8', env: env ? { ...process.env, ...env } : process.env },
5174
- (err, stdout, stderr) => {
5175
- if (err) {
5176
- err.stderrText = stderr;
5177
- reject(err);
5178
- return;
5179
- }
5180
- resolve(stdout || '');
5181
- },
5182
- );
5183
- });
5508
+ return sharedGitExec.execGit(cwd, args, { timeout, env });
5184
5509
  }
5185
5510
 
5186
5511
  async function pathExistsInTree(cwd, treeish, p) {
@@ -5423,6 +5748,26 @@ async function performLeftoverQuarantine(job, paths, headBefore = null) {
5423
5748
  });
5424
5749
  }
5425
5750
 
5751
+ // A headless `claude -p` run has no later turn. A tool that waits for a
5752
+ // later turn (ScheduleWakeup, the Cron tools, Monitor) or for a human
5753
+ // (AskUserQuestion, plan mode) stalls the run: it ends with no commit and no
5754
+ // verdict, and the job parks in needs_review. EnterWorktree/ExitWorktree wait
5755
+ // on a worktree handoff that never comes back in a headless run either.
5756
+ // Agent and Skill stay allowed: the finish protocol runs /code-review and
5757
+ // /security-review, both of which dispatch sub-agents.
5758
+ const HEADLESS_DISALLOWED_TOOLS = Object.freeze([
5759
+ 'ScheduleWakeup',
5760
+ 'CronCreate',
5761
+ 'CronDelete',
5762
+ 'CronList',
5763
+ 'Monitor',
5764
+ 'AskUserQuestion',
5765
+ 'EnterPlanMode',
5766
+ 'ExitPlanMode',
5767
+ 'EnterWorktree',
5768
+ 'ExitWorktree',
5769
+ ]);
5770
+
5426
5771
  /**
5427
5772
  * Pure argv builder for a `claude -p` child spawn, shared so the
5428
5773
  * resume-vs-fresh-session choice is made in exactly one place. `resume`
@@ -5434,6 +5779,9 @@ async function performLeftoverQuarantine(job, paths, headBefore = null) {
5434
5779
  * persona body, resolved by agentModelResolve.cjs's resolvePrdPersonaForSpawn),
5435
5780
  * is passed as `--append-system-prompt` so the executor IS that persona at
5436
5781
  * launch rather than being asked in prose to adopt one.
5782
+ * `--disallowedTools` takes a variadic list, so another flag must always
5783
+ * follow it to end that list — `--output-format` does, right after. Never
5784
+ * put `--disallowedTools` last or right before the prompt.
5437
5785
  */
5438
5786
  function buildClaudeSpawnArgs({ prompt, model, effort, sessionId, resume, systemPrompt }) {
5439
5787
  return [
@@ -5442,6 +5790,8 @@ function buildClaudeSpawnArgs({ prompt, model, effort, sessionId, resume, system
5442
5790
  ...effortArgs(effort),
5443
5791
  ...(systemPrompt ? ['--append-system-prompt', systemPrompt] : []),
5444
5792
  '--dangerously-skip-permissions',
5793
+ ...headlessPermissionArgs(),
5794
+ '--disallowedTools', HEADLESS_DISALLOWED_TOOLS.join(','),
5445
5795
  '--output-format', 'stream-json',
5446
5796
  '--verbose',
5447
5797
  ...(resume ? ['--resume', sessionId] : ['--session-id', sessionId]),
@@ -5464,6 +5814,36 @@ function pickRunDir() {
5464
5814
  return { runId: ts, dir };
5465
5815
  }
5466
5816
 
5817
+ // macOS ships no `timeout` command, but a PRD gate command starts with
5818
+ // `timeout <seconds> ...`. ensureTimeoutShimOnce() installs a dependency-free
5819
+ // stand-in (see lib/timeoutShim.cjs) once per process — not once per job — so
5820
+ // most executeJob calls reuse the same cached promise instead of re-checking
5821
+ // the files on disk. A failed install clears that cache, so the NEXT spawn
5822
+ // retries the install instead of running shim-less for the rest of the
5823
+ // process. It never throws and never blocks or fails a spawn: a failed
5824
+ // install just logs one line and leaves the PATH addition pointing at a shim
5825
+ // dir that may have nothing in it yet, no worse than today's
5826
+ // no-shim-at-all. SM_TIMEOUT_SHIM_DISABLE=1 skips it outright.
5827
+ let timeoutShimEnsured = null;
5828
+ function ensureTimeoutShimOnce() {
5829
+ if (process.env.SM_TIMEOUT_SHIM_DISABLE === '1') return Promise.resolve();
5830
+ if (!timeoutShimEnsured) {
5831
+ timeoutShimEnsured = ensureTimeoutShim().then(
5832
+ (result) => {
5833
+ if (!result.ok && !result.skipped) {
5834
+ console.error(`[scheduler] could not install the timeout shim: ${result.error}`);
5835
+ timeoutShimEnsured = null;
5836
+ }
5837
+ },
5838
+ (err) => {
5839
+ console.error(`[scheduler] could not install the timeout shim: ${err?.message ?? err}`);
5840
+ timeoutShimEnsured = null;
5841
+ },
5842
+ );
5843
+ }
5844
+ return timeoutShimEnsured;
5845
+ }
5846
+
5467
5847
  /**
5468
5848
  * Execute a single PRD job. Writes stdout/stderr to a log file and a meta
5469
5849
  * JSON sidecar. Accepts an optional onPid(pid) callback called synchronously
@@ -5541,11 +5921,12 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5541
5921
  prompt = buildResumeRecoveryPreamble({ dirtyPaths: resumeTarget.dirtyPaths, salvagePatch: resumeTarget.salvagePatch });
5542
5922
  } else {
5543
5923
  // Read full PRD body fresh from disk (queue stored only the preview).
5544
- // Resolve through findPrdDir's full candidate search (legacy flat dir +
5545
- // every project's Epic-scoped dirs) first, so the common case — a live
5546
- // Epic-scoped PRD — is a first-try hit instead of probing the retired flat
5547
- // dir and only then falling back.
5548
- const resolvedDir = await findPrdDir(job.slug);
5924
+ // Resolve through findPrdDirForJob's full candidate search (legacy flat
5925
+ // dir + every project's Epic-scoped dirs, then a stored absolute prd path,
5926
+ // then an ancestor-folder Epic walk above job.cwd) first, so the common
5927
+ // case — a live Epic-scoped PRD — is a first-try hit instead of probing
5928
+ // the retired flat dir and only then falling back.
5929
+ const resolvedDir = await findPrdDirForJob(job);
5549
5930
  prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
5550
5931
  try {
5551
5932
  const parsed = await parsePrd(prdPath);
@@ -5559,8 +5940,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5559
5940
  // The project-scoped dir isn't the only place a PRD source can live — a
5560
5941
  // writer that hasn't migrated to prdLocations.cjs yet (or a not-yet-run
5561
5942
  // boot migration) can leave it in the legacy global dir. Fall back to
5562
- // findPrdDir's full candidate search before failing the job outright.
5563
- const fallbackDir = await findPrdDir(job.slug);
5943
+ // findPrdDirForJob's full candidate search before failing the job outright.
5944
+ const fallbackDir = await findPrdDirForJob(job);
5564
5945
  if (fallbackDir) {
5565
5946
  const fallbackPath = path.join(fallbackDir, `${job.slug}.md`);
5566
5947
  safeLog(`[scheduler] PRD not in project dir; found ${job.slug}.md in ${fallbackDir}\n`);
@@ -5695,6 +6076,13 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5695
6076
  const personaEffort = resolveEpicEffort({ cwd, agentType: job.agentType }).effort;
5696
6077
  safeLog(`[scheduler] agentType=${job.agentType || '(none)'} persona=${personaResolution.personaPath || '(fallback — no persona applied)'} model=${personaResolution.model}${personaEffort ? ` effort=${personaEffort}` : ''}\n`);
5697
6078
 
6079
+ // Authoritative guarantee that buildClaudeSpawnArgs (a pure sync helper,
6080
+ // called below via withChildAndLog) sees a resolved --permission-prompts
6081
+ // probe result — the module-load priming above is a warm-cache head start,
6082
+ // not a correctness guarantee on its own.
6083
+ await ensureCliCapsProbed();
6084
+ await ensureTimeoutShimOnce();
6085
+
5698
6086
  return await new Promise((resolve) => {
5699
6087
  const claudeBin = resolveClaudeBin();
5700
6088
  // Strip Claude Code env and secrets that leak in when session-manager is
@@ -5715,8 +6103,11 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5715
6103
  // `launchEnv` is the launch circuit breaker's degraded-mode env (e.g.
5716
6104
  // MAX_THINKING_TOKENS=0 while an outdated CLI's thinking parameter is
5717
6105
  // being rejected — lib/launchFailure.cjs); applied last so it wins.
6106
+ // SM_TIMEOUT_SHIM_DISABLE=1 is the kill switch for the PATH addition
6107
+ // below — paired with the install skip inside ensureTimeoutShimOnce().
6108
+ const timeoutShimEnabled = process.env.SM_TIMEOUT_SHIM_DISABLE !== '1';
5718
6109
  const childEnv = cleanChildEnv({
5719
- PATH: pathWithUserBins(),
6110
+ PATH: timeoutShimEnabled ? withTimeoutShimOnPath(pathWithUserBins()) : pathWithUserBins(),
5720
6111
  SM_PROJECT_ROOT: cwd,
5721
6112
  SM_SCHEDULER_JOB_SLUG: job.slug,
5722
6113
  SM_SCHEDULER_JOB_MAY_QUEUE: job.agentType === 'architect' ? '1' : '0',
@@ -5724,6 +6115,13 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5724
6115
  BASH_MAX_TIMEOUT_MS: String(BASH_MAX_TIMEOUT_MS),
5725
6116
  ...(launchEnv && typeof launchEnv === 'object' ? launchEnv : {}),
5726
6117
  });
6118
+ // Set AFTER cleanChildEnv returns, never inside the object passed to it:
6119
+ // cleanChildEnv strips every CLAUDE_CODE_*-prefixed key from its own
6120
+ // merged result (see cleanEnv.cjs), so setting it there would delete its
6121
+ // own addition. A headless run has no later turn, so a background task
6122
+ // it starts has nothing to report back to — same stall as a tool that
6123
+ // waits for one.
6124
+ childEnv.CLAUDE_CODE_DISABLE_BACKGROUND_TASKS = '1';
5727
6125
  if (launchEnv && Object.keys(launchEnv).length) {
5728
6126
  safeLog(`[scheduler] launch mitigation env applied: ${Object.entries(launchEnv).map(([k, v]) => `${k}=${v}`).join(' ')}\n`);
5729
6127
  }
@@ -6295,6 +6693,14 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
6295
6693
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is mechanical-recovery eligible`);
6296
6694
  return { deferred: false };
6297
6695
  }
6696
+ // stray_checkout is environmental (the project's own checkout sits off the
6697
+ // expected base branch) — no fix-plan PRD can ever land against it; a human
6698
+ // must fix the checkout. Authoring one anyway just re-hits the same refusal
6699
+ // on every retry, forever blocking dependents.
6700
+ if (failedJob.integrationFailureKind === 'stray_checkout') {
6701
+ console.log(`[scheduler] skip investigation: ${failedJob.slug} integration blocked by stray checkout (environmental)`);
6702
+ return { deferred: false };
6703
+ }
6298
6704
  if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth, failedJob.isFixPlan)) {
6299
6705
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
6300
6706
  return { deferred: false };
@@ -6404,18 +6810,51 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
6404
6810
  // the whole probe duration — previously this left the job's persisted
6405
6811
  // status frozen at 'failed'/'needs_review' the entire time, which read as
6406
6812
  // "nothing is happening" even though an Opus process was actively running.
6813
+ //
6814
+ // Live-row check, in the SAME mutate: `failedJob` can be a stale snapshot
6815
+ // (the deferred-investigation path keeps one from the moment the slot was
6816
+ // busy) and the row underneath it can have moved on — most commonly the
6817
+ // background gate shadow completed it in this same pass. Re-check the row
6818
+ // actually read here, not the snapshot the caller passed in: it must
6819
+ // exist, its live status must still be 'needs_review' or 'failed', and
6820
+ // transitionJob must accept the move. Fail closed on any of the three —
6821
+ // never spawn a probe for a row that is already finished or gone.
6822
+ let liveCheck = { ok: false, status: 'gone' };
6407
6823
  await mutate((s) => {
6408
6824
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
6409
- if (j) transitionJob(j, 'investigating', { reason: 'spawning investigation probe', source: 'spawnInvestigation:start' });
6825
+ if (!j) return;
6826
+ const statusEligible = j.status === 'needs_review' || j.status === 'failed';
6827
+ const transitioned = transitionJob(j, 'investigating', { reason: 'spawning investigation probe', source: 'spawnInvestigation:start' });
6828
+ liveCheck = { ok: statusEligible && transitioned, status: j.status };
6410
6829
  });
6830
+ if (!liveCheck.ok) {
6831
+ const msg = `[scheduler] skip investigation: ${failedJob.slug} is now ${liveCheck.status} — not probing`;
6832
+ console.log(msg);
6833
+ safeLog(`${msg}\n`);
6834
+ closeFd();
6835
+ releaseSlot();
6836
+ return { deferred: false };
6837
+ }
6411
6838
  await broadcast({ flush: true });
6412
6839
 
6840
+ await ensureCliCapsProbed();
6841
+ await ensureTimeoutShimOnce();
6413
6842
  const claudeBin = resolveClaudeBin();
6843
+ // SM_TIMEOUT_SHIM_DISABLE=1 is the kill switch for the PATH addition below —
6844
+ // paired with the install skip inside ensureTimeoutShimOnce().
6845
+ const timeoutShimEnabled = process.env.SM_TIMEOUT_SHIM_DISABLE !== '1';
6414
6846
  const childEnv = cleanChildEnv({
6415
- PATH: pathWithUserBins(), // Homebrew/user bins for macOS
6847
+ PATH: timeoutShimEnabled ? withTimeoutShimOnPath(pathWithUserBins()) : pathWithUserBins(), // Homebrew/user bins for macOS
6416
6848
  BASH_DEFAULT_TIMEOUT_MS: String(BASH_DEFAULT_TIMEOUT_MS),
6417
6849
  BASH_MAX_TIMEOUT_MS: String(BASH_MAX_TIMEOUT_MS),
6418
6850
  });
6851
+ // Set AFTER cleanChildEnv returns, never inside the object passed to it:
6852
+ // cleanChildEnv strips every CLAUDE_CODE_*-prefixed key from its own merged
6853
+ // result (see cleanEnv.cjs), so setting it there would delete its own
6854
+ // addition. A headless run has no later turn, so a background task it
6855
+ // starts has nothing to report back to — same stall as a tool that waits
6856
+ // for one.
6857
+ childEnv.CLAUDE_CODE_DISABLE_BACKGROUND_TASKS = '1';
6419
6858
 
6420
6859
  // Investigation needs only a deadman watchdog — no idle-tail or result-tail
6421
6860
  // since investigations are short-running Opus probes with a hard ceiling.
@@ -6443,6 +6882,8 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
6443
6882
  '-p', prompt,
6444
6883
  '--model', 'opus',
6445
6884
  '--dangerously-skip-permissions',
6885
+ ...headlessPermissionArgs(),
6886
+ '--disallowedTools', HEADLESS_DISALLOWED_TOOLS.join(','),
6446
6887
  '--output-format', 'stream-json',
6447
6888
  '--verbose',
6448
6889
  '--session-id', sessionId,
@@ -6823,7 +7264,7 @@ async function finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPat
6823
7264
  console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
6824
7265
  }
6825
7266
  }
6826
- const integration = await jw.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
7267
+ const integration = await jw.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths, baseBranch: worktree.baseBranch });
6827
7268
  if (integration.ok && integration.reason === 'carried-wip-only') {
6828
7269
  console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
6829
7270
  }
@@ -7071,6 +7512,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null, sib
7071
7512
  source: 'spawnJob:dispatch',
7072
7513
  });
7073
7514
  delete s.jobs[idx].heldReason;
7515
+ // A fresh dispatch means any gateShadow, looksDone, or
7516
+ // evidenceScannedAt on this row was computed for an earlier
7517
+ // runId/startedAt — never let any of them linger and read as this
7518
+ // run's result (looksDone stale here would let rung 6 complete this
7519
+ // run on the PREVIOUS run's evidence).
7520
+ delete s.jobs[idx].gateShadow;
7521
+ delete s.jobs[idx].looksDone;
7522
+ delete s.jobs[idx].evidenceScannedAt;
7074
7523
  s.jobs[idx].runId = runId;
7075
7524
  s.jobs[idx].startedAt = new Date().toISOString();
7076
7525
  // Dispatch-phase breadcrumb (PRD: dispatch-region diagnostic
@@ -7202,12 +7651,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null, sib
7202
7651
  // recorded on the job row so integration can exclude these paths from
7203
7652
  // the branch diff below, and so it's queryable from the queue.
7204
7653
  const carriedPaths = (worktree.ok && Array.isArray(worktree.carriedPaths)) ? worktree.carriedPaths : [];
7205
- if (carriedPaths.length) {
7206
- await mutate((s) => {
7207
- const idx = s.jobs.findIndex((x) => x.slug === job.slug);
7208
- if (idx >= 0) s.jobs[idx].carriedPaths = carriedPaths;
7209
- });
7210
- }
7654
+ // baseBranch (createWorktree, this PRD) — the branch `cwd` was actually on
7655
+ // when the worktree forked, threaded onto the job row so a later mechanical
7656
+ // recovery retries integration against the SAME target, not whatever `cwd`
7657
+ // happens to sit on by then.
7658
+ const worktreeBaseBranch = (worktree.ok && typeof worktree.baseBranch === 'string' && worktree.baseBranch) ? worktree.baseBranch : null;
7659
+ await mutate((s) => {
7660
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
7661
+ if (idx < 0) return;
7662
+ if (carriedPaths.length) s.jobs[idx].carriedPaths = carriedPaths;
7663
+ if (worktreeBaseBranch) s.jobs[idx].worktreeBaseBranch = worktreeBaseBranch;
7664
+ else delete s.jobs[idx].worktreeBaseBranch;
7665
+ });
7211
7666
 
7212
7667
  // Integrate the job's branch back into guardCwd's own HEAD, THEN tear the
7213
7668
  // worktree checkout down — both must happen BEFORE any git read below
@@ -8094,14 +8549,56 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null, sib
8094
8549
  annotations: needsReviewRcaSnapshot.verifierAnnotations,
8095
8550
  })
8096
8551
  .then(async (report) => {
8097
- // Persist the classification onto the parked job row so the scheduler
8098
- // can route on it (e.g. selectAutoFixTargets excluding 'archive')
8099
- // without re-parsing the RCA markdown on every pass.
8552
+ // Resolve the authoring Epic the same way notifyNeedsReview does —
8553
+ // the review notice needs the same epicId so flushDueReviewNotices
8554
+ // can route the eventual grouped message to the right place.
8555
+ let epicId = null;
8556
+ try {
8557
+ const prd = await resolveNotifyPrd(needsReviewRcaSnapshot, prdParser.parsePrdRaw);
8558
+ epicId = prd?.sourcePromptId || needsReviewRcaSnapshot.epicId || null;
8559
+ } catch {
8560
+ epicId = null;
8561
+ }
8562
+ // Persist the RCA classification AND the review notice onto the
8563
+ // parked job row in the SAME mutate — the scheduler routes on the
8564
+ // classification (e.g. selectAutoFixTargets excluding 'archive')
8565
+ // and flushDueReviewNotices routes on the notice, neither re-parsing
8566
+ // the RCA markdown nor losing the notice to a lost race.
8567
+ let builtNotice = null;
8100
8568
  await mutate((s) => {
8101
8569
  const j = s.jobs.find((x) => x.slug === needsReviewRcaSnapshot.slug);
8102
8570
  applyRcaClassification(j, report);
8103
- }).catch(() => {});
8104
- return notifyNeedsReview(needsReviewRcaSnapshot, report);
8571
+ // Write the notice only while the row is still the SAME
8572
+ // needs_review episode this RCA was filed for — a row a human
8573
+ // already reset, or that moved on before this async write
8574
+ // lands, must not get a stale notice grafted back onto it.
8575
+ if (j && j.status === 'needs_review') {
8576
+ j.reviewNotice = buildReviewNotice({ job: j, report, epicId, now: new Date().toISOString(), prior: j.reviewNotice });
8577
+ builtNotice = j.reviewNotice;
8578
+ }
8579
+ }).catch((e) => {
8580
+ console.error('[scheduler] writeRcaReport reviewNotice mutate error', needsReviewRcaSnapshot.slug, e);
8581
+ });
8582
+ // Quiet by default (this PRD): the notice recorded above waits for
8583
+ // flushDueReviewNotices to send it, grouped, once the ladder gives
8584
+ // up or the hold time passes. SM_REVIEW_NOTICE_IMMEDIATE=1 restores
8585
+ // the old immediate-notify behavior, for local debugging.
8586
+ if (process.env.SM_REVIEW_NOTICE_IMMEDIATE === '1') {
8587
+ const sent = await notifyNeedsReview(needsReviewRcaSnapshot, report);
8588
+ if (sent && builtNotice) {
8589
+ await mutate((s) => {
8590
+ const j = s.jobs.find((x) => x.slug === needsReviewRcaSnapshot.slug);
8591
+ // Stamp only if the live notice is still the exact one just
8592
+ // sent — a re-park between the send and this mutate must not
8593
+ // be marked sent for a notice that never went out.
8594
+ if (j && j.reviewNotice && j.reviewNotice.firstParkedAt === builtNotice.firstParkedAt && j.reviewNotice.cause === builtNotice.cause) {
8595
+ j.reviewNotice.sentAt = new Date().toISOString();
8596
+ }
8597
+ }).catch((e) => {
8598
+ console.error('[scheduler] writeRcaReport immediate-stamp mutate error', needsReviewRcaSnapshot.slug, e);
8599
+ });
8600
+ }
8601
+ }
8105
8602
  })
8106
8603
  .catch((e) => {
8107
8604
  console.error('[scheduler] writeRcaReport error', job.slug, e);
@@ -8483,7 +8980,13 @@ async function tickBody(gen, { bypassLoadGate }) {
8483
8980
  if (batch.length === 0) {
8484
8981
  // Queue drained — run the definition-of-done gate fire-and-forget.
8485
8982
  // Non-blocking: does not hold the mutate lock; errors are logged, not thrown.
8486
- runDefinitionOfDoneOnDrain(state, { cancelToken }).catch((err) => {
8983
+ // resolvePrdPath mirrors every other verify call site in this file (e.g.
8984
+ // computeLooksDone, runGateShadow): try the live Epic-scoped dir first,
8985
+ // then the archived twin — never the retired flat dir.
8986
+ runDefinitionOfDoneOnDrain(state, {
8987
+ cancelToken,
8988
+ resolvePrdPath: async (job) => (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job),
8989
+ }).catch((err) => {
8487
8990
  console.log(`[scheduler] dod-drain: ${err?.message ?? String(err)}`);
8488
8991
  });
8489
8992
  if (holdReason) return recordTick({ fired: false, reason: 'held', detail: holdReason }, { holds });
@@ -10878,22 +11381,58 @@ async function computeLooksDone(job, fetchedCwds) {
10878
11381
  return { commits: attributed.commits, paths, detectedAt: new Date().toISOString(), rule: attributed.rule };
10879
11382
  }
10880
11383
 
11384
+ // Dirty, non-ignored tracked+untracked paths at `cwd`, app-owned churn
11385
+ // stripped (stripAppOwnedChurn) — null on any git failure, never thrown.
11386
+ // runGateShadow calls this once before the gate and once after, so it can
11387
+ // tell a tree that was clean throughout from one a foreign write (or the
11388
+ // gate command itself) touched mid-gate. Deliberately no
11389
+ // `--untracked-files=no`: an uncommitted new file left by a shared-tree job
11390
+ // must count as dirty here (see decideGateAuthority's cleanOk).
11391
+ async function gateTreeDirt(cwd) {
11392
+ try {
11393
+ return stripAppOwnedChurn(parsePorcelain(await execGitAt(cwd, ['status', '--porcelain'])));
11394
+ } catch {
11395
+ return null;
11396
+ }
11397
+ }
11398
+
10881
11399
  /**
10882
- * Shadow gate (observation only): run a needs_review row's authored gate at
10883
- * the project's current HEAD and record what it WOULD have decided as
10884
- * `gateShadow` on the verdicts sidecar and the row. Changes NO status, takes
10885
- * no slot (not a claude -p run — runGateSequence keeps one shadow gate in
10886
- * flight machine-wide). Never called from finalize: only the reverify pass.
10887
- * Returns the recorded gateShadow, or null when nothing was recorded (already
10888
- * recorded at this HEAD, PRD unreadable, or another shadow gate is running).
11400
+ * Shadow gate: run a needs_review row's authored gate at the project's
11401
+ * current HEAD and record what it decided as `gateShadow`, on the verdicts
11402
+ * sidecar and the row. For every verdict this is observation only — status
11403
+ * never changes from it alone. For the three verdicts in
11404
+ * GATE_AUTHORITY_VERDICTS, with a landed commit that is git-verified evidence
11405
+ * from THIS dispatch (never an older run's stale sha — see
11406
+ * resolveLandedCommitEvidence) that is also an ancestor of HEAD recorded
11407
+ * before the gate ran, a green re-run against that EXACT tree (HEAD unmoved,
11408
+ * tracked+untracked tree clean, both before and after the gate ran) is gate
11409
+ * AUTHORITY: it completes the row with no human in the loop. Any other
11410
+ * result — stale/missing evidence, the commit not an ancestor, a dirty tree,
11411
+ * HEAD moving mid-gate, no gate, or a non-green outcome — leaves status
11412
+ * untouched.
11413
+ *
11414
+ * Takes no slot (not a claude -p run — runGateSequence keeps one shadow gate
11415
+ * in flight machine-wide). Never called from finalize: only the reverify
11416
+ * pass. `gateShadow.definitive` marks whether this result is settled: true
11417
+ * only for a clean-throughout, HEAD-unmoved run with a real pass/fail outcome
11418
+ * — the only shape worth trusting indefinitely. Anything else (a dirty tree,
11419
+ * a moved HEAD, or an unavailable outcome) is recorded but re-run on the very
11420
+ * next pick instead of being memoized as settled.
11421
+ *
11422
+ * Returns the recorded gateShadow, or null when nothing was recorded (HEAD
11423
+ * unreadable — there is no tree to bind the result to —, already recorded
11424
+ * definitively at this HEAD, PRD unreadable, or another shadow gate is
11425
+ * running).
10889
11426
  */
10890
11427
  async function runGateShadow(job) {
10891
11428
  if (!job || !job.slug || !job.cwd) return null;
10892
11429
  const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
10893
11430
  let prdText;
10894
11431
  try { prdText = fs.readFileSync(prdPath, 'utf8'); } catch { return null; }
10895
- const head = await gitHead(job.cwd);
10896
- if (job.gateShadow && job.gateShadow.head === head) return null;
11432
+ const headBefore = await gitHead(job.cwd);
11433
+ if (!headBefore) return null; // no tree to bind the result to
11434
+ if (job.gateShadow && job.gateShadow.head === headBefore && job.gateShadow.definitive === true) return null;
11435
+ const dirtBefore = await gateTreeDirt(job.cwd);
10897
11436
  const gate = resolveGate(prdText);
10898
11437
  let outcome;
10899
11438
  if (gate.source === 'none') outcome = { status: 'unavailable', reason: 'gate-opt-out', results: [] };
@@ -10903,7 +11442,48 @@ async function runGateShadow(job) {
10903
11442
  if (r.status === 'busy') return null;
10904
11443
  outcome = r;
10905
11444
  }
10906
- const gateShadow = { ...outcome, head, source: gate.source, ranAt: new Date().toISOString() };
11445
+ const headAfter = await gitHead(job.cwd);
11446
+ const dirtAfter = await gateTreeDirt(job.cwd);
11447
+ const headStable = headAfter !== null && headAfter === headBefore;
11448
+ const cleanOk = Array.isArray(dirtBefore) && dirtBefore.length === 0
11449
+ && Array.isArray(dirtAfter) && dirtAfter.length === 0;
11450
+ // "A real pass or fail" — an unavailable outcome (opt-out, no parseable
11451
+ // gate) is never definitive even at a clean, unmoved tree: it says nothing
11452
+ // about done-ness, so it must stay eligible for retry once the PRD gains a
11453
+ // real gate. `busy` never reaches here (returns early above).
11454
+ const definitive = headStable && cleanOk && (outcome.status === 'green' || outcome.status === 'red');
11455
+ const gateShadow = { ...outcome, head: headBefore, source: gate.source, ranAt: new Date().toISOString(), definitive };
11456
+
11457
+ // Gate authority: a green re-run may complete the row outright, but only
11458
+ // for transcript-noise verdicts with a landed commit — every other park
11459
+ // (including no verdict at all) falls through to the plain observe-only
11460
+ // path above, unchanged. evidenceOk/ancestorOk/cleanOk/headStable are all
11461
+ // ground-truth git checks against job.startedAt and headBefore — never
11462
+ // trusted from anything the job itself reported.
11463
+ let decision = null;
11464
+ if (
11465
+ job.verifierVerdict
11466
+ && GATE_AUTHORITY_VERDICTS.includes(job.verifierVerdict)
11467
+ && typeof job.landedCommit === 'string'
11468
+ && job.landedCommit.length > 0
11469
+ ) {
11470
+ // resolveLandedCommitEvidence skips its own time check when sinceIso is
11471
+ // empty — so a missing/unparseable startedAt must refuse here, never
11472
+ // fall through to "no time bound at all".
11473
+ const startedAtMs = typeof job.startedAt === 'string' ? Date.parse(job.startedAt) : NaN;
11474
+ let evidenceOk = false;
11475
+ if (Number.isFinite(startedAtMs)) {
11476
+ try { evidenceOk = await resolveLandedCommitEvidence(job.cwd, job.landedCommit, job.startedAt); } catch { evidenceOk = false; }
11477
+ }
11478
+ let ancestorOk = false;
11479
+ try {
11480
+ await execGitAt(job.cwd, ['merge-base', '--is-ancestor', job.landedCommit, headBefore], { timeout: 10_000 });
11481
+ ancestorOk = true;
11482
+ } catch { ancestorOk = false; }
11483
+ decision = decideGateAuthority({ job, gate, outcome, evidenceOk, ancestorOk, cleanOk, headStable });
11484
+ gateShadow.authority = decision;
11485
+ }
11486
+
10907
11487
  const runId = job.runId || resolveRunId(job);
10908
11488
  if (runId) {
10909
11489
  const verdictsPath = path.join(schedulerPaths.runsDir(), runId, `${job.slug}.verdicts.json`);
@@ -10915,17 +11495,121 @@ async function runGateShadow(job) {
10915
11495
  try { atomicWriteJsonSync(verdictsPath, { ...existing, gateShadow }); } catch { /* best-effort */ }
10916
11496
  }
10917
11497
  }
11498
+
11499
+ const headShort = headBefore ? headBefore.slice(0, 7) : 'unknown';
11500
+
11501
+ // Resolve inside the SAME mutate that records the shadow, re-checking the
11502
+ // live row still matches the snapshot `job` on status, verdict, runId AND
11503
+ // startedAt — a park can change underneath this async function (heal,
11504
+ // human reset, a fresh dispatch of the same slug) between the gate run and
11505
+ // this write landing, and any such change invalidates both the plain
11506
+ // observation and the authority decision alike, so the whole result is
11507
+ // dropped rather than written onto a row it was never computed for.
11508
+ let resolved = null;
10918
11509
  await mutate((s) => {
10919
11510
  for (const j of s.jobs) {
10920
- if (j.slug === job.slug && j.status === 'needs_review') j.gateShadow = gateShadow;
11511
+ if (j.slug !== job.slug) continue;
11512
+ const unchanged = j.status === 'needs_review'
11513
+ && j.verifierVerdict === job.verifierVerdict
11514
+ && j.runId === job.runId
11515
+ && j.startedAt === job.startedAt;
11516
+ if (!unchanged) {
11517
+ console.log(`[scheduler] gate authority: ${job.slug} result dropped — row changed under the gate`);
11518
+ continue;
11519
+ }
11520
+ j.gateShadow = gateShadow;
11521
+ if (!decision?.complete) continue;
11522
+ const v = j.verifierVerdict;
11523
+ transitionJob(j, 'completed', {
11524
+ reason: `gate re-run green at ${headShort}; verdict ${v} overridden`,
11525
+ source: 'gateAuthoritative',
11526
+ });
11527
+ j.error = null;
11528
+ delete j.verifierVerdict;
11529
+ delete j.looksDone;
11530
+ resolved = { slug: j.slug, cwd: j.cwd, verdict: v };
10921
11531
  }
10922
11532
  });
11533
+ if (resolved) {
11534
+ try { await archiveCompletedPrd(resolved.slug, resolved.cwd); } catch (e) {
11535
+ console.error('[scheduler] gate authority archive error', resolved.slug, e);
11536
+ }
11537
+ appendAuditEvent('needs_review_gate_resolved', { slug: resolved.slug, cwd: resolved.cwd, verdict: resolved.verdict, head: headBefore });
11538
+ console.log(`[scheduler] gate authority: ${resolved.slug} completed — gate re-run green at ${headShort}, verdict ${resolved.verdict} overridden`);
11539
+ }
10923
11540
  await broadcast();
10924
11541
  return gateShadow;
10925
11542
  }
10926
11543
 
10927
11544
  // Tail of the last background shadow gate — lets tests (and only tests) await it.
10928
11545
  let gateShadowPending = null;
11546
+ // The slug the background shadow gate is currently deciding, from the pick
11547
+ // below until that gate run finishes. Lets the auto-fix loop (and the
11548
+ // deferred-investigation drain) skip this one row while the gate still has
11549
+ // it — one row gets one recovery attempt at a time.
11550
+ let gateShadowSlug = null;
11551
+
11552
+ /**
11553
+ * Pure: picks the one needs_review row reverifyNeedsReview's background
11554
+ * shadow gate should run this pass, or null when none qualifies. No I/O:
11555
+ * `headByCwd` is a Map of cwd -> HEAD sha (or null), pre-fetched by the
11556
+ * caller with `gitHead` (which never rejects, so building it needs no
11557
+ * try/catch — a cwd whose read failed just maps to null here).
11558
+ *
11559
+ * A row is never picked when its cwd's HEAD is unknown (null) — there is
11560
+ * nothing to compare a recorded gateShadow against.
11561
+ *
11562
+ * Tier 1, skipped when `authorityDisabled`: a stale, gate-authority-eligible
11563
+ * row — verdict in GATE_AUTHORITY_VERDICTS, a non-empty landedCommit, and
11564
+ * stale (no gateShadow, or one at a different HEAD, or one not marked
11565
+ * `definitive`). Ordered so a row that never got a shadow run goes first,
11566
+ * then the oldest `gateShadow.ranAt` (an unparseable or missing `ranAt`
11567
+ * sorts as oldest), then queue order. Rule first, why second: the old pick
11568
+ * took the first stale row in queue order with no rotation, so in a busy
11569
+ * plan one row whose gate stays red forever could crowd out every other
11570
+ * row's turn (review finding #8).
11571
+ *
11572
+ * Tier 2: the first needs_review row with no gateShadow at all, in queue
11573
+ * order — today's plain observe-only rule, unchanged.
11574
+ *
11575
+ * Returns the first tier-1 row, else the tier-2 row, else null.
11576
+ */
11577
+ function selectGateShadowTarget(jobs, headByCwd, { authorityDisabled = false } = {}) {
11578
+ const eligible = jobs.filter((j) => {
11579
+ if (j.status !== 'needs_review') return false;
11580
+ const head = headByCwd.get(j.cwd);
11581
+ return typeof head === 'string' && head.length > 0;
11582
+ });
11583
+
11584
+ const isStale = (j) => {
11585
+ const head = headByCwd.get(j.cwd);
11586
+ return !j.gateShadow || j.gateShadow.head !== head || j.gateShadow.definitive !== true;
11587
+ };
11588
+
11589
+ if (!authorityDisabled) {
11590
+ const ranAtMs = (j) => {
11591
+ const t = j.gateShadow ? Date.parse(j.gateShadow.ranAt) : NaN;
11592
+ return Number.isFinite(t) ? t : -Infinity; // never-run or unparseable sorts oldest
11593
+ };
11594
+ const tier1 = eligible.filter((j) => isStale(j)
11595
+ && j.verifierVerdict
11596
+ && GATE_AUTHORITY_VERDICTS.includes(j.verifierVerdict)
11597
+ && typeof j.landedCommit === 'string'
11598
+ && j.landedCommit.length > 0);
11599
+ if (tier1.length) {
11600
+ const ranked = tier1
11601
+ .map((j, i) => ({ j, i, neverRun: !j.gateShadow, ranAt: ranAtMs(j) }))
11602
+ .sort((a, b) => {
11603
+ if (a.neverRun !== b.neverRun) return a.neverRun ? -1 : 1;
11604
+ if (a.ranAt !== b.ranAt) return a.ranAt - b.ranAt;
11605
+ return a.i - b.i; // tie: queue order
11606
+ });
11607
+ return ranked[0].j;
11608
+ }
11609
+ }
11610
+
11611
+ return eligible.find((j) => !j.gateShadow) ?? null;
11612
+ }
10929
11613
 
10930
11614
  async function reverifyNeedsReview() {
10931
11615
  const snap = await readQueue();
@@ -11053,15 +11737,48 @@ async function reverifyNeedsReview() {
11053
11737
  }
11054
11738
  }
11055
11739
  }
11056
- // Shadow gate (observation only): at most ONE needs_review row per pass,
11057
- // fired in the background so a 15-minute gate never stalls this pass.
11740
+ // Shadow gate (observation only, or — for a transcript-noise verdict with
11741
+ // a landed commit — gate AUTHORITY: a green re-run completes the row). At
11742
+ // most one needs_review row per pass, fired in the background so a long
11743
+ // gate never stalls this pass. selectGateShadowTarget does the picking: a
11744
+ // never-run or stale authority-eligible row first (oldest shadow run
11745
+ // first), else the first row that never got a shadow run at all — so one
11746
+ // row whose gate stays red forever can't crowd out every other row's turn.
11747
+ // gateShadowSlug names the picked row for as long as its gate is running,
11748
+ // so the auto-fix loop below (and the deferred-investigation drain) can
11749
+ // leave that one row alone — the gate is cheap and goes first; a probe
11750
+ // only gets the row if the gate didn't resolve it.
11751
+ //
11752
+ // The pick itself is async (it reads HEAD per candidate's cwd), so it runs
11753
+ // inside its own promise, `pick`, resolved before `gateShadowPending`
11754
+ // chains the gate run after it. `gateShadowPending` is still assigned
11755
+ // synchronously, before any await, exactly as before — otherwise a second
11756
+ // pass starting before this one's pick resolves could choose a target of
11757
+ // its own too, breaking the single-flight rule.
11758
+ let pick = null;
11058
11759
  if (!gateShadowPending && process.env.SM_GATE_SHADOW_DISABLE !== '1') {
11059
- const gateTarget = snap.jobs.find((j) => j.status === 'needs_review' && !j.gateShadow);
11060
- if (gateTarget) {
11061
- gateShadowPending = runGateShadow(gateTarget)
11062
- .catch((e) => { console.error('[scheduler] gate shadow error', gateTarget.slug, e); })
11063
- .finally(() => { gateShadowPending = null; });
11064
- }
11760
+ pick = (async () => {
11761
+ const needsReview = snap.jobs.filter((j) => j.status === 'needs_review');
11762
+ const headByCwd = new Map();
11763
+ for (const cwd of new Set(needsReview.map((j) => j.cwd))) {
11764
+ headByCwd.set(cwd, await gitHead(cwd));
11765
+ }
11766
+ return selectGateShadowTarget(snap.jobs, headByCwd, {
11767
+ authorityDisabled: process.env.SM_GATE_AUTHORITATIVE_DISABLE === '1',
11768
+ });
11769
+ })();
11770
+ gateShadowPending = pick
11771
+ .then((gateTarget) => {
11772
+ if (!gateTarget) return;
11773
+ gateShadowSlug = gateTarget.slug;
11774
+ return runGateShadow(gateTarget);
11775
+ })
11776
+ .catch((e) => { console.error('[scheduler] gate shadow error', gateShadowSlug, e); })
11777
+ .finally(() => {
11778
+ gateShadowPending = null;
11779
+ gateShadowSlug = null;
11780
+ drainDeferredInvestigation();
11781
+ });
11065
11782
  }
11066
11783
  if (evidenceScanned.length) {
11067
11784
  const scannedSet = new Set(evidenceScanned);
@@ -11285,9 +12002,14 @@ async function reverifyNeedsReview() {
11285
12002
  // MAX_CONCURRENT_INVESTIGATIONS (queues the rest for retry), so this loop
11286
12003
  // cannot fan out past the cap regardless of how many targets are selected.
11287
12004
  if (process.env.SM_AUTOFIX_DISABLE !== '1') {
12005
+ // If this pass started a gate-shadow pick above, wait for it — it only
12006
+ // reads HEADs, so this is a short wait, not the gate run itself — then
12007
+ // drop its target from this loop's candidates. The gate gets first shot
12008
+ // at that one row; a probe only takes it if the gate didn't resolve it.
12009
+ if (pick) await pick;
11288
12010
  const targets = selectAutoFixTargets(queueForResumeAndAutofix.jobs, {
11289
12011
  fixSlugExists: (s) => candidatePrdsDirs().some((dir) => fs.existsSync(path.join(dir, `${s}.md`))),
11290
- });
12012
+ }).filter((job) => job.slug !== gateShadowSlug);
11291
12013
  for (const job of targets) {
11292
12014
  const runId = job.runId || resolveRunId(job);
11293
12015
  const runDir = path.join(schedulerPaths.runsDir(), runId);
@@ -11505,22 +12227,25 @@ function registerScheduleHandlers() {
11505
12227
  return { ok: true, config };
11506
12228
  }));
11507
12229
 
11508
- ipcMain.handle('schedule:reset-job', validated(schemas.scheduleSlug, async ({ slug }) => {
12230
+ ipcMain.handle('schedule:reset-job', validated(schemas.scheduleResetJob, async ({ slug, cwd }) => {
11509
12231
  if (!(await safeSlugPath(slug))) return { ok: false, error: 'invalid slug' };
11510
12232
  const outcome = await mutate((state) => {
11511
- const idx = state.jobs.findIndex((j) => j.slug === slug);
11512
- if (idx < 0) return 'not-found';
12233
+ const idx = state.jobs.findIndex((j) => jobMatchesSlugAndCwd(j, slug, cwd));
12234
+ if (idx < 0) return { kind: 'not-found' };
11513
12235
  // Guard is in resetJobFields: refuses to reset an already-'completed'
11514
- // job, which would otherwise re-fire a PRD whose deliverable already
11515
- // landed (see resetJobFields' doc comment for the incident).
11516
- return resetJobFields(state.jobs[idx], null, { source: 'ipc:schedule:reset-job' }) ? 'ok' : 'refused';
12236
+ // or already-'skipped' job without force:true (see its doc comment for
12237
+ // why each is refused). This handler never passes force — there is no
12238
+ // force option on the renderer IPC path — so resetRefusalMessage's
12239
+ // canForce:false tells the caller to use scheduler_reset_job instead.
12240
+ const status = state.jobs[idx].status;
12241
+ if (!resetJobFields(state.jobs[idx], null, { source: 'ipc:schedule:reset-job' })) {
12242
+ return { kind: 'refused', status };
12243
+ }
12244
+ return { kind: 'ok' };
11517
12245
  });
11518
- if (outcome === 'not-found') return { ok: false, error: 'not found' };
11519
- if (outcome === 'refused') {
11520
- return {
11521
- ok: false,
11522
- error: 'job already completed — resetting it would re-execute shipped work; archive the PRD instead',
11523
- };
12246
+ if (outcome.kind === 'not-found') return { ok: false, error: 'not found' };
12247
+ if (outcome.kind === 'refused') {
12248
+ return { ok: false, error: resetRefusalMessage(outcome.status, { canForce: false }) };
11524
12249
  }
11525
12250
  await broadcast({ flush: true });
11526
12251
  return { ok: true };
@@ -11952,7 +12677,18 @@ function rescheduleIntervalTick() {
11952
12677
  );
11953
12678
  }
11954
12679
  }
11955
- }).catch(() => {});
12680
+ }).catch(() => {})
12681
+ .finally(() => {
12682
+ // An auto-resolve skip just made a notice due in THIS pass — flush
12683
+ // right away instead of waiting for the next one. Fire-and-forget:
12684
+ // never blocks this tick, never throws (flushDueReviewNotices
12685
+ // catches internally).
12686
+ flushDueReviewNotices().catch(() => {});
12687
+ });
12688
+ } else if (s.jobs.some((j) => j.reviewNotice && !j.reviewNotice.sentAt
12689
+ && (j.status === 'needs_review' || (j.status === 'skipped' && j.needsReviewAutoResolvedSkip)))) {
12690
+ // Ladder didn't fire this tick (row ineligible for auto-resolve, or SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1) — flush any notice whose hold already expired, reusing this tick's own snapshot for the guard so a tick with nothing unsent reads nothing extra. Status-checked so a row that healed (e.g. a human reset, or a later run that completed) with a leftover unsent notice from a past episode no longer costs a queue read every tick — resetJobFields already deletes the notice outright, but this guard stays defensive in case a notice is ever left behind some other way.
12691
+ flushDueReviewNotices().catch(() => {});
11956
12692
  }
11957
12693
  }
11958
12694
 
@@ -12357,7 +13093,8 @@ const remote = {
12357
13093
  // realpath resolves symlinks; re-check boundary to block a rogue agent job
12358
13094
  // that places a symlink inside the PRDs dir pointing outside the safe root.
12359
13095
  const real = await fsp.realpath(filePath);
12360
- if (!real.startsWith(dir + path.sep)) {
13096
+ const realDir = await realPrdsDir(dir);
13097
+ if (!realDir || !real.startsWith(realDir + path.sep)) {
12361
13098
  return { ok: false, error: 'invalid slug' };
12362
13099
  }
12363
13100
  const text = await fsp.readFile(real, 'utf8');
@@ -12434,7 +13171,8 @@ const remote = {
12434
13171
  // re-assert containment; also reject the target if it is already a
12435
13172
  // symlink.
12436
13173
  const realParent = await fsp.realpath(path.dirname(resolved));
12437
- if (realParent !== dir && !realParent.startsWith(dir + path.sep)) {
13174
+ const realDir = await realPrdsDir(dir);
13175
+ if (!realDir || (realParent !== realDir && !realParent.startsWith(realDir + path.sep))) {
12438
13176
  return { ok: false, error: 'invalid slug' };
12439
13177
  }
12440
13178
  const existing = await fsp.lstat(resolved).catch(() => null);
@@ -12453,6 +13191,19 @@ const remote = {
12453
13191
  }
12454
13192
  },
12455
13193
 
13194
+ // Triggers an immediate reconcile pass through the SAME seam resetJob
13195
+ // already uses below (broadcast's coalescer, whose getPayload runs
13196
+ // module.exports.reconcile) — no second reconcile implementation. PRD
13197
+ // 1446: prdCreate.cjs's createPrd() calls this right after a successful
13198
+ // write so a fresh PRD becomes a pending queue row without waiting for
13199
+ // the next scheduled pass — the 60s pollLoop only reaches reconcile() via
13200
+ // maybeLaunchWhenAvailable, which returns early while zero rows are
13201
+ // pending, and the 10-minute rescheduleTimer can itself stall behind a
13202
+ // slow/hung billing fetch (see rescheduleTimer's own bounded race below).
13203
+ async requestReconcile() {
13204
+ await broadcast({ flush: true });
13205
+ },
13206
+
12456
13207
  // User-initiated pause/resume — the admin-route/MCP twins of the
12457
13208
  // schedule:pause / schedule:resume IPC handlers, through the same setPaused /
12458
13209
  // clearPause. Pause stops NEW dispatch only; running jobs are never touched.
@@ -12472,25 +13223,23 @@ const remote = {
12472
13223
  return { ok: false, error: resolved.reason === 'invalid-slug' ? 'invalid slug' : unknownSlugMessage(slug) };
12473
13224
  }
12474
13225
  const outcome = await mutate((state) => {
12475
- // Same cwd filter as resolveSlugOrReason's file lookup above — slugs are
12476
- // derived from title text with no cwd salt, so two different projects
12477
- // can independently produce the identical slug; an opts.cwd caller must
12478
- // reset THAT project's job, not just any queue row matching the string.
12479
- const idx = state.jobs.findIndex((j) => j.slug === slug && (!opts.cwd || j.cwd === opts.cwd));
13226
+ // Same cwd filter as the schedule:reset-job IPC handler above and
13227
+ // resolveSlugOrReason's file lookup — see jobMatchesSlugAndCwd's header.
13228
+ const idx = state.jobs.findIndex((j) => jobMatchesSlugAndCwd(j, slug, opts.cwd));
12480
13229
  if (idx < 0) return { kind: 'not-found' };
12481
13230
  // Terminal-status guard lives in resetJobFields itself; force:true
12482
- // threads through to override it.
13231
+ // threads through to override it. Capture the pre-reset status here
13232
+ // (inside the same mutate callback) so a refusal can report exactly
13233
+ // which status blocked it, via resetRefusalMessage below.
13234
+ const status = state.jobs[idx].status;
12483
13235
  if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true, source: 'remote:resetJob' })) {
12484
- return { kind: 'refused' };
13236
+ return { kind: 'refused', status };
12485
13237
  }
12486
13238
  return { kind: 'ok' };
12487
13239
  });
12488
13240
  if (outcome.kind === 'not-found') return { ok: false, error: 'not found' };
12489
13241
  if (outcome.kind === 'refused') {
12490
- return {
12491
- ok: false,
12492
- error: 'job already completed — resetting it would re-execute shipped work; archive the PRD instead, or pass force:true',
12493
- };
13242
+ return { ok: false, error: resetRefusalMessage(outcome.status, { canForce: true }) };
12494
13243
  }
12495
13244
  await broadcast({ flush: true });
12496
13245
  return { ok: true, slug, status: 'pending' };
@@ -12566,7 +13315,8 @@ const remote = {
12566
13315
  // Symlink defense, matching readPrd/writePrd's comment: safeSlugPathIn
12567
13316
  // is lexical and does not resolve symlinks.
12568
13317
  const real = await fsp.realpath(filePath);
12569
- if (!real.startsWith(dir + path.sep)) return { ok: false, error: 'invalid slug' };
13318
+ const realDir = await realPrdsDir(dir);
13319
+ if (!realDir || !real.startsWith(realDir + path.sep)) return { ok: false, error: 'invalid slug' };
12570
13320
  const [raw, parsed] = await Promise.all([fsp.readFile(real, 'utf8'), prdParser.parsePrdRaw(real)]);
12571
13321
  return {
12572
13322
  ok: true,
@@ -12591,20 +13341,32 @@ const remote = {
12591
13341
  }
12592
13342
  },
12593
13343
 
12594
- // Edits a NOT-yet-running PRD's frontmatter and/or body in place, refusing
12595
- // once a queue row exists for it and that row is anything but 'pending'
12596
- // (running/completed/failed/needs_review — editing the spec under a live
12597
- // or already-finished executor would silently rewrite history). Reuses
13344
+ // Edits a PRD's frontmatter and/or body in place. Works when there is no
13345
+ // queue row yet, and when the row's status is 'pending', 'quarantined',
13346
+ // 'needs_review', 'failed', or 'skipped'. None of those have a live
13347
+ // executor reading the file right now, so a rewrite is safe. A parked
13348
+ // (needs_review/failed) or skipped job is exactly the planner-repair case:
13349
+ // fix the spec here, then reset the job with scheduler_reset_job, which
13350
+ // clears the old run fields. 'quarantined' is also editable for a second
13351
+ // reason: it's the ONLY way a quarantined PRD's createdVia stamp gets
13352
+ // written (the adopt action below), so refusing it here would make
13353
+ // quarantine irreversible through the API. Refuses 'running' (a live
13354
+ // executor could read the file mid-edit) and 'completed' (the work already
13355
+ // landed — queue a new PRD instead of rewriting a finished one). Reuses
12598
13356
  // prdFrontmatter.cjs's parsePrdFile/serializePrdFile round-trip pair (PRD
12599
13357
  // 1024) so unrecognized keys (e.g. dependsOn) and untouched recognized
12600
13358
  // keys' original line formatting survive unchanged.
12601
13359
  async updatePrd({ slug, cwd, frontmatter, body }) {
12602
13360
  const job = await this.getJob(slug);
12603
- // 'quarantined' is also editable: it's the ONLY way a quarantined PRD's
12604
- // createdVia stamp gets written (the adopt action below), so refusing it
12605
- // here would make quarantine irreversible through the API.
12606
- if (job && job.status !== 'pending' && job.status !== 'quarantined') {
12607
- return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
13361
+ const EDITABLE_JOB_STATUSES = new Set(['pending', 'quarantined', 'needs_review', 'failed', 'skipped']);
13362
+ if (job && !EDITABLE_JOB_STATUSES.has(job.status)) {
13363
+ if (job.status === 'running') {
13364
+ return { ok: false, error: 'job status is "running" — wait for it to end, or stop it with scheduler_cancel_job, then edit it.' };
13365
+ }
13366
+ if (job.status === 'completed') {
13367
+ return { ok: false, error: 'job status is "completed" — its work already landed. Queue a new PRD for more work.' };
13368
+ }
13369
+ return { ok: false, error: `job status is "${job.status}" — this PRD cannot be edited now.` };
12608
13370
  }
12609
13371
 
12610
13372
  // Write-time FK check for a patched dependsOn (PRD 1124), reusing the
@@ -12658,7 +13420,8 @@ const remote = {
12658
13420
  // target that is itself already a symlink — a rogue job could plant
12659
13421
  // one inside the PRDs dir pointing outside the safe root.
12660
13422
  const real = await fsp.realpath(filePath);
12661
- if (!real.startsWith(dir + path.sep)) return { ok: false, error: 'invalid slug' };
13423
+ const realDir = await realPrdsDir(dir);
13424
+ if (!realDir || !real.startsWith(realDir + path.sep)) return { ok: false, error: 'invalid slug' };
12662
13425
  const existing = await fsp.lstat(filePath).catch(() => null);
12663
13426
  if (existing && existing.isSymbolicLink()) return { ok: false, error: 'invalid slug' };
12664
13427
  raw = await fsp.readFile(real, 'utf8');
@@ -12876,8 +13639,11 @@ module.exports = {
12876
13639
  computeDegradedBudget,
12877
13640
  healRefusalReason,
12878
13641
  writeQueue,
13642
+ _mutateForTests: mutate, // mutate() has no other exported call site; test-only seam.
12879
13643
  reconcile,
12880
13644
  broadcast,
13645
+ rescheduleTimer,
13646
+ RESCHEDULE_TIMER_BILLING_RACE_MS,
12881
13647
  reconcileSourcePromptId,
12882
13648
  allocateParallelGroup,
12883
13649
  selectHistoryJobs,
@@ -12899,6 +13665,7 @@ module.exports = {
12899
13665
  reverifyNeedsReview,
12900
13666
  runGateShadow,
12901
13667
  awaitGateShadowIdle: async () => { while (gateShadowPending) await gateShadowPending; },
13668
+ selectGateShadowTarget,
12902
13669
  shouldRunPeriodicReverify,
12903
13670
  findStuckFailedJobs,
12904
13671
  STUCK_FAILED_ESCALATE_MS,
@@ -12978,6 +13745,7 @@ module.exports = {
12978
13745
  registerAdminRoutes,
12979
13746
  notifyOriginatingTab,
12980
13747
  notifyNeedsReview,
13748
+ flushDueReviewNotices,
12981
13749
  isNotifiableTerminalStatus,
12982
13750
  extractResultTextFromLog,
12983
13751
  candidatePrdsDirs,
@@ -12988,6 +13756,7 @@ module.exports = {
12988
13756
  archivedPrdPathForJob,
12989
13757
  archivedTwinExists,
12990
13758
  findPrdDir,
13759
+ findPrdDirForJob,
12991
13760
  resolveVerifyPrdPath,
12992
13761
  resolveFixPlanPath,
12993
13762
  resolveNotifyPrd,
@@ -12999,6 +13768,8 @@ module.exports = {
12999
13768
  SCHEDULER_BOOTED_AT,
13000
13769
  SCHEDULER_CODE_SHA,
13001
13770
  resetJobFields,
13771
+ resetRefusalMessage,
13772
+ jobMatchesSlugAndCwd,
13002
13773
  executeJob,
13003
13774
  killOrphanClaudePid,
13004
13775
  prdArchivedSkipResult,
@@ -13030,6 +13801,7 @@ module.exports = {
13030
13801
  selectResumeRecoveryTarget,
13031
13802
  buildResumeRecoveryPreamble,
13032
13803
  buildClaudeSpawnArgs,
13804
+ HEADLESS_DISALLOWED_TOOLS,
13033
13805
  spawnResumeRecovery,
13034
13806
  selectMechanicalRecoveryTarget,
13035
13807
  isMechanicalRecoveryFutile,