claude-code-session-manager 0.85.0 → 0.87.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/dist/assets/AgentLibrary-DyLWzZDf.js +3 -0
  2. package/dist/assets/{DataModel-DRH-Ty20.js → DataModel--mISIJ6h.js} +1 -1
  3. package/dist/assets/{History-CfRhT1Im.js → History-C2ahUXTg.js} +2 -2
  4. package/dist/assets/{Hooks-wEmh_U6c.js → Hooks-BiC6oyR2.js} +3 -3
  5. package/dist/assets/{HostBilko-D_t7Rbi7.js → HostBilko-BPleEOld.js} +1 -1
  6. package/dist/assets/{Library-CpArQ-OJ.js → Library-Dc8Qst1R.js} +1 -1
  7. package/dist/assets/{ListDetail-pjaKYs84.js → ListDetail-DIXh-OLX.js} +1 -1
  8. package/dist/assets/MarkdownEditor-C90bkLXK.js +1 -0
  9. package/dist/assets/{McpServers-ftqaV3kn.js → McpServers-DqcbLOLZ.js} +2 -2
  10. package/dist/assets/{Memory-ChMWkNd0.js → Memory-CW62MXlh.js} +4 -4
  11. package/dist/assets/{Panel-D9Kr40Ai.js → Panel-Bw1FhRuF.js} +1 -1
  12. package/dist/assets/Permissions-BcUC-5y8.js +3 -0
  13. package/dist/assets/{Plugins-BtChISho.js → Plugins-BnKx9flD.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-DBA5EcYy.js → ProvenanceBadge-Bw5vNVPT.js} +1 -1
  15. package/dist/assets/{SaveBar-I0_dWNTX.js → SaveBar-CWr0O_w-.js} +1 -1
  16. package/dist/assets/Scheduler-DYdLuUqq.js +14 -0
  17. package/dist/assets/{ScopeSwitcher-5GTEveb2.js → ScopeSwitcher-CrBLbg8s.js} +1 -1
  18. package/dist/assets/Settings-DluB-vN1.js +3 -0
  19. package/dist/assets/{SkillReferenceGraph-DNBFGrYE.js → SkillReferenceGraph-CHLSseay.js} +1 -1
  20. package/dist/assets/{Skills-DJB6-bBM.js → Skills-gNdo_HNK.js} +2 -2
  21. package/dist/assets/{SystemPrompt-BiDDrJUA.js → SystemPrompt-Cru05-Ia.js} +1 -1
  22. package/dist/assets/{TagLibrary-_Wrevtop.js → TagLibrary-DNHY0xou.js} +1 -1
  23. package/dist/assets/{TiptapBody-OWWXdLRy.js → TiptapBody-I4lmbCgP.js} +1 -1
  24. package/dist/assets/{Toggle-B122N0HL.js → Toggle-bWMHjmRh.js} +1 -1
  25. package/dist/assets/{index-CDo9xBR9.css → index-DV3PorRY.css} +1 -1
  26. package/dist/assets/{index-DhvuQL4C.js → index-fc_JjdxL.js} +724 -724
  27. package/dist/assets/settingsSchema-BfhtZnGD.js +3 -0
  28. package/dist/index.html +2 -2
  29. package/package.json +15 -14
  30. package/plugins/CLAUDE.md +61 -0
  31. package/plugins/session-manager-dev/.claude-plugin/plugin.json +1 -1
  32. package/plugins/session-manager-dev/skills/builder/4-manual/SKILL.md +1 -1
  33. package/plugins/session-manager-dev/skills/ops-sweep/SKILL.md +1 -1
  34. package/scripts/scheduler-mcp-server.cjs +7 -0
  35. package/src/main/__tests__/agentModelResolve.test.cjs +100 -9
  36. package/src/main/__tests__/broadcastCoalescer.test.cjs +18 -0
  37. package/src/main/__tests__/epicMint.test.cjs +2 -2
  38. package/src/main/__tests__/health-delegation-chain.test.cjs +2 -1
  39. package/src/main/__tests__/health-starve-escalation.test.cjs +94 -0
  40. package/src/main/__tests__/loadGateDetailTick.test.cjs +31 -0
  41. package/src/main/__tests__/machineProfile.test.cjs +19 -1
  42. package/src/main/__tests__/needsReviewLedger.test.cjs +162 -0
  43. package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +3 -3
  44. package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +15 -1
  45. package/src/main/__tests__/prdCreateDisposition.test.cjs +201 -0
  46. package/src/main/__tests__/prdFrontmatterDisposition.test.cjs +125 -0
  47. package/src/main/__tests__/prdLocations.test.cjs +100 -2
  48. package/src/main/__tests__/prdLocationsArchived.test.cjs +43 -1
  49. package/src/main/__tests__/prdSetDisposition.test.cjs +222 -0
  50. package/src/main/__tests__/pty-session-open-telemetry.test.cjs +96 -0
  51. package/src/main/__tests__/queue-health-verdict.test.cjs +170 -0
  52. package/src/main/__tests__/queue-starvation-per-project.test.cjs +147 -0
  53. package/src/main/__tests__/queueHistory.test.cjs +63 -0
  54. package/src/main/__tests__/reconcileTiming.test.cjs +135 -0
  55. package/src/main/__tests__/scheduleJobTransitions.test.cjs +101 -1
  56. package/src/main/__tests__/scheduler-boot-orphans.test.cjs +2 -2
  57. package/src/main/__tests__/scheduler-broadcast-reconcile.test.cjs +121 -0
  58. package/src/main/__tests__/scheduler-cross-project-batch.test.cjs +43 -0
  59. package/src/main/__tests__/scheduler-failed-autoreset.test.cjs +121 -0
  60. package/src/main/__tests__/scheduler-guard-verdict-autoresolve.test.cjs +344 -0
  61. package/src/main/__tests__/scheduler-job-budget.test.cjs +172 -0
  62. package/src/main/__tests__/scheduler-looks-done.test.cjs +93 -3
  63. package/src/main/__tests__/scheduler-needs-review-autoresolve.test.cjs +189 -0
  64. package/src/main/__tests__/scheduler-no-dead-end-status.test.cjs +152 -0
  65. package/src/main/__tests__/scheduler-porcelain-rename.test.cjs +164 -0
  66. package/src/main/__tests__/scheduler-quarantine-autoresolve.test.cjs +165 -0
  67. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +160 -0
  68. package/src/main/__tests__/scheduler-reaper-helpers-basics.test.cjs +87 -0
  69. package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +88 -0
  70. package/src/main/__tests__/scheduler-starve-escalation.test.cjs +154 -0
  71. package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +14 -0
  72. package/src/main/__tests__/telemetryClient.test.cjs +75 -5
  73. package/src/main/__tests__/telemetrySettings.test.cjs +32 -0
  74. package/src/main/chatRunner.cjs +8 -5
  75. package/src/main/health.cjs +76 -3
  76. package/src/main/historyAggregator.cjs +5 -0
  77. package/src/main/index.cjs +95 -44
  78. package/src/main/ipcSchemas.cjs +47 -0
  79. package/src/main/lib/__tests__/active-sessions.test.cjs +251 -0
  80. package/src/main/lib/__tests__/bootSelfHeal.test.cjs +107 -0
  81. package/src/main/lib/__tests__/delegationReadiness.test.cjs +322 -43
  82. package/src/main/lib/__tests__/effectiveModelInfo.test.cjs +239 -0
  83. package/src/main/lib/__tests__/gitWorktree.test.cjs +89 -0
  84. package/src/main/lib/__tests__/guardShims.test.cjs +151 -0
  85. package/src/main/lib/__tests__/loadGate.test.cjs +103 -2
  86. package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +5 -5
  87. package/src/main/lib/__tests__/prdDisposition.test.cjs +224 -0
  88. package/src/main/lib/__tests__/reaperHelpers.test.cjs +179 -1
  89. package/src/main/lib/__tests__/telemetryBoot.test.cjs +11 -0
  90. package/src/main/lib/__tests__/usageCircuit.test.cjs +224 -0
  91. package/src/main/lib/__tests__/watchdog-helpers.test.cjs +312 -0
  92. package/src/main/lib/__tests__/watchdog-relaunch.test.cjs +193 -0
  93. package/{scripts → src/main}/lib/activeSessions.cjs +50 -4
  94. package/src/main/lib/agentModelResolve.cjs +65 -27
  95. package/src/main/lib/bootSelfHeal.cjs +88 -0
  96. package/src/main/lib/delegationReadiness.cjs +290 -225
  97. package/src/main/lib/effectiveModelInfo.cjs +333 -0
  98. package/src/main/lib/ephemeralCwd.cjs +1 -1
  99. package/src/main/lib/epicMint.cjs +3 -3
  100. package/src/main/lib/gitWorktree.cjs +42 -12
  101. package/src/main/lib/guardShims.cjs +156 -0
  102. package/src/main/lib/jobDirtFilter.cjs +7 -2
  103. package/src/main/lib/launchFailure.cjs +2 -1
  104. package/src/main/lib/loadGate.cjs +23 -1
  105. package/src/main/lib/machineProfile.cjs +15 -0
  106. package/src/main/lib/mcpToolCatalog.cjs +4 -1
  107. package/src/main/lib/needsReviewLedger.cjs +205 -0
  108. package/src/main/lib/opsErrorLog.cjs +1 -1
  109. package/src/main/lib/opsOwnership.cjs +1 -1
  110. package/src/main/lib/prdCreate.cjs +56 -1
  111. package/src/main/lib/prdDisposition.cjs +199 -0
  112. package/src/main/lib/prdFrontmatter.cjs +8 -2
  113. package/src/main/lib/prdLocations.cjs +167 -45
  114. package/src/main/lib/projectHomeAdminRoutes.cjs +4 -4
  115. package/src/main/lib/projectPageSummarySchema.cjs +1 -1
  116. package/src/main/lib/projectRootResolve.cjs +1 -1
  117. package/src/main/lib/queueHistory.cjs +19 -1
  118. package/src/main/lib/queueStore.cjs +6 -1
  119. package/src/main/lib/reaperHelpers.cjs +181 -15
  120. package/src/main/lib/scheduleJobSchema.cjs +8 -0
  121. package/src/main/lib/scheduleJobTransitions.cjs +33 -0
  122. package/src/main/lib/schedulerBatch.cjs +12 -1
  123. package/src/main/lib/schedulerConfig.cjs +37 -0
  124. package/src/main/lib/telemetryBoot.cjs +11 -8
  125. package/src/main/lib/telemetryClient.cjs +44 -2
  126. package/src/main/lib/telemetrySettings.cjs +20 -3
  127. package/src/main/lib/usageCircuit.cjs +159 -0
  128. package/{scripts → src/main}/lib/watchdogHelpers.cjs +1 -1
  129. package/src/main/pty.cjs +9 -0
  130. package/src/main/scheduler/prdParser.cjs +13 -0
  131. package/src/main/scheduler.cjs +1869 -193
  132. package/src/main/templates/PRD_AUTHORING.md +50 -0
  133. package/src/main/templates/project-pages-catalog.json +1 -1
  134. package/src/main/usage.cjs +21 -3
  135. package/src/preload/api.d.ts +92 -1
  136. package/src/preload/index.cjs +10 -0
  137. package/web/README.md +41 -0
  138. package/{scripts/render-project-pages.cjs → web/project-pages/render.cjs} +4 -4
  139. package/{scripts/render-project-pages → web/project-pages/renderer}/dist/renderer.cjs +1 -1
  140. package/{scripts/validate-project-pages-summary.cjs → web/project-pages/validate-summary.cjs} +5 -5
  141. package/dist/assets/AgentLibrary-Bkv-HcP1.js +0 -3
  142. package/dist/assets/MarkdownEditor-Xc141kjj.js +0 -1
  143. package/dist/assets/Permissions-DKoNVgzj.js +0 -3
  144. package/dist/assets/Scheduler-CbES7MC8.js +0 -14
  145. package/dist/assets/Settings-BX3FElXk.js +0 -3
  146. package/dist/assets/settingsSchema-sGoCTd7J.js +0 -3
  147. /package/{scripts/project-pages-logic → web/project-pages/logic}/dist/logic.cjs +0 -0
@@ -60,7 +60,9 @@ const { readTail } = require('./lib/fileTail.cjs');
60
60
  const {
61
61
  claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs,
62
62
  findLiveProcessForJob, logHasOutput, resolvePidlessGateOutcome, resolveCommitGuardOutcome,
63
+ readSpawnedPidFromLog, readLogMtimeMs,
63
64
  } = require('./lib/reaperHelpers.cjs');
65
+ const { resolveProjectRoot } = require('./lib/opsOwnership.cjs');
64
66
  const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
65
67
  const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
66
68
  const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
@@ -89,22 +91,42 @@ const {
89
91
  USAGE_REFRESH_INTERVAL_MS,
90
92
  MAX_JOB_DURATION_MS,
91
93
  BROADCAST_COALESCE_MS,
94
+ RECONCILE_SLOW_PASS_MS,
92
95
  QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
93
96
  JOB_OVERRUN_FACTOR: JOB_OVERRUN_FACTOR_DEFAULT,
94
97
  JOB_OVERRUN_FLOOR_MS: JOB_OVERRUN_FLOOR_MS_DEFAULT,
98
+ JOB_BUDGET_FACTOR: JOB_BUDGET_FACTOR_DEFAULT,
99
+ JOB_BUDGET_FLOOR_MS: JOB_BUDGET_FLOOR_MS_DEFAULT,
100
+ JOB_BUDGET_CEILING_MS: JOB_BUDGET_CEILING_MS_DEFAULT,
95
101
  PIDLESS_SPAWN_GRACE_MS,
96
102
  INVESTIGATION_MAX_MS,
97
103
  STARVATION_ESCALATE_MS,
104
+ STARVE_ESCALATION_MS,
98
105
  } = require('./lib/schedulerConfig.cjs');
99
106
  const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
100
107
  ? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
101
108
  : QUARANTINE_ESCALATE_MS_DEFAULT;
102
- const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
103
- ? Number(process.env.SM_JOB_OVERRUN_FACTOR)
104
- : JOB_OVERRUN_FACTOR_DEFAULT;
105
- const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
106
- ? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
107
- : JOB_OVERRUN_FLOOR_MS_DEFAULT;
109
+ // Shared by every SM_*-env-overridable numeric constant below (bare factors
110
+ // use unitMs=1; minute-denominated knobs use unitMs=60_000) — one parse rule
111
+ // instead of one hand-copied ternary per constant.
112
+ function numEnvOverride(envVar, unitMs, fallback) {
113
+ const raw = process.env[envVar];
114
+ return raw ? Number(raw) * unitMs : fallback;
115
+ }
116
+ const JOB_OVERRUN_FACTOR = numEnvOverride('SM_JOB_OVERRUN_FACTOR', 1, JOB_OVERRUN_FACTOR_DEFAULT);
117
+ const JOB_OVERRUN_FLOOR_MS = numEnvOverride('SM_JOB_OVERRUN_FLOOR_MINUTES', 60_000, JOB_OVERRUN_FLOOR_MS_DEFAULT);
118
+ // Same three numbers as JOB_OVERRUN_FACTOR/JOB_OVERRUN_FLOOR_MS today (3x,
119
+ // 45min) is coincidental, not structural — this triad ACTS (kills) where
120
+ // JOB_OVERRUN_* only ever escalates (see JOB_OVERRUN_FACTOR's own header);
121
+ // tune them independently, don't re-couple on a future pass just because the
122
+ // defaults happen to match right now.
123
+ const JOB_BUDGET_FACTOR = numEnvOverride('SM_JOB_BUDGET_FACTOR', 1, JOB_BUDGET_FACTOR_DEFAULT);
124
+ const JOB_BUDGET_FLOOR_MS = numEnvOverride('SM_JOB_BUDGET_FLOOR_MINUTES', 60_000, JOB_BUDGET_FLOOR_MS_DEFAULT);
125
+ const JOB_BUDGET_CEILING_MS = numEnvOverride('SM_JOB_BUDGET_CEILING_MINUTES', 60_000, JOB_BUDGET_CEILING_MS_DEFAULT);
126
+ // A running job past this fraction of its own budget gets a durable
127
+ // `budgetWarning` stamp on its row (see the budget watchdog below) so the
128
+ // renderer can warn BEFORE the kill, not only after.
129
+ const BUDGET_WARNING_FRACTION = 0.75;
108
130
  const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD, DEP_HISTORY_FAIL_OPEN } = require('./lib/schedulerBatch.cjs');
109
131
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
110
132
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
@@ -150,8 +172,9 @@ const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
150
172
  const queueStore = require('./lib/queueStore.cjs');
151
173
  const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
152
174
  const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
175
+ const { computeDispositionRewrite } = require('./lib/prdDisposition.cjs');
153
176
  const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
154
- const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
177
+ const { allProjectCwds } = require('./lib/activeSessions.cjs');
155
178
 
156
179
  // Captured once at module load so every run's meta sidecar can record how
157
180
  // stale the running process is relative to on-disk source (incident: PRD
@@ -310,22 +333,95 @@ only post-AC work. If a review finding can't be fixed within scope, commit what
310
333
  you have, describe the finding in the commit body, and note the follow-up in your
311
334
  final result.`;
312
335
 
313
- // Parse \`git status --porcelain\` output into a list of changed paths. Pure +
314
- // exported for unit testing. Each porcelain line is "XY<space>PATH" (2 status
315
- // chars + space), so the path starts at index 3; rename lines ("R a -> b")
316
- // keep the "a -> b" tail, which is fine for a human-facing dirty-file list.
317
- function parsePorcelain(stdout) {
336
+ // Unquote a single git porcelain v1 path token. Defined once in
337
+ // gitWorktree.cjs (which this file already requires — the reverse would be
338
+ // circular, since gitWorktree.cjs's own salvageDirtyDelta needs the exact
339
+ // same unquoting) and reused here rather than re-implemented, so the two
340
+ // porcelain consumers in this codebase can never drift apart.
341
+ const { unquotePorcelainPath } = gitWorktree;
342
+
343
+ // Split a rename/copy porcelain path field ("old -> new") into its two real
344
+ // paths. Each side is independently quoted per unquotePorcelainPath's rule —
345
+ // only the side that needs escaping is wrapped in quotes, the literal " -> "
346
+ // arrow between them never is. Returns null when no " -> " separator is
347
+ // found (a malformed/unexpected line) so the caller can fall back to treating
348
+ // the whole field as one opaque path rather than guessing.
349
+ function splitRenamePorcelainField(field) {
350
+ const arrow = ' -> ';
351
+ let head;
352
+ let rest;
353
+ if (field[0] === '"') {
354
+ let end = -1;
355
+ for (let i = 1; i < field.length; i += 1) {
356
+ if (field[i] === '\\') { i += 1; continue; }
357
+ if (field[i] === '"') { end = i; break; }
358
+ }
359
+ if (end === -1) return null;
360
+ head = field.slice(0, end + 1);
361
+ rest = field.slice(end + 1);
362
+ } else {
363
+ const idx = field.indexOf(arrow);
364
+ if (idx === -1) return null;
365
+ // An unquoted path containing a literal " -> " substring (git only
366
+ // quotes for a quote/backslash/control-byte/non-ASCII byte — a plain
367
+ // ASCII arrow inside a filename is never quoted) makes the true
368
+ // old/new boundary genuinely ambiguous from this text alone: the first
369
+ // occurrence could be the real separator, or it could be sitting
370
+ // inside the old path with the real separator later in the field.
371
+ // Guessing wrong silently corrupts oldPath/path for downstream
372
+ // fs.existsSync/Set-membership checks, which is worse than the
373
+ // existing "malformed line" fallback below — so more than one
374
+ // occurrence falls back to treating the whole field as one opaque
375
+ // path, same as any other line this function can't confidently parse.
376
+ if (field.indexOf(arrow, idx + arrow.length) !== -1) return null;
377
+ head = field.slice(0, idx);
378
+ rest = field.slice(idx);
379
+ }
380
+ if (!rest.startsWith(arrow)) return null;
381
+ return { oldPath: unquotePorcelainPath(head), path: unquotePorcelainPath(rest.slice(arrow.length)) };
382
+ }
383
+
384
+ // Parse `git status --porcelain` output into `{ code, path }` entries (plus
385
+ // `oldPath` for a rename/copy). Pure + exported for unit testing. Each
386
+ // porcelain line is "XY<space>PATH"; a staged rename/copy line is
387
+ // "XY<space>OLD -> NEW" instead — X (index status) is 'R' or 'C' — and NEW is
388
+ // the path git will report in any later `git status` call, so callers that
389
+ // key off `.path` (dirtyAfter membership, pathsCommittedDuringRun membership,
390
+ // fs.existsSync) must compare against NEW, never the fused "OLD -> NEW"
391
+ // string. `oldPath` is retained on the entry for callers that need the
392
+ // original path too. `code` is the raw 2-char status (e.g. '??' for
393
+ // untracked) — callers that need to distinguish "untracked" from
394
+ // "tracked-but-modified" (the shared-tree guard's revert-vs-now-ignored
395
+ // split) read it off the entry instead of re-deriving it later.
396
+ function parsePorcelainEntries(stdout) {
318
397
  return String(stdout || '')
319
398
  .split('\n')
320
399
  .filter((l) => l.length > 0)
321
- .map((l) => l.slice(3))
322
- .filter(Boolean);
400
+ .map((l) => {
401
+ const code = l.slice(0, 2);
402
+ const field = l.slice(3);
403
+ if (code.includes('R') || code.includes('C')) {
404
+ const split = splitRenamePorcelainField(field);
405
+ if (split) return { code, path: split.path, oldPath: split.oldPath };
406
+ }
407
+ return { code, path: unquotePorcelainPath(field) };
408
+ })
409
+ .filter((e) => e.path);
323
410
  }
324
411
 
325
- // Return the list of uncommitted paths in cwd, or null when the guard does not
326
- // apply (cwd is not a git work tree, git is missing, or the call errors). Never
327
- // throws — a guard failure must not fail an otherwise-successful job.
328
- function uncommittedChanges(cwd) {
412
+ // Parse `git status --porcelain` output into a list of changed paths. Pure +
413
+ // exported for unit testing.
414
+ function parsePorcelain(stdout) {
415
+ return parsePorcelainEntries(stdout).map((e) => e.path);
416
+ }
417
+
418
+ // Same as uncommittedChanges but keeps each path's porcelain status code —
419
+ // the shared-tree guard's baseline needs this to tell "was untracked" apart
420
+ // from "was tracked-and-modified" (see evaluateSharedTreeGuard). Returns null
421
+ // when the guard does not apply (cwd is not a git work tree, git is missing,
422
+ // or the call errors); never throws — a guard failure must not fail an
423
+ // otherwise-successful job.
424
+ function uncommittedChangesWithStatus(cwd) {
329
425
  return new Promise((resolve) => {
330
426
  if (!cwd) { resolve(null); return; }
331
427
  execFile(
@@ -333,13 +429,24 @@ function uncommittedChanges(cwd) {
333
429
  ['-C', cwd, 'status', '--porcelain'],
334
430
  { timeout: 10_000, windowsHide: true },
335
431
  (err, stdout) => {
336
- if (err) { resolve(null); return; } // not a repo / git missing → skip
337
- resolve(parsePorcelain(stdout));
432
+ if (err) { resolve(null); return; }
433
+ resolve(parsePorcelainEntries(stdout));
338
434
  },
339
435
  );
340
436
  });
341
437
  }
342
438
 
439
+ // Return the list of uncommitted paths in cwd, or null under the same
440
+ // conditions as uncommittedChangesWithStatus (never throws). Kept as a thin
441
+ // path-only projection of that call rather than its own execFile, so a future
442
+ // fix to the git invocation (timeout, error handling) can't land in one and
443
+ // silently miss the other.
444
+ function uncommittedChanges(cwd) {
445
+ return uncommittedChangesWithStatus(cwd).then((entries) => (
446
+ entries === null ? null : entries.map((e) => e.path)
447
+ ));
448
+ }
449
+
343
450
  // Return the current HEAD commit sha in cwd, or null on any error. Used by the
344
451
  // commit-guard to detect whether the job self-committed during its run (HEAD
345
452
  // moved) — in which case leftover working-tree dirt is presumptively from a
@@ -428,17 +535,52 @@ function restoreSpecificStash(cwd, ref) {
428
535
  // - reverted: a path that was dirty in the baseline, is clean now, and was
429
536
  // not touched by any commit landed during the run — the job reset/
430
537
  // checked-out over pre-existing uncommitted work without stashing it.
431
- // Pure/no I/O — the guard's git calls happen at the call site
432
- // (checkSharedTreeGuard). Exported for unit testing.
433
- function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAfter, pathsCommittedDuringRun }) {
538
+ //
539
+ // A THIRD outcome is not a revert at all: an untracked path can drop out of
540
+ // `git status` because the run committed a `.gitignore` change that now
541
+ // matches it — the file is untouched on disk, just no longer visible to git
542
+ // (Incident: 2026-09-12, PRD 1181 added a bare `logs/` ignore pattern, eleven
543
+ // untracked `session-manager-operations/logs/*` paths vanished from status,
544
+ // and an otherwise-perfect run was parked in needs_review for a human who had
545
+ // nothing to decide). `dirtyBefore` entries therefore carry each path's
546
+ // porcelain status code (`{ code, path }`, from parsePorcelainEntries) so this
547
+ // function can tell "was untracked" apart from "was tracked-and-modified":
548
+ // - a TRACKED path (any code other than '??') leaving the dirty set always
549
+ // means its content was restored to HEAD — still `reverted`, even though
550
+ // the file still exists on disk, because for a tracked file "exists" is
551
+ // not the question; "matches what the human left uncommitted" is.
552
+ // - an UNTRACKED path ('??') leaving the dirty set is `reverted` only if it
553
+ // no longer exists on disk; if it still exists, it merely became ignored
554
+ // and is reported separately as `nowIgnored`.
555
+ // Plain path strings are still accepted in `dirtyBefore` for callers that
556
+ // have no status code (e.g. the stash-detection pass, which always passes an
557
+ // empty array) — an entry with no `code` is treated as tracked, matching the
558
+ // old behavior exactly.
559
+ //
560
+ // Pure/no I/O — status-code parsing and on-disk existence checks both happen
561
+ // at the call site (checkSharedTreeGuard); this function never stats the
562
+ // filesystem. Exported for unit testing.
563
+ function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAfter, pathsCommittedDuringRun, existsAfter }) {
434
564
  const beforeHashes = new Set((stashBefore || []).map((l) => parseStashLine(l)?.hash).filter(Boolean));
435
565
  const newStashes = (stashAfter || [])
436
566
  .map(parseStashLine)
437
567
  .filter((e) => e && !beforeHashes.has(e.hash));
438
568
  const dirtyAfterSet = new Set(dirtyAfter || []);
439
569
  const committedSet = new Set(pathsCommittedDuringRun || []);
440
- const reverted = (dirtyBefore || []).filter((p) => !dirtyAfterSet.has(p) && !committedSet.has(p));
441
- return { newStashes, reverted };
570
+ const existsSet = new Set(existsAfter || []);
571
+ const reverted = [];
572
+ const nowIgnored = [];
573
+ for (const entry of dirtyBefore || []) {
574
+ const p = typeof entry === 'string' ? entry : entry.path;
575
+ const code = typeof entry === 'string' ? undefined : entry.code;
576
+ if (dirtyAfterSet.has(p) || committedSet.has(p)) continue;
577
+ if (code === '??' && existsSet.has(p)) {
578
+ nowIgnored.push(p);
579
+ } else {
580
+ reverted.push(p);
581
+ }
582
+ }
583
+ return { newStashes, reverted, nowIgnored };
442
584
  }
443
585
 
444
586
  // Post-run shared-tree guard for an IN-PLACE job (worktree.ok === false —
@@ -488,18 +630,40 @@ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBef
488
630
  // restored stash is not ALSO reported as an unexplained revert (it was
489
631
  // explained — by the stash this guard just restored).
490
632
  const dirtyAfter = await module.exports.uncommittedChanges(cwd);
491
- const { reverted } = module.exports.evaluateSharedTreeGuard({
633
+ // Existence check for the "now ignored, not reverted" split (2026-09-12
634
+ // incident) — only untracked baseline entries need it; a tracked path
635
+ // leaving the dirty set is always a revert regardless of disk state (see
636
+ // evaluateSharedTreeGuard). Scoped to entries carrying a status code —
637
+ // plain path strings (no code) fall back to the old always-reverted path.
638
+ const untrackedBaselinePaths = (dirtyBaseline || [])
639
+ .filter((e) => e && typeof e === 'object' && e.code === '??')
640
+ .map((e) => e.path);
641
+ // A large untracked baseline (the 2026-09-12 incident's shared tree had
642
+ // ~240 such paths) makes this a lot of stat calls — fs.promises.access
643
+ // run concurrently instead of fs.existsSync run synchronously one at a
644
+ // time keeps this off the event loop instead of blocking every other
645
+ // in-flight scheduler/IPC task for the duration.
646
+ const existsChecks = await Promise.all(
647
+ untrackedBaselinePaths.map((p) => fsp.access(path.join(cwd, p)).then(() => true, () => false)),
648
+ );
649
+ const existsAfter = untrackedBaselinePaths.filter((_, i) => existsChecks[i]);
650
+ const { reverted, nowIgnored } = module.exports.evaluateSharedTreeGuard({
492
651
  stashBefore: stashBaseline,
493
652
  stashAfter,
494
653
  dirtyBefore: dirtyBaseline,
495
654
  dirtyAfter,
496
655
  pathsCommittedDuringRun,
656
+ existsAfter,
497
657
  });
498
658
  if (reverted.length) {
499
659
  result.reverted = reverted;
500
660
  console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
501
661
  }
502
- return (result.restoredStash || result.restoreFailed || result.ambiguousStashes || result.reverted) ? result : null;
662
+ if (nowIgnored.length) {
663
+ result.nowIgnored = nowIgnored;
664
+ console.log(`[scheduler] ${slug}: shared-tree guard: ${nowIgnored.length} path(s) no longer shown by git status but still present on disk — likely a new ignore rule, not a revert (${nowIgnored.slice(0, 3).join(', ')})`);
665
+ }
666
+ return (result.restoredStash || result.restoreFailed || result.ambiguousStashes || result.reverted || result.nowIgnored) ? result : null;
503
667
  } catch (e) {
504
668
  console.error(`[scheduler] ${slug}: shared-tree guard error`, e);
505
669
  return null;
@@ -1151,8 +1315,9 @@ function ensureDirs() {
1151
1315
  * reconcile()-level call is what makes "anything written to the retired flat
1152
1316
  * prds/ dir is swept into prds-archived/ without being executed" actually
1153
1317
  * true regardless of which of reconcile's several callers (tickQueue's poll,
1154
- * job completion, the schedule:state/schedule:rescan IPC handlers,
1155
- * rescheduleTimer) triggers the pass: a PRD dropped in the flat dir has no
1318
+ * job completion, the schedule:rescan/schedule:adopt-prd IPC handlers,
1319
+ * broadcast()'s coalescer, rescheduleTimer) triggers the pass — schedule:state
1320
+ * no longer reconciles on read. A PRD dropped in the flat dir has no
1156
1321
  * queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
1157
1322
  * sweep archives it before reconcile can ever turn it into a pending job.
1158
1323
  */
@@ -1216,7 +1381,22 @@ async function runPrdMigration() {
1216
1381
  // on every pass, but stays here so a fresh boot's very first log line
1217
1382
  // still reports the initial sweep — see consolidateAllFlatPrds's own
1218
1383
  // comment for why reconcile() is the load-bearing call site.)
1219
- await consolidateAllFlatPrds(allProjectCwds());
1384
+ //
1385
+ // Deferred off scheduler.init()'s synchronous critical path: allProjectCwds()
1386
+ // is a synchronous ~270ms directory scan, and awaiting it inline here
1387
+ // competed with the renderer's first IPC round trips (schedule.state,
1388
+ // billing.fetch, teams.list) for the event loop during boot. Dropping the
1389
+ // await doesn't weaken the consolidation guarantee — consolidateAllFlatPrds
1390
+ // also runs at the top of every reconcile() (see its own comment above),
1391
+ // and a flat PRD can only ever execute via tickQueue, which always
1392
+ // reconciles first, so nothing dropped in the flat dir can run before a
1393
+ // reconcile() pass sweeps it regardless of whether this boot-time pass has
1394
+ // finished yet.
1395
+ setImmediate(() => {
1396
+ consolidateAllFlatPrds(allProjectCwds()).catch((e) => {
1397
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'deferred flat-PRD consolidation failed', meta: { error: e?.message } });
1398
+ });
1399
+ });
1220
1400
 
1221
1401
  // Rollout migration for the PRD-authoring-lockdown feature: stamp every
1222
1402
  // pre-existing PRD as legacy-adopted BEFORE reconcile() ever runs its
@@ -1484,14 +1664,15 @@ function computeBlockedChains(jobs) {
1484
1664
  * findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
1485
1665
  *
1486
1666
  * Pure, no IO. A 'quarantined' row (no createdVia provenance) can otherwise
1487
- * sit forever with nothing looking at it — quarantine only ever clears via a
1488
- * human adopting or archiving it. This is the escalation half of that gate:
1489
- * any quarantined row whose recorded quarantine timestamp (statusHistory's
1490
- * `to === 'quarantined'` entry — stamped at creation, or backfilled from the
1491
- * PRD file's mtime by reconcile() for rows quarantined before that stamp
1492
- * existed) is older than `thresholdMs` is reported so the caller can
1493
- * warn-log and surface it distinctly. A row with no recoverable timestamp is
1494
- * skipped rather than guessed at.
1667
+ * sit forever with nothing looking at it — quarantine used to clear only via
1668
+ * a human adopting or archiving it; autoResolveQuarantine below now gives it
1669
+ * a bounded automatic exit too. This function stays the escalation/warn half
1670
+ * of that gate: any quarantined row whose recorded quarantine timestamp
1671
+ * (statusHistory's `to === 'quarantined'` entry — stamped at creation, or
1672
+ * backfilled from the PRD file's mtime by reconcile() for rows quarantined
1673
+ * before that stamp existed) is older than `thresholdMs` is reported so the
1674
+ * caller can warn-log and surface it distinctly. A row with no recoverable
1675
+ * timestamp is skipped rather than guessed at.
1495
1676
  */
1496
1677
  function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
1497
1678
  const stale = [];
@@ -1507,6 +1688,106 @@ function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
1507
1688
  return stale;
1508
1689
  }
1509
1690
 
1691
+ // Bounded automatic exit for a quarantined row (this PRD): up to
1692
+ // QUARANTINE_RESOLVE_CAP auto-resolve attempts, each gated on having sat
1693
+ // `quarantined` for QUARANTINE_ESCALATE_MS, before autoResolveQuarantine
1694
+ // below settles the row to 'skipped' rather than leaving it as a dead end
1695
+ // only a human `scheduler_reset_job`/adopt action could ever clear. A single
1696
+ // attempt is enough in practice — the outcome is terminal — but the counter
1697
+ // still guards against two overlapping ticks both trying to resolve the
1698
+ // same row.
1699
+ const QUARANTINE_RESOLVE_CAP = 1;
1700
+
1701
+ /**
1702
+ * Kill-switch gate for the quarantine auto-resolve pass below
1703
+ * (SM_QUARANTINE_AUTORESOLVE_DISABLE=1), same shape as
1704
+ * failedAutoResetDisabled/needsReviewAutoResolveDisabled.
1705
+ */
1706
+ function quarantineAutoResolveDisabled() {
1707
+ return process.env.SM_QUARANTINE_AUTORESOLVE_DISABLE === '1';
1708
+ }
1709
+
1710
+ /**
1711
+ * selectQuarantineAutoResolveTargets(jobs, now, thresholdMs) →
1712
+ * [{ slug, cwd, ageMs }]
1713
+ *
1714
+ * Pure selector — no IO. Same age computation as findStaleQuarantinedJobs
1715
+ * above, bounded additionally by quarantineResolveAttempts so a row already
1716
+ * auto-resolved (or mid-resolve on a race) is never re-selected. Deliberately
1717
+ * does NOT check createdVia here — that requires a disk read of the PRD
1718
+ * file, and doing it at selection time would let this pass act on a
1719
+ * snapshot that's gone stale by the time the mutate() pass actually runs.
1720
+ * autoResolveQuarantine below re-reads createdVia fresh, immediately before
1721
+ * transitioning, inside the same mutate() callback that applies this
1722
+ * selector's targets — see that function's own header for why.
1723
+ */
1724
+ function selectQuarantineAutoResolveTargets(jobs, now, thresholdMs) {
1725
+ const targets = [];
1726
+ for (const j of jobs ?? []) {
1727
+ if (j.status !== 'quarantined') continue;
1728
+ if ((j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue;
1729
+ const entry = (j.statusHistory || []).find((h) => h.to === 'quarantined');
1730
+ if (!entry) continue;
1731
+ const since = Date.parse(entry.at);
1732
+ if (Number.isNaN(since)) continue;
1733
+ const ageMs = now - since;
1734
+ if (ageMs < thresholdMs) continue;
1735
+ targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs });
1736
+ }
1737
+ return targets;
1738
+ }
1739
+
1740
+ /**
1741
+ * autoResolveQuarantine(job, ageMs) → Promise<'skipped'|null>
1742
+ *
1743
+ * Applies the bounded automatic exit to a single quarantined row (mutates in
1744
+ * place; calls transitionJob + appendAuditEvent) — extracted so it's
1745
+ * unit-testable without going through mutate()/queue.json IO, same shape as
1746
+ * applyNeedsReviewAutoResolve above.
1747
+ *
1748
+ * Re-validates status + the attempts cap itself (race guard, mirrors the
1749
+ * other auto-resolve loops in the 10-minute interval body), THEN re-reads the
1750
+ * PRD file's createdVia frontmatter fresh from disk before doing anything
1751
+ * else. That ordering is load-bearing: reconcile()'s adopt path (the only
1752
+ * OTHER route off 'quarantined') promotes a row to 'pending' the instant it
1753
+ * observes a createdVia stamp, on its own independent pass — if this
1754
+ * function trusted a snapshot taken before its own turn to run, it could
1755
+ * transition a row to 'skipped' the same tick reconcile() already adopted it
1756
+ * to 'pending', silently discarding a PRD a human just fixed. Checking here,
1757
+ * immediately before the transition, inside the caller's mutate() callback,
1758
+ * closes that window.
1759
+ *
1760
+ * A PRD file that cannot be found or parsed at all is treated as still
1761
+ * lacking provenance — there is no proof it has one, and stalling forever on
1762
+ * an unreadable file would defeat the point of a bounded exit (same
1763
+ * can't-prove-it/don't-guess-but-don't-stall posture as the rest of this
1764
+ * file's stale-row detectors).
1765
+ */
1766
+ async function autoResolveQuarantine(job, ageMs) {
1767
+ if (!job || job.status !== 'quarantined') return null;
1768
+ if ((job.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) return null;
1769
+
1770
+ let createdVia = null;
1771
+ try {
1772
+ const resolvedDir = await findPrdDir(job.slug);
1773
+ const prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
1774
+ const parsed = await parsePrd(prdPath);
1775
+ createdVia = parsed.createdVia ?? null;
1776
+ } catch { /* unreadable/gone — no provenance found, so it stays "lacking" */ }
1777
+ if (createdVia) return null; // reconcile()'s own adopt path owns this row now
1778
+
1779
+ const attempt = (job.quarantineResolveAttempts ?? 0) + 1;
1780
+ job.quarantineResolveAttempts = attempt;
1781
+ job.error = `quarantined without createdVia provenance past the ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h `
1782
+ + 'escalation window — auto-resolved to skipped';
1783
+ transitionJob(job, 'skipped', {
1784
+ reason: 'quarantined without createdVia provenance past escalation window',
1785
+ source: 'autoResolveQuarantine',
1786
+ });
1787
+ appendAuditEvent('quarantine_auto_resolved', { slug: job.slug, cwd: job.cwd ?? null, ageMs: ageMs ?? null, attempt });
1788
+ return 'skipped';
1789
+ }
1790
+
1510
1791
  /**
1511
1792
  * findOverrunningJobs(jobs, now, { factor, floorMs }) → [{ slug, cwd, estimateMinutes, ranMs, ratio }]
1512
1793
  *
@@ -1553,6 +1834,99 @@ function findOverrunningJobs(jobs, now, { factor, floorMs } = {}) {
1553
1834
  return out;
1554
1835
  }
1555
1836
 
1837
+ /**
1838
+ * computeJobBudgetMs(estimateMinutes, { factor, floorMs, ceilingMs }) → number
1839
+ *
1840
+ * Pure. `budgetMs = clamp(estimateMinutes * factor, floorMs, ceilingMs)` — see
1841
+ * JOB_BUDGET_FACTOR's header comment (schedulerConfig.cjs) for the measured
1842
+ * p50/p90/max this is calibrated against. A missing/zero/non-finite estimate
1843
+ * is treated as 0, which the floor clamp then dominates — "jobs with a
1844
+ * missing estimate get the floor" falls straight out of the clamp, no
1845
+ * special-casing needed.
1846
+ */
1847
+ function computeJobBudgetMs(estimateMinutes, { factor, floorMs, ceilingMs } = {}) {
1848
+ const f = typeof factor === 'number' && factor > 0 ? factor : JOB_BUDGET_FACTOR;
1849
+ const floor = typeof floorMs === 'number' && floorMs >= 0 ? floorMs : JOB_BUDGET_FLOOR_MS;
1850
+ const ceiling = typeof ceilingMs === 'number' && ceilingMs > 0 ? ceilingMs : JOB_BUDGET_CEILING_MS;
1851
+ const est = Number(estimateMinutes);
1852
+ const minutes = Number.isFinite(est) && est > 0 ? est : 0;
1853
+ return Math.min(Math.max(minutes * f * 60_000, floor), ceiling);
1854
+ }
1855
+
1856
+ /**
1857
+ * classifyBudgetKill(res, landedCommitEvidence) → { status, reason, landedCommit } | null
1858
+ *
1859
+ * Pure. `res` is executeJob's resolved outcome — only fires when
1860
+ * `res.killedByWatchdog === 'budget'` (stamped by the budget watchdog inside
1861
+ * executeJob, never inferred from exit code/duration alone, so it can never
1862
+ * collide with an ordinary idle-tail/deadman/external kill). ALWAYS routes to
1863
+ * needs_review — never 'failed' (spawnJob's ordinary non-zero-exit default)
1864
+ * and never silently 'completed' (executeJob's onExit excludes
1865
+ * killedByWatchdog === 'budget' from the result=success → exit 0 mapping
1866
+ * idle-tail/deadman get) — so a budget kill is a visible, actionable park,
1867
+ * never a retry (classifyFailureOutcome/selectAutoFixTargets only ever see
1868
+ * 'failed'/ordinary needs_review rows, not this one — see
1869
+ * selectAutoFixTargets' own budget_exceeded exclusion) and never a discard:
1870
+ * `landedCommitEvidence`, when the caller resolved one via the SAME
1871
+ * commit-guard evidence check the plain sigterm/exit paths already use, is
1872
+ * threaded onto the row as `landedCommit` so a job that HAD already
1873
+ * committed before overrunning is still adjudicated on its git evidence.
1874
+ */
1875
+ function classifyBudgetKill(res, landedCommitEvidence) {
1876
+ if (!res || res.killedByWatchdog !== 'budget') return null;
1877
+ return {
1878
+ status: 'needs_review',
1879
+ reason: res.budgetKillReason || `wall-clock budget exceeded (exit ${res.exitCode})`,
1880
+ landedCommit: landedCommitEvidence || null,
1881
+ };
1882
+ }
1883
+
1884
+ /**
1885
+ * isJobBudgetExempt(job) → boolean
1886
+ *
1887
+ * Pure. `quietMachine: true` PRDs (their whole point is running alone,
1888
+ * un-contended, for a timing-sensitive measurement) and any PRD with an
1889
+ * explicit `budgetExempt: true` opt-out have no wall-clock kill ceiling.
1890
+ */
1891
+ function isJobBudgetExempt(job) {
1892
+ return job?.quietMachine === true || job?.budgetExempt === true;
1893
+ }
1894
+
1895
+ /** Pure predicate the budget watchdog's shouldFire calls — single source of
1896
+ * truth for "has this job run past its own budget" so it's unit-testable
1897
+ * without spinning up real timers. */
1898
+ function shouldKillForBudget(elapsedMs, budgetMs) {
1899
+ return elapsedMs >= budgetMs;
1900
+ }
1901
+
1902
+ /**
1903
+ * resolveBudgetKillOutcome({ killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes })
1904
+ * → { killedByWatchdog: 'budget'|null, budgetKillReason: string|null }
1905
+ *
1906
+ * Pure. `ctx.killedByWatchdog` is stamped by the budget watchdog's action()
1907
+ * the instant its periodic shouldFire() observes elapsedMs >= jobBudgetMs —
1908
+ * but that setInterval tick and the child's real 'exit' event both run on
1909
+ * the SAME single-threaded event loop, so it's possible to observe the
1910
+ * budget threshold crossed and call ctx.killTree() in the same window the
1911
+ * agent happens to exit cleanly (exit 0) or fails on its own for an
1912
+ * unrelated reason — ctx.killTree() against an already-exited pid is a
1913
+ * silent no-op (ESRCH, caught), but the flag would still read 'budget'
1914
+ * unless gated here. Only trusted when the exit SHAPE actually looks like a
1915
+ * signal kill (killedBySignal — mirrors the exact same check onExit already
1916
+ * uses for its own mappedToSuccess exclusion), so a clean exit=0 or an
1917
+ * ordinary unrelated non-zero failure racing the watchdog's tick is never
1918
+ * misclassified as a budget kill downstream.
1919
+ */
1920
+ function resolveBudgetKillOutcome({ killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes }) {
1921
+ if (killedByWatchdog !== 'budget' || !killedBySignal) {
1922
+ return { killedByWatchdog: killedByWatchdog === 'budget' ? null : (killedByWatchdog ?? null), budgetKillReason: null };
1923
+ }
1924
+ return {
1925
+ killedByWatchdog: 'budget',
1926
+ budgetKillReason: `wall-clock budget exceeded: ran ${Math.round(durationMs / 60_000)}m against a ${Math.round(jobBudgetMs / 60_000)}m budget (estimateMinutes=${estimateMinutes ?? 0})`,
1927
+ };
1928
+ }
1929
+
1556
1930
  /**
1557
1931
  * findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
1558
1932
  * → [{ slug, cwd, ageMs, restoreStatus }]
@@ -1772,7 +2146,7 @@ async function listPrdFiles() {
1772
2146
  ensureDirs();
1773
2147
  const dirs = candidatePrdsDirs();
1774
2148
  const perDir = await Promise.all(dirs.map((dir) => prdParser.listPrdFiles(dir)));
1775
- return perDir.flat().sort();
2149
+ return { files: perDir.flat().sort(), dirCount: dirs.length };
1776
2150
  }
1777
2151
 
1778
2152
  /**
@@ -1916,16 +2290,31 @@ async function reconcile(state) {
1916
2290
  if (state && state.unreadable) {
1917
2291
  throw new Error(`reconcile skipped: queue.json unreadable (${state.unreadable})`);
1918
2292
  }
2293
+ // Per-phase timing (PRD: reconcile evidence trail) — plain Date.now() diffs,
2294
+ // matching the ad-hoc elapsedMs idiom already used in health.cjs/
2295
+ // definitionOfDone.cjs. Only logged when the total exceeds
2296
+ // RECONCILE_SLOW_PASS_MS (see the warn emission at the bottom of this
2297
+ // function); a normal-speed pass logs nothing.
2298
+ const reconcileStartMs = Date.now();
2299
+ const phaseMs = {};
2300
+
1919
2301
  // Sweep the retired flat prds/ dir BEFORE scanning it below. reconcile()
1920
- // has several callers besides tickQueue's ~60s poll (broadcast,
1921
- // rescheduleTimer, the schedule:state IPC handler, schedule:rescan) — this
2302
+ // has several callers besides tickQueue's ~60s poll (broadcast()'s
2303
+ // coalescer, rescheduleTimer, schedule:rescan, schedule:adopt-prd) — this
1922
2304
  // lives here, not in any one caller, so the "a hand-written PRD in the flat
1923
2305
  // dir is swept before it can become a job" guarantee holds regardless of
1924
2306
  // which caller triggers this reconcile pass. A freshly hand-written file
1925
2307
  // has no queue row yet, so it is never "live" and gets archived here
1926
2308
  // instead of ever reaching the onDisk scan below.
2309
+ let phaseStartMs = Date.now();
1927
2310
  await consolidateAllFlatPrds(allProjectCwds());
1928
- const files = await listPrdFiles();
2311
+ phaseMs.flatPrdSweep = Date.now() - phaseStartMs;
2312
+
2313
+ phaseStartMs = Date.now();
2314
+ const { files, dirCount } = await listPrdFiles();
2315
+ phaseMs.prdDirResolve = Date.now() - phaseStartMs;
2316
+
2317
+ phaseStartMs = Date.now();
1929
2318
  const onDisk = new Map();
1930
2319
  for (const f of files) {
1931
2320
  try {
@@ -1937,6 +2326,7 @@ async function reconcile(state) {
1937
2326
  console.warn('[scheduler] failed to parse', f, e?.message);
1938
2327
  }
1939
2328
  }
2329
+ phaseMs.parseLoop = Date.now() - phaseStartMs;
1940
2330
 
1941
2331
  const next = [];
1942
2332
  const seen = new Set();
@@ -2002,7 +2392,9 @@ async function reconcile(state) {
2002
2392
  // membership, so moving the file between Epic dirs must re-point the row.
2003
2393
  epicId: p.epicId ?? job.epicId ?? null,
2004
2394
  dependsOn: p.dependsOn,
2395
+ disposition: p.disposition ?? null,
2005
2396
  quietMachine: p.quietMachine === true,
2397
+ budgetExempt: p.budgetExempt === true,
2006
2398
  originSessionId: job.originSessionId
2007
2399
  ?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
2008
2400
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
@@ -2057,9 +2449,11 @@ async function reconcile(state) {
2057
2449
  // ScheduleJobSchema (e.g. the 1021/1022 incident's `"status": "queued"`) —
2058
2450
  // see the repair pass below, right after historyBySlug is available.
2059
2451
  const invalidJobs = Array.isArray(state.invalidJobs) ? state.invalidJobs : [];
2452
+ phaseStartMs = Date.now();
2060
2453
  const historyBySlug = (unmatchedSlugs.length > 0 || terminalDroppedNeedingHistoryCheck.length > 0 || invalidJobs.length > 0)
2061
2454
  ? await queueHistory.historyTerminalBySlug()
2062
2455
  : new Map();
2456
+ phaseMs.historyLookup = Date.now() - phaseStartMs;
2063
2457
 
2064
2458
  // Backfill: any terminal job dropped above whose slug isn't already in
2065
2459
  // history.jsonl gets written now, before its row is gone for good. This is
@@ -2115,7 +2509,9 @@ async function reconcile(state) {
2115
2509
  sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
2116
2510
  epicId: p.epicId ?? inv.row?.epicId ?? null,
2117
2511
  dependsOn: p.dependsOn,
2512
+ disposition: p.disposition ?? null,
2118
2513
  quietMachine: p.quietMachine === true,
2514
+ budgetExempt: p.budgetExempt === true,
2119
2515
  originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
2120
2516
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
2121
2517
  agentType: p.agentType ?? inv.row?.agentType ?? null,
@@ -2237,7 +2633,9 @@ async function reconcile(state) {
2237
2633
  sourceTabId: p.sourceTabId,
2238
2634
  epicId: p.epicId ?? null,
2239
2635
  dependsOn: p.dependsOn,
2636
+ disposition: p.disposition ?? null,
2240
2637
  quietMachine: p.quietMachine === true,
2638
+ budgetExempt: p.budgetExempt === true,
2241
2639
  originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
2242
2640
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
2243
2641
  agentType: p.agentType ?? null,
@@ -2334,7 +2732,8 @@ async function reconcile(state) {
2334
2732
  // small. Append BEFORE dropping so a crash between the two can't lose a
2335
2733
  // record — appendHistory dedupes by slug+runId, so a replay of the same
2336
2734
  // batch on next boot is a safe no-op.
2337
- const nowMs = Date.now();
2735
+ phaseStartMs = Date.now();
2736
+ const nowMs = phaseStartMs;
2338
2737
  const { hot, toArchive } = queueHistory.partitionJobs(sorted, nowMs);
2339
2738
  if (toArchive.length > 0) {
2340
2739
  await queueHistory.appendHistory(toArchive);
@@ -2370,6 +2769,17 @@ async function reconcile(state) {
2370
2769
  } catch (e) {
2371
2770
  console.warn('[scheduler] autoArchiveCompleted failed', e?.message);
2372
2771
  }
2772
+ phaseMs.queueWrite = Date.now() - phaseStartMs;
2773
+
2774
+ const totalMs = Date.now() - reconcileStartMs;
2775
+ if (totalMs > RECONCILE_SLOW_PASS_MS) {
2776
+ logs.writeLine({
2777
+ level: 'warn',
2778
+ scope: 'scheduler',
2779
+ message: `reconcile() pass took ${totalMs}ms (threshold ${RECONCILE_SLOW_PASS_MS}ms)`,
2780
+ meta: { totalMs, phaseMs, prdFileCount: files.length, resolvedDirCount: dirCount },
2781
+ });
2782
+ }
2373
2783
 
2374
2784
  return state;
2375
2785
  }
@@ -2559,20 +2969,36 @@ function applyPauseCleared(wasPaused, token) {
2559
2969
  return token;
2560
2970
  }
2561
2971
 
2972
+ /**
2973
+ * Human-readable explanation for a `reason: 'load-deferred'` tick, surfaced
2974
+ * to the renderer via lastTick.detail. Names the gate, the measured ratio,
2975
+ * the threshold and how long the stretch has been held — the box could sit
2976
+ * gated for 80+ minutes with nothing in the UI naming why (PRD: load gate
2977
+ * hysteresis). Pure so it's unit-testable without driving tickQueue's full
2978
+ * fs/worktree machinery.
2979
+ */
2980
+ function formatLoadGateDetail(load) {
2981
+ const heldMinutes = Math.round(load.gatedSinceMs / 60_000);
2982
+ return `CPU load gate: loadavg1 ${load.loadavg1} / ${load.cores} cores = ${load.ratio} > threshold ${load.threshold}, held for ${heldMinutes}m`;
2983
+ }
2984
+
2562
2985
  function attachWindow(w) { mainWindow = w; }
2563
2986
 
2564
2987
  /**
2565
2988
  * Build the snapshot payload consumed by both the `schedule:state` IPC
2566
- * handler and the `schedule:state` broadcast event. The IPC return adds a
2567
- * `paths` map (renderer uses it for "open folder" actions); broadcast omits
2568
- * it because subscribers don't need to re-derive paths on every tick.
2989
+ * handler and the `schedule:state` broadcast event.
2569
2990
  */
2570
- function buildScheduleStatePayload(state, { withPaths = false } = {}) {
2991
+ function buildScheduleStatePayload(state) {
2571
2992
  const payload = {
2572
2993
  config: state.config,
2573
2994
  jobs: state.jobs,
2574
2995
  scheduledFor: state.scheduledFor,
2575
2996
  lastRunAt: state.lastRunAt,
2997
+ // Distinct from lastRunAt (only stamped when a batch actually launches):
2998
+ // stamped every time tickQueue reaches the picker at all. See
2999
+ // classifyQueueHealth/classifyQueueStarvation's header comments for why
3000
+ // the two must never merge.
3001
+ lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
2576
3002
  nextReset: getNextResetCached(),
2577
3003
  paused: state.paused,
2578
3004
  // Launch circuit breaker (issue #11): which personas cannot launch right
@@ -2603,9 +3029,6 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
2603
3029
  };
2604
3030
  })(),
2605
3031
  };
2606
- if (withPaths) {
2607
- payload.paths = { root: ROOT, prds: PRDS_DIR, runs: RUNS_DIR, queue: queueStore.MACHINE_STATE_PATH };
2608
- }
2609
3032
  return payload;
2610
3033
  }
2611
3034
 
@@ -2616,20 +3039,33 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
2616
3039
  // per mutation. Callers where latency matters (pause/resume, job
2617
3040
  // start/finish/reap/reset) pass `{ flush: true }` to bypass the window and
2618
3041
  // send immediately.
3042
+ // getPayload is the coalescer's ONLY entry point back into queue state, so
3043
+ // routing the reconcile+write pair through it (rather than broadcast() doing
3044
+ // its own bare readQueue/reconcile/writeQueue) is what makes a burst of
3045
+ // broadcast() calls cost exactly one reconcile + one write per coalesce
3046
+ // window. It goes through mutate() (not a bare read/write pair) so it
3047
+ // serializes against every other concurrent mutation and inherits mutate's
3048
+ // pre-fn `state.unreadable` bail — never write a state derived from a failed
3049
+ // read. `module.exports.reconcile` (not the bare local binding) is the seam
3050
+ // tests spy on, matching this file's existing testable-seam convention (see
3051
+ // module.exports.stashList/evaluateSharedTreeGuard/committedInWindow above).
2619
3052
  const broadcastCoalescer = createBroadcastCoalescer({
2620
3053
  delayMs: BROADCAST_COALESCE_MS,
2621
3054
  send: (payload) => {
2622
3055
  if (!mainWindow || mainWindow.isDestroyed()) return;
2623
3056
  sendIfAlive(mainWindow, 'schedule:state', payload);
2624
3057
  },
2625
- getPayload: async () => buildScheduleStatePayload(await readQueue()),
3058
+ getPayload: () => mutate(async (state) => {
3059
+ await module.exports.reconcile(state);
3060
+ return buildScheduleStatePayload(state);
3061
+ }),
2626
3062
  });
2627
3063
 
3064
+ // Reconcile is unconditional — even with no window attached (a
3065
+ // scheduler-passive or headless instance), discovery must still run so
3066
+ // on-disk PRDs get onboarded. Only the actual IPC push is window-gated,
3067
+ // inside the coalescer's own `send`.
2628
3068
  async function broadcast(opts = {}) {
2629
- if (!mainWindow || mainWindow.isDestroyed()) return;
2630
- const state = await readQueue();
2631
- await reconcile(state);
2632
- await writeQueue(state);
2633
3069
  if (opts.flush) {
2634
3070
  await broadcastCoalescer.flush();
2635
3071
  } else {
@@ -2929,7 +3365,7 @@ const BOOT_ORPHAN_KILL_GRACE_MS = 6000;
2929
3365
  * process may still be writing to it, so reading now risks misclassifying a
2930
3366
  * job that is about to emit result:success as no_result and double-running it.
2931
3367
  * Ported from reconcileQueueOffline's cross-tick escalation (see
2932
- * scripts/lib/watchdogHelpers.cjs) — here it's a single deferred window since
3368
+ * src/main/lib/watchdogHelpers.cjs) — here it's a single deferred window since
2933
3369
  * this process stays up to revisit it, rather than a separate short-lived
2934
3370
  * watchdog process needing another tick.
2935
3371
  */
@@ -2949,17 +3385,26 @@ function partitionBootOrphans(jobs, isAlive = claudePidAlive) {
2949
3385
  }
2950
3386
 
2951
3387
  /**
2952
- * applyOrphanOutcome(job, outcome, killNote?) → void
3388
+ * applyOrphanOutcome(job, outcome, killNote?, confirmedLandedCommit?) → void
2953
3389
  *
2954
3390
  * Mutates `job` in place to finalize a boot-orphaned 'running' job given its
2955
3391
  * classified run outcome: success/failed finalize terminally; no_result/unknown
2956
3392
  * re-queues to pending bounded by ORPHAN_REQUEUE_CAP. The status-mutation
2957
3393
  * semantics (and the cap-exhaustion boundary) match the now-deleted
2958
- * reconcileQueueOffline (scripts/lib/watchdogHelpers.cjs) verbatim; killNote
3394
+ * reconcileQueueOffline (src/main/lib/watchdogHelpers.cjs) verbatim; killNote
2959
3395
  * plumbing differs slightly (see call sites) since this path always knows
2960
3396
  * pid liveness up front rather than re-checking per tick.
3397
+ *
3398
+ * `confirmedLandedCommit` is the same evidence-before-failure gate
3399
+ * reapDeadRunningJobs applies (see resolveLandedCommitEvidence): a job that
3400
+ * dies while the app itself is offline is classified 'failed' from its log
3401
+ * tail alone, exactly like the pre-fix reap path was — so without this, an
3402
+ * orphaned job that actually landed a real commit is reachable via boot
3403
+ * reconciliation even though the live reap path is now guarded. Callers
3404
+ * must resolve this (a git spawn) BEFORE calling mutate(), never inside it —
3405
+ * pass null to skip the gate (e.g. when the outcome isn't 'failed').
2961
3406
  */
2962
- function applyOrphanOutcome(job, outcome, killNote = '') {
3407
+ function applyOrphanOutcome(job, outcome, killNote = '', confirmedLandedCommit = null) {
2963
3408
  const now = new Date().toISOString();
2964
3409
  if (outcome === 'success') {
2965
3410
  transitionJob(job, 'completed', { reason: 'boot orphan reconciliation: run succeeded', source: 'applyOrphanOutcome' });
@@ -2968,9 +3413,16 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
2968
3413
  job.finishedAt = now;
2969
3414
  delete job.runtime;
2970
3415
  } else if (outcome === 'failed') {
2971
- transitionJob(job, 'failed', { reason: `orphaned: app restarted while running${killNote}`, source: 'applyOrphanOutcome' });
2972
- job.exitCode = job.exitCode ?? 1;
2973
- job.error = `orphaned: app restarted while running${killNote}`;
3416
+ if (confirmedLandedCommit) {
3417
+ transitionJob(job, 'completed', { reason: `orphaned: app restarted while running${killNote}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence`, source: 'applyOrphanOutcome:landed' });
3418
+ job.exitCode = 0;
3419
+ job.error = null;
3420
+ job.landedCommit = confirmedLandedCommit;
3421
+ } else {
3422
+ transitionJob(job, 'failed', { reason: `orphaned: app restarted while running${killNote}`, source: 'applyOrphanOutcome' });
3423
+ job.exitCode = job.exitCode ?? 1;
3424
+ job.error = `orphaned: app restarted while running${killNote}`;
3425
+ }
2974
3426
  job.finishedAt = now;
2975
3427
  delete job.runtime;
2976
3428
  } else {
@@ -2978,6 +3430,13 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
2978
3430
  if (tries < ORPHAN_REQUEUE_CAP) {
2979
3431
  resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}`, { source: 'applyOrphanOutcome' });
2980
3432
  job.orphanRetries = tries + 1;
3433
+ } else if (confirmedLandedCommit) {
3434
+ transitionJob(job, 'completed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence`, source: 'applyOrphanOutcome:landed' });
3435
+ job.exitCode = 0;
3436
+ job.error = null;
3437
+ job.landedCommit = confirmedLandedCommit;
3438
+ job.finishedAt = now;
3439
+ delete job.runtime;
2981
3440
  } else {
2982
3441
  transitionJob(job, 'failed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`, source: 'applyOrphanOutcome' });
2983
3442
  job.exitCode = job.exitCode ?? 1;
@@ -3865,6 +4324,60 @@ async function pathExistsInTree(cwd, treeish, p) {
3865
4324
  }
3866
4325
  }
3867
4326
 
4327
+ /**
4328
+ * resolveLandedCommitEvidence(cwd, sha, sinceIso) → Promise<boolean>
4329
+ *
4330
+ * Bounded, non-fatal proof that `sha` is a real, resolvable commit in the
4331
+ * repo at `cwd`, committed no earlier than `sinceIso` — `git cat-file -e
4332
+ * <sha>^{commit}` plus a `git log -1 --format=%cI` timestamp check, both via
4333
+ * execGitAt's existing spawn+timeout bound (never shell:true, never an
4334
+ * unbounded execSync). This is the evidence gate reapDeadRunningJobs (PRD:
4335
+ * reaper must consult completion evidence) adds ahead of stamping a reaped
4336
+ * row 'failed': a landedCommit field being non-empty is not proof by itself
4337
+ * (job 1192 had one and still got reaped 'failed') — only a git-verified
4338
+ * resolution is.
4339
+ *
4340
+ * The timestamp bound matters because `landedCommit` deliberately survives
4341
+ * resetJobFields (see the comment there) so a re-fired run can consult it as
4342
+ * priorLandedCommit — which means a STALE landedCommit from an earlier
4343
+ * dispatch of the same slug can still be sitting on the row when a LATER
4344
+ * dispatch dies for real. Without `sinceIso`, that stale-but-real sha would
4345
+ * satisfy `cat-file -e` and wrongly promote a genuine failure to
4346
+ * 'completed'. Passing the current dispatch's `row.startedAt` as `sinceIso`
4347
+ * closes that: only a commit landed during THIS run counts as evidence.
4348
+ *
4349
+ * `cwd` is normalized through opsOwnership's resolveProjectRoot first (the
4350
+ * same "never trust a raw agent cwd" reasoning delegationReadiness.cjs
4351
+ * already relies on) so a row reaped while its cwd is an ephemeral worktree
4352
+ * checkout resolves the commit against the real project root instead.
4353
+ *
4354
+ * Never throws: a missing sha, a resolveProjectRoot failure (ephemeral cwd,
4355
+ * thrown error), a spawn failure, a timeout, or a cwd that no longer exists
4356
+ * on disk all resolve to `false` — the caller's safe default is 'failed',
4357
+ * exactly like today, whenever this can't positively prove landing.
4358
+ */
4359
+ async function resolveLandedCommitEvidence(cwd, sha, sinceIso) {
4360
+ if (!sha || typeof sha !== 'string') return false;
4361
+ try {
4362
+ const root = resolveProjectRoot(cwd);
4363
+ await execGitAt(root, ['cat-file', '-e', `${sha}^{commit}`], { timeout: 10_000 });
4364
+ if (sinceIso) {
4365
+ const since = new Date(sinceIso).getTime();
4366
+ if (Number.isFinite(since)) {
4367
+ const committedIso = (await execGitAt(root, ['log', '-1', '--format=%cI', sha], { timeout: 10_000 })).trim();
4368
+ const committedAt = new Date(committedIso).getTime();
4369
+ // A commit dated before this dispatch even started can only be a
4370
+ // stale sha surviving from an earlier life of the row — never
4371
+ // evidence that THIS dispatch landed anything.
4372
+ if (Number.isFinite(committedAt) && committedAt < since) return false;
4373
+ }
4374
+ }
4375
+ return true;
4376
+ } catch {
4377
+ return false;
4378
+ }
4379
+ }
4380
+
3868
4381
  /**
3869
4382
  * Commit exactly `paths` (must already be dirty on disk) onto a dedicated
3870
4383
  * `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
@@ -4407,6 +4920,56 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4407
4920
  },
4408
4921
  };
4409
4922
 
4923
+ // Wall-clock budget watchdog: unlike idleTailWatchdog above (which only
4924
+ // fires when the log mtime STALLS), this fires on total elapsed wall-
4925
+ // clock time regardless of whether the job keeps writing output — the
4926
+ // gap a chatty-but-runaway executor slips through (see
4927
+ // computeJobBudgetMs's header for the measured p50/p90/max this budget
4928
+ // is calibrated against). `quietMachine` jobs and any PRD with an
4929
+ // explicit `budgetExempt: true` opt out entirely — logged once here so
4930
+ // an unbounded job is never silently unbounded.
4931
+ const jobBudgetMs = computeJobBudgetMs(job.estimateMinutes);
4932
+ const budgetExempt = isJobBudgetExempt(job);
4933
+ if (budgetExempt) {
4934
+ safeLog(`[scheduler] wall-clock budget watchdog EXEMPT for ${job.slug} ` +
4935
+ `(${job.quietMachine === true ? 'quietMachine' : 'budgetExempt'}) — no wall-clock kill ceiling this run\n`);
4936
+ }
4937
+ let budgetWarningStamped = false;
4938
+ const budgetWatchdog = {
4939
+ label: 'budget',
4940
+ intervalMs: IDLE_CHECK_INTERVAL_MS,
4941
+ shouldFire(ctx) {
4942
+ if (budgetExempt) return false;
4943
+ const elapsedMs = Date.now() - ctx.startedAt;
4944
+ if (!budgetWarningStamped && elapsedMs >= jobBudgetMs * BUDGET_WARNING_FRACTION) {
4945
+ budgetWarningStamped = true;
4946
+ // Fire-and-forget (side effect inside a sync predicate, same pattern
4947
+ // resultTailWatchdog's shouldFire already uses for agentResultSubtype)
4948
+ // — exposes the warning on the row well before the kill fires, so
4949
+ // the renderer can show it without waiting for the next tick.
4950
+ mutate((state) => {
4951
+ const j = state.jobs.find((x) => x.slug === job.slug);
4952
+ if (!j) return;
4953
+ j.budgetWarning = { budgetMs: jobBudgetMs, elapsedMs, at: new Date().toISOString() };
4954
+ }).catch((e) => console.warn('[scheduler] budget-warning stamp failed', job.slug, e?.message));
4955
+ }
4956
+ return shouldKillForBudget(elapsedMs, jobBudgetMs);
4957
+ },
4958
+ action(ctx) {
4959
+ const elapsedMs = Date.now() - ctx.startedAt;
4960
+ ctx.safeLog(`\n[scheduler] wall-clock budget watchdog: ran ${Math.round(elapsedMs / 60_000)}m ` +
4961
+ `(> ${Math.round(jobBudgetMs / 60_000)}m budget, estimateMinutes=${job.estimateMinutes ?? 0}) — SIGTERM process group\n`);
4962
+ ctx.killedByWatchdog = 'budget';
4963
+ ctx.killTree('SIGTERM');
4964
+ const budgetKillTimer = setTimeout(() => {
4965
+ ctx.safeLog(`\n[scheduler] budget watchdog: still alive ${Math.round(POST_RESULT_KILL_MS/1000)}s after SIGTERM — SIGKILL\n`);
4966
+ ctx.killTree('SIGKILL');
4967
+ }, POST_RESULT_KILL_MS);
4968
+ if (budgetKillTimer.unref) budgetKillTimer.unref();
4969
+ ctx.addTimer(budgetKillTimer);
4970
+ },
4971
+ };
4972
+
4410
4973
  // ---------- spawn ----------
4411
4974
 
4412
4975
  const { child } = withChildAndLog({
@@ -4437,8 +5000,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4437
5000
  detached: true,
4438
5001
  },
4439
5002
  },
4440
- watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog],
4441
- onExit({ exitCode, signal, killedByWatchdog: _kbw, error, spawnFailed, leakedDescendants, safeLog: sl }) {
5003
+ watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog, budgetWatchdog],
5004
+ onExit({ exitCode, signal, killedByWatchdog, error, spawnFailed, leakedDescendants, safeLog: sl }) {
4442
5005
  const durationMs = Date.now() - startedAt;
4443
5006
  const leaked = leakedDescendants ?? [];
4444
5007
  if (leaked.length > 0) {
@@ -4468,7 +5031,11 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4468
5031
  // and 137 (128+SIGKILL) in case the process exited via signal-as-code.
4469
5032
  let effectiveCode = exitCode;
4470
5033
  const killedBySignal = signal === 'SIGTERM' || signal === 'SIGKILL' || exitCode === 143 || exitCode === 137 || exitCode === null;
4471
- const mappedToSuccess = agentResultSubtype === 'success' && killedBySignal;
5034
+ // A budget kill must NEVER be laundered into a clean exit=0, even when
5035
+ // the agent had already emitted result=success before it fired — the
5036
+ // AC requires it always park needs_review, never silently 'completed'.
5037
+ // idle-tail/deadman/result-tail kills keep the existing success-mapping.
5038
+ const mappedToSuccess = agentResultSubtype === 'success' && killedBySignal && killedByWatchdog !== 'budget';
4472
5039
  if (mappedToSuccess) {
4473
5040
  effectiveCode = 0;
4474
5041
  sl(`\n[scheduler] mapping exit code=${exitCode} signal=${signal} → 0 ` +
@@ -4491,6 +5058,16 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4491
5058
  sl(`\n[scheduler] LAUNCH FAILURE (${launchFailed.kind}${launchFailed.httpStatus ? ` HTTP ${launchFailed.httpStatus}` : ''}): ` +
4492
5059
  `${launchFailed.message} — no turn was taken; this is not a PRD failure\n`);
4493
5060
  }
5061
+ // Formatted once, here, off the FINAL durationMs (more accurate than
5062
+ // the watchdog action's own snapshot at kill time) — matches the
5063
+ // reason string format the AC requires verbatim. See
5064
+ // resolveBudgetKillOutcome's own header for why this is gated on
5065
+ // killedBySignal, not on killedByWatchdog alone.
5066
+ const budgetKillOutcome = resolveBudgetKillOutcome({
5067
+ killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes: job.estimateMinutes,
5068
+ });
5069
+ const { budgetKillReason } = budgetKillOutcome;
5070
+ const effectiveKilledByWatchdog = budgetKillOutcome.killedByWatchdog;
4494
5071
  // Sync write: child 'exit' handler must flush meta before resolve()
4495
5072
  // so the spawnJob mutate() that follows sees the persisted exit code.
4496
5073
  config.writeJsonSync(metaPath, {
@@ -4501,10 +5078,14 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4501
5078
  launchEnvApplied: launchEnv && Object.keys(launchEnv).length ? Object.keys(launchEnv) : [],
4502
5079
  startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
4503
5080
  agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
5081
+ killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
4504
5082
  schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
4505
5083
  originSessionId, contextDigestApplied,
4506
5084
  });
4507
- resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, launchFailure: launchFailed, resultStats, leakedDescendants: leaked, sessionId });
5085
+ resolve({
5086
+ exitCode: effectiveCode, durationMs, rateLimited, networkError, launchFailure: launchFailed, resultStats,
5087
+ leakedDescendants: leaked, sessionId, killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
5088
+ });
4508
5089
  },
4509
5090
  });
4510
5091
 
@@ -4512,8 +5093,27 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4512
5093
  safeLog(`[scheduler] spawned pid=${child.pid} sessionId=${sessionId} (process group)\n\n`);
4513
5094
  // Make this job the OOM killer's preferred victim over Electron.
4514
5095
  biasJobOomScore(child.pid);
4515
- // Fire-and-forget pid persistence — best effort.
4516
- if (onPid) onPid(child.pid, sessionId, cwd).catch(() => {});
5096
+ // Persist runtime.pid with one retry — still fire-and-forget (must
5097
+ // never block the spawn), but a final failure is now loud instead of
5098
+ // silently swallowed. A silent failure here is exactly what let the
5099
+ // pidless-grace reaper terminalize a live, working job (runtime.pid
5100
+ // never landed, so selectReapableJobs had no way to tell "never
5101
+ // spawned" from "spawned but unrecorded").
5102
+ if (onPid) {
5103
+ (async () => {
5104
+ try {
5105
+ await onPid(child.pid, sessionId, cwd);
5106
+ } catch (firstErr) {
5107
+ try {
5108
+ await onPid(child.pid, sessionId, cwd);
5109
+ } catch (finalErr) {
5110
+ const message = finalErr?.message ?? String(finalErr);
5111
+ console.error(`[scheduler] FAILED to persist runtime.pid for ${job.slug} pid=${child.pid}: ${message}`);
5112
+ appendAuditEvent('job_pid_persist_failed', { slug: job.slug, cwd, pid: child.pid, error: message });
5113
+ }
5114
+ }
5115
+ })();
5116
+ }
4517
5117
  }
4518
5118
  });
4519
5119
  }
@@ -5435,8 +6035,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5435
6035
 
5436
6036
  // Commit-guard baseline: snapshot the working tree BEFORE the run so the
5437
6037
  // post-run check flags only paths THIS job left dirty, not pre-existing WIP.
6038
+ // Captured once, with status codes, so the shared-tree guard below can
6039
+ // tell "was untracked" apart from "was tracked-and-modified" (2026-09-12
6040
+ // incident) without a second `git status` call; every other consumer of
6041
+ // `guardBaseline` still gets the plain path-string array it always did.
5438
6042
  const guardCwd = job.cwd || defaultCwd;
5439
- const guardBaseline = await uncommittedChanges(guardCwd);
6043
+ const guardBaselineEntries = await uncommittedChangesWithStatus(guardCwd);
6044
+ const guardBaseline = guardBaselineEntries ? guardBaselineEntries.map((e) => e.path) : guardBaselineEntries;
5440
6045
  const guardHeadBefore = await gitHead(guardCwd);
5441
6046
  // Shared-tree stash guard baseline (incident 2026-09-01): captured
5442
6047
  // unconditionally, before worktree isolation is even attempted, so an
@@ -5488,6 +6093,22 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5488
6093
  console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
5489
6094
  } else {
5490
6095
  console.log(`[scheduler] ${job.slug}: running in main tree (worktree not used: ${worktree.reason})`);
6096
+ // A job losing worktree isolation must never be a silent downgrade
6097
+ // discoverable only by reading queue.json afterwards — every genuine
6098
+ // fallback (never the deliberate SM_JOB_WORKTREE_DISABLE opt-out) is
6099
+ // logged at warn in the durable ops error log, with the job slug, cwd,
6100
+ // and specific reason attached.
6101
+ if (!jobWorktree.isWorktreeDisabled()) {
6102
+ try {
6103
+ appendError({
6104
+ cwd: job.cwd || defaultCwd,
6105
+ scope: 'scheduler',
6106
+ level: 'warn',
6107
+ message: `${job.slug}: worktree isolation fell back to the SHARED working tree — ${worktree.reason}`,
6108
+ meta: { slug: job.slug, cwd: job.cwd || defaultCwd, reason: worktree.reason },
6109
+ });
6110
+ } catch { /* durable logging must never break dispatch */ }
6111
+ }
5491
6112
  }
5492
6113
  // dispatchPhase stamp folded into a single unconditional mutate covering
5493
6114
  // both branches above — the degraded-isolation fallback flag (skipped
@@ -5882,7 +6503,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5882
6503
  sharedTreeGuard = await module.exports.checkSharedTreeGuard({
5883
6504
  cwd: guardCwd,
5884
6505
  stashBaseline,
5885
- dirtyBaseline: guardBaseline,
6506
+ dirtyBaseline: guardBaselineEntries,
5886
6507
  headBefore: guardHeadBefore,
5887
6508
  slug: job.slug,
5888
6509
  });
@@ -5909,6 +6530,10 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5909
6530
  // narrowly to exit 143 — never applied to other non-zero exit codes or to
5910
6531
  // rateLimited (already handled separately, above).
5911
6532
  let sigtermCommitFound = false;
6533
+ // Verified SHA twin of sigtermCommitFound's boolean — only the budget-kill
6534
+ // path (below) threads this onto the row's landedCommit; classifySigtermWithCommit's
6535
+ // own needs_review branch is unchanged and keeps using the boolean alone.
6536
+ let sigtermLandedCommitEvidence = null;
5912
6537
  if (res.exitCode === 143 && !res.rateLimited) {
5913
6538
  const guardHeadAtSigterm = await gitHead(guardCwd);
5914
6539
  sigtermCommitFound = await computeCommittedDuringRun(
@@ -5918,6 +6543,10 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5918
6543
  job.startedAt,
5919
6544
  new Date().toISOString(),
5920
6545
  );
6546
+ if (guardHeadBefore && guardHeadAtSigterm && guardHeadAtSigterm !== guardHeadBefore) {
6547
+ const verified = await resolveLandedCommitEvidence(guardCwd, guardHeadAtSigterm, job.startedAt);
6548
+ if (verified) sigtermLandedCommitEvidence = guardHeadAtSigterm;
6549
+ }
5921
6550
  }
5922
6551
 
5923
6552
  // BLOCKED_BY_FOREIGN_WIP claim scan: the executor exits non-zero for this
@@ -5939,6 +6568,27 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5939
6568
  }
5940
6569
  }
5941
6570
 
6571
+ // Evidence gate for a PLAIN non-zero exit — not SIGTERM-with-commit
6572
+ // (classifySigtermWithCommit above already routes exit 143 to
6573
+ // needs_review when a commit landed) and not rate-limited (handled
6574
+ // separately). A process that dies non-zero for any OTHER reason
6575
+ // (crash during post-commit cleanup, an overrun watchdog's SIGKILL) is
6576
+ // observed directly by THIS exit handler — it never reaches
6577
+ // reapDeadRunningJobs' own git-verified landedCommit evidence gate, so
6578
+ // without this check the exact bug that gate exists to prevent (job
6579
+ // 1192: a landedCommit non-empty is not proof by itself, but discarding
6580
+ // proof of real landed work with no evidence check at all is worse)
6581
+ // recurs here, one call site over. Computed outside mutate() (I/O) like
6582
+ // every other pre-finalize git check above.
6583
+ let plainExitLandedCommitEvidence = null;
6584
+ if (res.exitCode !== 0 && res.exitCode !== 143 && !res.rateLimited) {
6585
+ const headAtPlainExit = await gitHead(guardCwd);
6586
+ if (guardHeadBefore && headAtPlainExit && headAtPlainExit !== guardHeadBefore) {
6587
+ const verified = await resolveLandedCommitEvidence(guardCwd, headAtPlainExit, job.startedAt);
6588
+ if (verified) plainExitLandedCommitEvidence = headAtPlainExit;
6589
+ }
6590
+ }
6591
+
5942
6592
  let actuallyFailed = false;
5943
6593
  let failedJobSnapshot = null;
5944
6594
  let needsInvestigationNow = false;
@@ -6009,7 +6659,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6009
6659
  // Determine effective status, applying the verifier verdict for exit=0 runs.
6010
6660
  let effectiveStatus;
6011
6661
  let sigtermOverrideReason = null;
6012
- const sigtermOverride = res.exitCode !== 0
6662
+ // Wall-clock budget kill — checked FIRST and unconditionally wins:
6663
+ // never 'failed', never silently 'completed', and (via
6664
+ // sigtermLandedCommitEvidence/plainExitLandedCommitEvidence, whichever
6665
+ // this exit code populated) still adjudicated on git evidence rather
6666
+ // than discarded. See classifyBudgetKill's own header.
6667
+ const budgetKill = classifyBudgetKill(res, sigtermLandedCommitEvidence || plainExitLandedCommitEvidence);
6668
+ const sigtermOverride = (!budgetKill && res.exitCode !== 0)
6013
6669
  ? classifySigtermWithCommit(res.exitCode, sigtermCommitFound)
6014
6670
  : null;
6015
6671
  // Validated against the LIVE row's own foreign-WIP manifest — never
@@ -6017,7 +6673,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6017
6673
  // launder a real regression into a block (PRD: give the executor a
6018
6674
  // first-class verdict for "the gate failed on a sibling's in-flight
6019
6675
  // file", but VALIDATE the claim rather than trust it).
6020
- const foreignWipValidation = (!sigtermOverride && foreignWipClaimedPaths !== null)
6676
+ const foreignWipValidation = (!budgetKill && !sigtermOverride && foreignWipClaimedPaths !== null)
6021
6677
  ? validateForeignWipBlockClaim(foreignWipClaimedPaths, s.jobs[i2])
6022
6678
  : null;
6023
6679
  // Consecutive-block streak: cleared by default on every outcome and
@@ -6026,7 +6682,12 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6026
6682
  // between two blocks always resets "in a row" back to zero.
6027
6683
  const priorForeignWipBlockCount = s.jobs[i2].foreignWipBlockCount ?? 0;
6028
6684
  delete s.jobs[i2].foreignWipBlockCount;
6029
- if (sigtermOverride) {
6685
+ if (budgetKill) {
6686
+ effectiveStatus = budgetKill.status;
6687
+ sigtermOverrideReason = budgetKill.reason;
6688
+ s.jobs[i2].verifierVerdict = 'budget_exceeded';
6689
+ if (budgetKill.landedCommit) jobLandedCommitThisRun = budgetKill.landedCommit;
6690
+ } else if (sigtermOverride) {
6030
6691
  effectiveStatus = sigtermOverride.status;
6031
6692
  sigtermOverrideReason = sigtermOverride.reason;
6032
6693
  } else if (foreignWipValidation && foreignWipValidation.ok) {
@@ -6061,7 +6722,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6061
6722
  effectiveStatus = 'failed';
6062
6723
  sigtermOverrideReason = `SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP rejected — unlisted path(s) not in the disclosed foreign-WIP manifest: ${foreignWipValidation.invalidPaths.join(', ') || '(no FOREIGN_WIP_PATHS line)'}`;
6063
6724
  } else if (res.exitCode !== 0) {
6064
- effectiveStatus = 'failed';
6725
+ if (plainExitLandedCommitEvidence) {
6726
+ // Same conservative posture as the SIGTERM+commit case above:
6727
+ // a landed commit doesn't prove every AC line passed, so this
6728
+ // still routes to needs_review for a human/reverify pass,
6729
+ // never silently to completed.
6730
+ effectiveStatus = 'needs_review';
6731
+ sigtermOverrideReason = `exited ${res.exitCode} after landing a git-verified commit — verify AC before treating as done`;
6732
+ jobLandedCommitThisRun = plainExitLandedCommitEvidence;
6733
+ } else {
6734
+ effectiveStatus = 'failed';
6735
+ }
6065
6736
  } else if (
6066
6737
  !verifyResult
6067
6738
  || COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict)
@@ -6090,15 +6761,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6090
6761
  const finalizeReason = (effectiveStatus === 'completed' && verifyResult?.verdict === 'already_satisfied_on_main')
6091
6762
  ? verifyResult.reason
6092
6763
  : (sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`);
6093
- transitionJob(s.jobs[i2], effectiveStatus, { reason: finalizeReason, source: 'spawnJob:finalize' });
6094
- s.jobs[i2].finishedAt = new Date().toISOString();
6095
- s.jobs[i2].exitCode = res.exitCode;
6096
- s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
6097
- if (salvagePatch) {
6098
- s.jobs[i2].salvagePatch = salvagePatch;
6099
- } else {
6100
- delete s.jobs[i2].salvagePatch;
6101
- }
6764
+ // error/verifierVerdict are stamped BEFORE transitionJob() below —
6765
+ // needsReviewLedger's buildNeedsReviewEntryLine reads job.
6766
+ // verifierVerdict/heldReason/error synchronously off `job` the
6767
+ // instant transitionJob() runs (it's called inside transitionJob,
6768
+ // not deferred), so setting these after that call fed the durable
6769
+ // needs_review ledger stale/leftover values from before this run,
6770
+ // defeating its whole `byReason` rollup for the two escalation
6771
+ // paths that land here.
6102
6772
  s.jobs[i2].error = (effectiveStatus === 'needs_review' || s.jobs[i2].blockedByForeignWip === true)
6103
6773
  ? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
6104
6774
  // A failed job (non-zero exit) never consults verifyResult above,
@@ -6116,18 +6786,28 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6116
6786
  s.jobs[i2].landedCommit = jobLandedCommitThisRun;
6117
6787
  }
6118
6788
  // Persist the verifier's verdict string so the renderer can show it.
6119
- // 'blocked_by_foreign_wip_streak' is set above from the sigterm/
6120
- // exit-code path, never from verifyResult (which stays null on a
6121
- // non-zero exit) — never clobber it here.
6789
+ // 'blocked_by_foreign_wip_streak'/'budget_exceeded' are set above
6790
+ // from the sigterm/exit-code path, never from verifyResult (which
6791
+ // stays null on a non-zero exit) — never clobber either here.
6122
6792
  if (verifyResult?.verdict && verifyResult.verdict !== 'clean') {
6123
6793
  s.jobs[i2].verifierVerdict = verifyResult.verdict;
6124
- } else if (s.jobs[i2].verifierVerdict !== 'blocked_by_foreign_wip_streak') {
6794
+ } else if (!['blocked_by_foreign_wip_streak', 'budget_exceeded'].includes(s.jobs[i2].verifierVerdict)) {
6125
6795
  delete s.jobs[i2].verifierVerdict;
6126
6796
  }
6797
+ transitionJob(s.jobs[i2], effectiveStatus, { reason: finalizeReason, source: 'spawnJob:finalize' });
6798
+ s.jobs[i2].finishedAt = new Date().toISOString();
6799
+ s.jobs[i2].exitCode = res.exitCode;
6800
+ s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
6801
+ if (salvagePatch) {
6802
+ s.jobs[i2].salvagePatch = salvagePatch;
6803
+ } else {
6804
+ delete s.jobs[i2].salvagePatch;
6805
+ }
6127
6806
  // Closed-set outcome taxonomy (issue #11 list A2) so a queue row
6128
6807
  // says WHY it ended without anyone opening the transcript.
6129
6808
  s.jobs[i2].terminalReason = launchFailure.deriveTerminalReason({
6130
6809
  effectiveStatus, exitCode: res.exitCode, verifyResult, sigtermOverride, worktreeIntegrationFailure,
6810
+ budgetKill: !!budgetKill,
6131
6811
  });
6132
6812
  delete s.jobs[i2].launchFailure;
6133
6813
  delete s.jobs[i2].heldReason;
@@ -6714,7 +7394,7 @@ function tickQueue({ bypassLoadGate = false } = {}) {
6714
7394
  }
6715
7395
  return recordTick(
6716
7396
  { fired: false, reason: 'load-deferred', deferredCount: gatedBatch.length, ratio: load.ratio, threshold: load.threshold },
6717
- { detail: `load gate: ${load.loadavg1} / ${load.cores} cores = ${load.ratio} > ${load.threshold}`, holds },
7397
+ { detail: formatLoadGateDetail(load), holds },
6718
7398
  );
6719
7399
  }
6720
7400
 
@@ -6860,43 +7540,220 @@ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now,
6860
7540
  }
6861
7541
 
6862
7542
  /**
6863
- * The watchdog half: acts on classifyQueueStarvation. Called from the
6864
- * heartbeat, which already runs on its own timer independent of the billing
6865
- * poll loop — so a wedged or never-succeeding poll (the /api/oauth/usage
6866
- * endpoint was itself 429ing all of 2026-09-05) can no longer leave a queue
6867
- * with ready work idle indefinitely.
7543
+ * classifyQueueStarvationByProject({ jobs, paused, runningSet, lastRunAtMs, now, thresholdMs })
7544
+ * → [{ cwd, kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }]
7545
+ *
7546
+ * Per-project driver around classifyQueueStarvation's pure single-project
7547
+ * core. `runningCount > 0` inside that core used to be fed the MACHINE-WIDE
7548
+ * `runningSet.size`, which meant one long-lived job in ANY project disarmed
7549
+ * the watchdog for EVERY other project on the box — observed live
7550
+ * 2026-09-12: a job in starry-night-ships ran 80+ minutes while two other
7551
+ * projects sat starved/blocked for hours, and the watchdog never fired once
7552
+ * because "work is flowing" was true somewhere else. Partitioning by cwd
7553
+ * (the same grouping computeBlockedChains already does) fixes DETECTION only
7554
+ * — the idle clock (`lastRunAtMs`) stays machine-wide, since
7555
+ * `lastDispatchAttemptAt` is machine-level state, and only one tick is ever
7556
+ * forced per watchdog pass regardless of how many cwds are starved.
7557
+ *
7558
+ * Pure, no IO. Returns [] when paused (a DECISION, not a stall) or when no
7559
+ * project has a verdict.
7560
+ */
7561
+ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
7562
+ if (paused) return [];
7563
+ const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
7564
+ const byCwd = new Map();
7565
+ for (const j of rows) {
7566
+ const key = j.cwd || '(unknown)';
7567
+ if (!byCwd.has(key)) byCwd.set(key, []);
7568
+ byCwd.get(key).push(j);
7569
+ }
7570
+
7571
+ const verdicts = [];
7572
+ for (const [cwd, projectJobs] of byCwd) {
7573
+ // Same source of truth tickQueue itself uses for "is anything running":
7574
+ // the in-process runningSet OR a row already stamped status:'running'.
7575
+ const projRunningCount = projectJobs.filter(
7576
+ (j) => j.status === 'running' || runningSlugs?.has?.(j.slug),
7577
+ ).length;
7578
+ const verdict = classifyQueueStarvation({
7579
+ jobs: projectJobs,
7580
+ paused: false,
7581
+ runningCount: projRunningCount,
7582
+ lastRunAtMs,
7583
+ now,
7584
+ thresholdMs,
7585
+ });
7586
+ if (verdict) verdicts.push({ cwd, ...verdict });
7587
+ }
7588
+ return verdicts;
7589
+ }
7590
+
7591
+ /**
7592
+ * classifyQueueHealth({ jobs, paused, launchBlocks, runningSet, freeSlots,
7593
+ * totalSlots, lastDispatchAttemptAtMs, now, cwd, thresholdMs })
7594
+ * → { kind, cwd, pending, dispatchable, blockedChains, needsReviewCount, runningCount, ... }
7595
+ *
7596
+ * Single source of truth for the Scheduler page's queue-health header: the
7597
+ * one thing a human staring at a stale-looking queue needs is "which of the
7598
+ * genuinely different causes is this" (all slots busy? every pending row
7599
+ * blocked on a dependency? the dispatch driver itself never ticked?) — this
7600
+ * function names that cause instead of leaving the renderer to re-derive it.
7601
+ *
7602
+ * Reuses classifyQueueStarvation for the blocked/stalled read so the header
7603
+ * can never disagree with runQueueStarvationWatchdog's own decision to force
7604
+ * a tick: both are handed the same lastDispatchAttemptAt-based idle clock and
7605
+ * the same computeBlockedChains walk under the hood. Called here with
7606
+ * `thresholdMs: 0` first (a live header must say "blocked" the instant every
7607
+ * pending row is dependency-stuck, not wait out the watchdog's own 10-minute
7608
+ * grace period) — the returned `idleMs` is then compared against the REAL
7609
+ * `thresholdMs` to decide 'stalled' vs the healthy 'running' default, which
7610
+ * is exactly the comparison classifyQueueStarvation would make internally.
7611
+ *
7612
+ * `pending`/`dispatchable`/`blockedChains`/`needsReviewCount` are always
7613
+ * populated (via computeBlockedChains — the exact primitive
7614
+ * classifyQueueStarvation itself calls) regardless of kind, so a 'saturated'
7615
+ * or 'running' header can still say how much of the backlog is dependency-
7616
+ * blocked, not just the kinds where that's the headline cause.
7617
+ *
7618
+ * Kinds, in the priority order they're checked (paused is a decision, not a
7619
+ * stall; an open launch breaker explains an otherwise-inexplicable
7620
+ * non-dispatch before slot/dependency causes are even considered):
7621
+ * 'paused' — the scheduler itself is paused.
7622
+ * 'launch-blocked' — a pending row's persona has an active circuit-breaker
7623
+ * entry (lib/launchFailure.cjs).
7624
+ * 'idle' — nothing pending in this scope.
7625
+ * 'saturated' — pending work exists but every session slot is in use.
7626
+ * 'blocked' — nothing running, slots free, every pending row's
7627
+ * dependsOn chain terminates in a non-completed row.
7628
+ * 'stalled' — nothing running, slots free, at least one row is
7629
+ * dispatchable right now, and the dispatch driver has
7630
+ * been idle >= thresholdMs (agrees with the watchdog).
7631
+ * 'running' — the healthy default: work is flowing, or the driver
7632
+ * hasn't been idle long enough to call a stall yet.
7633
+ *
7634
+ * Pure, no IO. `cwd` scopes jobs/pending/blocked/needsReview to one project
7635
+ * (the Scheduler nav row is PROJECT-face — see CLAUDE.md); `freeSlots` /
7636
+ * `totalSlots` / `launchBlocks` stay machine-wide inputs by design, same as
7637
+ * WindowStrip's existing scopeCwd split.
7638
+ */
7639
+ function classifyQueueHealth({
7640
+ jobs, paused, launchBlocks, runningSet: runningSlugs, freeSlots, totalSlots,
7641
+ lastDispatchAttemptAtMs, now, cwd = null, thresholdMs = QUEUE_STARVATION_MS,
7642
+ } = {}) {
7643
+ const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
7644
+ const projectJobs = cwd ? rows.filter((j) => j.cwd === cwd) : rows;
7645
+ const pendingRows = projectJobs.filter((j) => j.status === 'pending');
7646
+ const runningRows = projectJobs.filter((j) => j.status === 'running' || runningSlugs?.has?.(j.slug));
7647
+ const needsReviewCount = projectJobs.filter((j) => j.status === 'needs_review').length;
7648
+
7649
+ // computeBlockedChains is the SAME primitive classifyQueueStarvation calls
7650
+ // internally, computed once here so EVERY kind (not just 'blocked'/
7651
+ // 'stalled') carries real dispatchable-vs-blocked counts instead of a null.
7652
+ const blockedChains = computeBlockedChains(projectJobs);
7653
+ const blockedTotal = blockedChains.reduce((n, c) => n + c.blocked, 0);
7654
+ const dispatchable = Math.max(0, pendingRows.length - blockedTotal);
7655
+ const base = {
7656
+ cwd, pending: pendingRows.length, dispatchable, blockedChains, needsReviewCount, runningCount: runningRows.length,
7657
+ };
7658
+
7659
+ if (paused) return { ...base, kind: 'paused', reason: paused.reason ?? null };
7660
+
7661
+ // launch-blocked: only a persona a PENDING row in this scope actually uses
7662
+ // — a breaker open for a persona nothing here needs is not this scope's
7663
+ // problem (matches WindowStrip's own unconditional-banner-per-block read).
7664
+ const neededAgentTypes = new Set(pendingRows.map((j) => launchFailure.launchBlockKeyFor(j)));
7665
+ for (const [key, block] of Object.entries(launchBlocks ?? {})) {
7666
+ if (block && neededAgentTypes.has(key)) {
7667
+ return { ...base, kind: 'launch-blocked', agentType: key, block };
7668
+ }
7669
+ }
7670
+
7671
+ if (pendingRows.length === 0) return { ...base, kind: 'idle' };
7672
+
7673
+ // classifyQueueStarvation only ever classifies while nothing is running
7674
+ // (its own runningCount > 0 guard) — that boundary is also exactly where
7675
+ // slot saturation, not dependency shape, is the honest cause.
7676
+ if (runningRows.length > 0) {
7677
+ if (Number.isFinite(freeSlots) && freeSlots <= 0) {
7678
+ return { ...base, kind: 'saturated', totalSlots: totalSlots ?? null };
7679
+ }
7680
+ return { ...base, kind: 'running' };
7681
+ }
7682
+
7683
+ // Nothing running: hand the SAME rows + idle clock to classifyQueueStarvation
7684
+ // (thresholdMs: 0 — a live header must say "blocked" the instant every
7685
+ // pending row is dependency-stuck, not wait out the watchdog's own grace
7686
+ // period) purely for its idleMs reading; its own dispatchable/blockedChains
7687
+ // are mathematically identical to `base`'s (same computeBlockedChains walk
7688
+ // over the same rows), so `base` already carries them.
7689
+ const immediate = classifyQueueStarvation({
7690
+ jobs: projectJobs, paused: false, runningCount: 0,
7691
+ lastRunAtMs: lastDispatchAttemptAtMs, now, thresholdMs: 0,
7692
+ });
7693
+ // pending.length is already > 0 above, so `immediate` can only be null when
7694
+ // lastDispatchAttemptAtMs is itself in the future (clock skew) — fall back
7695
+ // to computing idleMs the same way rather than asserting a kind we can't
7696
+ // back up with a real number.
7697
+ const idleMs = immediate ? immediate.idleMs
7698
+ : (Number.isFinite(lastDispatchAttemptAtMs) ? now - lastDispatchAttemptAtMs : Infinity);
7699
+ if (dispatchable === 0) return { ...base, kind: 'blocked', idleMs };
7700
+ const kind = idleMs >= thresholdMs ? 'stalled' : 'running';
7701
+ return { ...base, kind, idleMs };
7702
+ }
7703
+
7704
+ /**
7705
+ * The watchdog half: acts on classifyQueueStarvationByProject. Called from
7706
+ * the heartbeat, which already runs on its own timer independent of the
7707
+ * billing poll loop — so a wedged or never-succeeding poll (the
7708
+ * /api/oauth/usage endpoint was itself 429ing all of 2026-09-05) can no
7709
+ * longer leave a queue with ready work idle indefinitely.
7710
+ *
7711
+ * Logs and audits one event PER starved/blocked cwd (each carrying that
7712
+ * cwd), but still forces at most one machine-wide tickQueue() per pass —
7713
+ * the tick itself is machine-wide (it drives whatever the picker finds
7714
+ * across every project), only the DETECTION is per-project.
6868
7715
  */
6869
7716
  async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
6870
7717
  // lastDispatchAttemptAt, not lastRunAt: the latter only advances when a
6871
7718
  // batch actually launches, so a poll that keeps succeeding while dispatch
6872
7719
  // itself never gets invoked would otherwise mask a stall behind a fresh-
6873
7720
  // looking timestamp that was never actually tracking dispatch liveness.
6874
- const verdict = classifyQueueStarvation({
7721
+ const verdicts = classifyQueueStarvationByProject({
6875
7722
  jobs: state?.jobs,
6876
7723
  paused: state?.paused,
6877
- runningCount: runningSet.size,
7724
+ runningSet,
6878
7725
  lastRunAtMs: Date.parse(state?.lastDispatchAttemptAt ?? ''),
6879
7726
  now,
6880
7727
  thresholdMs,
6881
7728
  });
6882
- if (!verdict) return null;
7729
+ if (verdicts.length === 0) return null;
7730
+
7731
+ let anyStarved = false;
7732
+ let primary = null;
7733
+ for (const verdict of verdicts) {
7734
+ const mins = Math.round(verdict.idleMs / 60_000);
7735
+ if (verdict.kind === 'blocked') {
7736
+ console.warn(
7737
+ `[scheduler] QUEUE BLOCKED (${verdict.cwd}): ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
7738
+ + `terminal or parked dependency, so ticking cannot help. Blockers: `
7739
+ + verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
7740
+ );
7741
+ appendAuditEvent('queue_blocked_stall', { cwd: verdict.cwd, pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
7742
+ if (!primary) primary = verdict;
7743
+ continue;
7744
+ }
6883
7745
 
6884
- const mins = Math.round(verdict.idleMs / 60_000);
6885
- if (verdict.kind === 'blocked') {
6886
7746
  console.warn(
6887
- `[scheduler] QUEUE BLOCKED: ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
6888
- + `terminal or parked dependency, so ticking cannot help. Blockers: `
6889
- + verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
7747
+ `[scheduler] QUEUE STARVED (${verdict.cwd}): ${verdict.dispatchable} dispatchable job(s) of ${verdict.pending} pending, 0 running, `
7748
+ + `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
6890
7749
  );
6891
- appendAuditEvent('queue_blocked_stall', { pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
6892
- return verdict;
7750
+ appendAuditEvent('queue_starvation_forced_tick', { cwd: verdict.cwd, pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
7751
+ anyStarved = true;
7752
+ primary = verdict;
6893
7753
  }
6894
7754
 
6895
- console.warn(
6896
- `[scheduler] QUEUE STARVED: ${verdict.dispatchable} dispatchable job(s) of ${verdict.pending} pending, 0 running, `
6897
- + `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
6898
- );
6899
- appendAuditEvent('queue_starvation_forced_tick', { pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
7755
+ if (!anyStarved) return primary;
7756
+
6900
7757
  // A never-populated utilization reading is itself one of the ways the
6901
7758
  // when-available path silently never fires (maybeLaunchWhenAvailable
6902
7759
  // returns early on null). Treat unknown as safe here, exactly as the
@@ -6911,8 +7768,76 @@ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs
6911
7768
  // actually ticked. The watchdog is the last line of defence against a
6912
7769
  // wedged dispatcher, so it must be able to un-wedge this too.
6913
7770
  cancelToken.cancelled = false;
7771
+ // A forced tick is machine-wide by construction (the picker considers
7772
+ // every project's rows) — one call here services every starved cwd found
7773
+ // this pass, not one call per cwd.
6914
7774
  await tickQueue({ bypassLoadGate: false }).catch((e) => console.error('[scheduler] starvation tick error', e));
6915
- return verdict;
7775
+ return primary;
7776
+ }
7777
+
7778
+ // One-shot latch, keyed per cwd, for the starve-escalation consequence below —
7779
+ // never escalate the same starve stretch twice. Cleared the moment that cwd
7780
+ // stops appearing in findStarvedProjects at all (dispatched, or the machine
7781
+ // went idle/paused), mirroring the heartbeat's stallSince/stallToasted pair.
7782
+ const starveEscalated = new Set();
7783
+
7784
+ /**
7785
+ * selectStarveEscalations(starvedProjects, escalatedCwds, thresholdMs)
7786
+ * → { toEscalate: [...sp], toClear: [cwd, ...] }
7787
+ *
7788
+ * Pure. `starvedProjects` is this sweep's findStarvedProjects() output (the
7789
+ * per-cwd STARVED verdict — cwd, pendingCount, oldestPendingSlug, ageMs);
7790
+ * `escalatedCwds` is the Set already latched from a prior sweep.
7791
+ *
7792
+ * toEscalate: rows crossing thresholdMs for the FIRST time this stretch —
7793
+ * i.e. old enough AND not already latched.
7794
+ * toClear: previously-latched cwds no longer reported as starved at all this
7795
+ * sweep, so a LATER starve on that project escalates again instead of being
7796
+ * silently suppressed forever by a stale latch.
7797
+ */
7798
+ function selectStarveEscalations(starvedProjects, escalatedCwds, thresholdMs = STARVE_ESCALATION_MS) {
7799
+ const stillStarved = new Set(starvedProjects.map((sp) => sp.cwd));
7800
+ const toClear = [...escalatedCwds].filter((cwd) => !stillStarved.has(cwd));
7801
+ const toEscalate = starvedProjects.filter((sp) => sp.ageMs >= thresholdMs && !escalatedCwds.has(sp.cwd));
7802
+ return { toEscalate, toClear };
7803
+ }
7804
+
7805
+ /**
7806
+ * runStarveEscalationSweep(starvedProjects) — acts on selectStarveEscalations'
7807
+ * verdict: audits a DISTINCT 'project_starve_escalated' event (once per starve
7808
+ * stretch, per cwd) and pushes the same toast-channel error the heartbeat's
7809
+ * stall detector already uses ('schedule:stall' → renderer toast.error), so a
7810
+ * starve that has gone on long enough to matter is visible without grepping
7811
+ * the audit log. The hold reason is read from `lastTick` (recordTick's own
7812
+ * last-computed outcome) — never re-evaluated here, so this can never
7813
+ * disagree with what actually happened on the last tick.
7814
+ *
7815
+ * Escalation only: never mutates a job, never dispatches, never bypasses a
7816
+ * gate. Exported for direct unit testing (attach a fake window via
7817
+ * attachWindow() first to assert the toast send).
7818
+ */
7819
+ function runStarveEscalationSweep(starvedProjects) {
7820
+ const { toEscalate, toClear } = selectStarveEscalations(starvedProjects, starveEscalated, STARVE_ESCALATION_MS);
7821
+ for (const cwd of toClear) starveEscalated.delete(cwd);
7822
+ for (const sp of toEscalate) {
7823
+ starveEscalated.add(sp.cwd);
7824
+ const holdReason = lastTick?.reason ?? 'unknown';
7825
+ const mins = Math.round(sp.ageMs / 60_000);
7826
+ console.error(
7827
+ `[scheduler] PROJECT STARVE ESCALATED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
7828
+ + `waiting=${mins}m (>= ${Math.round(STARVE_ESCALATION_MS / 60_000)}m escalation threshold), hold reason=${holdReason} — `
7829
+ + 'a bounded escalation only; nothing was auto-reset, cancelled, or dispatched',
7830
+ );
7831
+ appendAuditEvent('project_starve_escalated', {
7832
+ cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs, holdReason,
7833
+ });
7834
+ sendIfAlive(mainWindow, 'schedule:stall', {
7835
+ message: `Project starved: ${sp.cwd} has ${sp.pendingCount} pending PRD(s), oldest waiting ~${mins}m `
7836
+ + `(hold reason: ${holdReason}) — check the Scheduler tab.`,
7837
+ total: sp.pendingCount,
7838
+ byProject: { [sp.cwd]: { starved: sp.pendingCount } },
7839
+ });
7840
+ }
6916
7841
  }
6917
7842
 
6918
7843
  // ---------- dead-process reaper ----------
@@ -7007,12 +7932,19 @@ async function reapDeadRunningJobs() {
7007
7932
  // status:"running" with no slug left in runningSet to trigger reconciliation.
7008
7933
  // queue.json is the source of truth for which jobs are actually running.
7009
7934
  const state = await readQueue();
7935
+ // Shared by the log-evidence injections below and the reapable-processing
7936
+ // loop further down — same `j.runId` → run log path formula either way.
7937
+ const logPathForJob = (j) => (j?.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null);
7010
7938
  const { reapable, warnings, recovered } = selectReapableJobs(state.jobs, Date.now(), {
7011
7939
  pidAlive: claudePidAlive,
7012
7940
  grace: PIDLESS_SPAWN_GRACE_MS,
7013
7941
  findLiveProcess: (j) => findLiveProcessForJob(j, {
7014
7942
  worktreeDir: jobWorktree.worktreeDirFor(j.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD, j.slug),
7943
+ runCwd: j.runtime?.cwd || j.cwd,
7015
7944
  }),
7945
+ getLogPid: (j) => readSpawnedPidFromLog(logPathForJob(j)),
7946
+ getLogMtimeMs: (j) => readLogMtimeMs(logPathForJob(j)),
7947
+ logFreshWindowMs: IDLE_OUTPUT_KILL_MS,
7016
7948
  });
7017
7949
  for (const w of warnings) {
7018
7950
  console.warn(`[scheduler] reapDeadRunningJobs: ${w.reason} slug=${w.slug} — leaving row alone`);
@@ -7037,11 +7969,9 @@ async function reapDeadRunningJobs() {
7037
7969
  }
7038
7970
 
7039
7971
  const dead = [];
7040
- for (const { slug, pid, pidless, reason } of reapable) {
7972
+ for (const { slug, pid, pidless, reason, failureOverride } of reapable) {
7041
7973
  const j = state.jobs.find((x) => x.slug === slug);
7042
- const logPath = j?.runId
7043
- ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`)
7044
- : null;
7974
+ const logPath = logPathForJob(j);
7045
7975
  // Absent/empty run dir → classifyRunOutcome finds no result event →
7046
7976
  // 'no_result' → non-success below → filed as failed, never completed.
7047
7977
  const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
@@ -7060,7 +7990,7 @@ async function reapDeadRunningJobs() {
7060
7990
  // re-derived so a phantom link never survives the reap.
7061
7991
  const hasOwnArtifact = pidless ? logHasOutput(logPath) : true;
7062
7992
  const gateOutcome = pidless ? resolvePidlessGateOutcome(outcome, hasOwnArtifact) : mapOutcomeToGateOutcome(outcome);
7063
- dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath, noOwnArtifact: pidless && !hasOwnArtifact });
7993
+ dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath, noOwnArtifact: pidless && !hasOwnArtifact, failureOverride });
7064
7994
  }
7065
7995
 
7066
7996
  queueHealthSweepCycle += 1;
@@ -7158,8 +8088,44 @@ async function reapDeadRunningJobs() {
7158
8088
  integrationResults.set(d.slug, { effectiveSuccess, landedCommit, notLandedInfo });
7159
8089
  }
7160
8090
 
8091
+ // Evidence-before-failure guard for a row about to be stamped 'failed'
8092
+ // (this PRD — job 1192 shipped a real 3-file commit and was still
8093
+ // reaped 'failed' because this check did not exist): a `landedCommit`
8094
+ // already recorded on the row is only ever stamped from an actual HEAD
8095
+ // advance or a proven branch-integration (jobLandedCommitThisRun / the
8096
+ // dead-pid integration proof above / the dispatch-time sidecar
8097
+ // backfill) — never speculative — but it can still be STALE by the time
8098
+ // this row is reaped (the branch it named could have been force-pushed
8099
+ // over, or the row could be carrying a sidecar-backfilled sha from a
8100
+ // run that was later discarded). git-resolving it here is what turns
8101
+ // "the field is non-empty" into "this sha is a real commit in this
8102
+ // repo right now". Computed OUTSIDE mutate() for the same reason
8103
+ // integrationResults is above: git spawn work must never run inside
8104
+ // mutate()'s single global serialization chain.
8105
+ //
8106
+ // Scoped to exactly the rows that would otherwise fall through to
8107
+ // 'failed' below: a 'success' outcome is already resolved by
8108
+ // integrationResults above (never reaches 'failed'), a rate-limited
8109
+ // death is retryable and never terminal, and a pidless row that already
8110
+ // carries a `failureOverride` (PRD 1173) is already diverted to
8111
+ // needs_review — this gate must never re-litigate either of those.
8112
+ const landedCommitEvidence = new Map();
8113
+ // Each row's evidence check is an independent read-only `git cat-file`/
8114
+ // `git log` pair with no shared mutable state between iterations, so
8115
+ // this runs the whole dead-job batch concurrently rather than one
8116
+ // dispatch's git-spawn latency at a time.
8117
+ await Promise.all(dead.map(async (d) => {
8118
+ if (d.outcome === 'rate_limited' || d.outcome === 'success') return;
8119
+ if (d.pidless && d.failureOverride) return;
8120
+ const row = state.jobs.find((x) => x.slug === d.slug);
8121
+ if (!row?.landedCommit) return;
8122
+ const rowCwd = row.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD;
8123
+ const resolved = await resolveLandedCommitEvidence(rowCwd, row.landedCommit, row.startedAt);
8124
+ if (resolved) landedCommitEvidence.set(d.slug, row.landedCommit);
8125
+ }));
8126
+
7161
8127
  await mutate(async (s) => {
7162
- for (const { slug, pid, outcome, gateOutcome, pidless, reason, noOwnArtifact } of dead) {
8128
+ for (const { slug, pid, outcome, gateOutcome, pidless, reason, noOwnArtifact, failureOverride } of dead) {
7163
8129
  const idx = s.jobs.findIndex((x) => x.slug === slug);
7164
8130
  if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
7165
8131
  const rateLimited = outcome === 'rate_limited';
@@ -7218,6 +8184,20 @@ async function reapDeadRunningJobs() {
7218
8184
  notLandedInfo = ir.notLandedInfo;
7219
8185
  }
7220
8186
  }
8187
+ // Evidence-before-failure guard for the pidless-reap path (PRD 1173):
8188
+ // a pidless reap about to stamp 'failed' purely because runtime.pid
8189
+ // was never recorded must first check whether this row already
8190
+ // carries a landedCommit from an earlier dispatch of the same slug
8191
+ // (landedCommit survives a reset — see the comment near
8192
+ // resetJobFields). Diverted to needs_review, never silently
8193
+ // 'completed' — see resolvePidlessFailureOverride's header in
8194
+ // reaperHelpers.cjs for why needs_review is the correct destination.
8195
+ // Scoped strictly to the pidless branch: dead-pid and rate-limited
8196
+ // rows are untouched, and a pidless row that already resolved to
8197
+ // effectiveSuccess/notLandedInfo above is left alone too.
8198
+ if (pidless && !effectiveSuccess && !rateLimited && !notLandedInfo && failureOverride) {
8199
+ notLandedInfo = { verdict: failureOverride.verdict, reason: failureOverride.reason };
8200
+ }
7221
8201
 
7222
8202
  const leftoverSuffix = deltaPaths && deltaPaths.length
7223
8203
  ? ` — left ${deltaPaths.length} files uncommitted`
@@ -7225,9 +8205,20 @@ async function reapDeadRunningJobs() {
7225
8205
  const baseReason = notLandedInfo
7226
8206
  ? `reaped: ${notLandedInfo.reason}`
7227
8207
  : (pidless ? reason : `reaped: process gone (outcome=${outcome})`);
8208
+ // Evidence gate (this PRD): a row that would otherwise fall through
8209
+ // to 'failed' below, but whose landedCommit was proven to resolve
8210
+ // via git cat-file BEFORE this mutate() ran (see landedCommitEvidence
8211
+ // above), gets promoted to 'completed' instead — the row already
8212
+ // shipped real work, so a bookkeeping gap (no runtime.pid recorded)
8213
+ // must never override git-verified evidence with a false failure.
8214
+ const confirmedLandedCommit = (!effectiveSuccess && !notLandedInfo && !rateLimited)
8215
+ ? (landedCommitEvidence.get(slug) || null)
8216
+ : null;
7228
8217
  const transitionReason = rateLimited
7229
8218
  ? `reaped: rate limit detected — reset to pending, not failed (outcome=${outcome})${leftoverSuffix}`
7230
- : baseReason + leftoverSuffix;
8219
+ : confirmedLandedCommit
8220
+ ? `${baseReason}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence${leftoverSuffix}`
8221
+ : baseReason + leftoverSuffix;
7231
8222
 
7232
8223
  if (rateLimited) {
7233
8224
  // Retryable, never terminal (PRD 1117) — same resetJobFields path
@@ -7236,17 +8227,27 @@ async function reapDeadRunningJobs() {
7236
8227
  // paused-for-rate-limit reset: fresh runId/startedAt/exitCode.
7237
8228
  resetJobFields(s.jobs[idx], transitionReason, { source: 'reapDeadRunningJobs:rate-limit' });
7238
8229
  } else {
7239
- const targetStatus = effectiveSuccess ? 'completed' : (notLandedInfo ? 'needs_review' : 'failed');
7240
- transitionJob(s.jobs[idx], targetStatus, { reason: transitionReason, source: 'reapDeadRunningJobs' });
7241
- s.jobs[idx].exitCode = effectiveSuccess ? 0 : (s.jobs[idx].exitCode ?? 1);
7242
- s.jobs[idx].finishedAt = new Date().toISOString();
7243
- s.jobs[idx].error = effectiveSuccess ? null : `${transitionReason} (outcome=${outcome})`;
7244
- s.jobs[idx].gateOutcome = gateOutcome;
8230
+ const landed = effectiveSuccess || Boolean(confirmedLandedCommit);
8231
+ const targetStatus = effectiveSuccess
8232
+ ? 'completed'
8233
+ : (notLandedInfo ? 'needs_review' : (confirmedLandedCommit ? 'completed' : 'failed'));
8234
+ const source = confirmedLandedCommit ? 'reapDeadRunningJobs:landed' : 'reapDeadRunningJobs';
8235
+ // error/verifierVerdict are stamped BEFORE transitionJob() below —
8236
+ // see the identical ordering fix (and its rationale) in spawnJob's
8237
+ // finalize path: transitionJob's needs_review ledger entry reads
8238
+ // these fields off `job` synchronously the instant it runs, so
8239
+ // setting them after fed the ledger a stale/leftover reason.
8240
+ s.jobs[idx].error = landed ? null : `${transitionReason} (outcome=${outcome})`;
7245
8241
  if (notLandedInfo) {
7246
8242
  s.jobs[idx].verifierVerdict = notLandedInfo.verdict;
7247
8243
  } else {
7248
8244
  delete s.jobs[idx].verifierVerdict;
7249
8245
  }
8246
+ transitionJob(s.jobs[idx], targetStatus, { reason: transitionReason, source });
8247
+ s.jobs[idx].exitCode = landed ? 0 : (s.jobs[idx].exitCode ?? 1);
8248
+ s.jobs[idx].finishedAt = new Date().toISOString();
8249
+ s.jobs[idx].gateOutcome = gateOutcome;
8250
+ if (confirmedLandedCommit) s.jobs[idx].landedCommit = confirmedLandedCommit;
7250
8251
  if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
7251
8252
  }
7252
8253
  // A pidless spawn that never wrote its own '<slug>.log' into the
@@ -7282,7 +8283,14 @@ async function reapDeadRunningJobs() {
7282
8283
  appendAuditEvent('job_reaped_rate_limited', { slug, cwd: s.jobs[idx].cwd ?? null });
7283
8284
  } else if (pidless) {
7284
8285
  console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
7285
- appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
8286
+ appendAuditEvent('job_reaped_pidless', {
8287
+ slug,
8288
+ cwd: s.jobs[idx].cwd ?? null,
8289
+ outcome,
8290
+ graceMs: PIDLESS_SPAWN_GRACE_MS,
8291
+ landedCommit: s.jobs[idx].landedCommit ?? null,
8292
+ verifierVerdict: s.jobs[idx].verifierVerdict ?? null,
8293
+ });
7286
8294
  } else {
7287
8295
  console.log(`[scheduler] reaped dead job slug=${slug} pid=${pid} outcome=${outcome}`);
7288
8296
  }
@@ -7477,6 +8485,12 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
7477
8485
  const seen = new Set(hot.map((j) => `${j.slug}|${j.runId ?? ''}`));
7478
8486
  const archived = (Array.isArray(historyEntries) ? historyEntries : []).filter((j) => {
7479
8487
  if (!j) return false;
8488
+ // needs_review_entry/needs_review_resolution lines (needsReviewLedger.cjs)
8489
+ // share history.jsonl with terminal job rows but carry no `status` — the
8490
+ // History view (SchedulerHistoryView.tsx) renders ScheduleJob rows, so a
8491
+ // ledger line slipping through here would show up as a statusless,
8492
+ // meaningless row in that table.
8493
+ if (j.kind && j.kind !== 'terminal') return false;
7480
8494
  const key = `${j.slug}|${j.runId ?? ''}`;
7481
8495
  if (seen.has(key)) return false;
7482
8496
  seen.add(key);
@@ -7602,6 +8616,69 @@ function isExhaustedAutoFix(job) {
7602
8616
  return job.autoFixOutcome !== 'plan' && (job.autoFixRetries ?? 0) >= 1;
7603
8617
  }
7604
8618
 
8619
+ /**
8620
+ * Verdicts from the POST-RUN GUARDS (commit-guard / shared-tree guard) that a
8621
+ * later, independently-checkable commit/looksDone signal can meaningfully
8622
+ * confirm or refute. Auto-fix investigations only ever launch for FAILING
8623
+ * runs (see isExhaustedAutoFix above and selectAutoFixTargets) — a job parked
8624
+ * by one of these GUARD verdicts exits 0 and never has autoFixAttempted set,
8625
+ * so it is invisible to isExhaustedAutoFix and can sit in needs_review
8626
+ * forever with nothing to spend and nothing to exhaust (PRD 1181, 2026-09-12:
8627
+ * exit 0, commit aff5607 landed, parked on a shared-tree verdict, cleared
8628
+ * only by a human).
8629
+ *
8630
+ * 'worktree_integration_failed' is deliberately EXCLUDED — its damage IS a
8631
+ * commit: one stranded on an unmerged `sm-job/<slug>` branch. A `landedCommit`
8632
+ * existing is not evidence against that verdict, it is a restatement of it,
8633
+ * so admitting it here would auto-complete a row whose work never actually
8634
+ * reached the target branch. It already has its own dedicated, git-native
8635
+ * resolution path (selectMechanicalRecoveryTarget / performMechanicalRecovery
8636
+ * — a real re-attempted merge) and must never be pulled into this ladder.
8637
+ *
8638
+ * 'pidless_reap_with_landed_commit' (PRD 1173, resolvePidlessFailureOverride
8639
+ * in reaperHelpers.cjs) is included: it parks on the exact same shape (exit
8640
+ * never observed / no autoFixAttempted, real landedCommit evidence) as
8641
+ * 'silent_no_op' and 'shared_tree_reverted', and this ladder never trusts
8642
+ * landedCommit alone anyway — applyNeedsReviewAutoResolve only resolves once
8643
+ * job.looksDone independently reconfirms via a fresh commits-since-this-run
8644
+ * scan, which is exactly the "does the commit correspond to THIS dispatch"
8645
+ * re-verification resolvePidlessFailureOverride's own header says the
8646
+ * pidless-reap path itself cannot do. Omitting it here reproduces the same
8647
+ * "nothing to spend, nothing to exhaust" needs_review stall this PRD exists
8648
+ * to fix, just for a third verdict.
8649
+ */
8650
+ const GUARD_VERDICT_EVIDENCE_ELIGIBLE = new Set(['silent_no_op', 'shared_tree_reverted', 'pidless_reap_with_landed_commit']);
8651
+
8652
+ /**
8653
+ * Pure predicate, no I/O: a needs_review row parked directly by one of the
8654
+ * GUARD_VERDICT_EVIDENCE_ELIGIBLE verdicts, that never went through an
8655
+ * auto-fix investigation at all (autoFixAttempted is not true) — the
8656
+ * structural gap this PRD closes, distinct from isExhaustedAutoFix's "went
8657
+ * through auto-fix and spent it" case. A job that DID get an auto-fix
8658
+ * investigation is left to isExhaustedAutoFix's own ladder rather than this
8659
+ * one, even if its verifierVerdict happens to also be in the eligible set.
8660
+ * Exported for tests.
8661
+ */
8662
+ function isGuardParkedWithoutAutoFix(job) {
8663
+ if (!job || job.status !== 'needs_review') return false;
8664
+ if (job.autoFixAttempted === true) return false;
8665
+ return GUARD_VERDICT_EVIDENCE_ELIGIBLE.has(job.verifierVerdict);
8666
+ }
8667
+
8668
+ /**
8669
+ * Pure predicate, no I/O: is this needs_review row eligible for the bounded
8670
+ * auto-resolve ladder at all — either because its auto-fix path is genuinely
8671
+ * spent (isExhaustedAutoFix), or because it was parked by a GUARD verdict
8672
+ * that never entered auto-fix in the first place (isGuardParkedWithoutAutoFix).
8673
+ * Both classes share ONE ladder (applyNeedsReviewAutoResolve) rather than a
8674
+ * duplicated one — the ladder itself doesn't care which door a row came
8675
+ * through, only whether it now carries completion evidence (job.looksDone).
8676
+ * Exported for tests.
8677
+ */
8678
+ function isEligibleForNeedsReviewAutoResolve(job) {
8679
+ return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job);
8680
+ }
8681
+
7605
8682
  /**
7606
8683
  * Pure predicate: an investigation produced a fix plan (autoFixOutcome ===
7607
8684
  * 'plan') but its fix-plan slug is not present among `queuedSlugs` — the
@@ -7749,17 +8826,26 @@ function isRescanCandidate(job) {
7749
8826
  * selectResumeRecoveryTarget / selectAutoFixTargets) so the guard can never
7750
8827
  * again be narrower than the work reverifyNeedsReview performs.
7751
8828
  *
7752
- * Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget are pure
7753
- * (no I/O). selectAutoFixTargets is called with an injected fixSlugExists
7754
- * that always returns false — cheap and deliberately over-inclusive (a false
7755
- * positive here just means one extra periodic pass, never a missed one) so
7756
- * this guard never pays selectAutoFixTargets's production fs.existsSync scan
7757
- * per tick. resolveRunId's IO only fires for rows missing job.runId, same as
8829
+ * Widened again (this PRD): a `needs_review` row parked directly by a GUARD
8830
+ * verdict with no auto-fix history (isGuardParkedWithoutAutoFix) is not an
8831
+ * isRescanCandidate either — RESCANNABLE_VERDICTS covers transcript-verifier
8832
+ * verdicts, not commit-guard/shared-tree-guard verdicts — but
8833
+ * reverifyNeedsReview's looksDone-annotation pass now runs for it too (see
8834
+ * that function). Same rule as always: never let this guard be narrower than
8835
+ * the work reverifyNeedsReview actually performs.
8836
+ *
8837
+ * Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget and
8838
+ * isGuardParkedWithoutAutoFix are pure (no I/O). selectAutoFixTargets is
8839
+ * called with an injected fixSlugExists that always returns false — cheap
8840
+ * and deliberately over-inclusive (a false positive here just means one
8841
+ * extra periodic pass, never a missed one) so this guard never pays
8842
+ * selectAutoFixTargets's production fs.existsSync scan per tick.
8843
+ * resolveRunId's IO only fires for rows missing job.runId, same as
7758
8844
  * isRescanCandidate already incurs above.
7759
8845
  */
7760
8846
  function shouldRunPeriodicReverify(jobs) {
7761
8847
  if (!Array.isArray(jobs)) return false;
7762
- if (jobs.some((j) => isRescanCandidate(j))) return true;
8848
+ if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j))) return true;
7763
8849
  if (jobs.some((j) => selectMechanicalRecoveryTarget(j) || selectResumeRecoveryTarget(j))) return true;
7764
8850
  return selectAutoFixTargets(jobs, { fixSlugExists: () => false }).length > 0;
7765
8851
  }
@@ -7781,24 +8867,90 @@ function stuckFailedEscalationDisabled() {
7781
8867
  return process.env.SM_STUCK_FAILED_ESCALATE_DISABLE === '1';
7782
8868
  }
7783
8869
 
8870
+ // Bounded automatic failed -> pending recovery (PRD 1151): a failed row gets
8871
+ // up to FAILED_AUTORESET_CAP auto-reset attempts, each gated on having sat
8872
+ // `failed` for FAILED_AUTORESET_MS, before the stuck-failed escalation below
8873
+ // is allowed to page a human. Same env-override shape as
8874
+ // STUCK_FAILED_ESCALATE_MS/QUARANTINE_ESCALATE_MS above.
8875
+ const FAILED_AUTORESET_CAP = 3;
8876
+ const FAILED_AUTORESET_MS = process.env.SM_FAILED_AUTORESET_MINUTES
8877
+ ? Number(process.env.SM_FAILED_AUTORESET_MINUTES) * 60_000
8878
+ : 10 * 60_000;
8879
+
8880
+ /**
8881
+ * Kill-switch gate for the failed-autoreset pass below
8882
+ * (SM_FAILED_AUTORESET_DISABLE=1), same shape as stuckFailedEscalationDisabled
8883
+ * above.
8884
+ */
8885
+ function failedAutoResetDisabled() {
8886
+ return process.env.SM_FAILED_AUTORESET_DISABLE === '1';
8887
+ }
8888
+
8889
+ /**
8890
+ * selectFailedAutoResetTargets(jobs, now, thresholdMs) →
8891
+ * [{ slug, cwd, ageMs, attempts }]
8892
+ *
8893
+ * Pure selector — no IO, no `require` inside the function. Selects `failed`
8894
+ * rows whose newest statusHistory entry with `to === 'failed'` is older than
8895
+ * `thresholdMs` and whose failedAutoResetAttempts counter hasn't yet spent
8896
+ * FAILED_AUTORESET_CAP attempts. "Newest" (not first) matters because a row
8897
+ * can have failed more than once across its lifetime (an earlier auto-reset
8898
+ * attempt that itself failed again) — only the most recent failed-since
8899
+ * timestamp should gate the next attempt.
8900
+ *
8901
+ * Excludes a row whose newest failed-entry came from spawnJob:fail-dirty —
8902
+ * that source means a transient failure left genuinely uncommitted work in
8903
+ * the job's worktree and the system already decided once, deliberately, not
8904
+ * to auto-requeue it (see that call site's own comment: "could discard
8905
+ * uncommitted work left by the failed run"). This bounded auto-reset is a
8906
+ * different, slower mechanism and must not quietly override that decision
8907
+ * 10 minutes later — a human should look at a dirty worktree before it gets
8908
+ * re-driven.
8909
+ */
8910
+ function selectFailedAutoResetTargets(jobs, now, thresholdMs) {
8911
+ const targets = [];
8912
+ for (const j of jobs ?? []) {
8913
+ if (j.status !== 'failed') continue;
8914
+ const attempts = j.failedAutoResetAttempts ?? 0;
8915
+ if (attempts >= FAILED_AUTORESET_CAP) continue;
8916
+ const history = j.statusHistory || [];
8917
+ let entry = null;
8918
+ for (let i = history.length - 1; i >= 0; i--) {
8919
+ if (history[i].to === 'failed') { entry = history[i]; break; }
8920
+ }
8921
+ if (!entry) continue;
8922
+ if (entry.source === 'spawnJob:fail-dirty') continue;
8923
+ const since = Date.parse(entry.at);
8924
+ if (Number.isNaN(since)) continue;
8925
+ const ageMs = now - since;
8926
+ if (ageMs < thresholdMs) continue;
8927
+ targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, attempts });
8928
+ }
8929
+ return targets;
8930
+ }
8931
+
7784
8932
  /**
7785
8933
  * findStuckFailedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
7786
8934
  *
7787
8935
  * Pure predicate (isRescanCandidate's own resolveRunId/classifyRunOutcome log
7788
8936
  * read is the only IO, gated per-job exactly like shouldRunPeriodicReverify
7789
- * above). `failed` is a fully terminal state for every automated recovery
7790
- * path — selectResumeRecoveryTarget/selectAutoFixTargets both require
7791
- * needs_review, reapDeadRunningJobs only ever writes running → failed, and
7792
- * reconcile-repair's to-pending is for structurally invalid rows. Only a
7793
- * human's scheduler_reset_job ever takes failed → pending (LEGAL_TRANSITIONS).
7794
- * A rescan candidate (isRescanCandidate) that has sat failed longer than
7795
- * `thresholdMs` can therefore go silently stuck forever — job
7796
- * 4056-outcome-stats sat `failed` for five days with no operator signal
7797
- * (reported 2026-09-10, social-signals-trader) even though the periodic
7798
- * reverify pass (once shouldRunPeriodicReverify's guard was fixed) WAS firing
7799
- * on it — reverifyNeedsReview's failed branch can annotate looksDone but can
7800
- * never resolve a failed row itself (see its own header). This is the
7801
- * visibility half that guard fix was missing: escalate once, never requeue.
8937
+ * above). `failed` used to be a fully terminal state for every automated
8938
+ * recovery path — selectResumeRecoveryTarget/selectAutoFixTargets both
8939
+ * require needs_review, reapDeadRunningJobs only ever writes running →
8940
+ * failed, and reconcile-repair's to-pending is for structurally invalid rows.
8941
+ * That is no longer true: selectFailedAutoResetTargets above now drives a
8942
+ * bounded failed → pending auto-reset (LEGAL_TRANSITIONS already allowed the
8943
+ * edge). This escalation now only fires once that auto-reset budget is
8944
+ * genuinely spent (see the interval body's filter on failedAutoResetAttempts)
8945
+ * — a rescan candidate (isRescanCandidate) that has sat failed longer than
8946
+ * `thresholdMs` AND exhausted its auto-reset attempts can therefore go
8947
+ * silently stuck forever — job 4056-outcome-stats sat `failed` for five days
8948
+ * with no operator signal (reported 2026-09-10, social-signals-trader) even
8949
+ * though the periodic reverify pass (once shouldRunPeriodicReverify's guard
8950
+ * was fixed) WAS firing on it — reverifyNeedsReview's failed branch can
8951
+ * annotate looksDone but can never resolve a failed row itself (see its own
8952
+ * header). This is the visibility half that guard fix was missing: escalate
8953
+ * once per exhausted row, never requeue from here.
7802
8954
  *
7803
8955
  * `stuckFailedNotified` gates this to exactly once per row — once the caller
7804
8956
  * stamps it, this always excludes that row so a human is never re-paged on
@@ -7812,7 +8964,18 @@ function findStuckFailedJobs(jobs, now, thresholdMs) {
7812
8964
  if (j.status !== 'failed') continue;
7813
8965
  if (j.stuckFailedNotified === true) continue;
7814
8966
  if (!isRescanCandidate(j)) continue;
7815
- const entry = (j.statusHistory || []).find((h) => h.to === 'failed');
8967
+ // Newest (not first) to === 'failed' entry — same rationale as
8968
+ // selectFailedAutoResetTargets above: a row can have failed more than
8969
+ // once across its lifetime (an earlier auto-reset attempt that itself
8970
+ // failed again), and only the CURRENT failure episode's age should gate
8971
+ // escalation. Using the first/oldest entry would report a stale age
8972
+ // (and become instantly escalation-eligible) for a row that failed
8973
+ // months ago, recovered, and has only just failed again.
8974
+ const history = j.statusHistory || [];
8975
+ let entry = null;
8976
+ for (let i = history.length - 1; i >= 0; i--) {
8977
+ if (history[i].to === 'failed') { entry = history[i]; break; }
8978
+ }
7816
8979
  if (!entry) continue;
7817
8980
  const since = Date.parse(entry.at);
7818
8981
  if (Number.isNaN(since)) continue;
@@ -7822,6 +8985,138 @@ function findStuckFailedJobs(jobs, now, thresholdMs) {
7822
8985
  return stuck;
7823
8986
  }
7824
8987
 
8988
+ // Bounded automatic terminal decision for an EXHAUSTED needs_review row
8989
+ // (isExhaustedAutoFix === true — auto-fix attempted, no plan produced,
8990
+ // retries spent): up to NEEDS_REVIEW_RESOLVE_CAP requeue attempts (a
8991
+ // needs_review -> pending -> ... -> needs_review round trip counts as one
8992
+ // spent attempt), each gated on having sat exhausted-needs_review for
8993
+ // NEEDS_REVIEW_RESOLVE_MS, before the row is auto-skipped so a `dependsOn`
8994
+ // chain behind it always drains without an operator. Same env-override
8995
+ // shape as FAILED_AUTORESET_MS above.
8996
+ const NEEDS_REVIEW_RESOLVE_CAP = 2;
8997
+ const NEEDS_REVIEW_RESOLVE_MS = process.env.SM_NEEDS_REVIEW_RESOLVE_MINUTES
8998
+ ? Number(process.env.SM_NEEDS_REVIEW_RESOLVE_MINUTES) * 60_000
8999
+ : 30 * 60_000;
9000
+
9001
+ /**
9002
+ * Kill-switch gate for the needs_review auto-resolve pass below
9003
+ * (SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1), same shape as
9004
+ * failedAutoResetDisabled/stuckFailedEscalationDisabled above.
9005
+ */
9006
+ function needsReviewAutoResolveDisabled() {
9007
+ return process.env.SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE === '1';
9008
+ }
9009
+
9010
+ /**
9011
+ * selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) →
9012
+ * [{ slug, cwd, ageMs, attempts }]
9013
+ *
9014
+ * Pure selector — no IO. Selects `needs_review` rows eligible for the
9015
+ * bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve — either
9016
+ * auto-fix genuinely spent, or parked by a GUARD verdict that never entered
9017
+ * auto-fix at all), whose newest statusHistory entry with `to ===
9018
+ * 'needs_review'` is older than `thresholdMs`, and whose
9019
+ * exhaustedResolveAttempts counter has not yet spent its cap.
9020
+ *
9021
+ * The inclusion bound is inclusive of the cap itself (`<= CAP`, not `<
9022
+ * CAP`): NEEDS_REVIEW_RESOLVE_CAP counts REQUEUE attempts already spent, and
9023
+ * the pass that observes attempts === CAP is exactly the one that must fire
9024
+ * the terminal skip (see the interval body's branch below) — excluding that
9025
+ * row here would mean the cap-exhausted row is never selected again and the
9026
+ * dependsOn chain behind it never drains, defeating this PRD's own purpose.
9027
+ */
9028
+ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
9029
+ const targets = [];
9030
+ for (const j of jobs ?? []) {
9031
+ if (j.status !== 'needs_review') continue;
9032
+ if (!isEligibleForNeedsReviewAutoResolve(j)) continue;
9033
+ if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) continue;
9034
+ const history = j.statusHistory || [];
9035
+ let entry = null;
9036
+ for (let i = history.length - 1; i >= 0; i--) {
9037
+ if (history[i].to === 'needs_review') { entry = history[i]; break; }
9038
+ }
9039
+ if (!entry) continue;
9040
+ const since = Date.parse(entry.at);
9041
+ if (Number.isNaN(since)) continue;
9042
+ const ageMs = now - since;
9043
+ if (ageMs < thresholdMs) continue;
9044
+ targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, attempts: j.exhaustedResolveAttempts ?? 0 });
9045
+ }
9046
+ return targets;
9047
+ }
9048
+
9049
+ /**
9050
+ * Applies the needs_review auto-resolve decision to a single job (mutates in
9051
+ * place; calls transitionJob + appendAuditEvent). Extracted from the
9052
+ * interval body so the three branches are unit-testable without going
9053
+ * through mutate()/queue.json IO. Order of decision:
9054
+ * 1. job.looksDone (the annotation reverifyNeedsReview writes when a
9055
+ * later commit touches the PRD's declared paths) -> 'completed'.
9056
+ * 2. otherwise, one more bounded requeue -> 'pending', incrementing
9057
+ * exhaustedResolveAttempts.
9058
+ * 3. once NEEDS_REVIEW_RESOLVE_CAP requeue attempts are spent -> 'skipped',
9059
+ * with job.error naming the exhausted path so the Queue UI still shows
9060
+ * why, and a marker (needsReviewAutoResolvedSkip) that findBlockingDep
9061
+ * reads to stop treating this SPECIFIC skip as a permanent dependsOn
9062
+ * block — unlike a generic "PRD source vanished" skip, this row was
9063
+ * given every bounded chance to resolve itself.
9064
+ * Re-validates status/exhaustion/cap itself (same race-guard shape as the
9065
+ * failed-autoreset loop above) so a stale target computed before this
9066
+ * mutate() pass can never double-apply. Returns the outcome, or null if the
9067
+ * race guard rejected it.
9068
+ *
9069
+ * A row can reach here through either door (isEligibleForNeedsReviewAutoResolve):
9070
+ * auto-fix genuinely exhausted, or parked directly by a GUARD_VERDICT_EVIDENCE_
9071
+ * ELIGIBLE verdict with no auto-fix history at all. `originIsGuardParked` picks
9072
+ * which door this particular row came through, purely to make the requeue/skip
9073
+ * reason text (and the Queue UI's job.error) name the RIGHT evidence — a
9074
+ * guard-parked row was never "exhausted auto-fix" and must never claim to be.
9075
+ */
9076
+ function applyNeedsReviewAutoResolve(j) {
9077
+ if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j)) return null;
9078
+ if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) return null;
9079
+ const originIsGuardParked = !isExhaustedAutoFix(j) && isGuardParkedWithoutAutoFix(j);
9080
+
9081
+ if (j.looksDone) {
9082
+ const attempt = j.exhaustedResolveAttempts ?? 0;
9083
+ const reason = originIsGuardParked
9084
+ ? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with landed commit and looksDone evidence (${j.looksDone.rule}) — `
9085
+ + `${j.looksDone.commits.length} commit(s) attributable to this job's own run touch the PRD's declared paths, work landed`
9086
+ : `needs_review auto-resolve: verifier annotation shows work landed (${j.looksDone.rule} — ${j.looksDone.commits.length} commit(s) attributable to this job's own run touch the PRD's declared paths)`;
9087
+ transitionJob(j, 'completed', { reason, source: 'needsReviewAutoResolve' });
9088
+ appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'completed', attempt });
9089
+ return 'completed';
9090
+ }
9091
+
9092
+ const attemptsSoFar = j.exhaustedResolveAttempts ?? 0;
9093
+ if (attemptsSoFar < NEEDS_REVIEW_RESOLVE_CAP) {
9094
+ const attempt = attemptsSoFar + 1;
9095
+ j.exhaustedResolveAttempts = attempt;
9096
+ const reason = originIsGuardParked
9097
+ ? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with no completion evidence yet — `
9098
+ + `requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`
9099
+ : `needs_review auto-resolve: exhausted auto-fix, no completion evidence — requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`;
9100
+ transitionJob(j, 'pending', { reason, source: 'needsReviewAutoResolve' });
9101
+ appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'requeued', attempt });
9102
+ return 'requeued';
9103
+ }
9104
+
9105
+ j.needsReviewAutoResolvedSkip = true;
9106
+ j.error = originIsGuardParked
9107
+ ? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with no completion evidence after `
9108
+ + `${NEEDS_REVIEW_RESOLVE_CAP} requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`
9109
+ : `needs_review auto-resolve: exhausted auto-fix path (autoFixOutcome=${j.autoFixOutcome ?? 'none'}, `
9110
+ + `autoFixRetries=${j.autoFixRetries ?? 0}) with no completion evidence after ${NEEDS_REVIEW_RESOLVE_CAP} `
9111
+ + `requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`;
9112
+ transitionJob(j, 'skipped', {
9113
+ reason: `needs_review auto-resolve: cap exhausted (${NEEDS_REVIEW_RESOLVE_CAP}/${NEEDS_REVIEW_RESOLVE_CAP} requeue attempts) — auto-skipped`,
9114
+ source: 'needsReviewAutoResolve',
9115
+ });
9116
+ appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'skipped', attempt: attemptsSoFar });
9117
+ return 'skipped';
9118
+ }
9119
+
7825
9120
  /**
7826
9121
  * Self-healing pass over needs_review jobs. The verifier runs in-process, so a
7827
9122
  * fix to runVerify.cjs only takes effect for jobs verified AFTER an app
@@ -7879,6 +9174,11 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
7879
9174
  // fix-plan investigation to diagnose — there is no code defect to
7880
9175
  // author a PRD against, only another job's still-uncommitted tree.
7881
9176
  if (job.blockedByForeignWip === true) return false;
9177
+ // A budget-killed job parks for a human/ladder decision, never an
9178
+ // auto-fix investigation or auto-retry — the run didn't fail, it simply
9179
+ // overran its own estimate; there's no code defect to diagnose (Out of
9180
+ // scope: "retrying or auto-resuming a budget-killed job" for this PRD).
9181
+ if (job.verifierVerdict === 'budget_exceeded') return false;
7882
9182
  // A stale re-run whose work already shipped (rcaReport's 'already-shipped'
7883
9183
  // class) must never buy a fix-plan PRD — there is nothing to fix, and the
7884
9184
  // correct recovery (archiving the PRD) is a human/reconcile action, not
@@ -7938,35 +9238,153 @@ function isEligibleForImmediateAutoFix(job, allJobs, fixSlugExists) {
7938
9238
  }
7939
9239
 
7940
9240
  /**
7941
- * Widened evidence check (PRD 1102): does at least one commit land AFTER
7942
- * this job's run window that touches a path the PRD itself declares? Scoped
7943
- * to the PRD's own declared paths (never the whole repo) so a sibling job's
7944
- * unrelated commit is not credited to this one — see healRefusalReason's own
9241
+ * attributeLandedCommits(job, pathCommits, cwd) → { commits, rule } | null
9242
+ *
9243
+ * Narrows a set of PATH-overlapping commits (computeLooksDone's
9244
+ * `landedSinceRun` result — any commit touching the PRD's declared paths,
9245
+ * regardless of who authored it) down to the subset actually attributable to
9246
+ * THIS job's own run. Path overlap alone is not attribution: inside one Epic,
9247
+ * sibling PRDs routinely declare the same hot file, so a sibling's commit is
9248
+ * indistinguishable from this job's own by path alone (the incident this
9249
+ * function exists to close — PRD 1204's parked row cited PRD 1205's commit
9250
+ * 5dadf3c as its own evidence).
9251
+ *
9252
+ * Three rules, tried strongest-first, first match wins:
9253
+ *
9254
+ * 1. 'landedCommit' — the row's own `job.landedCommit`, re-verified here via
9255
+ * `resolveLandedCommitEvidence` against THIS job's `startedAt`. This is
9256
+ * the strongest signal because it is not inferred from `git log` at all:
9257
+ * it is the sha spawnJob's own finalize step observed THIS dispatch's
9258
+ * worktree/branch landing (see resolveLandedCommitEvidence's own header
9259
+ * for why it also guards against a stale sha surviving a reset). Trusted
9260
+ * independent of whether it appears in `pathCommits` — it is definitionally
9261
+ * this job's own work, not something discovered by scanning history.
9262
+ * 2. 'job branch' — a path-overlapping commit reachable from (an ancestor of
9263
+ * or equal to) this job's own `sm-job/<slug>` branch tip. Still
9264
+ * job-specific even though it IS a `git log` scan: a sibling's commit can
9265
+ * never be an ancestor of THIS job's own branch ref. In practice this
9266
+ * branch is deleted on successful integration (gitWorktree.cjs's
9267
+ * `cleanupWorktree`), so this mainly fires when integration failed and
9268
+ * the branch was deliberately kept for recovery, or reverify runs before
9269
+ * cleanup — a narrower window than rule 1, hence checked second.
9270
+ * 3. 'slug trailer' — a path-overlapping commit whose message contains this
9271
+ * job's slug verbatim. Weakest of the three (a coincidental substring
9272
+ * match is possible, and nothing stamps this automatically today), so it
9273
+ * is the last resort when the two structural signals above found
9274
+ * nothing.
9275
+ *
9276
+ * Deliberately NOT a rule: raw path overlap by itself (the bug this function
9277
+ * fixes) and `committedInWindow`-style time-window-only evidence — a sibling
9278
+ * job running concurrently in the very same window is exactly as invisible to
9279
+ * a time bound as it is to a path filter, so neither narrows attribution.
9280
+ *
9281
+ * Never throws: a missing ref, an unresolvable sha, or any git failure for a
9282
+ * given commit/rule is treated as "that commit doesn't satisfy this rule",
9283
+ * never as a fabricated match.
9284
+ */
9285
+ async function attributeLandedCommits(job, pathCommits, cwd) {
9286
+ if (job?.landedCommit && await resolveLandedCommitEvidence(cwd, job.landedCommit, job.startedAt)) {
9287
+ return { commits: [job.landedCommit], rule: 'landedCommit' };
9288
+ }
9289
+
9290
+ const branch = `sm-job/${job?.slug}`;
9291
+ const branchCommits = [];
9292
+ for (const sha of pathCommits) {
9293
+ try {
9294
+ await execGitAt(cwd, ['merge-base', '--is-ancestor', sha, branch], { timeout: 10_000 });
9295
+ branchCommits.push(sha);
9296
+ } catch { /* not an ancestor of this job's own branch, or branch doesn't exist */ }
9297
+ }
9298
+ if (branchCommits.length) return { commits: branchCommits, rule: 'job branch' };
9299
+
9300
+ if (job?.slug) {
9301
+ const trailerCommits = [];
9302
+ for (const sha of pathCommits) {
9303
+ try {
9304
+ const msg = await execGitAt(cwd, ['log', '-1', '--format=%B', sha], { timeout: 10_000 });
9305
+ if (msg.includes(job.slug)) trailerCommits.push(sha);
9306
+ } catch { /* unresolvable sha */ }
9307
+ }
9308
+ if (trailerCommits.length) return { commits: trailerCommits, rule: 'slug trailer' };
9309
+ }
9310
+
9311
+ return null;
9312
+ }
9313
+
9314
+ /**
9315
+ * Widened evidence check (PRD 1102, narrowed to per-job attribution by a
9316
+ * later PRD): does at least one commit ATTRIBUTABLE TO THIS JOB land AFTER
9317
+ * its run window and touch a path the PRD itself declares? Scoped to the
9318
+ * PRD's own declared paths (never the whole repo) so a sibling job's
9319
+ * unrelated commit is never even considered — see healRefusalReason's own
7945
9320
  * rationale for why unscoped, repo-wide evidence is not attribution.
7946
9321
  *
9322
+ * Path overlap alone is NOT evidence (see attributeLandedCommits's header):
9323
+ * a sibling PRD in the same Epic routinely declares the same hot file, so
9324
+ * `landedSinceRun`'s raw result is only a candidate list — the returned
9325
+ * annotation is null unless `attributeLandedCommits` narrows it to at least
9326
+ * one commit this job can actually claim.
9327
+ *
7947
9328
  * Returns null (no annotation, never fabricated) when the PRD names no
7948
- * paths — the caller then has only the existing, already-computed
7949
- * committedInWindow signal to go on, same as before this PRD.
9329
+ * paths, when no commit touches a declared path at all, or when
9330
+ * path-overlapping commits exist but none are attributable to this job — the
9331
+ * caller then has only the existing, already-computed committedInWindow
9332
+ * signal to go on, same as before this PRD.
7950
9333
  *
7951
- * @returns {Promise<{commits: string[], paths: string[], detectedAt: string} | null>}
9334
+ * `fetchedCwds` (optional) lets a caller iterating many candidates in one
9335
+ * pass (reverifyNeedsReview) dedupe the `git fetch --all --prune` across
9336
+ * candidates that share a `cwd` — several `needs_review` rows for the same
9337
+ * project is the common case a backlog produces, and each fetch is up to
9338
+ * ~20s, so re-fetching the same repo once per row multiplies that pass's
9339
+ * wall-clock cost for zero new evidence. Omitted (or a fresh Set per call)
9340
+ * simply always fetches, unchanged from before this cache existed.
9341
+ *
9342
+ * @returns {Promise<{commits: string[], paths: string[], detectedAt: string, rule: string} | null>}
7952
9343
  */
7953
- async function computeLooksDone(job) {
9344
+ async function computeLooksDone(job, fetchedCwds) {
7954
9345
  const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
7955
9346
  const paths = declaredPathsForPrd(prdPath);
7956
9347
  if (!paths.length) return null;
7957
- await fetchAllRefs(job.cwd);
9348
+ if (!fetchedCwds || !fetchedCwds.has(job.cwd)) {
9349
+ await fetchAllRefs(job.cwd);
9350
+ if (fetchedCwds) fetchedCwds.add(job.cwd);
9351
+ }
7958
9352
  const commits = await landedSinceRun(job.cwd, job.startedAt, paths);
7959
9353
  if (!commits.length) return null;
7960
- return { commits, paths, detectedAt: new Date().toISOString() };
9354
+ const attributed = await attributeLandedCommits(job, commits, job.cwd);
9355
+ if (!attributed) return null;
9356
+ return { commits: attributed.commits, paths, detectedAt: new Date().toISOString(), rule: attributed.rule };
7961
9357
  }
7962
9358
 
7963
9359
  async function reverifyNeedsReview() {
7964
9360
  const snap = await readQueue();
7965
- const candidates = snap.jobs.filter(isRescanCandidate);
9361
+ // isGuardParkedWithoutAutoFix rows are NOT isRescanCandidate (their
9362
+ // verifierVerdict is a commit-guard/shared-tree-guard verdict, not a
9363
+ // RESCANNABLE_VERDICTS transcript-verifier one) — included here so this
9364
+ // pass also computes their looksDone evidence, the widened half of the
9365
+ // guard-verdict auto-resolve gap this PRD closes. Handled in its own
9366
+ // branch below (no transcript rescan — there is no transcript verdict to
9367
+ // rescan) rather than through the isRescanCandidate machinery.
9368
+ const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j));
7966
9369
  const healed = [];
7967
9370
  const leftForReview = [];
7968
9371
  const looksDoneUpdates = [];
9372
+ // Shared across every computeLooksDone call in this one pass — dedupes
9373
+ // the `git fetch --all --prune` per distinct cwd (see computeLooksDone's
9374
+ // header) rather than re-fetching the same repo once per candidate row.
9375
+ const fetchedCwds = new Set();
7969
9376
  for (const job of candidates) {
9377
+ if (!isRescanCandidate(job) && isGuardParkedWithoutAutoFix(job)) {
9378
+ // Guard-verdict park, never auto-fixed: only evidence gathering, never
9379
+ // a transcript rescan (there was never a transcript-verifier verdict
9380
+ // here) and never a direct heal — applyNeedsReviewAutoResolve is the
9381
+ // sole place that turns this annotation into a status change.
9382
+ const looksDone = await computeLooksDone(job, fetchedCwds);
9383
+ if (looksDone) {
9384
+ looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
9385
+ }
9386
+ continue;
9387
+ }
7970
9388
  if (job.status === 'failed') {
7971
9389
  // A failed row never runs the transcript-verifier rescan below — that
7972
9390
  // machinery (verifyRun/COMPLETED_EQUIVALENT_VERDICTS) exists to
@@ -7975,7 +9393,7 @@ async function reverifyNeedsReview() {
7975
9393
  // completing-direction constraint). The only thing a failed candidate
7976
9394
  // can gain here is a looksDone annotation + a failed → needs_review
7977
9395
  // transition, for a human to confirm.
7978
- const looksDone = await computeLooksDone(job);
9396
+ const looksDone = await computeLooksDone(job, fetchedCwds);
7979
9397
  if (looksDone) {
7980
9398
  looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: true });
7981
9399
  } else {
@@ -8030,7 +9448,7 @@ async function reverifyNeedsReview() {
8030
9448
  // always before this periodic/boot pass can run against the same row, so
8031
9449
  // this check reliably catches the only order that can occur.
8032
9450
  if (stillOpen && job.autoFixAttempted !== true) {
8033
- const looksDone = await computeLooksDone(job);
9451
+ const looksDone = await computeLooksDone(job, fetchedCwds);
8034
9452
  if (looksDone) {
8035
9453
  looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
8036
9454
  }
@@ -8044,14 +9462,14 @@ async function reverifyNeedsReview() {
8044
9462
  if (!u) continue;
8045
9463
  if (u.fromFailed) {
8046
9464
  transitionJob(j, 'needs_review', {
8047
- reason: 'looks done — commit(s) since this run touch this PRD\'s declared paths; confirm before archiving',
9465
+ reason: `looks done (${u.looksDone.rule}) — commit(s) attributable to this job's own run touch this PRD's declared paths; confirm before archiving`,
8048
9466
  source: 'reverifyNeedsReview:looksDone',
8049
9467
  });
8050
9468
  }
8051
9469
  if (j.status !== 'needs_review') continue;
8052
9470
  j.looksDone = u.looksDone;
8053
9471
  const shaList = u.looksDone.commits.slice(0, 5).map((c) => c.slice(0, 7)).join(', ');
8054
- j.error = `looks done — ${u.looksDone.commits.length} commit(s) since this run touch this PRD's paths (${shaList}); confirm before archiving`;
9472
+ j.error = `looks done (${u.looksDone.rule}) — ${u.looksDone.commits.length} commit(s) attributable to this job's own run touch this PRD's paths (${shaList}); confirm before archiving`;
8055
9473
  }
8056
9474
  });
8057
9475
  console.log(`[scheduler] boot reverify: looksDone annotated for ${looksDoneUpdates.length} row(s): ${looksDoneUpdates.map((u) => u.slug).join(', ')}`);
@@ -8323,11 +9741,21 @@ function registerScheduleHandlers() {
8323
9741
  ensureDirs();
8324
9742
  supervisor.registerHandlers();
8325
9743
 
9744
+ // Cheap read only — no reconcile(), no writeQueue(). The renderer treats
9745
+ // this as a fast call behind a 5s deadline (scheduleState.ts's
9746
+ // withTimeout), but reconcile() does a cross-project PRD discovery walk
9747
+ // plus a disk write, which could blow that deadline and, worse, throw
9748
+ // outright on a torn queue.json (reconcile refuses to run against
9749
+ // `state.unreadable`) — turning a recoverable read into a rejected IPC and
9750
+ // an error toast. Discovery still runs on a fixed cadence elsewhere:
9751
+ // tickQueue (every POLL_INTERVAL_MS, 60s), rescheduleTimer, broadcast()'s
9752
+ // coalescer (getPayload), schedule:rescan and schedule:adopt-prd. Worst
9753
+ // case, a PRD dropped on disk while the Scheduler tab is open surfaces
9754
+ // here within ~POLL_INTERVAL_MS + BROADCAST_COALESCE_MS (~60.2s) — via
9755
+ // tickQueue's reconcile + its trailing broadcast() — not via this handler.
8326
9756
  ipcMain.handle('schedule:state', async () => {
8327
9757
  const state = await readQueue();
8328
- await reconcile(state);
8329
- await writeQueue(state);
8330
- return buildScheduleStatePayload(state, { withPaths: true });
9758
+ return buildScheduleStatePayload(state);
8331
9759
  });
8332
9760
 
8333
9761
  // Session-Manager-wide claude -p slot pool (lib/sessionSlots.cjs) —
@@ -8370,6 +9798,57 @@ function registerScheduleHandlers() {
8370
9798
  };
8371
9799
  });
8372
9800
 
9801
+ // Queue-health header (PRD): the one honest read of "why does the queue
9802
+ // look stale" — reuses classifyQueueHealth so the UI and the starvation
9803
+ // watchdog can never disagree. `cwd` is optional (null = machine-wide,
9804
+ // matching WindowStrip's own scopeCwd fallback).
9805
+ ipcMain.handle('schedule:queue-health', async (_e, payload) => {
9806
+ const cwd = (payload && typeof payload.cwd === 'string') ? payload.cwd : null;
9807
+ const state = await readQueue();
9808
+ if (state.unreadable) {
9809
+ return { unknown: true, reason: state.unreadable };
9810
+ }
9811
+ const now = Date.now();
9812
+ const slotSnapshot = sessionSlots.snapshot();
9813
+ const freeSlots = Math.max(0, slotSnapshot.total - slotSnapshot.inUse);
9814
+ const verdict = classifyQueueHealth({
9815
+ jobs: state.jobs,
9816
+ paused: state.paused,
9817
+ launchBlocks: state.launchBlocks,
9818
+ runningSet,
9819
+ freeSlots,
9820
+ totalSlots: slotSnapshot.total,
9821
+ lastDispatchAttemptAtMs: Date.parse(state.lastDispatchAttemptAt ?? ''),
9822
+ now,
9823
+ cwd,
9824
+ });
9825
+ // Oldest running job across the whole machine (any project) — the
9826
+ // number that actually explains slot saturation, alongside the
9827
+ // machine-wide slot pool itself.
9828
+ let oldestRunningAgeMs = null;
9829
+ for (const j of state.jobs) {
9830
+ if (j.status !== 'running' && !runningSet.has(j.slug)) continue;
9831
+ const startedAtMs = j.startedAt ? Date.parse(j.startedAt) : NaN;
9832
+ if (!Number.isFinite(startedAtMs)) continue;
9833
+ const age = now - startedAtMs;
9834
+ if (oldestRunningAgeMs === null || age > oldestRunningAgeMs) oldestRunningAgeMs = age;
9835
+ }
9836
+ return {
9837
+ unknown: false,
9838
+ now,
9839
+ verdict,
9840
+ slots: {
9841
+ inUse: slotSnapshot.inUse,
9842
+ total: slotSnapshot.total,
9843
+ free: freeSlots,
9844
+ source: slotSnapshot.envOverride ? 'env' : 'pool',
9845
+ },
9846
+ oldestRunningAgeMs,
9847
+ lastRunAt: state.lastRunAt ?? null,
9848
+ lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
9849
+ };
9850
+ });
9851
+
8373
9852
  ipcMain.handle('schedule:force-tick', async () => {
8374
9853
  // Bypass the billing-poll gate entirely — fire pending jobs immediately regardless of meter state.
8375
9854
  // Clears any existing pause first (same semantics as run-now).
@@ -8448,6 +9927,21 @@ function registerScheduleHandlers() {
8448
9927
  return { ok: true, kind: 'info', message: `Adopted ${slug} — it will run as a normal pending job` };
8449
9928
  }));
8450
9929
 
9930
+ // Scheduler UI's "change disposition" action (scheduler wave-disposition
9931
+ // PRD): promotes an appended wave to its own head, or re-attaches a head
9932
+ // behind another chain. Thin wrapper over remote.setPrdDisposition, which
9933
+ // validates the rewrite (cycle-safety, running/completed rows untouched)
9934
+ // before delegating to the same remote.updatePrd every other PRD edit
9935
+ // path uses — see that method's own comment in this file.
9936
+ ipcMain.handle('schedule:set-prd-disposition', validated(schemas.scheduleSetPrdDisposition, async ({ slug, cwd, disposition, dependsOn }) => {
9937
+ if (!(await safeSlugPath(slug))) return { ok: false, kind: 'error', message: 'invalid slug' };
9938
+ const result = await remote.setPrdDisposition({ slug, cwd, disposition, dependsOn });
9939
+ if (!result.ok) return { ok: false, kind: 'error', message: result.error ?? 'disposition change failed' };
9940
+ appendAuditEvent('scheduler_prd_disposition_set', { slug, cwd: cwd ?? null, disposition, source: 'ipc:schedule:set-prd-disposition' });
9941
+ await broadcast({ flush: true });
9942
+ return { ok: true, kind: 'info', message: `${slug} is now ${disposition === 'new-head' ? 'an independent head' : 'attached behind the chosen chain'}` };
9943
+ }));
9944
+
8451
9945
  ipcMain.handle('schedule:run-now', async () => {
8452
9946
  // Manual run-now overrides any auto-pause. Clear it first.
8453
9947
  await clearPause('run-now');
@@ -8460,9 +9954,11 @@ function registerScheduleHandlers() {
8460
9954
  return { ok: true };
8461
9955
  });
8462
9956
 
8463
- // Re-scan prds/ folder and merge into queue.json. The `schedule:state`
8464
- // handler already reconciles on read, but this gives the renderer an
8465
- // explicit refresh path that also broadcasts so all views update.
9957
+ // Re-scan prds/ folder and merge into queue.json. `schedule:state` is a
9958
+ // cheap read with no reconcile of its own — this is the renderer's
9959
+ // explicit, immediate discovery path (mutate() + reconcile() + broadcast())
9960
+ // for "I just dropped a PRD on disk and want it to show up now" rather than
9961
+ // waiting for tickQueue's next ~60s pass.
8466
9962
  ipcMain.handle('schedule:rescan', async () => {
8467
9963
  const { added, removed } = await mutate(async (state) => {
8468
9964
  const before = new Set(state.jobs.map((j) => j.slug));
@@ -8691,6 +10187,18 @@ async function init() {
8691
10187
  const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
8692
10188
  bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
8693
10189
  }
10190
+ // Same evidence-before-failure gate reapDeadRunningJobs applies, resolved
10191
+ // BEFORE mutate() for the same reason (git spawn work must never run
10192
+ // inside mutate()'s single global serialization chain) — an orphaned job
10193
+ // classified 'failed'/'unknown' from its log tail alone can still have
10194
+ // actually landed a real commit before the app restarted mid-run.
10195
+ const bootLandedCommitEvidence = new Map();
10196
+ await Promise.all(bootSnap.jobs.map(async (j) => {
10197
+ if (!immediateSlugs.includes(j.slug) || j.status !== 'running') return;
10198
+ if (bootOutcomes.get(j.slug) === 'success' || !j.landedCommit) return;
10199
+ const resolved = await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt);
10200
+ if (resolved) bootLandedCommitEvidence.set(j.slug, j.landedCommit);
10201
+ }));
8694
10202
  const bootReconciledCompletions = [];
8695
10203
  await mutate((state) => {
8696
10204
  for (const j of state.jobs) {
@@ -8698,7 +10206,7 @@ async function init() {
8698
10206
  const outcome = bootOutcomes.get(j.slug) ?? 'unknown';
8699
10207
  const pid = j.runtime?.pid;
8700
10208
  const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
8701
- applyOrphanOutcome(j, outcome, killNote);
10209
+ applyOrphanOutcome(j, outcome, killNote, bootLandedCommitEvidence.get(j.slug) || null);
8702
10210
  if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
8703
10211
  console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
8704
10212
  }
@@ -8722,9 +10230,18 @@ async function init() {
8722
10230
  if (result === 'killed') {
8723
10231
  console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
8724
10232
  }
8725
- setTimeout(() => {
10233
+ setTimeout(async () => {
8726
10234
  const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
8727
10235
  const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
10236
+ // Same evidence-before-failure gate as the immediate-orphan path
10237
+ // above, resolved before mutate() for the same reason (git spawn
10238
+ // work must never run inside mutate()'s serialization chain). Uses
10239
+ // the captured pre-kill snapshot's landedCommit/cwd/startedAt — the
10240
+ // race guard below already confirms `cur` is still this same run
10241
+ // (runId === bootRunId) before this evidence is applied.
10242
+ const confirmedLandedCommit = (outcome !== 'success' && j.landedCommit)
10243
+ ? (await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt) ? j.landedCommit : null)
10244
+ : null;
8728
10245
  let deferredCompletedCwd;
8729
10246
  mutate((state) => {
8730
10247
  const cur = state.jobs.find((x) => x.slug === slug);
@@ -8733,7 +10250,7 @@ async function init() {
8733
10250
  // that new run is not the boot orphan we SIGTERM'd and must not be
8734
10251
  // touched by this stale classification.
8735
10252
  if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
8736
- applyOrphanOutcome(cur, outcome, killNote);
10253
+ applyOrphanOutcome(cur, outcome, killNote, confirmedLandedCommit);
8737
10254
  console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
8738
10255
  deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
8739
10256
  }).then(() => {
@@ -8879,7 +10396,8 @@ async function init() {
8879
10396
  // else distinguishes "no pending work" from "pending work, never
8880
10397
  // started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
8881
10398
  // Escalation only, same shape as the quarantine/overrun warnings above.
8882
- for (const sp of findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS)) {
10399
+ const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
10400
+ for (const sp of starvedProjects) {
8883
10401
  console.warn(
8884
10402
  `[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
8885
10403
  + `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
@@ -8887,30 +10405,104 @@ async function init() {
8887
10405
  );
8888
10406
  appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
8889
10407
  }
8890
-
8891
- // Stuck-failed escalation (2026-09-10, social-signals-trader): see
8892
- // findStuckFailedJobs' header for why `failed` has no automated way
8893
- // back to pending. Escalation only, same shape as the three warnings
8894
- // above — never an automatic failed → pending requeue (that could
8895
- // discard uncommitted work left by the failed run; see
8896
- // spawnJob:fail-dirty). Kill-switch: SM_STUCK_FAILED_ESCALATE_DISABLE=1.
8897
- if (!stuckFailedEscalationDisabled()) {
8898
- const stuckFailed = findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
8899
- if (stuckFailed.length > 0) {
8900
- mutate((ms) => {
8901
- for (const stuck of stuckFailed) {
8902
- const j = ms.jobs.find((x) => x.slug === stuck.slug);
8903
- if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
8904
- j.stuckFailedNotified = true;
10408
+ // Bounded, automated consequence for a starve that outlives the WARN
10409
+ // above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
10410
+ // project_starved rows and zero consequence). STARVE_ESCALATION_MS is
10411
+ // strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
10412
+ // a subset of the rows already reported above — same verdict, no
10413
+ // re-derivation.
10414
+ runStarveEscalationSweep(starvedProjects);
10415
+
10416
+ // Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
10417
+ // escalation now narrowed to only the rows that auto-reset gave up on.
10418
+ // See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
10419
+ // Computed together, acted on in the SAME mutate(...) pass, so the
10420
+ // stuckFailedNotified race guard below and the auto-reset race guard
10421
+ // above it can never observe two different snapshots of the same row.
10422
+ // Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
10423
+ const autoResetTargets = failedAutoResetDisabled()
10424
+ ? []
10425
+ : selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
10426
+ const stuckFailed = stuckFailedEscalationDisabled()
10427
+ ? []
10428
+ : findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
10429
+ // Bounded automatic terminal decision for exhausted needs_review rows
10430
+ // (this PRD): computed alongside the failed-row passes above and acted
10431
+ // on in the SAME mutate(...) pass below, for the same race-guard reason
10432
+ // — a row's exhaustedResolveAttempts counter must never be read from one
10433
+ // snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
10434
+ const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
10435
+ ? []
10436
+ : selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
10437
+ // Bounded automatic exit for quarantined rows (this PRD): computed
10438
+ // alongside the passes above and acted on in the SAME mutate(...) pass
10439
+ // below, for the same race-guard reason — quarantineResolveAttempts must
10440
+ // never be read from one snapshot and written from another, and the
10441
+ // createdVia re-check inside autoResolveQuarantine must happen in the
10442
+ // same turn as the transition it gates. Kill-switch:
10443
+ // SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
10444
+ const quarantineTargets = quarantineAutoResolveDisabled()
10445
+ ? []
10446
+ : selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
10447
+ if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
10448
+ mutate(async (ms) => {
10449
+ for (const target of autoResetTargets) {
10450
+ const j = ms.jobs.find((x) => x.slug === target.slug);
10451
+ if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
10452
+ const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
10453
+ j.failedAutoResetAttempts = attempt;
10454
+ const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
10455
+ // resetJobFields is the same field-clearing list the admin
10456
+ // scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
10457
+ // it rather than inventing a second list. It also sets job.error to
10458
+ // the reason text passed in; we clear that back to null right
10459
+ // after since this is a clean auto-reset, not a recorded error.
10460
+ if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
10461
+ j.error = null;
10462
+ delete j.stuckFailedNotified;
10463
+ console.warn(
10464
+ `[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
10465
+ + `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
10466
+ );
10467
+ appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
10468
+ }
10469
+ for (const stuck of stuckFailed) {
10470
+ const j = ms.jobs.find((x) => x.slug === stuck.slug);
10471
+ if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
10472
+ // Still has auto-reset attempts left — it will be (or already was,
10473
+ // earlier this same pass) picked up by the loop above instead.
10474
+ // Never log "reset it by hand" for a row that isn't actually stuck.
10475
+ if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
10476
+ j.stuckFailedNotified = true;
10477
+ console.warn(
10478
+ `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
10479
+ + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
10480
+ + `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
10481
+ );
10482
+ appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
10483
+ }
10484
+ for (const target of exhaustedNeedsReviewTargets) {
10485
+ const j = ms.jobs.find((x) => x.slug === target.slug);
10486
+ const outcome = applyNeedsReviewAutoResolve(j);
10487
+ if (outcome) {
8905
10488
  console.warn(
8906
- `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
8907
- + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
8908
- + `no automated recovery reaches a failed row; reset it by hand via scheduler_reset_job`,
10489
+ `[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
10490
+ + `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
8909
10491
  );
8910
- appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
8911
10492
  }
8912
- }).catch(() => {});
8913
- }
10493
+ }
10494
+ for (const target of quarantineTargets) {
10495
+ const j = ms.jobs.find((x) => x.slug === target.slug);
10496
+ if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
10497
+ const outcome = await autoResolveQuarantine(j, target.ageMs);
10498
+ if (outcome) {
10499
+ console.warn(
10500
+ `[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
10501
+ + `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
10502
+ );
10503
+ }
10504
+ }
10505
+ }).catch(() => {});
8914
10506
  }
8915
10507
  }, 10 * 60_000);
8916
10508
 
@@ -9094,7 +10686,9 @@ async function listPrdsInternal() {
9094
10686
  estimateMinutes: parsed.estimateMinutes,
9095
10687
  sourcePromptId: parsed.sourcePromptId,
9096
10688
  epicId: parsed.epicId ?? null,
10689
+ dependsOn: parsed.dependsOn ?? null,
9097
10690
  agentType: parsed.agentType ?? null,
10691
+ disposition: parsed.disposition ?? null,
9098
10692
  mtimeMs: stat.mtimeMs,
9099
10693
  archived,
9100
10694
  };
@@ -9491,6 +11085,33 @@ const remote = {
9491
11085
  }
9492
11086
  },
9493
11087
 
11088
+ // Backs the Scheduler UI's "change disposition" action (scheduler
11089
+ // wave-disposition PRD): promoting an appended wave to its own head, or
11090
+ // re-attaching a head behind another chain. `dependsOn` for a 'new-head'
11091
+ // disposition is ignored (cleared unconditionally); for 'append' it's the
11092
+ // caller's chosen target chain's terminal slug(s) — the renderer computes
11093
+ // that from the SAME backlog tree (lib/backlogTree.ts) it already renders,
11094
+ // so this function only has to validate the rewrite is safe, never
11095
+ // re-derive "the" terminal itself.
11096
+ //
11097
+ // Validates via prdDisposition.cjs's computeDispositionRewrite (row not
11098
+ // running/completed, no already-satisfied blocker being rewritten out from
11099
+ // under it, no dependsOn cycle) BEFORE delegating the actual write to this
11100
+ // SAME updatePrd — so a rejected rewrite never reaches the filesystem, and
11101
+ // an accepted one gets updatePrd's own dependsOn FK re-validation for free.
11102
+ async setPrdDisposition({ slug, cwd, disposition, dependsOn }) {
11103
+ let listing;
11104
+ try {
11105
+ listing = await this.listPrds({ cwd, fields: 'full', limit: Number.MAX_SAFE_INTEGER });
11106
+ } catch (e) {
11107
+ return { ok: false, error: `could not read project PRDs: ${e?.message ?? e}` };
11108
+ }
11109
+ const rows = listing.prds ?? [];
11110
+ const rewrite = computeDispositionRewrite({ slug, disposition, dependsOn: dependsOn ?? [], rows });
11111
+ if (!rewrite.ok) return rewrite;
11112
+ return this.updatePrd({ slug, cwd, frontmatter: { dependsOn: rewrite.dependsOn, disposition } });
11113
+ },
11114
+
9494
11115
  // Cancels a job that hasn't finished yet. A 'running' job's process group
9495
11116
  // is SIGTERM'd (reusing killOrphanClaudePid — the same kill path boot
9496
11117
  // reconciliation uses for an orphaned running job) before its queue row is
@@ -9518,18 +11139,38 @@ const remote = {
9518
11139
  if (wasRunning && pid) {
9519
11140
  killOrphanClaudePid(pid);
9520
11141
  }
11142
+ // Evidence-before-failure guard, scoped to an actually-running job being
11143
+ // killed here (a 'pending' cancel has no live process, so nothing new
11144
+ // could have landed since its last stamp — and 'needs_review' is not
11145
+ // even a legal transition from 'pending', see LEGAL_TRANSITIONS): the
11146
+ // same reapDeadRunningJobs evidence gate (job 1192 — a landedCommit
11147
+ // being non-empty is not proof by itself, but discarding proof of real
11148
+ // landed work with no check at all is worse) applies here too. A
11149
+ // dead-pid reap of a job that landed a commit (e.g. via the
11150
+ // dispatch-time sidecar backfill) is routed to needs_review/completed;
11151
+ // a deliberate cancel of that same state deserves no less.
11152
+ const confirmedLandedCommit = (wasRunning && job.landedCommit)
11153
+ ? ((await resolveLandedCommitEvidence(job.cwd || DEFAULT_PROJECT_CWD, job.landedCommit, job.startedAt))
11154
+ ? job.landedCommit
11155
+ : null)
11156
+ : null;
11157
+ const targetStatus = confirmedLandedCommit ? 'needs_review' : 'failed';
11158
+ const cancelReason = confirmedLandedCommit
11159
+ ? `cancelled via admin API, but landedCommit ${confirmedLandedCommit} resolves — verify before treating as done`
11160
+ : 'cancelled via admin API';
9521
11161
  await mutate((s) => {
9522
11162
  const idx = s.jobs.findIndex((j) => j.slug === slug);
9523
11163
  if (idx < 0) return;
9524
11164
  const j = s.jobs[idx];
9525
- transitionJob(j, 'failed', { reason: 'cancelled via admin API', source: 'remote:cancelJob' });
9526
- j.error = 'cancelled via admin API';
11165
+ transitionJob(j, targetStatus, { reason: cancelReason, source: 'remote:cancelJob' });
11166
+ j.error = cancelReason;
9527
11167
  j.finishedAt = new Date().toISOString();
9528
11168
  j.exitCode = j.exitCode ?? null;
11169
+ if (confirmedLandedCommit) j.verifierVerdict = 'cancelled_with_landed_commit';
9529
11170
  delete j.runtime;
9530
11171
  });
9531
11172
  await broadcast({ flush: true });
9532
- return { ok: true, slug, status: 'failed', wasRunning, cwd: job.cwd ?? null };
11173
+ return { ok: true, slug, status: targetStatus, wasRunning, cwd: job.cwd ?? null };
9533
11174
  },
9534
11175
 
9535
11176
  // Exposes the module-level allocateParallelGroup (PRD 548) to callers that
@@ -9575,13 +11216,27 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
9575
11216
 
9576
11217
  module.exports = {
9577
11218
  classifyQueueStarvation,
11219
+ classifyQueueStarvationByProject,
11220
+ classifyQueueHealth,
9578
11221
  runQueueStarvationWatchdog,
9579
11222
  QUEUE_STARVATION_MS,
11223
+ selectStarveEscalations,
11224
+ runStarveEscalationSweep,
11225
+ STARVE_ESCALATION_MS,
9580
11226
  computeBlockedChains,
9581
11227
  stripAppOwnedChurn,
9582
11228
  findOverrunningJobs,
9583
11229
  JOB_OVERRUN_FACTOR,
9584
11230
  JOB_OVERRUN_FLOOR_MS,
11231
+ computeJobBudgetMs,
11232
+ classifyBudgetKill,
11233
+ isJobBudgetExempt,
11234
+ shouldKillForBudget,
11235
+ resolveBudgetKillOutcome,
11236
+ JOB_BUDGET_FACTOR,
11237
+ JOB_BUDGET_FLOOR_MS,
11238
+ JOB_BUDGET_CEILING_MS,
11239
+ BUDGET_WARNING_FRACTION,
9585
11240
  registerScheduleHandlers,
9586
11241
  attachWindow,
9587
11242
  init,
@@ -9595,10 +11250,12 @@ module.exports = {
9595
11250
  healRefusalReason,
9596
11251
  writeQueue,
9597
11252
  reconcile,
11253
+ broadcast,
9598
11254
  reconcileSourcePromptId,
9599
11255
  allocateParallelGroup,
9600
11256
  selectHistoryJobs,
9601
11257
  parsePorcelain,
11258
+ parsePorcelainEntries,
9602
11259
  FINISH_PROTOCOL,
9603
11260
  IDLE_OUTPUT_KILL_MS,
9604
11261
  BASH_DEFAULT_TIMEOUT_MS,
@@ -9616,9 +11273,19 @@ module.exports = {
9616
11273
  findStuckFailedJobs,
9617
11274
  STUCK_FAILED_ESCALATE_MS,
9618
11275
  stuckFailedEscalationDisabled,
11276
+ selectFailedAutoResetTargets,
11277
+ FAILED_AUTORESET_CAP,
11278
+ FAILED_AUTORESET_MS,
11279
+ failedAutoResetDisabled,
11280
+ selectExhaustedNeedsReviewTargets,
11281
+ applyNeedsReviewAutoResolve,
11282
+ NEEDS_REVIEW_RESOLVE_CAP,
11283
+ NEEDS_REVIEW_RESOLVE_MS,
11284
+ needsReviewAutoResolveDisabled,
9619
11285
  isRescanCandidate,
9620
11286
  isFailedUnverifiedShaped,
9621
11287
  computeLooksDone,
11288
+ attributeLandedCommits,
9622
11289
  isPromotableOriginal,
9623
11290
  selectAutoFixTargets,
9624
11291
  applyRcaClassification,
@@ -9626,6 +11293,9 @@ module.exports = {
9626
11293
  resolveRunId,
9627
11294
  isUnresolvableNeedsReview,
9628
11295
  isExhaustedAutoFix,
11296
+ GUARD_VERDICT_EVIDENCE_ELIGIBLE,
11297
+ isGuardParkedWithoutAutoFix,
11298
+ isEligibleForNeedsReviewAutoResolve,
9629
11299
  isPlanUnqueued,
9630
11300
  isFixPlanDead,
9631
11301
  fixSlugFor,
@@ -9646,6 +11316,7 @@ module.exports = {
9646
11316
  MAX_INVESTIGATION_DEPTH,
9647
11317
  forceTickOutcome,
9648
11318
  applyPauseCleared,
11319
+ formatLoadGateDetail,
9649
11320
  detectNetworkErrorInLog,
9650
11321
  detectRateLimitInLog,
9651
11322
  classifyFailureOutcome,
@@ -9695,6 +11366,10 @@ module.exports = {
9695
11366
  computeStallSummary,
9696
11367
  findStaleQuarantinedJobs,
9697
11368
  QUARANTINE_ESCALATE_MS,
11369
+ selectQuarantineAutoResolveTargets,
11370
+ autoResolveQuarantine,
11371
+ QUARANTINE_RESOLVE_CAP,
11372
+ quarantineAutoResolveDisabled,
9698
11373
  applyClearQueueVictims,
9699
11374
  PIDLESS_SPAWN_GRACE_MS,
9700
11375
  findStrandedInvestigations,
@@ -9706,6 +11381,7 @@ module.exports = {
9706
11381
  evaluateSharedTreeGuard,
9707
11382
  checkSharedTreeGuard,
9708
11383
  uncommittedChanges,
11384
+ uncommittedChangesWithStatus,
9709
11385
  gitHead,
9710
11386
  isBranchAlreadyIntegrated,
9711
11387
  selectResumeRecoveryTarget,