claude-code-session-manager 0.86.0 → 0.87.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/dist/assets/AgentLibrary-DyLWzZDf.js +3 -0
  2. package/dist/assets/{DataModel-Q4jhl24R.js → DataModel--mISIJ6h.js} +1 -1
  3. package/dist/assets/{History-Cj2FejEo.js → History-C2ahUXTg.js} +2 -2
  4. package/dist/assets/{Hooks-CaelQI6t.js → Hooks-BiC6oyR2.js} +3 -3
  5. package/dist/assets/{HostBilko--v7cMR8I.js → HostBilko-BPleEOld.js} +1 -1
  6. package/dist/assets/{Library-DgI9oCCZ.js → Library-Dc8Qst1R.js} +1 -1
  7. package/dist/assets/{ListDetail-DYUZN-x-.js → ListDetail-DIXh-OLX.js} +1 -1
  8. package/dist/assets/MarkdownEditor-C90bkLXK.js +1 -0
  9. package/dist/assets/{McpServers-ypCYURh3.js → McpServers-DqcbLOLZ.js} +2 -2
  10. package/dist/assets/{Memory-C2qYp-3M.js → Memory-CW62MXlh.js} +4 -4
  11. package/dist/assets/{Panel-Cj2kw-Zv.js → Panel-Bw1FhRuF.js} +1 -1
  12. package/dist/assets/Permissions-BcUC-5y8.js +3 -0
  13. package/dist/assets/{Plugins-C1Vj8_dU.js → Plugins-BnKx9flD.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-DczPNM5U.js → ProvenanceBadge-Bw5vNVPT.js} +1 -1
  15. package/dist/assets/{SaveBar-Cd_7U6Gb.js → SaveBar-CWr0O_w-.js} +1 -1
  16. package/dist/assets/Scheduler-DYdLuUqq.js +14 -0
  17. package/dist/assets/{ScopeSwitcher-DVSyI44-.js → ScopeSwitcher-CrBLbg8s.js} +1 -1
  18. package/dist/assets/Settings-DluB-vN1.js +3 -0
  19. package/dist/assets/{SkillReferenceGraph-CUv1_Q2c.js → SkillReferenceGraph-CHLSseay.js} +1 -1
  20. package/dist/assets/{Skills-C_YHkAy-.js → Skills-gNdo_HNK.js} +2 -2
  21. package/dist/assets/{SystemPrompt-B8R7T9xn.js → SystemPrompt-Cru05-Ia.js} +1 -1
  22. package/dist/assets/{TagLibrary-dj9YHWyy.js → TagLibrary-DNHY0xou.js} +1 -1
  23. package/dist/assets/{TiptapBody-DnSBUjHE.js → TiptapBody-I4lmbCgP.js} +1 -1
  24. package/dist/assets/{Toggle-CjV_BJn6.js → Toggle-bWMHjmRh.js} +1 -1
  25. package/dist/assets/{index-CDo9xBR9.css → index-DV3PorRY.css} +1 -1
  26. package/dist/assets/{index-CXFQIPhO.js → index-fc_JjdxL.js} +724 -724
  27. package/dist/assets/settingsSchema-BfhtZnGD.js +3 -0
  28. package/dist/index.html +2 -2
  29. package/package.json +14 -14
  30. package/plugins/CLAUDE.md +61 -0
  31. package/plugins/session-manager-dev/.claude-plugin/plugin.json +1 -1
  32. package/plugins/session-manager-dev/skills/builder/4-manual/SKILL.md +1 -1
  33. package/plugins/session-manager-dev/skills/ops-sweep/SKILL.md +1 -1
  34. package/scripts/scheduler-mcp-server.cjs +7 -0
  35. package/src/main/__tests__/agentModelResolve.test.cjs +100 -9
  36. package/src/main/__tests__/broadcastCoalescer.test.cjs +18 -0
  37. package/src/main/__tests__/epicMint.test.cjs +2 -2
  38. package/src/main/__tests__/health-delegation-chain.test.cjs +2 -1
  39. package/src/main/__tests__/needsReviewLedger.test.cjs +162 -0
  40. package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +3 -3
  41. package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +15 -1
  42. package/src/main/__tests__/prdCreateDisposition.test.cjs +201 -0
  43. package/src/main/__tests__/prdFrontmatterDisposition.test.cjs +125 -0
  44. package/src/main/__tests__/prdLocations.test.cjs +100 -2
  45. package/src/main/__tests__/prdLocationsArchived.test.cjs +43 -1
  46. package/src/main/__tests__/prdSetDisposition.test.cjs +222 -0
  47. package/src/main/__tests__/queue-health-verdict.test.cjs +170 -0
  48. package/src/main/__tests__/queue-starvation-per-project.test.cjs +14 -2
  49. package/src/main/__tests__/queueHistory.test.cjs +63 -0
  50. package/src/main/__tests__/reconcileTiming.test.cjs +135 -0
  51. package/src/main/__tests__/scheduleJobTransitions.test.cjs +101 -1
  52. package/src/main/__tests__/scheduler-boot-orphans.test.cjs +2 -2
  53. package/src/main/__tests__/scheduler-broadcast-reconcile.test.cjs +121 -0
  54. package/src/main/__tests__/scheduler-cross-project-batch.test.cjs +43 -0
  55. package/src/main/__tests__/scheduler-guard-verdict-autoresolve.test.cjs +344 -0
  56. package/src/main/__tests__/scheduler-job-budget.test.cjs +172 -0
  57. package/src/main/__tests__/scheduler-looks-done.test.cjs +93 -3
  58. package/src/main/__tests__/scheduler-porcelain-rename.test.cjs +164 -0
  59. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +160 -0
  60. package/src/main/__tests__/scheduler-reaper-helpers-basics.test.cjs +87 -0
  61. package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +88 -0
  62. package/src/main/__tests__/scheduler-starve-escalation.test.cjs +14 -4
  63. package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +14 -0
  64. package/src/main/chatRunner.cjs +8 -5
  65. package/src/main/health.cjs +1 -1
  66. package/src/main/historyAggregator.cjs +5 -0
  67. package/src/main/index.cjs +95 -44
  68. package/src/main/ipcSchemas.cjs +47 -0
  69. package/src/main/lib/__tests__/active-sessions.test.cjs +251 -0
  70. package/src/main/lib/__tests__/bootSelfHeal.test.cjs +107 -0
  71. package/src/main/lib/__tests__/delegationReadiness.test.cjs +322 -43
  72. package/src/main/lib/__tests__/effectiveModelInfo.test.cjs +239 -0
  73. package/src/main/lib/__tests__/gitWorktree.test.cjs +89 -0
  74. package/src/main/lib/__tests__/guardShims.test.cjs +151 -0
  75. package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +5 -5
  76. package/src/main/lib/__tests__/prdDisposition.test.cjs +224 -0
  77. package/src/main/lib/__tests__/reaperHelpers.test.cjs +179 -1
  78. package/src/main/lib/__tests__/usageCircuit.test.cjs +224 -0
  79. package/src/main/lib/__tests__/watchdog-helpers.test.cjs +312 -0
  80. package/src/main/lib/__tests__/watchdog-relaunch.test.cjs +193 -0
  81. package/{scripts → src/main}/lib/activeSessions.cjs +50 -4
  82. package/src/main/lib/agentModelResolve.cjs +65 -27
  83. package/src/main/lib/bootSelfHeal.cjs +88 -0
  84. package/src/main/lib/delegationReadiness.cjs +290 -225
  85. package/src/main/lib/effectiveModelInfo.cjs +333 -0
  86. package/src/main/lib/ephemeralCwd.cjs +1 -1
  87. package/src/main/lib/epicMint.cjs +3 -3
  88. package/src/main/lib/gitWorktree.cjs +42 -12
  89. package/src/main/lib/guardShims.cjs +156 -0
  90. package/src/main/lib/jobDirtFilter.cjs +7 -2
  91. package/src/main/lib/launchFailure.cjs +2 -1
  92. package/src/main/lib/mcpToolCatalog.cjs +4 -1
  93. package/src/main/lib/needsReviewLedger.cjs +205 -0
  94. package/src/main/lib/opsErrorLog.cjs +1 -1
  95. package/src/main/lib/opsOwnership.cjs +1 -1
  96. package/src/main/lib/prdCreate.cjs +56 -1
  97. package/src/main/lib/prdDisposition.cjs +199 -0
  98. package/src/main/lib/prdFrontmatter.cjs +8 -2
  99. package/src/main/lib/prdLocations.cjs +167 -45
  100. package/src/main/lib/projectHomeAdminRoutes.cjs +4 -4
  101. package/src/main/lib/projectPageSummarySchema.cjs +1 -1
  102. package/src/main/lib/projectRootResolve.cjs +1 -1
  103. package/src/main/lib/queueHistory.cjs +19 -1
  104. package/src/main/lib/queueStore.cjs +6 -1
  105. package/src/main/lib/reaperHelpers.cjs +181 -15
  106. package/src/main/lib/scheduleJobSchema.cjs +8 -0
  107. package/src/main/lib/scheduleJobTransitions.cjs +33 -0
  108. package/src/main/lib/schedulerConfig.cjs +24 -0
  109. package/src/main/lib/usageCircuit.cjs +159 -0
  110. package/{scripts → src/main}/lib/watchdogHelpers.cjs +1 -1
  111. package/src/main/scheduler/prdParser.cjs +13 -0
  112. package/src/main/scheduler.cjs +1285 -143
  113. package/src/main/templates/PRD_AUTHORING.md +50 -0
  114. package/src/main/templates/project-pages-catalog.json +1 -1
  115. package/src/main/usage.cjs +21 -3
  116. package/src/preload/api.d.ts +92 -1
  117. package/src/preload/index.cjs +10 -0
  118. package/web/README.md +41 -0
  119. package/{scripts/render-project-pages.cjs → web/project-pages/render.cjs} +4 -4
  120. package/{scripts/render-project-pages → web/project-pages/renderer}/dist/renderer.cjs +1 -1
  121. package/{scripts/validate-project-pages-summary.cjs → web/project-pages/validate-summary.cjs} +5 -5
  122. package/dist/assets/AgentLibrary-DTFL7y8G.js +0 -3
  123. package/dist/assets/MarkdownEditor-DCIubYWf.js +0 -1
  124. package/dist/assets/Permissions-BiYZNGYW.js +0 -3
  125. package/dist/assets/Scheduler-DcLBiJBq.js +0 -14
  126. package/dist/assets/Settings-Cv-pRyms.js +0 -3
  127. package/dist/assets/settingsSchema-BJVciriw.js +0 -3
  128. /package/{scripts/project-pages-logic → web/project-pages/logic}/dist/logic.cjs +0 -0
@@ -60,7 +60,9 @@ const { readTail } = require('./lib/fileTail.cjs');
60
60
  const {
61
61
  claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs,
62
62
  findLiveProcessForJob, logHasOutput, resolvePidlessGateOutcome, resolveCommitGuardOutcome,
63
+ readSpawnedPidFromLog, readLogMtimeMs,
63
64
  } = require('./lib/reaperHelpers.cjs');
65
+ const { resolveProjectRoot } = require('./lib/opsOwnership.cjs');
64
66
  const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
65
67
  const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
66
68
  const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
@@ -89,9 +91,13 @@ const {
89
91
  USAGE_REFRESH_INTERVAL_MS,
90
92
  MAX_JOB_DURATION_MS,
91
93
  BROADCAST_COALESCE_MS,
94
+ RECONCILE_SLOW_PASS_MS,
92
95
  QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
93
96
  JOB_OVERRUN_FACTOR: JOB_OVERRUN_FACTOR_DEFAULT,
94
97
  JOB_OVERRUN_FLOOR_MS: JOB_OVERRUN_FLOOR_MS_DEFAULT,
98
+ JOB_BUDGET_FACTOR: JOB_BUDGET_FACTOR_DEFAULT,
99
+ JOB_BUDGET_FLOOR_MS: JOB_BUDGET_FLOOR_MS_DEFAULT,
100
+ JOB_BUDGET_CEILING_MS: JOB_BUDGET_CEILING_MS_DEFAULT,
95
101
  PIDLESS_SPAWN_GRACE_MS,
96
102
  INVESTIGATION_MAX_MS,
97
103
  STARVATION_ESCALATE_MS,
@@ -100,12 +106,27 @@ const {
100
106
  const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
101
107
  ? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
102
108
  : QUARANTINE_ESCALATE_MS_DEFAULT;
103
- const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
104
- ? Number(process.env.SM_JOB_OVERRUN_FACTOR)
105
- : JOB_OVERRUN_FACTOR_DEFAULT;
106
- const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
107
- ? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
108
- : JOB_OVERRUN_FLOOR_MS_DEFAULT;
109
+ // Shared by every SM_*-env-overridable numeric constant below (bare factors
110
+ // use unitMs=1; minute-denominated knobs use unitMs=60_000) — one parse rule
111
+ // instead of one hand-copied ternary per constant.
112
+ function numEnvOverride(envVar, unitMs, fallback) {
113
+ const raw = process.env[envVar];
114
+ return raw ? Number(raw) * unitMs : fallback;
115
+ }
116
+ const JOB_OVERRUN_FACTOR = numEnvOverride('SM_JOB_OVERRUN_FACTOR', 1, JOB_OVERRUN_FACTOR_DEFAULT);
117
+ const JOB_OVERRUN_FLOOR_MS = numEnvOverride('SM_JOB_OVERRUN_FLOOR_MINUTES', 60_000, JOB_OVERRUN_FLOOR_MS_DEFAULT);
118
+ // Same three numbers as JOB_OVERRUN_FACTOR/JOB_OVERRUN_FLOOR_MS today (3x,
119
+ // 45min) is coincidental, not structural — this triad ACTS (kills) where
120
+ // JOB_OVERRUN_* only ever escalates (see JOB_OVERRUN_FACTOR's own header);
121
+ // tune them independently, don't re-couple on a future pass just because the
122
+ // defaults happen to match right now.
123
+ const JOB_BUDGET_FACTOR = numEnvOverride('SM_JOB_BUDGET_FACTOR', 1, JOB_BUDGET_FACTOR_DEFAULT);
124
+ const JOB_BUDGET_FLOOR_MS = numEnvOverride('SM_JOB_BUDGET_FLOOR_MINUTES', 60_000, JOB_BUDGET_FLOOR_MS_DEFAULT);
125
+ const JOB_BUDGET_CEILING_MS = numEnvOverride('SM_JOB_BUDGET_CEILING_MINUTES', 60_000, JOB_BUDGET_CEILING_MS_DEFAULT);
126
+ // A running job past this fraction of its own budget gets a durable
127
+ // `budgetWarning` stamp on its row (see the budget watchdog below) so the
128
+ // renderer can warn BEFORE the kill, not only after.
129
+ const BUDGET_WARNING_FRACTION = 0.75;
109
130
  const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD, DEP_HISTORY_FAIL_OPEN } = require('./lib/schedulerBatch.cjs');
110
131
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
111
132
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
@@ -151,8 +172,9 @@ const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
151
172
  const queueStore = require('./lib/queueStore.cjs');
152
173
  const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
153
174
  const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
175
+ const { computeDispositionRewrite } = require('./lib/prdDisposition.cjs');
154
176
  const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
155
- const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
177
+ const { allProjectCwds } = require('./lib/activeSessions.cjs');
156
178
 
157
179
  // Captured once at module load so every run's meta sidecar can record how
158
180
  // stale the running process is relative to on-disk source (incident: PRD
@@ -311,22 +333,95 @@ only post-AC work. If a review finding can't be fixed within scope, commit what
311
333
  you have, describe the finding in the commit body, and note the follow-up in your
312
334
  final result.`;
313
335
 
314
- // Parse \`git status --porcelain\` output into a list of changed paths. Pure +
315
- // exported for unit testing. Each porcelain line is "XY<space>PATH" (2 status
316
- // chars + space), so the path starts at index 3; rename lines ("R a -> b")
317
- // keep the "a -> b" tail, which is fine for a human-facing dirty-file list.
318
- function parsePorcelain(stdout) {
336
+ // Unquote a single git porcelain v1 path token. Defined once in
337
+ // gitWorktree.cjs (which this file already requires — the reverse would be
338
+ // circular, since gitWorktree.cjs's own salvageDirtyDelta needs the exact
339
+ // same unquoting) and reused here rather than re-implemented, so the two
340
+ // porcelain consumers in this codebase can never drift apart.
341
+ const { unquotePorcelainPath } = gitWorktree;
342
+
343
+ // Split a rename/copy porcelain path field ("old -> new") into its two real
344
+ // paths. Each side is independently quoted per unquotePorcelainPath's rule —
345
+ // only the side that needs escaping is wrapped in quotes, the literal " -> "
346
+ // arrow between them never is. Returns null when no " -> " separator is
347
+ // found (a malformed/unexpected line) so the caller can fall back to treating
348
+ // the whole field as one opaque path rather than guessing.
349
+ function splitRenamePorcelainField(field) {
350
+ const arrow = ' -> ';
351
+ let head;
352
+ let rest;
353
+ if (field[0] === '"') {
354
+ let end = -1;
355
+ for (let i = 1; i < field.length; i += 1) {
356
+ if (field[i] === '\\') { i += 1; continue; }
357
+ if (field[i] === '"') { end = i; break; }
358
+ }
359
+ if (end === -1) return null;
360
+ head = field.slice(0, end + 1);
361
+ rest = field.slice(end + 1);
362
+ } else {
363
+ const idx = field.indexOf(arrow);
364
+ if (idx === -1) return null;
365
+ // An unquoted path containing a literal " -> " substring (git only
366
+ // quotes for a quote/backslash/control-byte/non-ASCII byte — a plain
367
+ // ASCII arrow inside a filename is never quoted) makes the true
368
+ // old/new boundary genuinely ambiguous from this text alone: the first
369
+ // occurrence could be the real separator, or it could be sitting
370
+ // inside the old path with the real separator later in the field.
371
+ // Guessing wrong silently corrupts oldPath/path for downstream
372
+ // fs.existsSync/Set-membership checks, which is worse than the
373
+ // existing "malformed line" fallback below — so more than one
374
+ // occurrence falls back to treating the whole field as one opaque
375
+ // path, same as any other line this function can't confidently parse.
376
+ if (field.indexOf(arrow, idx + arrow.length) !== -1) return null;
377
+ head = field.slice(0, idx);
378
+ rest = field.slice(idx);
379
+ }
380
+ if (!rest.startsWith(arrow)) return null;
381
+ return { oldPath: unquotePorcelainPath(head), path: unquotePorcelainPath(rest.slice(arrow.length)) };
382
+ }
383
+
384
+ // Parse `git status --porcelain` output into `{ code, path }` entries (plus
385
+ // `oldPath` for a rename/copy). Pure + exported for unit testing. Each
386
+ // porcelain line is "XY<space>PATH"; a staged rename/copy line is
387
+ // "XY<space>OLD -> NEW" instead — X (index status) is 'R' or 'C' — and NEW is
388
+ // the path git will report in any later `git status` call, so callers that
389
+ // key off `.path` (dirtyAfter membership, pathsCommittedDuringRun membership,
390
+ // fs.existsSync) must compare against NEW, never the fused "OLD -> NEW"
391
+ // string. `oldPath` is retained on the entry for callers that need the
392
+ // original path too. `code` is the raw 2-char status (e.g. '??' for
393
+ // untracked) — callers that need to distinguish "untracked" from
394
+ // "tracked-but-modified" (the shared-tree guard's revert-vs-now-ignored
395
+ // split) read it off the entry instead of re-deriving it later.
396
+ function parsePorcelainEntries(stdout) {
319
397
  return String(stdout || '')
320
398
  .split('\n')
321
399
  .filter((l) => l.length > 0)
322
- .map((l) => l.slice(3))
323
- .filter(Boolean);
400
+ .map((l) => {
401
+ const code = l.slice(0, 2);
402
+ const field = l.slice(3);
403
+ if (code.includes('R') || code.includes('C')) {
404
+ const split = splitRenamePorcelainField(field);
405
+ if (split) return { code, path: split.path, oldPath: split.oldPath };
406
+ }
407
+ return { code, path: unquotePorcelainPath(field) };
408
+ })
409
+ .filter((e) => e.path);
324
410
  }
325
411
 
326
- // Return the list of uncommitted paths in cwd, or null when the guard does not
327
- // apply (cwd is not a git work tree, git is missing, or the call errors). Never
328
- // throws — a guard failure must not fail an otherwise-successful job.
329
- function uncommittedChanges(cwd) {
412
+ // Parse `git status --porcelain` output into a list of changed paths. Pure +
413
+ // exported for unit testing.
414
+ function parsePorcelain(stdout) {
415
+ return parsePorcelainEntries(stdout).map((e) => e.path);
416
+ }
417
+
418
+ // Same as uncommittedChanges but keeps each path's porcelain status code —
419
+ // the shared-tree guard's baseline needs this to tell "was untracked" apart
420
+ // from "was tracked-and-modified" (see evaluateSharedTreeGuard). Returns null
421
+ // when the guard does not apply (cwd is not a git work tree, git is missing,
422
+ // or the call errors); never throws — a guard failure must not fail an
423
+ // otherwise-successful job.
424
+ function uncommittedChangesWithStatus(cwd) {
330
425
  return new Promise((resolve) => {
331
426
  if (!cwd) { resolve(null); return; }
332
427
  execFile(
@@ -334,13 +429,24 @@ function uncommittedChanges(cwd) {
334
429
  ['-C', cwd, 'status', '--porcelain'],
335
430
  { timeout: 10_000, windowsHide: true },
336
431
  (err, stdout) => {
337
- if (err) { resolve(null); return; } // not a repo / git missing → skip
338
- resolve(parsePorcelain(stdout));
432
+ if (err) { resolve(null); return; }
433
+ resolve(parsePorcelainEntries(stdout));
339
434
  },
340
435
  );
341
436
  });
342
437
  }
343
438
 
439
+ // Return the list of uncommitted paths in cwd, or null under the same
440
+ // conditions as uncommittedChangesWithStatus (never throws). Kept as a thin
441
+ // path-only projection of that call rather than its own execFile, so a future
442
+ // fix to the git invocation (timeout, error handling) can't land in one and
443
+ // silently miss the other.
444
+ function uncommittedChanges(cwd) {
445
+ return uncommittedChangesWithStatus(cwd).then((entries) => (
446
+ entries === null ? null : entries.map((e) => e.path)
447
+ ));
448
+ }
449
+
344
450
  // Return the current HEAD commit sha in cwd, or null on any error. Used by the
345
451
  // commit-guard to detect whether the job self-committed during its run (HEAD
346
452
  // moved) — in which case leftover working-tree dirt is presumptively from a
@@ -429,17 +535,52 @@ function restoreSpecificStash(cwd, ref) {
429
535
  // - reverted: a path that was dirty in the baseline, is clean now, and was
430
536
  // not touched by any commit landed during the run — the job reset/
431
537
  // checked-out over pre-existing uncommitted work without stashing it.
432
- // Pure/no I/O — the guard's git calls happen at the call site
433
- // (checkSharedTreeGuard). Exported for unit testing.
434
- function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAfter, pathsCommittedDuringRun }) {
538
+ //
539
+ // A THIRD outcome is not a revert at all: an untracked path can drop out of
540
+ // `git status` because the run committed a `.gitignore` change that now
541
+ // matches it — the file is untouched on disk, just no longer visible to git
542
+ // (Incident: 2026-09-12, PRD 1181 added a bare `logs/` ignore pattern, eleven
543
+ // untracked `session-manager-operations/logs/*` paths vanished from status,
544
+ // and an otherwise-perfect run was parked in needs_review for a human who had
545
+ // nothing to decide). `dirtyBefore` entries therefore carry each path's
546
+ // porcelain status code (`{ code, path }`, from parsePorcelainEntries) so this
547
+ // function can tell "was untracked" apart from "was tracked-and-modified":
548
+ // - a TRACKED path (any code other than '??') leaving the dirty set always
549
+ // means its content was restored to HEAD — still `reverted`, even though
550
+ // the file still exists on disk, because for a tracked file "exists" is
551
+ // not the question; "matches what the human left uncommitted" is.
552
+ // - an UNTRACKED path ('??') leaving the dirty set is `reverted` only if it
553
+ // no longer exists on disk; if it still exists, it merely became ignored
554
+ // and is reported separately as `nowIgnored`.
555
+ // Plain path strings are still accepted in `dirtyBefore` for callers that
556
+ // have no status code (e.g. the stash-detection pass, which always passes an
557
+ // empty array) — an entry with no `code` is treated as tracked, matching the
558
+ // old behavior exactly.
559
+ //
560
+ // Pure/no I/O — status-code parsing and on-disk existence checks both happen
561
+ // at the call site (checkSharedTreeGuard); this function never stats the
562
+ // filesystem. Exported for unit testing.
563
+ function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAfter, pathsCommittedDuringRun, existsAfter }) {
435
564
  const beforeHashes = new Set((stashBefore || []).map((l) => parseStashLine(l)?.hash).filter(Boolean));
436
565
  const newStashes = (stashAfter || [])
437
566
  .map(parseStashLine)
438
567
  .filter((e) => e && !beforeHashes.has(e.hash));
439
568
  const dirtyAfterSet = new Set(dirtyAfter || []);
440
569
  const committedSet = new Set(pathsCommittedDuringRun || []);
441
- const reverted = (dirtyBefore || []).filter((p) => !dirtyAfterSet.has(p) && !committedSet.has(p));
442
- return { newStashes, reverted };
570
+ const existsSet = new Set(existsAfter || []);
571
+ const reverted = [];
572
+ const nowIgnored = [];
573
+ for (const entry of dirtyBefore || []) {
574
+ const p = typeof entry === 'string' ? entry : entry.path;
575
+ const code = typeof entry === 'string' ? undefined : entry.code;
576
+ if (dirtyAfterSet.has(p) || committedSet.has(p)) continue;
577
+ if (code === '??' && existsSet.has(p)) {
578
+ nowIgnored.push(p);
579
+ } else {
580
+ reverted.push(p);
581
+ }
582
+ }
583
+ return { newStashes, reverted, nowIgnored };
443
584
  }
444
585
 
445
586
  // Post-run shared-tree guard for an IN-PLACE job (worktree.ok === false —
@@ -489,18 +630,40 @@ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBef
489
630
  // restored stash is not ALSO reported as an unexplained revert (it was
490
631
  // explained — by the stash this guard just restored).
491
632
  const dirtyAfter = await module.exports.uncommittedChanges(cwd);
492
- const { reverted } = module.exports.evaluateSharedTreeGuard({
633
+ // Existence check for the "now ignored, not reverted" split (2026-09-12
634
+ // incident) — only untracked baseline entries need it; a tracked path
635
+ // leaving the dirty set is always a revert regardless of disk state (see
636
+ // evaluateSharedTreeGuard). Scoped to entries carrying a status code —
637
+ // plain path strings (no code) fall back to the old always-reverted path.
638
+ const untrackedBaselinePaths = (dirtyBaseline || [])
639
+ .filter((e) => e && typeof e === 'object' && e.code === '??')
640
+ .map((e) => e.path);
641
+ // A large untracked baseline (the 2026-09-12 incident's shared tree had
642
+ // ~240 such paths) makes this a lot of stat calls — fs.promises.access
643
+ // run concurrently instead of fs.existsSync run synchronously one at a
644
+ // time keeps this off the event loop instead of blocking every other
645
+ // in-flight scheduler/IPC task for the duration.
646
+ const existsChecks = await Promise.all(
647
+ untrackedBaselinePaths.map((p) => fsp.access(path.join(cwd, p)).then(() => true, () => false)),
648
+ );
649
+ const existsAfter = untrackedBaselinePaths.filter((_, i) => existsChecks[i]);
650
+ const { reverted, nowIgnored } = module.exports.evaluateSharedTreeGuard({
493
651
  stashBefore: stashBaseline,
494
652
  stashAfter,
495
653
  dirtyBefore: dirtyBaseline,
496
654
  dirtyAfter,
497
655
  pathsCommittedDuringRun,
656
+ existsAfter,
498
657
  });
499
658
  if (reverted.length) {
500
659
  result.reverted = reverted;
501
660
  console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
502
661
  }
503
- return (result.restoredStash || result.restoreFailed || result.ambiguousStashes || result.reverted) ? result : null;
662
+ if (nowIgnored.length) {
663
+ result.nowIgnored = nowIgnored;
664
+ console.log(`[scheduler] ${slug}: shared-tree guard: ${nowIgnored.length} path(s) no longer shown by git status but still present on disk — likely a new ignore rule, not a revert (${nowIgnored.slice(0, 3).join(', ')})`);
665
+ }
666
+ return (result.restoredStash || result.restoreFailed || result.ambiguousStashes || result.reverted || result.nowIgnored) ? result : null;
504
667
  } catch (e) {
505
668
  console.error(`[scheduler] ${slug}: shared-tree guard error`, e);
506
669
  return null;
@@ -1152,8 +1315,9 @@ function ensureDirs() {
1152
1315
  * reconcile()-level call is what makes "anything written to the retired flat
1153
1316
  * prds/ dir is swept into prds-archived/ without being executed" actually
1154
1317
  * true regardless of which of reconcile's several callers (tickQueue's poll,
1155
- * job completion, the schedule:state/schedule:rescan IPC handlers,
1156
- * rescheduleTimer) triggers the pass: a PRD dropped in the flat dir has no
1318
+ * job completion, the schedule:rescan/schedule:adopt-prd IPC handlers,
1319
+ * broadcast()'s coalescer, rescheduleTimer) triggers the pass — schedule:state
1320
+ * no longer reconciles on read. A PRD dropped in the flat dir has no
1157
1321
  * queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
1158
1322
  * sweep archives it before reconcile can ever turn it into a pending job.
1159
1323
  */
@@ -1217,7 +1381,22 @@ async function runPrdMigration() {
1217
1381
  // on every pass, but stays here so a fresh boot's very first log line
1218
1382
  // still reports the initial sweep — see consolidateAllFlatPrds's own
1219
1383
  // comment for why reconcile() is the load-bearing call site.)
1220
- await consolidateAllFlatPrds(allProjectCwds());
1384
+ //
1385
+ // Deferred off scheduler.init()'s synchronous critical path: allProjectCwds()
1386
+ // is a synchronous ~270ms directory scan, and awaiting it inline here
1387
+ // competed with the renderer's first IPC round trips (schedule.state,
1388
+ // billing.fetch, teams.list) for the event loop during boot. Dropping the
1389
+ // await doesn't weaken the consolidation guarantee — consolidateAllFlatPrds
1390
+ // also runs at the top of every reconcile() (see its own comment above),
1391
+ // and a flat PRD can only ever execute via tickQueue, which always
1392
+ // reconciles first, so nothing dropped in the flat dir can run before a
1393
+ // reconcile() pass sweeps it regardless of whether this boot-time pass has
1394
+ // finished yet.
1395
+ setImmediate(() => {
1396
+ consolidateAllFlatPrds(allProjectCwds()).catch((e) => {
1397
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'deferred flat-PRD consolidation failed', meta: { error: e?.message } });
1398
+ });
1399
+ });
1221
1400
 
1222
1401
  // Rollout migration for the PRD-authoring-lockdown feature: stamp every
1223
1402
  // pre-existing PRD as legacy-adopted BEFORE reconcile() ever runs its
@@ -1655,6 +1834,99 @@ function findOverrunningJobs(jobs, now, { factor, floorMs } = {}) {
1655
1834
  return out;
1656
1835
  }
1657
1836
 
1837
+ /**
1838
+ * computeJobBudgetMs(estimateMinutes, { factor, floorMs, ceilingMs }) → number
1839
+ *
1840
+ * Pure. `budgetMs = clamp(estimateMinutes * factor, floorMs, ceilingMs)` — see
1841
+ * JOB_BUDGET_FACTOR's header comment (schedulerConfig.cjs) for the measured
1842
+ * p50/p90/max this is calibrated against. A missing/zero/non-finite estimate
1843
+ * is treated as 0, which the floor clamp then dominates — "jobs with a
1844
+ * missing estimate get the floor" falls straight out of the clamp, no
1845
+ * special-casing needed.
1846
+ */
1847
+ function computeJobBudgetMs(estimateMinutes, { factor, floorMs, ceilingMs } = {}) {
1848
+ const f = typeof factor === 'number' && factor > 0 ? factor : JOB_BUDGET_FACTOR;
1849
+ const floor = typeof floorMs === 'number' && floorMs >= 0 ? floorMs : JOB_BUDGET_FLOOR_MS;
1850
+ const ceiling = typeof ceilingMs === 'number' && ceilingMs > 0 ? ceilingMs : JOB_BUDGET_CEILING_MS;
1851
+ const est = Number(estimateMinutes);
1852
+ const minutes = Number.isFinite(est) && est > 0 ? est : 0;
1853
+ return Math.min(Math.max(minutes * f * 60_000, floor), ceiling);
1854
+ }
1855
+
1856
+ /**
1857
+ * classifyBudgetKill(res, landedCommitEvidence) → { status, reason, landedCommit } | null
1858
+ *
1859
+ * Pure. `res` is executeJob's resolved outcome — only fires when
1860
+ * `res.killedByWatchdog === 'budget'` (stamped by the budget watchdog inside
1861
+ * executeJob, never inferred from exit code/duration alone, so it can never
1862
+ * collide with an ordinary idle-tail/deadman/external kill). ALWAYS routes to
1863
+ * needs_review — never 'failed' (spawnJob's ordinary non-zero-exit default)
1864
+ * and never silently 'completed' (executeJob's onExit excludes
1865
+ * killedByWatchdog === 'budget' from the result=success → exit 0 mapping
1866
+ * idle-tail/deadman get) — so a budget kill is a visible, actionable park,
1867
+ * never a retry (classifyFailureOutcome/selectAutoFixTargets only ever see
1868
+ * 'failed'/ordinary needs_review rows, not this one — see
1869
+ * selectAutoFixTargets' own budget_exceeded exclusion) and never a discard:
1870
+ * `landedCommitEvidence`, when the caller resolved one via the SAME
1871
+ * commit-guard evidence check the plain sigterm/exit paths already use, is
1872
+ * threaded onto the row as `landedCommit` so a job that HAD already
1873
+ * committed before overrunning is still adjudicated on its git evidence.
1874
+ */
1875
+ function classifyBudgetKill(res, landedCommitEvidence) {
1876
+ if (!res || res.killedByWatchdog !== 'budget') return null;
1877
+ return {
1878
+ status: 'needs_review',
1879
+ reason: res.budgetKillReason || `wall-clock budget exceeded (exit ${res.exitCode})`,
1880
+ landedCommit: landedCommitEvidence || null,
1881
+ };
1882
+ }
1883
+
1884
+ /**
1885
+ * isJobBudgetExempt(job) → boolean
1886
+ *
1887
+ * Pure. `quietMachine: true` PRDs (their whole point is running alone,
1888
+ * un-contended, for a timing-sensitive measurement) and any PRD with an
1889
+ * explicit `budgetExempt: true` opt-out have no wall-clock kill ceiling.
1890
+ */
1891
+ function isJobBudgetExempt(job) {
1892
+ return job?.quietMachine === true || job?.budgetExempt === true;
1893
+ }
1894
+
1895
+ /** Pure predicate the budget watchdog's shouldFire calls — single source of
1896
+ * truth for "has this job run past its own budget" so it's unit-testable
1897
+ * without spinning up real timers. */
1898
+ function shouldKillForBudget(elapsedMs, budgetMs) {
1899
+ return elapsedMs >= budgetMs;
1900
+ }
1901
+
1902
+ /**
1903
+ * resolveBudgetKillOutcome({ killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes })
1904
+ * → { killedByWatchdog: 'budget'|null, budgetKillReason: string|null }
1905
+ *
1906
+ * Pure. `ctx.killedByWatchdog` is stamped by the budget watchdog's action()
1907
+ * the instant its periodic shouldFire() observes elapsedMs >= jobBudgetMs —
1908
+ * but that setInterval tick and the child's real 'exit' event both run on
1909
+ * the SAME single-threaded event loop, so it's possible to observe the
1910
+ * budget threshold crossed and call ctx.killTree() in the same window the
1911
+ * agent happens to exit cleanly (exit 0) or fails on its own for an
1912
+ * unrelated reason — ctx.killTree() against an already-exited pid is a
1913
+ * silent no-op (ESRCH, caught), but the flag would still read 'budget'
1914
+ * unless gated here. Only trusted when the exit SHAPE actually looks like a
1915
+ * signal kill (killedBySignal — mirrors the exact same check onExit already
1916
+ * uses for its own mappedToSuccess exclusion), so a clean exit=0 or an
1917
+ * ordinary unrelated non-zero failure racing the watchdog's tick is never
1918
+ * misclassified as a budget kill downstream.
1919
+ */
1920
+ function resolveBudgetKillOutcome({ killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes }) {
1921
+ if (killedByWatchdog !== 'budget' || !killedBySignal) {
1922
+ return { killedByWatchdog: killedByWatchdog === 'budget' ? null : (killedByWatchdog ?? null), budgetKillReason: null };
1923
+ }
1924
+ return {
1925
+ killedByWatchdog: 'budget',
1926
+ budgetKillReason: `wall-clock budget exceeded: ran ${Math.round(durationMs / 60_000)}m against a ${Math.round(jobBudgetMs / 60_000)}m budget (estimateMinutes=${estimateMinutes ?? 0})`,
1927
+ };
1928
+ }
1929
+
1658
1930
  /**
1659
1931
  * findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
1660
1932
  * → [{ slug, cwd, ageMs, restoreStatus }]
@@ -1874,7 +2146,7 @@ async function listPrdFiles() {
1874
2146
  ensureDirs();
1875
2147
  const dirs = candidatePrdsDirs();
1876
2148
  const perDir = await Promise.all(dirs.map((dir) => prdParser.listPrdFiles(dir)));
1877
- return perDir.flat().sort();
2149
+ return { files: perDir.flat().sort(), dirCount: dirs.length };
1878
2150
  }
1879
2151
 
1880
2152
  /**
@@ -2018,16 +2290,31 @@ async function reconcile(state) {
2018
2290
  if (state && state.unreadable) {
2019
2291
  throw new Error(`reconcile skipped: queue.json unreadable (${state.unreadable})`);
2020
2292
  }
2293
+ // Per-phase timing (PRD: reconcile evidence trail) — plain Date.now() diffs,
2294
+ // matching the ad-hoc elapsedMs idiom already used in health.cjs/
2295
+ // definitionOfDone.cjs. Only logged when the total exceeds
2296
+ // RECONCILE_SLOW_PASS_MS (see the warn emission at the bottom of this
2297
+ // function); a normal-speed pass logs nothing.
2298
+ const reconcileStartMs = Date.now();
2299
+ const phaseMs = {};
2300
+
2021
2301
  // Sweep the retired flat prds/ dir BEFORE scanning it below. reconcile()
2022
- // has several callers besides tickQueue's ~60s poll (broadcast,
2023
- // rescheduleTimer, the schedule:state IPC handler, schedule:rescan) — this
2302
+ // has several callers besides tickQueue's ~60s poll (broadcast()'s
2303
+ // coalescer, rescheduleTimer, schedule:rescan, schedule:adopt-prd) — this
2024
2304
  // lives here, not in any one caller, so the "a hand-written PRD in the flat
2025
2305
  // dir is swept before it can become a job" guarantee holds regardless of
2026
2306
  // which caller triggers this reconcile pass. A freshly hand-written file
2027
2307
  // has no queue row yet, so it is never "live" and gets archived here
2028
2308
  // instead of ever reaching the onDisk scan below.
2309
+ let phaseStartMs = Date.now();
2029
2310
  await consolidateAllFlatPrds(allProjectCwds());
2030
- const files = await listPrdFiles();
2311
+ phaseMs.flatPrdSweep = Date.now() - phaseStartMs;
2312
+
2313
+ phaseStartMs = Date.now();
2314
+ const { files, dirCount } = await listPrdFiles();
2315
+ phaseMs.prdDirResolve = Date.now() - phaseStartMs;
2316
+
2317
+ phaseStartMs = Date.now();
2031
2318
  const onDisk = new Map();
2032
2319
  for (const f of files) {
2033
2320
  try {
@@ -2039,6 +2326,7 @@ async function reconcile(state) {
2039
2326
  console.warn('[scheduler] failed to parse', f, e?.message);
2040
2327
  }
2041
2328
  }
2329
+ phaseMs.parseLoop = Date.now() - phaseStartMs;
2042
2330
 
2043
2331
  const next = [];
2044
2332
  const seen = new Set();
@@ -2104,7 +2392,9 @@ async function reconcile(state) {
2104
2392
  // membership, so moving the file between Epic dirs must re-point the row.
2105
2393
  epicId: p.epicId ?? job.epicId ?? null,
2106
2394
  dependsOn: p.dependsOn,
2395
+ disposition: p.disposition ?? null,
2107
2396
  quietMachine: p.quietMachine === true,
2397
+ budgetExempt: p.budgetExempt === true,
2108
2398
  originSessionId: job.originSessionId
2109
2399
  ?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
2110
2400
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
@@ -2159,9 +2449,11 @@ async function reconcile(state) {
2159
2449
  // ScheduleJobSchema (e.g. the 1021/1022 incident's `"status": "queued"`) —
2160
2450
  // see the repair pass below, right after historyBySlug is available.
2161
2451
  const invalidJobs = Array.isArray(state.invalidJobs) ? state.invalidJobs : [];
2452
+ phaseStartMs = Date.now();
2162
2453
  const historyBySlug = (unmatchedSlugs.length > 0 || terminalDroppedNeedingHistoryCheck.length > 0 || invalidJobs.length > 0)
2163
2454
  ? await queueHistory.historyTerminalBySlug()
2164
2455
  : new Map();
2456
+ phaseMs.historyLookup = Date.now() - phaseStartMs;
2165
2457
 
2166
2458
  // Backfill: any terminal job dropped above whose slug isn't already in
2167
2459
  // history.jsonl gets written now, before its row is gone for good. This is
@@ -2217,7 +2509,9 @@ async function reconcile(state) {
2217
2509
  sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
2218
2510
  epicId: p.epicId ?? inv.row?.epicId ?? null,
2219
2511
  dependsOn: p.dependsOn,
2512
+ disposition: p.disposition ?? null,
2220
2513
  quietMachine: p.quietMachine === true,
2514
+ budgetExempt: p.budgetExempt === true,
2221
2515
  originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
2222
2516
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
2223
2517
  agentType: p.agentType ?? inv.row?.agentType ?? null,
@@ -2339,7 +2633,9 @@ async function reconcile(state) {
2339
2633
  sourceTabId: p.sourceTabId,
2340
2634
  epicId: p.epicId ?? null,
2341
2635
  dependsOn: p.dependsOn,
2636
+ disposition: p.disposition ?? null,
2342
2637
  quietMachine: p.quietMachine === true,
2638
+ budgetExempt: p.budgetExempt === true,
2343
2639
  originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
2344
2640
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
2345
2641
  agentType: p.agentType ?? null,
@@ -2436,7 +2732,8 @@ async function reconcile(state) {
2436
2732
  // small. Append BEFORE dropping so a crash between the two can't lose a
2437
2733
  // record — appendHistory dedupes by slug+runId, so a replay of the same
2438
2734
  // batch on next boot is a safe no-op.
2439
- const nowMs = Date.now();
2735
+ phaseStartMs = Date.now();
2736
+ const nowMs = phaseStartMs;
2440
2737
  const { hot, toArchive } = queueHistory.partitionJobs(sorted, nowMs);
2441
2738
  if (toArchive.length > 0) {
2442
2739
  await queueHistory.appendHistory(toArchive);
@@ -2472,6 +2769,17 @@ async function reconcile(state) {
2472
2769
  } catch (e) {
2473
2770
  console.warn('[scheduler] autoArchiveCompleted failed', e?.message);
2474
2771
  }
2772
+ phaseMs.queueWrite = Date.now() - phaseStartMs;
2773
+
2774
+ const totalMs = Date.now() - reconcileStartMs;
2775
+ if (totalMs > RECONCILE_SLOW_PASS_MS) {
2776
+ logs.writeLine({
2777
+ level: 'warn',
2778
+ scope: 'scheduler',
2779
+ message: `reconcile() pass took ${totalMs}ms (threshold ${RECONCILE_SLOW_PASS_MS}ms)`,
2780
+ meta: { totalMs, phaseMs, prdFileCount: files.length, resolvedDirCount: dirCount },
2781
+ });
2782
+ }
2475
2783
 
2476
2784
  return state;
2477
2785
  }
@@ -2678,16 +2986,19 @@ function attachWindow(w) { mainWindow = w; }
2678
2986
 
2679
2987
  /**
2680
2988
  * Build the snapshot payload consumed by both the `schedule:state` IPC
2681
- * handler and the `schedule:state` broadcast event. The IPC return adds a
2682
- * `paths` map (renderer uses it for "open folder" actions); broadcast omits
2683
- * it because subscribers don't need to re-derive paths on every tick.
2989
+ * handler and the `schedule:state` broadcast event.
2684
2990
  */
2685
- function buildScheduleStatePayload(state, { withPaths = false } = {}) {
2991
+ function buildScheduleStatePayload(state) {
2686
2992
  const payload = {
2687
2993
  config: state.config,
2688
2994
  jobs: state.jobs,
2689
2995
  scheduledFor: state.scheduledFor,
2690
2996
  lastRunAt: state.lastRunAt,
2997
+ // Distinct from lastRunAt (only stamped when a batch actually launches):
2998
+ // stamped every time tickQueue reaches the picker at all. See
2999
+ // classifyQueueHealth/classifyQueueStarvation's header comments for why
3000
+ // the two must never merge.
3001
+ lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
2691
3002
  nextReset: getNextResetCached(),
2692
3003
  paused: state.paused,
2693
3004
  // Launch circuit breaker (issue #11): which personas cannot launch right
@@ -2718,9 +3029,6 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
2718
3029
  };
2719
3030
  })(),
2720
3031
  };
2721
- if (withPaths) {
2722
- payload.paths = { root: ROOT, prds: PRDS_DIR, runs: RUNS_DIR, queue: queueStore.MACHINE_STATE_PATH };
2723
- }
2724
3032
  return payload;
2725
3033
  }
2726
3034
 
@@ -2731,20 +3039,33 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
2731
3039
  // per mutation. Callers where latency matters (pause/resume, job
2732
3040
  // start/finish/reap/reset) pass `{ flush: true }` to bypass the window and
2733
3041
  // send immediately.
3042
+ // getPayload is the coalescer's ONLY entry point back into queue state, so
3043
+ // routing the reconcile+write pair through it (rather than broadcast() doing
3044
+ // its own bare readQueue/reconcile/writeQueue) is what makes a burst of
3045
+ // broadcast() calls cost exactly one reconcile + one write per coalesce
3046
+ // window. It goes through mutate() (not a bare read/write pair) so it
3047
+ // serializes against every other concurrent mutation and inherits mutate's
3048
+ // pre-fn `state.unreadable` bail — never write a state derived from a failed
3049
+ // read. `module.exports.reconcile` (not the bare local binding) is the seam
3050
+ // tests spy on, matching this file's existing testable-seam convention (see
3051
+ // module.exports.stashList/evaluateSharedTreeGuard/committedInWindow above).
2734
3052
  const broadcastCoalescer = createBroadcastCoalescer({
2735
3053
  delayMs: BROADCAST_COALESCE_MS,
2736
3054
  send: (payload) => {
2737
3055
  if (!mainWindow || mainWindow.isDestroyed()) return;
2738
3056
  sendIfAlive(mainWindow, 'schedule:state', payload);
2739
3057
  },
2740
- getPayload: async () => buildScheduleStatePayload(await readQueue()),
3058
+ getPayload: () => mutate(async (state) => {
3059
+ await module.exports.reconcile(state);
3060
+ return buildScheduleStatePayload(state);
3061
+ }),
2741
3062
  });
2742
3063
 
3064
+ // Reconcile is unconditional — even with no window attached (a
3065
+ // scheduler-passive or headless instance), discovery must still run so
3066
+ // on-disk PRDs get onboarded. Only the actual IPC push is window-gated,
3067
+ // inside the coalescer's own `send`.
2743
3068
  async function broadcast(opts = {}) {
2744
- if (!mainWindow || mainWindow.isDestroyed()) return;
2745
- const state = await readQueue();
2746
- await reconcile(state);
2747
- await writeQueue(state);
2748
3069
  if (opts.flush) {
2749
3070
  await broadcastCoalescer.flush();
2750
3071
  } else {
@@ -3044,7 +3365,7 @@ const BOOT_ORPHAN_KILL_GRACE_MS = 6000;
3044
3365
  * process may still be writing to it, so reading now risks misclassifying a
3045
3366
  * job that is about to emit result:success as no_result and double-running it.
3046
3367
  * Ported from reconcileQueueOffline's cross-tick escalation (see
3047
- * scripts/lib/watchdogHelpers.cjs) — here it's a single deferred window since
3368
+ * src/main/lib/watchdogHelpers.cjs) — here it's a single deferred window since
3048
3369
  * this process stays up to revisit it, rather than a separate short-lived
3049
3370
  * watchdog process needing another tick.
3050
3371
  */
@@ -3064,17 +3385,26 @@ function partitionBootOrphans(jobs, isAlive = claudePidAlive) {
3064
3385
  }
3065
3386
 
3066
3387
  /**
3067
- * applyOrphanOutcome(job, outcome, killNote?) → void
3388
+ * applyOrphanOutcome(job, outcome, killNote?, confirmedLandedCommit?) → void
3068
3389
  *
3069
3390
  * Mutates `job` in place to finalize a boot-orphaned 'running' job given its
3070
3391
  * classified run outcome: success/failed finalize terminally; no_result/unknown
3071
3392
  * re-queues to pending bounded by ORPHAN_REQUEUE_CAP. The status-mutation
3072
3393
  * semantics (and the cap-exhaustion boundary) match the now-deleted
3073
- * reconcileQueueOffline (scripts/lib/watchdogHelpers.cjs) verbatim; killNote
3394
+ * reconcileQueueOffline (src/main/lib/watchdogHelpers.cjs) verbatim; killNote
3074
3395
  * plumbing differs slightly (see call sites) since this path always knows
3075
3396
  * pid liveness up front rather than re-checking per tick.
3397
+ *
3398
+ * `confirmedLandedCommit` is the same evidence-before-failure gate
3399
+ * reapDeadRunningJobs applies (see resolveLandedCommitEvidence): a job that
3400
+ * dies while the app itself is offline is classified 'failed' from its log
3401
+ * tail alone, exactly like the pre-fix reap path was — so without this, an
3402
+ * orphaned job that actually landed a real commit is reachable via boot
3403
+ * reconciliation even though the live reap path is now guarded. Callers
3404
+ * must resolve this (a git spawn) BEFORE calling mutate(), never inside it —
3405
+ * pass null to skip the gate (e.g. when the outcome isn't 'failed').
3076
3406
  */
3077
- function applyOrphanOutcome(job, outcome, killNote = '') {
3407
+ function applyOrphanOutcome(job, outcome, killNote = '', confirmedLandedCommit = null) {
3078
3408
  const now = new Date().toISOString();
3079
3409
  if (outcome === 'success') {
3080
3410
  transitionJob(job, 'completed', { reason: 'boot orphan reconciliation: run succeeded', source: 'applyOrphanOutcome' });
@@ -3083,9 +3413,16 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
3083
3413
  job.finishedAt = now;
3084
3414
  delete job.runtime;
3085
3415
  } else if (outcome === 'failed') {
3086
- transitionJob(job, 'failed', { reason: `orphaned: app restarted while running${killNote}`, source: 'applyOrphanOutcome' });
3087
- job.exitCode = job.exitCode ?? 1;
3088
- job.error = `orphaned: app restarted while running${killNote}`;
3416
+ if (confirmedLandedCommit) {
3417
+ transitionJob(job, 'completed', { reason: `orphaned: app restarted while running${killNote}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence`, source: 'applyOrphanOutcome:landed' });
3418
+ job.exitCode = 0;
3419
+ job.error = null;
3420
+ job.landedCommit = confirmedLandedCommit;
3421
+ } else {
3422
+ transitionJob(job, 'failed', { reason: `orphaned: app restarted while running${killNote}`, source: 'applyOrphanOutcome' });
3423
+ job.exitCode = job.exitCode ?? 1;
3424
+ job.error = `orphaned: app restarted while running${killNote}`;
3425
+ }
3089
3426
  job.finishedAt = now;
3090
3427
  delete job.runtime;
3091
3428
  } else {
@@ -3093,6 +3430,13 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
3093
3430
  if (tries < ORPHAN_REQUEUE_CAP) {
3094
3431
  resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}`, { source: 'applyOrphanOutcome' });
3095
3432
  job.orphanRetries = tries + 1;
3433
+ } else if (confirmedLandedCommit) {
3434
+ transitionJob(job, 'completed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence`, source: 'applyOrphanOutcome:landed' });
3435
+ job.exitCode = 0;
3436
+ job.error = null;
3437
+ job.landedCommit = confirmedLandedCommit;
3438
+ job.finishedAt = now;
3439
+ delete job.runtime;
3096
3440
  } else {
3097
3441
  transitionJob(job, 'failed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`, source: 'applyOrphanOutcome' });
3098
3442
  job.exitCode = job.exitCode ?? 1;
@@ -3980,6 +4324,60 @@ async function pathExistsInTree(cwd, treeish, p) {
3980
4324
  }
3981
4325
  }
3982
4326
 
4327
+ /**
4328
+ * resolveLandedCommitEvidence(cwd, sha, sinceIso) → Promise<boolean>
4329
+ *
4330
+ * Bounded, non-fatal proof that `sha` is a real, resolvable commit in the
4331
+ * repo at `cwd`, committed no earlier than `sinceIso` — `git cat-file -e
4332
+ * <sha>^{commit}` plus a `git log -1 --format=%cI` timestamp check, both via
4333
+ * execGitAt's existing spawn+timeout bound (never shell:true, never an
4334
+ * unbounded execSync). This is the evidence gate reapDeadRunningJobs (PRD:
4335
+ * reaper must consult completion evidence) adds ahead of stamping a reaped
4336
+ * row 'failed': a landedCommit field being non-empty is not proof by itself
4337
+ * (job 1192 had one and still got reaped 'failed') — only a git-verified
4338
+ * resolution is.
4339
+ *
4340
+ * The timestamp bound matters because `landedCommit` deliberately survives
4341
+ * resetJobFields (see the comment there) so a re-fired run can consult it as
4342
+ * priorLandedCommit — which means a STALE landedCommit from an earlier
4343
+ * dispatch of the same slug can still be sitting on the row when a LATER
4344
+ * dispatch dies for real. Without `sinceIso`, that stale-but-real sha would
4345
+ * satisfy `cat-file -e` and wrongly promote a genuine failure to
4346
+ * 'completed'. Passing the current dispatch's `row.startedAt` as `sinceIso`
4347
+ * closes that: only a commit landed during THIS run counts as evidence.
4348
+ *
4349
+ * `cwd` is normalized through opsOwnership's resolveProjectRoot first (the
4350
+ * same "never trust a raw agent cwd" reasoning delegationReadiness.cjs
4351
+ * already relies on) so a row reaped while its cwd is an ephemeral worktree
4352
+ * checkout resolves the commit against the real project root instead.
4353
+ *
4354
+ * Never throws: a missing sha, a resolveProjectRoot failure (ephemeral cwd,
4355
+ * thrown error), a spawn failure, a timeout, or a cwd that no longer exists
4356
+ * on disk all resolve to `false` — the caller's safe default is 'failed',
4357
+ * exactly like today, whenever this can't positively prove landing.
4358
+ */
4359
+ async function resolveLandedCommitEvidence(cwd, sha, sinceIso) {
4360
+ if (!sha || typeof sha !== 'string') return false;
4361
+ try {
4362
+ const root = resolveProjectRoot(cwd);
4363
+ await execGitAt(root, ['cat-file', '-e', `${sha}^{commit}`], { timeout: 10_000 });
4364
+ if (sinceIso) {
4365
+ const since = new Date(sinceIso).getTime();
4366
+ if (Number.isFinite(since)) {
4367
+ const committedIso = (await execGitAt(root, ['log', '-1', '--format=%cI', sha], { timeout: 10_000 })).trim();
4368
+ const committedAt = new Date(committedIso).getTime();
4369
+ // A commit dated before this dispatch even started can only be a
4370
+ // stale sha surviving from an earlier life of the row — never
4371
+ // evidence that THIS dispatch landed anything.
4372
+ if (Number.isFinite(committedAt) && committedAt < since) return false;
4373
+ }
4374
+ }
4375
+ return true;
4376
+ } catch {
4377
+ return false;
4378
+ }
4379
+ }
4380
+
3983
4381
  /**
3984
4382
  * Commit exactly `paths` (must already be dirty on disk) onto a dedicated
3985
4383
  * `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
@@ -4522,6 +4920,56 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4522
4920
  },
4523
4921
  };
4524
4922
 
4923
+ // Wall-clock budget watchdog: unlike idleTailWatchdog above (which only
4924
+ // fires when the log mtime STALLS), this fires on total elapsed wall-
4925
+ // clock time regardless of whether the job keeps writing output — the
4926
+ // gap a chatty-but-runaway executor slips through (see
4927
+ // computeJobBudgetMs's header for the measured p50/p90/max this budget
4928
+ // is calibrated against). `quietMachine` jobs and any PRD with an
4929
+ // explicit `budgetExempt: true` opt out entirely — logged once here so
4930
+ // an unbounded job is never silently unbounded.
4931
+ const jobBudgetMs = computeJobBudgetMs(job.estimateMinutes);
4932
+ const budgetExempt = isJobBudgetExempt(job);
4933
+ if (budgetExempt) {
4934
+ safeLog(`[scheduler] wall-clock budget watchdog EXEMPT for ${job.slug} ` +
4935
+ `(${job.quietMachine === true ? 'quietMachine' : 'budgetExempt'}) — no wall-clock kill ceiling this run\n`);
4936
+ }
4937
+ let budgetWarningStamped = false;
4938
+ const budgetWatchdog = {
4939
+ label: 'budget',
4940
+ intervalMs: IDLE_CHECK_INTERVAL_MS,
4941
+ shouldFire(ctx) {
4942
+ if (budgetExempt) return false;
4943
+ const elapsedMs = Date.now() - ctx.startedAt;
4944
+ if (!budgetWarningStamped && elapsedMs >= jobBudgetMs * BUDGET_WARNING_FRACTION) {
4945
+ budgetWarningStamped = true;
4946
+ // Fire-and-forget (side effect inside a sync predicate, same pattern
4947
+ // resultTailWatchdog's shouldFire already uses for agentResultSubtype)
4948
+ // — exposes the warning on the row well before the kill fires, so
4949
+ // the renderer can show it without waiting for the next tick.
4950
+ mutate((state) => {
4951
+ const j = state.jobs.find((x) => x.slug === job.slug);
4952
+ if (!j) return;
4953
+ j.budgetWarning = { budgetMs: jobBudgetMs, elapsedMs, at: new Date().toISOString() };
4954
+ }).catch((e) => console.warn('[scheduler] budget-warning stamp failed', job.slug, e?.message));
4955
+ }
4956
+ return shouldKillForBudget(elapsedMs, jobBudgetMs);
4957
+ },
4958
+ action(ctx) {
4959
+ const elapsedMs = Date.now() - ctx.startedAt;
4960
+ ctx.safeLog(`\n[scheduler] wall-clock budget watchdog: ran ${Math.round(elapsedMs / 60_000)}m ` +
4961
+ `(> ${Math.round(jobBudgetMs / 60_000)}m budget, estimateMinutes=${job.estimateMinutes ?? 0}) — SIGTERM process group\n`);
4962
+ ctx.killedByWatchdog = 'budget';
4963
+ ctx.killTree('SIGTERM');
4964
+ const budgetKillTimer = setTimeout(() => {
4965
+ ctx.safeLog(`\n[scheduler] budget watchdog: still alive ${Math.round(POST_RESULT_KILL_MS/1000)}s after SIGTERM — SIGKILL\n`);
4966
+ ctx.killTree('SIGKILL');
4967
+ }, POST_RESULT_KILL_MS);
4968
+ if (budgetKillTimer.unref) budgetKillTimer.unref();
4969
+ ctx.addTimer(budgetKillTimer);
4970
+ },
4971
+ };
4972
+
4525
4973
  // ---------- spawn ----------
4526
4974
 
4527
4975
  const { child } = withChildAndLog({
@@ -4552,8 +5000,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4552
5000
  detached: true,
4553
5001
  },
4554
5002
  },
4555
- watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog],
4556
- onExit({ exitCode, signal, killedByWatchdog: _kbw, error, spawnFailed, leakedDescendants, safeLog: sl }) {
5003
+ watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog, budgetWatchdog],
5004
+ onExit({ exitCode, signal, killedByWatchdog, error, spawnFailed, leakedDescendants, safeLog: sl }) {
4557
5005
  const durationMs = Date.now() - startedAt;
4558
5006
  const leaked = leakedDescendants ?? [];
4559
5007
  if (leaked.length > 0) {
@@ -4583,7 +5031,11 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4583
5031
  // and 137 (128+SIGKILL) in case the process exited via signal-as-code.
4584
5032
  let effectiveCode = exitCode;
4585
5033
  const killedBySignal = signal === 'SIGTERM' || signal === 'SIGKILL' || exitCode === 143 || exitCode === 137 || exitCode === null;
4586
- const mappedToSuccess = agentResultSubtype === 'success' && killedBySignal;
5034
+ // A budget kill must NEVER be laundered into a clean exit=0, even when
5035
+ // the agent had already emitted result=success before it fired — the
5036
+ // AC requires it always park needs_review, never silently 'completed'.
5037
+ // idle-tail/deadman/result-tail kills keep the existing success-mapping.
5038
+ const mappedToSuccess = agentResultSubtype === 'success' && killedBySignal && killedByWatchdog !== 'budget';
4587
5039
  if (mappedToSuccess) {
4588
5040
  effectiveCode = 0;
4589
5041
  sl(`\n[scheduler] mapping exit code=${exitCode} signal=${signal} → 0 ` +
@@ -4606,6 +5058,16 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4606
5058
  sl(`\n[scheduler] LAUNCH FAILURE (${launchFailed.kind}${launchFailed.httpStatus ? ` HTTP ${launchFailed.httpStatus}` : ''}): ` +
4607
5059
  `${launchFailed.message} — no turn was taken; this is not a PRD failure\n`);
4608
5060
  }
5061
+ // Formatted once, here, off the FINAL durationMs (more accurate than
5062
+ // the watchdog action's own snapshot at kill time) — matches the
5063
+ // reason string format the AC requires verbatim. See
5064
+ // resolveBudgetKillOutcome's own header for why this is gated on
5065
+ // killedBySignal, not on killedByWatchdog alone.
5066
+ const budgetKillOutcome = resolveBudgetKillOutcome({
5067
+ killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes: job.estimateMinutes,
5068
+ });
5069
+ const { budgetKillReason } = budgetKillOutcome;
5070
+ const effectiveKilledByWatchdog = budgetKillOutcome.killedByWatchdog;
4609
5071
  // Sync write: child 'exit' handler must flush meta before resolve()
4610
5072
  // so the spawnJob mutate() that follows sees the persisted exit code.
4611
5073
  config.writeJsonSync(metaPath, {
@@ -4616,10 +5078,14 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4616
5078
  launchEnvApplied: launchEnv && Object.keys(launchEnv).length ? Object.keys(launchEnv) : [],
4617
5079
  startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
4618
5080
  agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
5081
+ killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
4619
5082
  schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
4620
5083
  originSessionId, contextDigestApplied,
4621
5084
  });
4622
- resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, launchFailure: launchFailed, resultStats, leakedDescendants: leaked, sessionId });
5085
+ resolve({
5086
+ exitCode: effectiveCode, durationMs, rateLimited, networkError, launchFailure: launchFailed, resultStats,
5087
+ leakedDescendants: leaked, sessionId, killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
5088
+ });
4623
5089
  },
4624
5090
  });
4625
5091
 
@@ -4627,8 +5093,27 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4627
5093
  safeLog(`[scheduler] spawned pid=${child.pid} sessionId=${sessionId} (process group)\n\n`);
4628
5094
  // Make this job the OOM killer's preferred victim over Electron.
4629
5095
  biasJobOomScore(child.pid);
4630
- // Fire-and-forget pid persistence — best effort.
4631
- if (onPid) onPid(child.pid, sessionId, cwd).catch(() => {});
5096
+ // Persist runtime.pid with one retry — still fire-and-forget (must
5097
+ // never block the spawn), but a final failure is now loud instead of
5098
+ // silently swallowed. A silent failure here is exactly what let the
5099
+ // pidless-grace reaper terminalize a live, working job (runtime.pid
5100
+ // never landed, so selectReapableJobs had no way to tell "never
5101
+ // spawned" from "spawned but unrecorded").
5102
+ if (onPid) {
5103
+ (async () => {
5104
+ try {
5105
+ await onPid(child.pid, sessionId, cwd);
5106
+ } catch (firstErr) {
5107
+ try {
5108
+ await onPid(child.pid, sessionId, cwd);
5109
+ } catch (finalErr) {
5110
+ const message = finalErr?.message ?? String(finalErr);
5111
+ console.error(`[scheduler] FAILED to persist runtime.pid for ${job.slug} pid=${child.pid}: ${message}`);
5112
+ appendAuditEvent('job_pid_persist_failed', { slug: job.slug, cwd, pid: child.pid, error: message });
5113
+ }
5114
+ }
5115
+ })();
5116
+ }
4632
5117
  }
4633
5118
  });
4634
5119
  }
@@ -5550,8 +6035,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5550
6035
 
5551
6036
  // Commit-guard baseline: snapshot the working tree BEFORE the run so the
5552
6037
  // post-run check flags only paths THIS job left dirty, not pre-existing WIP.
6038
+ // Captured once, with status codes, so the shared-tree guard below can
6039
+ // tell "was untracked" apart from "was tracked-and-modified" (2026-09-12
6040
+ // incident) without a second `git status` call; every other consumer of
6041
+ // `guardBaseline` still gets the plain path-string array it always did.
5553
6042
  const guardCwd = job.cwd || defaultCwd;
5554
- const guardBaseline = await uncommittedChanges(guardCwd);
6043
+ const guardBaselineEntries = await uncommittedChangesWithStatus(guardCwd);
6044
+ const guardBaseline = guardBaselineEntries ? guardBaselineEntries.map((e) => e.path) : guardBaselineEntries;
5555
6045
  const guardHeadBefore = await gitHead(guardCwd);
5556
6046
  // Shared-tree stash guard baseline (incident 2026-09-01): captured
5557
6047
  // unconditionally, before worktree isolation is even attempted, so an
@@ -5603,6 +6093,22 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5603
6093
  console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
5604
6094
  } else {
5605
6095
  console.log(`[scheduler] ${job.slug}: running in main tree (worktree not used: ${worktree.reason})`);
6096
+ // A job losing worktree isolation must never be a silent downgrade
6097
+ // discoverable only by reading queue.json afterwards — every genuine
6098
+ // fallback (never the deliberate SM_JOB_WORKTREE_DISABLE opt-out) is
6099
+ // logged at warn in the durable ops error log, with the job slug, cwd,
6100
+ // and specific reason attached.
6101
+ if (!jobWorktree.isWorktreeDisabled()) {
6102
+ try {
6103
+ appendError({
6104
+ cwd: job.cwd || defaultCwd,
6105
+ scope: 'scheduler',
6106
+ level: 'warn',
6107
+ message: `${job.slug}: worktree isolation fell back to the SHARED working tree — ${worktree.reason}`,
6108
+ meta: { slug: job.slug, cwd: job.cwd || defaultCwd, reason: worktree.reason },
6109
+ });
6110
+ } catch { /* durable logging must never break dispatch */ }
6111
+ }
5606
6112
  }
5607
6113
  // dispatchPhase stamp folded into a single unconditional mutate covering
5608
6114
  // both branches above — the degraded-isolation fallback flag (skipped
@@ -5997,7 +6503,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5997
6503
  sharedTreeGuard = await module.exports.checkSharedTreeGuard({
5998
6504
  cwd: guardCwd,
5999
6505
  stashBaseline,
6000
- dirtyBaseline: guardBaseline,
6506
+ dirtyBaseline: guardBaselineEntries,
6001
6507
  headBefore: guardHeadBefore,
6002
6508
  slug: job.slug,
6003
6509
  });
@@ -6024,6 +6530,10 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6024
6530
  // narrowly to exit 143 — never applied to other non-zero exit codes or to
6025
6531
  // rateLimited (already handled separately, above).
6026
6532
  let sigtermCommitFound = false;
6533
+ // Verified SHA twin of sigtermCommitFound's boolean — only the budget-kill
6534
+ // path (below) threads this onto the row's landedCommit; classifySigtermWithCommit's
6535
+ // own needs_review branch is unchanged and keeps using the boolean alone.
6536
+ let sigtermLandedCommitEvidence = null;
6027
6537
  if (res.exitCode === 143 && !res.rateLimited) {
6028
6538
  const guardHeadAtSigterm = await gitHead(guardCwd);
6029
6539
  sigtermCommitFound = await computeCommittedDuringRun(
@@ -6033,6 +6543,10 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6033
6543
  job.startedAt,
6034
6544
  new Date().toISOString(),
6035
6545
  );
6546
+ if (guardHeadBefore && guardHeadAtSigterm && guardHeadAtSigterm !== guardHeadBefore) {
6547
+ const verified = await resolveLandedCommitEvidence(guardCwd, guardHeadAtSigterm, job.startedAt);
6548
+ if (verified) sigtermLandedCommitEvidence = guardHeadAtSigterm;
6549
+ }
6036
6550
  }
6037
6551
 
6038
6552
  // BLOCKED_BY_FOREIGN_WIP claim scan: the executor exits non-zero for this
@@ -6054,6 +6568,27 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6054
6568
  }
6055
6569
  }
6056
6570
 
6571
+ // Evidence gate for a PLAIN non-zero exit — not SIGTERM-with-commit
6572
+ // (classifySigtermWithCommit above already routes exit 143 to
6573
+ // needs_review when a commit landed) and not rate-limited (handled
6574
+ // separately). A process that dies non-zero for any OTHER reason
6575
+ // (crash during post-commit cleanup, an overrun watchdog's SIGKILL) is
6576
+ // observed directly by THIS exit handler — it never reaches
6577
+ // reapDeadRunningJobs' own git-verified landedCommit evidence gate, so
6578
+ // without this check the exact bug that gate exists to prevent (job
6579
+ // 1192: a landedCommit non-empty is not proof by itself, but discarding
6580
+ // proof of real landed work with no evidence check at all is worse)
6581
+ // recurs here, one call site over. Computed outside mutate() (I/O) like
6582
+ // every other pre-finalize git check above.
6583
+ let plainExitLandedCommitEvidence = null;
6584
+ if (res.exitCode !== 0 && res.exitCode !== 143 && !res.rateLimited) {
6585
+ const headAtPlainExit = await gitHead(guardCwd);
6586
+ if (guardHeadBefore && headAtPlainExit && headAtPlainExit !== guardHeadBefore) {
6587
+ const verified = await resolveLandedCommitEvidence(guardCwd, headAtPlainExit, job.startedAt);
6588
+ if (verified) plainExitLandedCommitEvidence = headAtPlainExit;
6589
+ }
6590
+ }
6591
+
6057
6592
  let actuallyFailed = false;
6058
6593
  let failedJobSnapshot = null;
6059
6594
  let needsInvestigationNow = false;
@@ -6124,7 +6659,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6124
6659
  // Determine effective status, applying the verifier verdict for exit=0 runs.
6125
6660
  let effectiveStatus;
6126
6661
  let sigtermOverrideReason = null;
6127
- const sigtermOverride = res.exitCode !== 0
6662
+ // Wall-clock budget kill — checked FIRST and unconditionally wins:
6663
+ // never 'failed', never silently 'completed', and (via
6664
+ // sigtermLandedCommitEvidence/plainExitLandedCommitEvidence, whichever
6665
+ // this exit code populated) still adjudicated on git evidence rather
6666
+ // than discarded. See classifyBudgetKill's own header.
6667
+ const budgetKill = classifyBudgetKill(res, sigtermLandedCommitEvidence || plainExitLandedCommitEvidence);
6668
+ const sigtermOverride = (!budgetKill && res.exitCode !== 0)
6128
6669
  ? classifySigtermWithCommit(res.exitCode, sigtermCommitFound)
6129
6670
  : null;
6130
6671
  // Validated against the LIVE row's own foreign-WIP manifest — never
@@ -6132,7 +6673,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6132
6673
  // launder a real regression into a block (PRD: give the executor a
6133
6674
  // first-class verdict for "the gate failed on a sibling's in-flight
6134
6675
  // file", but VALIDATE the claim rather than trust it).
6135
- const foreignWipValidation = (!sigtermOverride && foreignWipClaimedPaths !== null)
6676
+ const foreignWipValidation = (!budgetKill && !sigtermOverride && foreignWipClaimedPaths !== null)
6136
6677
  ? validateForeignWipBlockClaim(foreignWipClaimedPaths, s.jobs[i2])
6137
6678
  : null;
6138
6679
  // Consecutive-block streak: cleared by default on every outcome and
@@ -6141,7 +6682,12 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6141
6682
  // between two blocks always resets "in a row" back to zero.
6142
6683
  const priorForeignWipBlockCount = s.jobs[i2].foreignWipBlockCount ?? 0;
6143
6684
  delete s.jobs[i2].foreignWipBlockCount;
6144
- if (sigtermOverride) {
6685
+ if (budgetKill) {
6686
+ effectiveStatus = budgetKill.status;
6687
+ sigtermOverrideReason = budgetKill.reason;
6688
+ s.jobs[i2].verifierVerdict = 'budget_exceeded';
6689
+ if (budgetKill.landedCommit) jobLandedCommitThisRun = budgetKill.landedCommit;
6690
+ } else if (sigtermOverride) {
6145
6691
  effectiveStatus = sigtermOverride.status;
6146
6692
  sigtermOverrideReason = sigtermOverride.reason;
6147
6693
  } else if (foreignWipValidation && foreignWipValidation.ok) {
@@ -6176,7 +6722,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6176
6722
  effectiveStatus = 'failed';
6177
6723
  sigtermOverrideReason = `SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP rejected — unlisted path(s) not in the disclosed foreign-WIP manifest: ${foreignWipValidation.invalidPaths.join(', ') || '(no FOREIGN_WIP_PATHS line)'}`;
6178
6724
  } else if (res.exitCode !== 0) {
6179
- effectiveStatus = 'failed';
6725
+ if (plainExitLandedCommitEvidence) {
6726
+ // Same conservative posture as the SIGTERM+commit case above:
6727
+ // a landed commit doesn't prove every AC line passed, so this
6728
+ // still routes to needs_review for a human/reverify pass,
6729
+ // never silently to completed.
6730
+ effectiveStatus = 'needs_review';
6731
+ sigtermOverrideReason = `exited ${res.exitCode} after landing a git-verified commit — verify AC before treating as done`;
6732
+ jobLandedCommitThisRun = plainExitLandedCommitEvidence;
6733
+ } else {
6734
+ effectiveStatus = 'failed';
6735
+ }
6180
6736
  } else if (
6181
6737
  !verifyResult
6182
6738
  || COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict)
@@ -6205,15 +6761,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6205
6761
  const finalizeReason = (effectiveStatus === 'completed' && verifyResult?.verdict === 'already_satisfied_on_main')
6206
6762
  ? verifyResult.reason
6207
6763
  : (sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`);
6208
- transitionJob(s.jobs[i2], effectiveStatus, { reason: finalizeReason, source: 'spawnJob:finalize' });
6209
- s.jobs[i2].finishedAt = new Date().toISOString();
6210
- s.jobs[i2].exitCode = res.exitCode;
6211
- s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
6212
- if (salvagePatch) {
6213
- s.jobs[i2].salvagePatch = salvagePatch;
6214
- } else {
6215
- delete s.jobs[i2].salvagePatch;
6216
- }
6764
+ // error/verifierVerdict are stamped BEFORE transitionJob() below —
6765
+ // needsReviewLedger's buildNeedsReviewEntryLine reads job.
6766
+ // verifierVerdict/heldReason/error synchronously off `job` the
6767
+ // instant transitionJob() runs (it's called inside transitionJob,
6768
+ // not deferred), so setting these after that call fed the durable
6769
+ // needs_review ledger stale/leftover values from before this run,
6770
+ // defeating its whole `byReason` rollup for the two escalation
6771
+ // paths that land here.
6217
6772
  s.jobs[i2].error = (effectiveStatus === 'needs_review' || s.jobs[i2].blockedByForeignWip === true)
6218
6773
  ? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
6219
6774
  // A failed job (non-zero exit) never consults verifyResult above,
@@ -6231,18 +6786,28 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6231
6786
  s.jobs[i2].landedCommit = jobLandedCommitThisRun;
6232
6787
  }
6233
6788
  // Persist the verifier's verdict string so the renderer can show it.
6234
- // 'blocked_by_foreign_wip_streak' is set above from the sigterm/
6235
- // exit-code path, never from verifyResult (which stays null on a
6236
- // non-zero exit) — never clobber it here.
6789
+ // 'blocked_by_foreign_wip_streak'/'budget_exceeded' are set above
6790
+ // from the sigterm/exit-code path, never from verifyResult (which
6791
+ // stays null on a non-zero exit) — never clobber either here.
6237
6792
  if (verifyResult?.verdict && verifyResult.verdict !== 'clean') {
6238
6793
  s.jobs[i2].verifierVerdict = verifyResult.verdict;
6239
- } else if (s.jobs[i2].verifierVerdict !== 'blocked_by_foreign_wip_streak') {
6794
+ } else if (!['blocked_by_foreign_wip_streak', 'budget_exceeded'].includes(s.jobs[i2].verifierVerdict)) {
6240
6795
  delete s.jobs[i2].verifierVerdict;
6241
6796
  }
6797
+ transitionJob(s.jobs[i2], effectiveStatus, { reason: finalizeReason, source: 'spawnJob:finalize' });
6798
+ s.jobs[i2].finishedAt = new Date().toISOString();
6799
+ s.jobs[i2].exitCode = res.exitCode;
6800
+ s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
6801
+ if (salvagePatch) {
6802
+ s.jobs[i2].salvagePatch = salvagePatch;
6803
+ } else {
6804
+ delete s.jobs[i2].salvagePatch;
6805
+ }
6242
6806
  // Closed-set outcome taxonomy (issue #11 list A2) so a queue row
6243
6807
  // says WHY it ended without anyone opening the transcript.
6244
6808
  s.jobs[i2].terminalReason = launchFailure.deriveTerminalReason({
6245
6809
  effectiveStatus, exitCode: res.exitCode, verifyResult, sigtermOverride, worktreeIntegrationFailure,
6810
+ budgetKill: !!budgetKill,
6246
6811
  });
6247
6812
  delete s.jobs[i2].launchFailure;
6248
6813
  delete s.jobs[i2].heldReason;
@@ -7023,6 +7588,119 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
7023
7588
  return verdicts;
7024
7589
  }
7025
7590
 
7591
+ /**
7592
+ * classifyQueueHealth({ jobs, paused, launchBlocks, runningSet, freeSlots,
7593
+ * totalSlots, lastDispatchAttemptAtMs, now, cwd, thresholdMs })
7594
+ * → { kind, cwd, pending, dispatchable, blockedChains, needsReviewCount, runningCount, ... }
7595
+ *
7596
+ * Single source of truth for the Scheduler page's queue-health header: the
7597
+ * one thing a human staring at a stale-looking queue needs is "which of the
7598
+ * genuinely different causes is this" (all slots busy? every pending row
7599
+ * blocked on a dependency? the dispatch driver itself never ticked?) — this
7600
+ * function names that cause instead of leaving the renderer to re-derive it.
7601
+ *
7602
+ * Reuses classifyQueueStarvation for the blocked/stalled read so the header
7603
+ * can never disagree with runQueueStarvationWatchdog's own decision to force
7604
+ * a tick: both are handed the same lastDispatchAttemptAt-based idle clock and
7605
+ * the same computeBlockedChains walk under the hood. Called here with
7606
+ * `thresholdMs: 0` first (a live header must say "blocked" the instant every
7607
+ * pending row is dependency-stuck, not wait out the watchdog's own 10-minute
7608
+ * grace period) — the returned `idleMs` is then compared against the REAL
7609
+ * `thresholdMs` to decide 'stalled' vs the healthy 'running' default, which
7610
+ * is exactly the comparison classifyQueueStarvation would make internally.
7611
+ *
7612
+ * `pending`/`dispatchable`/`blockedChains`/`needsReviewCount` are always
7613
+ * populated (via computeBlockedChains — the exact primitive
7614
+ * classifyQueueStarvation itself calls) regardless of kind, so a 'saturated'
7615
+ * or 'running' header can still say how much of the backlog is dependency-
7616
+ * blocked, not just the kinds where that's the headline cause.
7617
+ *
7618
+ * Kinds, in the priority order they're checked (paused is a decision, not a
7619
+ * stall; an open launch breaker explains an otherwise-inexplicable
7620
+ * non-dispatch before slot/dependency causes are even considered):
7621
+ * 'paused' — the scheduler itself is paused.
7622
+ * 'launch-blocked' — a pending row's persona has an active circuit-breaker
7623
+ * entry (lib/launchFailure.cjs).
7624
+ * 'idle' — nothing pending in this scope.
7625
+ * 'saturated' — pending work exists but every session slot is in use.
7626
+ * 'blocked' — nothing running, slots free, every pending row's
7627
+ * dependsOn chain terminates in a non-completed row.
7628
+ * 'stalled' — nothing running, slots free, at least one row is
7629
+ * dispatchable right now, and the dispatch driver has
7630
+ * been idle >= thresholdMs (agrees with the watchdog).
7631
+ * 'running' — the healthy default: work is flowing, or the driver
7632
+ * hasn't been idle long enough to call a stall yet.
7633
+ *
7634
+ * Pure, no IO. `cwd` scopes jobs/pending/blocked/needsReview to one project
7635
+ * (the Scheduler nav row is PROJECT-face — see CLAUDE.md); `freeSlots` /
7636
+ * `totalSlots` / `launchBlocks` stay machine-wide inputs by design, same as
7637
+ * WindowStrip's existing scopeCwd split.
7638
+ */
7639
+ function classifyQueueHealth({
7640
+ jobs, paused, launchBlocks, runningSet: runningSlugs, freeSlots, totalSlots,
7641
+ lastDispatchAttemptAtMs, now, cwd = null, thresholdMs = QUEUE_STARVATION_MS,
7642
+ } = {}) {
7643
+ const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
7644
+ const projectJobs = cwd ? rows.filter((j) => j.cwd === cwd) : rows;
7645
+ const pendingRows = projectJobs.filter((j) => j.status === 'pending');
7646
+ const runningRows = projectJobs.filter((j) => j.status === 'running' || runningSlugs?.has?.(j.slug));
7647
+ const needsReviewCount = projectJobs.filter((j) => j.status === 'needs_review').length;
7648
+
7649
+ // computeBlockedChains is the SAME primitive classifyQueueStarvation calls
7650
+ // internally, computed once here so EVERY kind (not just 'blocked'/
7651
+ // 'stalled') carries real dispatchable-vs-blocked counts instead of a null.
7652
+ const blockedChains = computeBlockedChains(projectJobs);
7653
+ const blockedTotal = blockedChains.reduce((n, c) => n + c.blocked, 0);
7654
+ const dispatchable = Math.max(0, pendingRows.length - blockedTotal);
7655
+ const base = {
7656
+ cwd, pending: pendingRows.length, dispatchable, blockedChains, needsReviewCount, runningCount: runningRows.length,
7657
+ };
7658
+
7659
+ if (paused) return { ...base, kind: 'paused', reason: paused.reason ?? null };
7660
+
7661
+ // launch-blocked: only a persona a PENDING row in this scope actually uses
7662
+ // — a breaker open for a persona nothing here needs is not this scope's
7663
+ // problem (matches WindowStrip's own unconditional-banner-per-block read).
7664
+ const neededAgentTypes = new Set(pendingRows.map((j) => launchFailure.launchBlockKeyFor(j)));
7665
+ for (const [key, block] of Object.entries(launchBlocks ?? {})) {
7666
+ if (block && neededAgentTypes.has(key)) {
7667
+ return { ...base, kind: 'launch-blocked', agentType: key, block };
7668
+ }
7669
+ }
7670
+
7671
+ if (pendingRows.length === 0) return { ...base, kind: 'idle' };
7672
+
7673
+ // classifyQueueStarvation only ever classifies while nothing is running
7674
+ // (its own runningCount > 0 guard) — that boundary is also exactly where
7675
+ // slot saturation, not dependency shape, is the honest cause.
7676
+ if (runningRows.length > 0) {
7677
+ if (Number.isFinite(freeSlots) && freeSlots <= 0) {
7678
+ return { ...base, kind: 'saturated', totalSlots: totalSlots ?? null };
7679
+ }
7680
+ return { ...base, kind: 'running' };
7681
+ }
7682
+
7683
+ // Nothing running: hand the SAME rows + idle clock to classifyQueueStarvation
7684
+ // (thresholdMs: 0 — a live header must say "blocked" the instant every
7685
+ // pending row is dependency-stuck, not wait out the watchdog's own grace
7686
+ // period) purely for its idleMs reading; its own dispatchable/blockedChains
7687
+ // are mathematically identical to `base`'s (same computeBlockedChains walk
7688
+ // over the same rows), so `base` already carries them.
7689
+ const immediate = classifyQueueStarvation({
7690
+ jobs: projectJobs, paused: false, runningCount: 0,
7691
+ lastRunAtMs: lastDispatchAttemptAtMs, now, thresholdMs: 0,
7692
+ });
7693
+ // pending.length is already > 0 above, so `immediate` can only be null when
7694
+ // lastDispatchAttemptAtMs is itself in the future (clock skew) — fall back
7695
+ // to computing idleMs the same way rather than asserting a kind we can't
7696
+ // back up with a real number.
7697
+ const idleMs = immediate ? immediate.idleMs
7698
+ : (Number.isFinite(lastDispatchAttemptAtMs) ? now - lastDispatchAttemptAtMs : Infinity);
7699
+ if (dispatchable === 0) return { ...base, kind: 'blocked', idleMs };
7700
+ const kind = idleMs >= thresholdMs ? 'stalled' : 'running';
7701
+ return { ...base, kind, idleMs };
7702
+ }
7703
+
7026
7704
  /**
7027
7705
  * The watchdog half: acts on classifyQueueStarvationByProject. Called from
7028
7706
  * the heartbeat, which already runs on its own timer independent of the
@@ -7254,12 +7932,19 @@ async function reapDeadRunningJobs() {
7254
7932
  // status:"running" with no slug left in runningSet to trigger reconciliation.
7255
7933
  // queue.json is the source of truth for which jobs are actually running.
7256
7934
  const state = await readQueue();
7935
+ // Shared by the log-evidence injections below and the reapable-processing
7936
+ // loop further down — same `j.runId` → run log path formula either way.
7937
+ const logPathForJob = (j) => (j?.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null);
7257
7938
  const { reapable, warnings, recovered } = selectReapableJobs(state.jobs, Date.now(), {
7258
7939
  pidAlive: claudePidAlive,
7259
7940
  grace: PIDLESS_SPAWN_GRACE_MS,
7260
7941
  findLiveProcess: (j) => findLiveProcessForJob(j, {
7261
7942
  worktreeDir: jobWorktree.worktreeDirFor(j.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD, j.slug),
7943
+ runCwd: j.runtime?.cwd || j.cwd,
7262
7944
  }),
7945
+ getLogPid: (j) => readSpawnedPidFromLog(logPathForJob(j)),
7946
+ getLogMtimeMs: (j) => readLogMtimeMs(logPathForJob(j)),
7947
+ logFreshWindowMs: IDLE_OUTPUT_KILL_MS,
7263
7948
  });
7264
7949
  for (const w of warnings) {
7265
7950
  console.warn(`[scheduler] reapDeadRunningJobs: ${w.reason} slug=${w.slug} — leaving row alone`);
@@ -7284,11 +7969,9 @@ async function reapDeadRunningJobs() {
7284
7969
  }
7285
7970
 
7286
7971
  const dead = [];
7287
- for (const { slug, pid, pidless, reason } of reapable) {
7972
+ for (const { slug, pid, pidless, reason, failureOverride } of reapable) {
7288
7973
  const j = state.jobs.find((x) => x.slug === slug);
7289
- const logPath = j?.runId
7290
- ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`)
7291
- : null;
7974
+ const logPath = logPathForJob(j);
7292
7975
  // Absent/empty run dir → classifyRunOutcome finds no result event →
7293
7976
  // 'no_result' → non-success below → filed as failed, never completed.
7294
7977
  const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
@@ -7307,7 +7990,7 @@ async function reapDeadRunningJobs() {
7307
7990
  // re-derived so a phantom link never survives the reap.
7308
7991
  const hasOwnArtifact = pidless ? logHasOutput(logPath) : true;
7309
7992
  const gateOutcome = pidless ? resolvePidlessGateOutcome(outcome, hasOwnArtifact) : mapOutcomeToGateOutcome(outcome);
7310
- dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath, noOwnArtifact: pidless && !hasOwnArtifact });
7993
+ dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath, noOwnArtifact: pidless && !hasOwnArtifact, failureOverride });
7311
7994
  }
7312
7995
 
7313
7996
  queueHealthSweepCycle += 1;
@@ -7405,8 +8088,44 @@ async function reapDeadRunningJobs() {
7405
8088
  integrationResults.set(d.slug, { effectiveSuccess, landedCommit, notLandedInfo });
7406
8089
  }
7407
8090
 
8091
+ // Evidence-before-failure guard for a row about to be stamped 'failed'
8092
+ // (this PRD — job 1192 shipped a real 3-file commit and was still
8093
+ // reaped 'failed' because this check did not exist): a `landedCommit`
8094
+ // already recorded on the row is only ever stamped from an actual HEAD
8095
+ // advance or a proven branch-integration (jobLandedCommitThisRun / the
8096
+ // dead-pid integration proof above / the dispatch-time sidecar
8097
+ // backfill) — never speculative — but it can still be STALE by the time
8098
+ // this row is reaped (the branch it named could have been force-pushed
8099
+ // over, or the row could be carrying a sidecar-backfilled sha from a
8100
+ // run that was later discarded). git-resolving it here is what turns
8101
+ // "the field is non-empty" into "this sha is a real commit in this
8102
+ // repo right now". Computed OUTSIDE mutate() for the same reason
8103
+ // integrationResults is above: git spawn work must never run inside
8104
+ // mutate()'s single global serialization chain.
8105
+ //
8106
+ // Scoped to exactly the rows that would otherwise fall through to
8107
+ // 'failed' below: a 'success' outcome is already resolved by
8108
+ // integrationResults above (never reaches 'failed'), a rate-limited
8109
+ // death is retryable and never terminal, and a pidless row that already
8110
+ // carries a `failureOverride` (PRD 1173) is already diverted to
8111
+ // needs_review — this gate must never re-litigate either of those.
8112
+ const landedCommitEvidence = new Map();
8113
+ // Each row's evidence check is an independent read-only `git cat-file`/
8114
+ // `git log` pair with no shared mutable state between iterations, so
8115
+ // this runs the whole dead-job batch concurrently rather than one
8116
+ // dispatch's git-spawn latency at a time.
8117
+ await Promise.all(dead.map(async (d) => {
8118
+ if (d.outcome === 'rate_limited' || d.outcome === 'success') return;
8119
+ if (d.pidless && d.failureOverride) return;
8120
+ const row = state.jobs.find((x) => x.slug === d.slug);
8121
+ if (!row?.landedCommit) return;
8122
+ const rowCwd = row.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD;
8123
+ const resolved = await resolveLandedCommitEvidence(rowCwd, row.landedCommit, row.startedAt);
8124
+ if (resolved) landedCommitEvidence.set(d.slug, row.landedCommit);
8125
+ }));
8126
+
7408
8127
  await mutate(async (s) => {
7409
- for (const { slug, pid, outcome, gateOutcome, pidless, reason, noOwnArtifact } of dead) {
8128
+ for (const { slug, pid, outcome, gateOutcome, pidless, reason, noOwnArtifact, failureOverride } of dead) {
7410
8129
  const idx = s.jobs.findIndex((x) => x.slug === slug);
7411
8130
  if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
7412
8131
  const rateLimited = outcome === 'rate_limited';
@@ -7465,6 +8184,20 @@ async function reapDeadRunningJobs() {
7465
8184
  notLandedInfo = ir.notLandedInfo;
7466
8185
  }
7467
8186
  }
8187
+ // Evidence-before-failure guard for the pidless-reap path (PRD 1173):
8188
+ // a pidless reap about to stamp 'failed' purely because runtime.pid
8189
+ // was never recorded must first check whether this row already
8190
+ // carries a landedCommit from an earlier dispatch of the same slug
8191
+ // (landedCommit survives a reset — see the comment near
8192
+ // resetJobFields). Diverted to needs_review, never silently
8193
+ // 'completed' — see resolvePidlessFailureOverride's header in
8194
+ // reaperHelpers.cjs for why needs_review is the correct destination.
8195
+ // Scoped strictly to the pidless branch: dead-pid and rate-limited
8196
+ // rows are untouched, and a pidless row that already resolved to
8197
+ // effectiveSuccess/notLandedInfo above is left alone too.
8198
+ if (pidless && !effectiveSuccess && !rateLimited && !notLandedInfo && failureOverride) {
8199
+ notLandedInfo = { verdict: failureOverride.verdict, reason: failureOverride.reason };
8200
+ }
7468
8201
 
7469
8202
  const leftoverSuffix = deltaPaths && deltaPaths.length
7470
8203
  ? ` — left ${deltaPaths.length} files uncommitted`
@@ -7472,9 +8205,20 @@ async function reapDeadRunningJobs() {
7472
8205
  const baseReason = notLandedInfo
7473
8206
  ? `reaped: ${notLandedInfo.reason}`
7474
8207
  : (pidless ? reason : `reaped: process gone (outcome=${outcome})`);
8208
+ // Evidence gate (this PRD): a row that would otherwise fall through
8209
+ // to 'failed' below, but whose landedCommit was proven to resolve
8210
+ // via git cat-file BEFORE this mutate() ran (see landedCommitEvidence
8211
+ // above), gets promoted to 'completed' instead — the row already
8212
+ // shipped real work, so a bookkeeping gap (no runtime.pid recorded)
8213
+ // must never override git-verified evidence with a false failure.
8214
+ const confirmedLandedCommit = (!effectiveSuccess && !notLandedInfo && !rateLimited)
8215
+ ? (landedCommitEvidence.get(slug) || null)
8216
+ : null;
7475
8217
  const transitionReason = rateLimited
7476
8218
  ? `reaped: rate limit detected — reset to pending, not failed (outcome=${outcome})${leftoverSuffix}`
7477
- : baseReason + leftoverSuffix;
8219
+ : confirmedLandedCommit
8220
+ ? `${baseReason}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence${leftoverSuffix}`
8221
+ : baseReason + leftoverSuffix;
7478
8222
 
7479
8223
  if (rateLimited) {
7480
8224
  // Retryable, never terminal (PRD 1117) — same resetJobFields path
@@ -7483,17 +8227,27 @@ async function reapDeadRunningJobs() {
7483
8227
  // paused-for-rate-limit reset: fresh runId/startedAt/exitCode.
7484
8228
  resetJobFields(s.jobs[idx], transitionReason, { source: 'reapDeadRunningJobs:rate-limit' });
7485
8229
  } else {
7486
- const targetStatus = effectiveSuccess ? 'completed' : (notLandedInfo ? 'needs_review' : 'failed');
7487
- transitionJob(s.jobs[idx], targetStatus, { reason: transitionReason, source: 'reapDeadRunningJobs' });
7488
- s.jobs[idx].exitCode = effectiveSuccess ? 0 : (s.jobs[idx].exitCode ?? 1);
7489
- s.jobs[idx].finishedAt = new Date().toISOString();
7490
- s.jobs[idx].error = effectiveSuccess ? null : `${transitionReason} (outcome=${outcome})`;
7491
- s.jobs[idx].gateOutcome = gateOutcome;
8230
+ const landed = effectiveSuccess || Boolean(confirmedLandedCommit);
8231
+ const targetStatus = effectiveSuccess
8232
+ ? 'completed'
8233
+ : (notLandedInfo ? 'needs_review' : (confirmedLandedCommit ? 'completed' : 'failed'));
8234
+ const source = confirmedLandedCommit ? 'reapDeadRunningJobs:landed' : 'reapDeadRunningJobs';
8235
+ // error/verifierVerdict are stamped BEFORE transitionJob() below —
8236
+ // see the identical ordering fix (and its rationale) in spawnJob's
8237
+ // finalize path: transitionJob's needs_review ledger entry reads
8238
+ // these fields off `job` synchronously the instant it runs, so
8239
+ // setting them after fed the ledger a stale/leftover reason.
8240
+ s.jobs[idx].error = landed ? null : `${transitionReason} (outcome=${outcome})`;
7492
8241
  if (notLandedInfo) {
7493
8242
  s.jobs[idx].verifierVerdict = notLandedInfo.verdict;
7494
8243
  } else {
7495
8244
  delete s.jobs[idx].verifierVerdict;
7496
8245
  }
8246
+ transitionJob(s.jobs[idx], targetStatus, { reason: transitionReason, source });
8247
+ s.jobs[idx].exitCode = landed ? 0 : (s.jobs[idx].exitCode ?? 1);
8248
+ s.jobs[idx].finishedAt = new Date().toISOString();
8249
+ s.jobs[idx].gateOutcome = gateOutcome;
8250
+ if (confirmedLandedCommit) s.jobs[idx].landedCommit = confirmedLandedCommit;
7497
8251
  if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
7498
8252
  }
7499
8253
  // A pidless spawn that never wrote its own '<slug>.log' into the
@@ -7529,7 +8283,14 @@ async function reapDeadRunningJobs() {
7529
8283
  appendAuditEvent('job_reaped_rate_limited', { slug, cwd: s.jobs[idx].cwd ?? null });
7530
8284
  } else if (pidless) {
7531
8285
  console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
7532
- appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
8286
+ appendAuditEvent('job_reaped_pidless', {
8287
+ slug,
8288
+ cwd: s.jobs[idx].cwd ?? null,
8289
+ outcome,
8290
+ graceMs: PIDLESS_SPAWN_GRACE_MS,
8291
+ landedCommit: s.jobs[idx].landedCommit ?? null,
8292
+ verifierVerdict: s.jobs[idx].verifierVerdict ?? null,
8293
+ });
7533
8294
  } else {
7534
8295
  console.log(`[scheduler] reaped dead job slug=${slug} pid=${pid} outcome=${outcome}`);
7535
8296
  }
@@ -7724,6 +8485,12 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
7724
8485
  const seen = new Set(hot.map((j) => `${j.slug}|${j.runId ?? ''}`));
7725
8486
  const archived = (Array.isArray(historyEntries) ? historyEntries : []).filter((j) => {
7726
8487
  if (!j) return false;
8488
+ // needs_review_entry/needs_review_resolution lines (needsReviewLedger.cjs)
8489
+ // share history.jsonl with terminal job rows but carry no `status` — the
8490
+ // History view (SchedulerHistoryView.tsx) renders ScheduleJob rows, so a
8491
+ // ledger line slipping through here would show up as a statusless,
8492
+ // meaningless row in that table.
8493
+ if (j.kind && j.kind !== 'terminal') return false;
7727
8494
  const key = `${j.slug}|${j.runId ?? ''}`;
7728
8495
  if (seen.has(key)) return false;
7729
8496
  seen.add(key);
@@ -7849,6 +8616,69 @@ function isExhaustedAutoFix(job) {
7849
8616
  return job.autoFixOutcome !== 'plan' && (job.autoFixRetries ?? 0) >= 1;
7850
8617
  }
7851
8618
 
8619
+ /**
8620
+ * Verdicts from the POST-RUN GUARDS (commit-guard / shared-tree guard) that a
8621
+ * later, independently-checkable commit/looksDone signal can meaningfully
8622
+ * confirm or refute. Auto-fix investigations only ever launch for FAILING
8623
+ * runs (see isExhaustedAutoFix above and selectAutoFixTargets) — a job parked
8624
+ * by one of these GUARD verdicts exits 0 and never has autoFixAttempted set,
8625
+ * so it is invisible to isExhaustedAutoFix and can sit in needs_review
8626
+ * forever with nothing to spend and nothing to exhaust (PRD 1181, 2026-09-12:
8627
+ * exit 0, commit aff5607 landed, parked on a shared-tree verdict, cleared
8628
+ * only by a human).
8629
+ *
8630
+ * 'worktree_integration_failed' is deliberately EXCLUDED — its damage IS a
8631
+ * commit: one stranded on an unmerged `sm-job/<slug>` branch. A `landedCommit`
8632
+ * existing is not evidence against that verdict, it is a restatement of it,
8633
+ * so admitting it here would auto-complete a row whose work never actually
8634
+ * reached the target branch. It already has its own dedicated, git-native
8635
+ * resolution path (selectMechanicalRecoveryTarget / performMechanicalRecovery
8636
+ * — a real re-attempted merge) and must never be pulled into this ladder.
8637
+ *
8638
+ * 'pidless_reap_with_landed_commit' (PRD 1173, resolvePidlessFailureOverride
8639
+ * in reaperHelpers.cjs) is included: it parks on the exact same shape (exit
8640
+ * never observed / no autoFixAttempted, real landedCommit evidence) as
8641
+ * 'silent_no_op' and 'shared_tree_reverted', and this ladder never trusts
8642
+ * landedCommit alone anyway — applyNeedsReviewAutoResolve only resolves once
8643
+ * job.looksDone independently reconfirms via a fresh commits-since-this-run
8644
+ * scan, which is exactly the "does the commit correspond to THIS dispatch"
8645
+ * re-verification resolvePidlessFailureOverride's own header says the
8646
+ * pidless-reap path itself cannot do. Omitting it here reproduces the same
8647
+ * "nothing to spend, nothing to exhaust" needs_review stall this PRD exists
8648
+ * to fix, just for a third verdict.
8649
+ */
8650
+ const GUARD_VERDICT_EVIDENCE_ELIGIBLE = new Set(['silent_no_op', 'shared_tree_reverted', 'pidless_reap_with_landed_commit']);
8651
+
8652
+ /**
8653
+ * Pure predicate, no I/O: a needs_review row parked directly by one of the
8654
+ * GUARD_VERDICT_EVIDENCE_ELIGIBLE verdicts, that never went through an
8655
+ * auto-fix investigation at all (autoFixAttempted is not true) — the
8656
+ * structural gap this PRD closes, distinct from isExhaustedAutoFix's "went
8657
+ * through auto-fix and spent it" case. A job that DID get an auto-fix
8658
+ * investigation is left to isExhaustedAutoFix's own ladder rather than this
8659
+ * one, even if its verifierVerdict happens to also be in the eligible set.
8660
+ * Exported for tests.
8661
+ */
8662
+ function isGuardParkedWithoutAutoFix(job) {
8663
+ if (!job || job.status !== 'needs_review') return false;
8664
+ if (job.autoFixAttempted === true) return false;
8665
+ return GUARD_VERDICT_EVIDENCE_ELIGIBLE.has(job.verifierVerdict);
8666
+ }
8667
+
8668
+ /**
8669
+ * Pure predicate, no I/O: is this needs_review row eligible for the bounded
8670
+ * auto-resolve ladder at all — either because its auto-fix path is genuinely
8671
+ * spent (isExhaustedAutoFix), or because it was parked by a GUARD verdict
8672
+ * that never entered auto-fix in the first place (isGuardParkedWithoutAutoFix).
8673
+ * Both classes share ONE ladder (applyNeedsReviewAutoResolve) rather than a
8674
+ * duplicated one — the ladder itself doesn't care which door a row came
8675
+ * through, only whether it now carries completion evidence (job.looksDone).
8676
+ * Exported for tests.
8677
+ */
8678
+ function isEligibleForNeedsReviewAutoResolve(job) {
8679
+ return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job);
8680
+ }
8681
+
7852
8682
  /**
7853
8683
  * Pure predicate: an investigation produced a fix plan (autoFixOutcome ===
7854
8684
  * 'plan') but its fix-plan slug is not present among `queuedSlugs` — the
@@ -7996,17 +8826,26 @@ function isRescanCandidate(job) {
7996
8826
  * selectResumeRecoveryTarget / selectAutoFixTargets) so the guard can never
7997
8827
  * again be narrower than the work reverifyNeedsReview performs.
7998
8828
  *
7999
- * Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget are pure
8000
- * (no I/O). selectAutoFixTargets is called with an injected fixSlugExists
8001
- * that always returns false — cheap and deliberately over-inclusive (a false
8002
- * positive here just means one extra periodic pass, never a missed one) so
8003
- * this guard never pays selectAutoFixTargets's production fs.existsSync scan
8004
- * per tick. resolveRunId's IO only fires for rows missing job.runId, same as
8829
+ * Widened again (this PRD): a `needs_review` row parked directly by a GUARD
8830
+ * verdict with no auto-fix history (isGuardParkedWithoutAutoFix) is not an
8831
+ * isRescanCandidate either — RESCANNABLE_VERDICTS covers transcript-verifier
8832
+ * verdicts, not commit-guard/shared-tree-guard verdicts — but
8833
+ * reverifyNeedsReview's looksDone-annotation pass now runs for it too (see
8834
+ * that function). Same rule as always: never let this guard be narrower than
8835
+ * the work reverifyNeedsReview actually performs.
8836
+ *
8837
+ * Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget and
8838
+ * isGuardParkedWithoutAutoFix are pure (no I/O). selectAutoFixTargets is
8839
+ * called with an injected fixSlugExists that always returns false — cheap
8840
+ * and deliberately over-inclusive (a false positive here just means one
8841
+ * extra periodic pass, never a missed one) so this guard never pays
8842
+ * selectAutoFixTargets's production fs.existsSync scan per tick.
8843
+ * resolveRunId's IO only fires for rows missing job.runId, same as
8005
8844
  * isRescanCandidate already incurs above.
8006
8845
  */
8007
8846
  function shouldRunPeriodicReverify(jobs) {
8008
8847
  if (!Array.isArray(jobs)) return false;
8009
- if (jobs.some((j) => isRescanCandidate(j))) return true;
8848
+ if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j))) return true;
8010
8849
  if (jobs.some((j) => selectMechanicalRecoveryTarget(j) || selectResumeRecoveryTarget(j))) return true;
8011
8850
  return selectAutoFixTargets(jobs, { fixSlugExists: () => false }).length > 0;
8012
8851
  }
@@ -8172,9 +9011,11 @@ function needsReviewAutoResolveDisabled() {
8172
9011
  * selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) →
8173
9012
  * [{ slug, cwd, ageMs, attempts }]
8174
9013
  *
8175
- * Pure selector — no IO. Selects `needs_review` rows whose auto-fix path is
8176
- * genuinely spent (isExhaustedAutoFix), whose newest statusHistory entry
8177
- * with `to === 'needs_review'` is older than `thresholdMs`, and whose
9014
+ * Pure selector — no IO. Selects `needs_review` rows eligible for the
9015
+ * bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve — either
9016
+ * auto-fix genuinely spent, or parked by a GUARD verdict that never entered
9017
+ * auto-fix at all), whose newest statusHistory entry with `to ===
9018
+ * 'needs_review'` is older than `thresholdMs`, and whose
8178
9019
  * exhaustedResolveAttempts counter has not yet spent its cap.
8179
9020
  *
8180
9021
  * The inclusion bound is inclusive of the cap itself (`<= CAP`, not `<
@@ -8188,7 +9029,7 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
8188
9029
  const targets = [];
8189
9030
  for (const j of jobs ?? []) {
8190
9031
  if (j.status !== 'needs_review') continue;
8191
- if (!isExhaustedAutoFix(j)) continue;
9032
+ if (!isEligibleForNeedsReviewAutoResolve(j)) continue;
8192
9033
  if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) continue;
8193
9034
  const history = j.statusHistory || [];
8194
9035
  let entry = null;
@@ -8224,17 +9065,26 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
8224
9065
  * failed-autoreset loop above) so a stale target computed before this
8225
9066
  * mutate() pass can never double-apply. Returns the outcome, or null if the
8226
9067
  * race guard rejected it.
9068
+ *
9069
+ * A row can reach here through either door (isEligibleForNeedsReviewAutoResolve):
9070
+ * auto-fix genuinely exhausted, or parked directly by a GUARD_VERDICT_EVIDENCE_
9071
+ * ELIGIBLE verdict with no auto-fix history at all. `originIsGuardParked` picks
9072
+ * which door this particular row came through, purely to make the requeue/skip
9073
+ * reason text (and the Queue UI's job.error) name the RIGHT evidence — a
9074
+ * guard-parked row was never "exhausted auto-fix" and must never claim to be.
8227
9075
  */
8228
9076
  function applyNeedsReviewAutoResolve(j) {
8229
- if (!j || j.status !== 'needs_review' || !isExhaustedAutoFix(j)) return null;
9077
+ if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j)) return null;
8230
9078
  if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) return null;
9079
+ const originIsGuardParked = !isExhaustedAutoFix(j) && isGuardParkedWithoutAutoFix(j);
8231
9080
 
8232
9081
  if (j.looksDone) {
8233
9082
  const attempt = j.exhaustedResolveAttempts ?? 0;
8234
- transitionJob(j, 'completed', {
8235
- reason: `needs_review auto-resolve: verifier annotation shows work landed (${j.looksDone.commits.length} commit(s) since this run touch the PRD's declared paths)`,
8236
- source: 'needsReviewAutoResolve',
8237
- });
9083
+ const reason = originIsGuardParked
9084
+ ? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with landed commit and looksDone evidence (${j.looksDone.rule}) — `
9085
+ + `${j.looksDone.commits.length} commit(s) attributable to this job's own run touch the PRD's declared paths, work landed`
9086
+ : `needs_review auto-resolve: verifier annotation shows work landed (${j.looksDone.rule} — ${j.looksDone.commits.length} commit(s) attributable to this job's own run touch the PRD's declared paths)`;
9087
+ transitionJob(j, 'completed', { reason, source: 'needsReviewAutoResolve' });
8238
9088
  appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'completed', attempt });
8239
9089
  return 'completed';
8240
9090
  }
@@ -8243,18 +9093,22 @@ function applyNeedsReviewAutoResolve(j) {
8243
9093
  if (attemptsSoFar < NEEDS_REVIEW_RESOLVE_CAP) {
8244
9094
  const attempt = attemptsSoFar + 1;
8245
9095
  j.exhaustedResolveAttempts = attempt;
8246
- transitionJob(j, 'pending', {
8247
- reason: `needs_review auto-resolve: exhausted auto-fix, no completion evidence — requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`,
8248
- source: 'needsReviewAutoResolve',
8249
- });
9096
+ const reason = originIsGuardParked
9097
+ ? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with no completion evidence yet — `
9098
+ + `requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`
9099
+ : `needs_review auto-resolve: exhausted auto-fix, no completion evidence — requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`;
9100
+ transitionJob(j, 'pending', { reason, source: 'needsReviewAutoResolve' });
8250
9101
  appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'requeued', attempt });
8251
9102
  return 'requeued';
8252
9103
  }
8253
9104
 
8254
9105
  j.needsReviewAutoResolvedSkip = true;
8255
- j.error = `needs_review auto-resolve: exhausted auto-fix path (autoFixOutcome=${j.autoFixOutcome ?? 'none'}, `
8256
- + `autoFixRetries=${j.autoFixRetries ?? 0}) with no completion evidence after ${NEEDS_REVIEW_RESOLVE_CAP} `
8257
- + `requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`;
9106
+ j.error = originIsGuardParked
9107
+ ? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with no completion evidence after `
9108
+ + `${NEEDS_REVIEW_RESOLVE_CAP} requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`
9109
+ : `needs_review auto-resolve: exhausted auto-fix path (autoFixOutcome=${j.autoFixOutcome ?? 'none'}, `
9110
+ + `autoFixRetries=${j.autoFixRetries ?? 0}) with no completion evidence after ${NEEDS_REVIEW_RESOLVE_CAP} `
9111
+ + `requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`;
8258
9112
  transitionJob(j, 'skipped', {
8259
9113
  reason: `needs_review auto-resolve: cap exhausted (${NEEDS_REVIEW_RESOLVE_CAP}/${NEEDS_REVIEW_RESOLVE_CAP} requeue attempts) — auto-skipped`,
8260
9114
  source: 'needsReviewAutoResolve',
@@ -8320,6 +9174,11 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
8320
9174
  // fix-plan investigation to diagnose — there is no code defect to
8321
9175
  // author a PRD against, only another job's still-uncommitted tree.
8322
9176
  if (job.blockedByForeignWip === true) return false;
9177
+ // A budget-killed job parks for a human/ladder decision, never an
9178
+ // auto-fix investigation or auto-retry — the run didn't fail, it simply
9179
+ // overran its own estimate; there's no code defect to diagnose (Out of
9180
+ // scope: "retrying or auto-resuming a budget-killed job" for this PRD).
9181
+ if (job.verifierVerdict === 'budget_exceeded') return false;
8323
9182
  // A stale re-run whose work already shipped (rcaReport's 'already-shipped'
8324
9183
  // class) must never buy a fix-plan PRD — there is nothing to fix, and the
8325
9184
  // correct recovery (archiving the PRD) is a human/reconcile action, not
@@ -8379,35 +9238,153 @@ function isEligibleForImmediateAutoFix(job, allJobs, fixSlugExists) {
8379
9238
  }
8380
9239
 
8381
9240
  /**
8382
- * Widened evidence check (PRD 1102): does at least one commit land AFTER
8383
- * this job's run window that touches a path the PRD itself declares? Scoped
8384
- * to the PRD's own declared paths (never the whole repo) so a sibling job's
8385
- * unrelated commit is not credited to this one — see healRefusalReason's own
9241
+ * attributeLandedCommits(job, pathCommits, cwd) → { commits, rule } | null
9242
+ *
9243
+ * Narrows a set of PATH-overlapping commits (computeLooksDone's
9244
+ * `landedSinceRun` result — any commit touching the PRD's declared paths,
9245
+ * regardless of who authored it) down to the subset actually attributable to
9246
+ * THIS job's own run. Path overlap alone is not attribution: inside one Epic,
9247
+ * sibling PRDs routinely declare the same hot file, so a sibling's commit is
9248
+ * indistinguishable from this job's own by path alone (the incident this
9249
+ * function exists to close — PRD 1204's parked row cited PRD 1205's commit
9250
+ * 5dadf3c as its own evidence).
9251
+ *
9252
+ * Three rules, tried strongest-first, first match wins:
9253
+ *
9254
+ * 1. 'landedCommit' — the row's own `job.landedCommit`, re-verified here via
9255
+ * `resolveLandedCommitEvidence` against THIS job's `startedAt`. This is
9256
+ * the strongest signal because it is not inferred from `git log` at all:
9257
+ * it is the sha spawnJob's own finalize step observed THIS dispatch's
9258
+ * worktree/branch landing (see resolveLandedCommitEvidence's own header
9259
+ * for why it also guards against a stale sha surviving a reset). Trusted
9260
+ * independent of whether it appears in `pathCommits` — it is definitionally
9261
+ * this job's own work, not something discovered by scanning history.
9262
+ * 2. 'job branch' — a path-overlapping commit reachable from (an ancestor of
9263
+ * or equal to) this job's own `sm-job/<slug>` branch tip. Still
9264
+ * job-specific even though it IS a `git log` scan: a sibling's commit can
9265
+ * never be an ancestor of THIS job's own branch ref. In practice this
9266
+ * branch is deleted on successful integration (gitWorktree.cjs's
9267
+ * `cleanupWorktree`), so this mainly fires when integration failed and
9268
+ * the branch was deliberately kept for recovery, or reverify runs before
9269
+ * cleanup — a narrower window than rule 1, hence checked second.
9270
+ * 3. 'slug trailer' — a path-overlapping commit whose message contains this
9271
+ * job's slug verbatim. Weakest of the three (a coincidental substring
9272
+ * match is possible, and nothing stamps this automatically today), so it
9273
+ * is the last resort when the two structural signals above found
9274
+ * nothing.
9275
+ *
9276
+ * Deliberately NOT a rule: raw path overlap by itself (the bug this function
9277
+ * fixes) and `committedInWindow`-style time-window-only evidence — a sibling
9278
+ * job running concurrently in the very same window is exactly as invisible to
9279
+ * a time bound as it is to a path filter, so neither narrows attribution.
9280
+ *
9281
+ * Never throws: a missing ref, an unresolvable sha, or any git failure for a
9282
+ * given commit/rule is treated as "that commit doesn't satisfy this rule",
9283
+ * never as a fabricated match.
9284
+ */
9285
+ async function attributeLandedCommits(job, pathCommits, cwd) {
9286
+ if (job?.landedCommit && await resolveLandedCommitEvidence(cwd, job.landedCommit, job.startedAt)) {
9287
+ return { commits: [job.landedCommit], rule: 'landedCommit' };
9288
+ }
9289
+
9290
+ const branch = `sm-job/${job?.slug}`;
9291
+ const branchCommits = [];
9292
+ for (const sha of pathCommits) {
9293
+ try {
9294
+ await execGitAt(cwd, ['merge-base', '--is-ancestor', sha, branch], { timeout: 10_000 });
9295
+ branchCommits.push(sha);
9296
+ } catch { /* not an ancestor of this job's own branch, or branch doesn't exist */ }
9297
+ }
9298
+ if (branchCommits.length) return { commits: branchCommits, rule: 'job branch' };
9299
+
9300
+ if (job?.slug) {
9301
+ const trailerCommits = [];
9302
+ for (const sha of pathCommits) {
9303
+ try {
9304
+ const msg = await execGitAt(cwd, ['log', '-1', '--format=%B', sha], { timeout: 10_000 });
9305
+ if (msg.includes(job.slug)) trailerCommits.push(sha);
9306
+ } catch { /* unresolvable sha */ }
9307
+ }
9308
+ if (trailerCommits.length) return { commits: trailerCommits, rule: 'slug trailer' };
9309
+ }
9310
+
9311
+ return null;
9312
+ }
9313
+
9314
+ /**
9315
+ * Widened evidence check (PRD 1102, narrowed to per-job attribution by a
9316
+ * later PRD): does at least one commit ATTRIBUTABLE TO THIS JOB land AFTER
9317
+ * its run window and touch a path the PRD itself declares? Scoped to the
9318
+ * PRD's own declared paths (never the whole repo) so a sibling job's
9319
+ * unrelated commit is never even considered — see healRefusalReason's own
8386
9320
  * rationale for why unscoped, repo-wide evidence is not attribution.
8387
9321
  *
9322
+ * Path overlap alone is NOT evidence (see attributeLandedCommits's header):
9323
+ * a sibling PRD in the same Epic routinely declares the same hot file, so
9324
+ * `landedSinceRun`'s raw result is only a candidate list — the returned
9325
+ * annotation is null unless `attributeLandedCommits` narrows it to at least
9326
+ * one commit this job can actually claim.
9327
+ *
8388
9328
  * Returns null (no annotation, never fabricated) when the PRD names no
8389
- * paths — the caller then has only the existing, already-computed
8390
- * committedInWindow signal to go on, same as before this PRD.
9329
+ * paths, when no commit touches a declared path at all, or when
9330
+ * path-overlapping commits exist but none are attributable to this job — the
9331
+ * caller then has only the existing, already-computed committedInWindow
9332
+ * signal to go on, same as before this PRD.
9333
+ *
9334
+ * `fetchedCwds` (optional) lets a caller iterating many candidates in one
9335
+ * pass (reverifyNeedsReview) dedupe the `git fetch --all --prune` across
9336
+ * candidates that share a `cwd` — several `needs_review` rows for the same
9337
+ * project is the common case a backlog produces, and each fetch is up to
9338
+ * ~20s, so re-fetching the same repo once per row multiplies that pass's
9339
+ * wall-clock cost for zero new evidence. Omitted (or a fresh Set per call)
9340
+ * simply always fetches, unchanged from before this cache existed.
8391
9341
  *
8392
- * @returns {Promise<{commits: string[], paths: string[], detectedAt: string} | null>}
9342
+ * @returns {Promise<{commits: string[], paths: string[], detectedAt: string, rule: string} | null>}
8393
9343
  */
8394
- async function computeLooksDone(job) {
9344
+ async function computeLooksDone(job, fetchedCwds) {
8395
9345
  const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
8396
9346
  const paths = declaredPathsForPrd(prdPath);
8397
9347
  if (!paths.length) return null;
8398
- await fetchAllRefs(job.cwd);
9348
+ if (!fetchedCwds || !fetchedCwds.has(job.cwd)) {
9349
+ await fetchAllRefs(job.cwd);
9350
+ if (fetchedCwds) fetchedCwds.add(job.cwd);
9351
+ }
8399
9352
  const commits = await landedSinceRun(job.cwd, job.startedAt, paths);
8400
9353
  if (!commits.length) return null;
8401
- return { commits, paths, detectedAt: new Date().toISOString() };
9354
+ const attributed = await attributeLandedCommits(job, commits, job.cwd);
9355
+ if (!attributed) return null;
9356
+ return { commits: attributed.commits, paths, detectedAt: new Date().toISOString(), rule: attributed.rule };
8402
9357
  }
8403
9358
 
8404
9359
  async function reverifyNeedsReview() {
8405
9360
  const snap = await readQueue();
8406
- const candidates = snap.jobs.filter(isRescanCandidate);
9361
+ // isGuardParkedWithoutAutoFix rows are NOT isRescanCandidate (their
9362
+ // verifierVerdict is a commit-guard/shared-tree-guard verdict, not a
9363
+ // RESCANNABLE_VERDICTS transcript-verifier one) — included here so this
9364
+ // pass also computes their looksDone evidence, the widened half of the
9365
+ // guard-verdict auto-resolve gap this PRD closes. Handled in its own
9366
+ // branch below (no transcript rescan — there is no transcript verdict to
9367
+ // rescan) rather than through the isRescanCandidate machinery.
9368
+ const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j));
8407
9369
  const healed = [];
8408
9370
  const leftForReview = [];
8409
9371
  const looksDoneUpdates = [];
9372
+ // Shared across every computeLooksDone call in this one pass — dedupes
9373
+ // the `git fetch --all --prune` per distinct cwd (see computeLooksDone's
9374
+ // header) rather than re-fetching the same repo once per candidate row.
9375
+ const fetchedCwds = new Set();
8410
9376
  for (const job of candidates) {
9377
+ if (!isRescanCandidate(job) && isGuardParkedWithoutAutoFix(job)) {
9378
+ // Guard-verdict park, never auto-fixed: only evidence gathering, never
9379
+ // a transcript rescan (there was never a transcript-verifier verdict
9380
+ // here) and never a direct heal — applyNeedsReviewAutoResolve is the
9381
+ // sole place that turns this annotation into a status change.
9382
+ const looksDone = await computeLooksDone(job, fetchedCwds);
9383
+ if (looksDone) {
9384
+ looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
9385
+ }
9386
+ continue;
9387
+ }
8411
9388
  if (job.status === 'failed') {
8412
9389
  // A failed row never runs the transcript-verifier rescan below — that
8413
9390
  // machinery (verifyRun/COMPLETED_EQUIVALENT_VERDICTS) exists to
@@ -8416,7 +9393,7 @@ async function reverifyNeedsReview() {
8416
9393
  // completing-direction constraint). The only thing a failed candidate
8417
9394
  // can gain here is a looksDone annotation + a failed → needs_review
8418
9395
  // transition, for a human to confirm.
8419
- const looksDone = await computeLooksDone(job);
9396
+ const looksDone = await computeLooksDone(job, fetchedCwds);
8420
9397
  if (looksDone) {
8421
9398
  looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: true });
8422
9399
  } else {
@@ -8471,7 +9448,7 @@ async function reverifyNeedsReview() {
8471
9448
  // always before this periodic/boot pass can run against the same row, so
8472
9449
  // this check reliably catches the only order that can occur.
8473
9450
  if (stillOpen && job.autoFixAttempted !== true) {
8474
- const looksDone = await computeLooksDone(job);
9451
+ const looksDone = await computeLooksDone(job, fetchedCwds);
8475
9452
  if (looksDone) {
8476
9453
  looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
8477
9454
  }
@@ -8485,14 +9462,14 @@ async function reverifyNeedsReview() {
8485
9462
  if (!u) continue;
8486
9463
  if (u.fromFailed) {
8487
9464
  transitionJob(j, 'needs_review', {
8488
- reason: 'looks done — commit(s) since this run touch this PRD\'s declared paths; confirm before archiving',
9465
+ reason: `looks done (${u.looksDone.rule}) — commit(s) attributable to this job's own run touch this PRD's declared paths; confirm before archiving`,
8489
9466
  source: 'reverifyNeedsReview:looksDone',
8490
9467
  });
8491
9468
  }
8492
9469
  if (j.status !== 'needs_review') continue;
8493
9470
  j.looksDone = u.looksDone;
8494
9471
  const shaList = u.looksDone.commits.slice(0, 5).map((c) => c.slice(0, 7)).join(', ');
8495
- j.error = `looks done — ${u.looksDone.commits.length} commit(s) since this run touch this PRD's paths (${shaList}); confirm before archiving`;
9472
+ j.error = `looks done (${u.looksDone.rule}) — ${u.looksDone.commits.length} commit(s) attributable to this job's own run touch this PRD's paths (${shaList}); confirm before archiving`;
8496
9473
  }
8497
9474
  });
8498
9475
  console.log(`[scheduler] boot reverify: looksDone annotated for ${looksDoneUpdates.length} row(s): ${looksDoneUpdates.map((u) => u.slug).join(', ')}`);
@@ -8764,11 +9741,21 @@ function registerScheduleHandlers() {
8764
9741
  ensureDirs();
8765
9742
  supervisor.registerHandlers();
8766
9743
 
9744
+ // Cheap read only — no reconcile(), no writeQueue(). The renderer treats
9745
+ // this as a fast call behind a 5s deadline (scheduleState.ts's
9746
+ // withTimeout), but reconcile() does a cross-project PRD discovery walk
9747
+ // plus a disk write, which could blow that deadline and, worse, throw
9748
+ // outright on a torn queue.json (reconcile refuses to run against
9749
+ // `state.unreadable`) — turning a recoverable read into a rejected IPC and
9750
+ // an error toast. Discovery still runs on a fixed cadence elsewhere:
9751
+ // tickQueue (every POLL_INTERVAL_MS, 60s), rescheduleTimer, broadcast()'s
9752
+ // coalescer (getPayload), schedule:rescan and schedule:adopt-prd. Worst
9753
+ // case, a PRD dropped on disk while the Scheduler tab is open surfaces
9754
+ // here within ~POLL_INTERVAL_MS + BROADCAST_COALESCE_MS (~60.2s) — via
9755
+ // tickQueue's reconcile + its trailing broadcast() — not via this handler.
8767
9756
  ipcMain.handle('schedule:state', async () => {
8768
9757
  const state = await readQueue();
8769
- await reconcile(state);
8770
- await writeQueue(state);
8771
- return buildScheduleStatePayload(state, { withPaths: true });
9758
+ return buildScheduleStatePayload(state);
8772
9759
  });
8773
9760
 
8774
9761
  // Session-Manager-wide claude -p slot pool (lib/sessionSlots.cjs) —
@@ -8811,6 +9798,57 @@ function registerScheduleHandlers() {
8811
9798
  };
8812
9799
  });
8813
9800
 
9801
+ // Queue-health header (PRD): the one honest read of "why does the queue
9802
+ // look stale" — reuses classifyQueueHealth so the UI and the starvation
9803
+ // watchdog can never disagree. `cwd` is optional (null = machine-wide,
9804
+ // matching WindowStrip's own scopeCwd fallback).
9805
+ ipcMain.handle('schedule:queue-health', async (_e, payload) => {
9806
+ const cwd = (payload && typeof payload.cwd === 'string') ? payload.cwd : null;
9807
+ const state = await readQueue();
9808
+ if (state.unreadable) {
9809
+ return { unknown: true, reason: state.unreadable };
9810
+ }
9811
+ const now = Date.now();
9812
+ const slotSnapshot = sessionSlots.snapshot();
9813
+ const freeSlots = Math.max(0, slotSnapshot.total - slotSnapshot.inUse);
9814
+ const verdict = classifyQueueHealth({
9815
+ jobs: state.jobs,
9816
+ paused: state.paused,
9817
+ launchBlocks: state.launchBlocks,
9818
+ runningSet,
9819
+ freeSlots,
9820
+ totalSlots: slotSnapshot.total,
9821
+ lastDispatchAttemptAtMs: Date.parse(state.lastDispatchAttemptAt ?? ''),
9822
+ now,
9823
+ cwd,
9824
+ });
9825
+ // Oldest running job across the whole machine (any project) — the
9826
+ // number that actually explains slot saturation, alongside the
9827
+ // machine-wide slot pool itself.
9828
+ let oldestRunningAgeMs = null;
9829
+ for (const j of state.jobs) {
9830
+ if (j.status !== 'running' && !runningSet.has(j.slug)) continue;
9831
+ const startedAtMs = j.startedAt ? Date.parse(j.startedAt) : NaN;
9832
+ if (!Number.isFinite(startedAtMs)) continue;
9833
+ const age = now - startedAtMs;
9834
+ if (oldestRunningAgeMs === null || age > oldestRunningAgeMs) oldestRunningAgeMs = age;
9835
+ }
9836
+ return {
9837
+ unknown: false,
9838
+ now,
9839
+ verdict,
9840
+ slots: {
9841
+ inUse: slotSnapshot.inUse,
9842
+ total: slotSnapshot.total,
9843
+ free: freeSlots,
9844
+ source: slotSnapshot.envOverride ? 'env' : 'pool',
9845
+ },
9846
+ oldestRunningAgeMs,
9847
+ lastRunAt: state.lastRunAt ?? null,
9848
+ lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
9849
+ };
9850
+ });
9851
+
8814
9852
  ipcMain.handle('schedule:force-tick', async () => {
8815
9853
  // Bypass the billing-poll gate entirely — fire pending jobs immediately regardless of meter state.
8816
9854
  // Clears any existing pause first (same semantics as run-now).
@@ -8889,6 +9927,21 @@ function registerScheduleHandlers() {
8889
9927
  return { ok: true, kind: 'info', message: `Adopted ${slug} — it will run as a normal pending job` };
8890
9928
  }));
8891
9929
 
9930
+ // Scheduler UI's "change disposition" action (scheduler wave-disposition
9931
+ // PRD): promotes an appended wave to its own head, or re-attaches a head
9932
+ // behind another chain. Thin wrapper over remote.setPrdDisposition, which
9933
+ // validates the rewrite (cycle-safety, running/completed rows untouched)
9934
+ // before delegating to the same remote.updatePrd every other PRD edit
9935
+ // path uses — see that method's own comment in this file.
9936
+ ipcMain.handle('schedule:set-prd-disposition', validated(schemas.scheduleSetPrdDisposition, async ({ slug, cwd, disposition, dependsOn }) => {
9937
+ if (!(await safeSlugPath(slug))) return { ok: false, kind: 'error', message: 'invalid slug' };
9938
+ const result = await remote.setPrdDisposition({ slug, cwd, disposition, dependsOn });
9939
+ if (!result.ok) return { ok: false, kind: 'error', message: result.error ?? 'disposition change failed' };
9940
+ appendAuditEvent('scheduler_prd_disposition_set', { slug, cwd: cwd ?? null, disposition, source: 'ipc:schedule:set-prd-disposition' });
9941
+ await broadcast({ flush: true });
9942
+ return { ok: true, kind: 'info', message: `${slug} is now ${disposition === 'new-head' ? 'an independent head' : 'attached behind the chosen chain'}` };
9943
+ }));
9944
+
8892
9945
  ipcMain.handle('schedule:run-now', async () => {
8893
9946
  // Manual run-now overrides any auto-pause. Clear it first.
8894
9947
  await clearPause('run-now');
@@ -8901,9 +9954,11 @@ function registerScheduleHandlers() {
8901
9954
  return { ok: true };
8902
9955
  });
8903
9956
 
8904
- // Re-scan prds/ folder and merge into queue.json. The `schedule:state`
8905
- // handler already reconciles on read, but this gives the renderer an
8906
- // explicit refresh path that also broadcasts so all views update.
9957
+ // Re-scan prds/ folder and merge into queue.json. `schedule:state` is a
9958
+ // cheap read with no reconcile of its own — this is the renderer's
9959
+ // explicit, immediate discovery path (mutate() + reconcile() + broadcast())
9960
+ // for "I just dropped a PRD on disk and want it to show up now" rather than
9961
+ // waiting for tickQueue's next ~60s pass.
8907
9962
  ipcMain.handle('schedule:rescan', async () => {
8908
9963
  const { added, removed } = await mutate(async (state) => {
8909
9964
  const before = new Set(state.jobs.map((j) => j.slug));
@@ -9132,6 +10187,18 @@ async function init() {
9132
10187
  const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
9133
10188
  bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
9134
10189
  }
10190
+ // Same evidence-before-failure gate reapDeadRunningJobs applies, resolved
10191
+ // BEFORE mutate() for the same reason (git spawn work must never run
10192
+ // inside mutate()'s single global serialization chain) — an orphaned job
10193
+ // classified 'failed'/'unknown' from its log tail alone can still have
10194
+ // actually landed a real commit before the app restarted mid-run.
10195
+ const bootLandedCommitEvidence = new Map();
10196
+ await Promise.all(bootSnap.jobs.map(async (j) => {
10197
+ if (!immediateSlugs.includes(j.slug) || j.status !== 'running') return;
10198
+ if (bootOutcomes.get(j.slug) === 'success' || !j.landedCommit) return;
10199
+ const resolved = await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt);
10200
+ if (resolved) bootLandedCommitEvidence.set(j.slug, j.landedCommit);
10201
+ }));
9135
10202
  const bootReconciledCompletions = [];
9136
10203
  await mutate((state) => {
9137
10204
  for (const j of state.jobs) {
@@ -9139,7 +10206,7 @@ async function init() {
9139
10206
  const outcome = bootOutcomes.get(j.slug) ?? 'unknown';
9140
10207
  const pid = j.runtime?.pid;
9141
10208
  const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
9142
- applyOrphanOutcome(j, outcome, killNote);
10209
+ applyOrphanOutcome(j, outcome, killNote, bootLandedCommitEvidence.get(j.slug) || null);
9143
10210
  if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
9144
10211
  console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
9145
10212
  }
@@ -9163,9 +10230,18 @@ async function init() {
9163
10230
  if (result === 'killed') {
9164
10231
  console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
9165
10232
  }
9166
- setTimeout(() => {
10233
+ setTimeout(async () => {
9167
10234
  const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
9168
10235
  const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
10236
+ // Same evidence-before-failure gate as the immediate-orphan path
10237
+ // above, resolved before mutate() for the same reason (git spawn
10238
+ // work must never run inside mutate()'s serialization chain). Uses
10239
+ // the captured pre-kill snapshot's landedCommit/cwd/startedAt — the
10240
+ // race guard below already confirms `cur` is still this same run
10241
+ // (runId === bootRunId) before this evidence is applied.
10242
+ const confirmedLandedCommit = (outcome !== 'success' && j.landedCommit)
10243
+ ? (await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt) ? j.landedCommit : null)
10244
+ : null;
9169
10245
  let deferredCompletedCwd;
9170
10246
  mutate((state) => {
9171
10247
  const cur = state.jobs.find((x) => x.slug === slug);
@@ -9174,7 +10250,7 @@ async function init() {
9174
10250
  // that new run is not the boot orphan we SIGTERM'd and must not be
9175
10251
  // touched by this stale classification.
9176
10252
  if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
9177
- applyOrphanOutcome(cur, outcome, killNote);
10253
+ applyOrphanOutcome(cur, outcome, killNote, confirmedLandedCommit);
9178
10254
  console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
9179
10255
  deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
9180
10256
  }).then(() => {
@@ -9610,7 +10686,9 @@ async function listPrdsInternal() {
9610
10686
  estimateMinutes: parsed.estimateMinutes,
9611
10687
  sourcePromptId: parsed.sourcePromptId,
9612
10688
  epicId: parsed.epicId ?? null,
10689
+ dependsOn: parsed.dependsOn ?? null,
9613
10690
  agentType: parsed.agentType ?? null,
10691
+ disposition: parsed.disposition ?? null,
9614
10692
  mtimeMs: stat.mtimeMs,
9615
10693
  archived,
9616
10694
  };
@@ -10007,6 +11085,33 @@ const remote = {
10007
11085
  }
10008
11086
  },
10009
11087
 
11088
+ // Backs the Scheduler UI's "change disposition" action (scheduler
11089
+ // wave-disposition PRD): promoting an appended wave to its own head, or
11090
+ // re-attaching a head behind another chain. `dependsOn` for a 'new-head'
11091
+ // disposition is ignored (cleared unconditionally); for 'append' it's the
11092
+ // caller's chosen target chain's terminal slug(s) — the renderer computes
11093
+ // that from the SAME backlog tree (lib/backlogTree.ts) it already renders,
11094
+ // so this function only has to validate the rewrite is safe, never
11095
+ // re-derive "the" terminal itself.
11096
+ //
11097
+ // Validates via prdDisposition.cjs's computeDispositionRewrite (row not
11098
+ // running/completed, no already-satisfied blocker being rewritten out from
11099
+ // under it, no dependsOn cycle) BEFORE delegating the actual write to this
11100
+ // SAME updatePrd — so a rejected rewrite never reaches the filesystem, and
11101
+ // an accepted one gets updatePrd's own dependsOn FK re-validation for free.
11102
+ async setPrdDisposition({ slug, cwd, disposition, dependsOn }) {
11103
+ let listing;
11104
+ try {
11105
+ listing = await this.listPrds({ cwd, fields: 'full', limit: Number.MAX_SAFE_INTEGER });
11106
+ } catch (e) {
11107
+ return { ok: false, error: `could not read project PRDs: ${e?.message ?? e}` };
11108
+ }
11109
+ const rows = listing.prds ?? [];
11110
+ const rewrite = computeDispositionRewrite({ slug, disposition, dependsOn: dependsOn ?? [], rows });
11111
+ if (!rewrite.ok) return rewrite;
11112
+ return this.updatePrd({ slug, cwd, frontmatter: { dependsOn: rewrite.dependsOn, disposition } });
11113
+ },
11114
+
10010
11115
  // Cancels a job that hasn't finished yet. A 'running' job's process group
10011
11116
  // is SIGTERM'd (reusing killOrphanClaudePid — the same kill path boot
10012
11117
  // reconciliation uses for an orphaned running job) before its queue row is
@@ -10034,18 +11139,38 @@ const remote = {
10034
11139
  if (wasRunning && pid) {
10035
11140
  killOrphanClaudePid(pid);
10036
11141
  }
11142
+ // Evidence-before-failure guard, scoped to an actually-running job being
11143
+ // killed here (a 'pending' cancel has no live process, so nothing new
11144
+ // could have landed since its last stamp — and 'needs_review' is not
11145
+ // even a legal transition from 'pending', see LEGAL_TRANSITIONS): the
11146
+ // same reapDeadRunningJobs evidence gate (job 1192 — a landedCommit
11147
+ // being non-empty is not proof by itself, but discarding proof of real
11148
+ // landed work with no check at all is worse) applies here too. A
11149
+ // dead-pid reap of a job that landed a commit (e.g. via the
11150
+ // dispatch-time sidecar backfill) is routed to needs_review/completed;
11151
+ // a deliberate cancel of that same state deserves no less.
11152
+ const confirmedLandedCommit = (wasRunning && job.landedCommit)
11153
+ ? ((await resolveLandedCommitEvidence(job.cwd || DEFAULT_PROJECT_CWD, job.landedCommit, job.startedAt))
11154
+ ? job.landedCommit
11155
+ : null)
11156
+ : null;
11157
+ const targetStatus = confirmedLandedCommit ? 'needs_review' : 'failed';
11158
+ const cancelReason = confirmedLandedCommit
11159
+ ? `cancelled via admin API, but landedCommit ${confirmedLandedCommit} resolves — verify before treating as done`
11160
+ : 'cancelled via admin API';
10037
11161
  await mutate((s) => {
10038
11162
  const idx = s.jobs.findIndex((j) => j.slug === slug);
10039
11163
  if (idx < 0) return;
10040
11164
  const j = s.jobs[idx];
10041
- transitionJob(j, 'failed', { reason: 'cancelled via admin API', source: 'remote:cancelJob' });
10042
- j.error = 'cancelled via admin API';
11165
+ transitionJob(j, targetStatus, { reason: cancelReason, source: 'remote:cancelJob' });
11166
+ j.error = cancelReason;
10043
11167
  j.finishedAt = new Date().toISOString();
10044
11168
  j.exitCode = j.exitCode ?? null;
11169
+ if (confirmedLandedCommit) j.verifierVerdict = 'cancelled_with_landed_commit';
10045
11170
  delete j.runtime;
10046
11171
  });
10047
11172
  await broadcast({ flush: true });
10048
- return { ok: true, slug, status: 'failed', wasRunning, cwd: job.cwd ?? null };
11173
+ return { ok: true, slug, status: targetStatus, wasRunning, cwd: job.cwd ?? null };
10049
11174
  },
10050
11175
 
10051
11176
  // Exposes the module-level allocateParallelGroup (PRD 548) to callers that
@@ -10092,6 +11217,7 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
10092
11217
  module.exports = {
10093
11218
  classifyQueueStarvation,
10094
11219
  classifyQueueStarvationByProject,
11220
+ classifyQueueHealth,
10095
11221
  runQueueStarvationWatchdog,
10096
11222
  QUEUE_STARVATION_MS,
10097
11223
  selectStarveEscalations,
@@ -10102,6 +11228,15 @@ module.exports = {
10102
11228
  findOverrunningJobs,
10103
11229
  JOB_OVERRUN_FACTOR,
10104
11230
  JOB_OVERRUN_FLOOR_MS,
11231
+ computeJobBudgetMs,
11232
+ classifyBudgetKill,
11233
+ isJobBudgetExempt,
11234
+ shouldKillForBudget,
11235
+ resolveBudgetKillOutcome,
11236
+ JOB_BUDGET_FACTOR,
11237
+ JOB_BUDGET_FLOOR_MS,
11238
+ JOB_BUDGET_CEILING_MS,
11239
+ BUDGET_WARNING_FRACTION,
10105
11240
  registerScheduleHandlers,
10106
11241
  attachWindow,
10107
11242
  init,
@@ -10115,10 +11250,12 @@ module.exports = {
10115
11250
  healRefusalReason,
10116
11251
  writeQueue,
10117
11252
  reconcile,
11253
+ broadcast,
10118
11254
  reconcileSourcePromptId,
10119
11255
  allocateParallelGroup,
10120
11256
  selectHistoryJobs,
10121
11257
  parsePorcelain,
11258
+ parsePorcelainEntries,
10122
11259
  FINISH_PROTOCOL,
10123
11260
  IDLE_OUTPUT_KILL_MS,
10124
11261
  BASH_DEFAULT_TIMEOUT_MS,
@@ -10148,6 +11285,7 @@ module.exports = {
10148
11285
  isRescanCandidate,
10149
11286
  isFailedUnverifiedShaped,
10150
11287
  computeLooksDone,
11288
+ attributeLandedCommits,
10151
11289
  isPromotableOriginal,
10152
11290
  selectAutoFixTargets,
10153
11291
  applyRcaClassification,
@@ -10155,6 +11293,9 @@ module.exports = {
10155
11293
  resolveRunId,
10156
11294
  isUnresolvableNeedsReview,
10157
11295
  isExhaustedAutoFix,
11296
+ GUARD_VERDICT_EVIDENCE_ELIGIBLE,
11297
+ isGuardParkedWithoutAutoFix,
11298
+ isEligibleForNeedsReviewAutoResolve,
10158
11299
  isPlanUnqueued,
10159
11300
  isFixPlanDead,
10160
11301
  fixSlugFor,
@@ -10240,6 +11381,7 @@ module.exports = {
10240
11381
  evaluateSharedTreeGuard,
10241
11382
  checkSharedTreeGuard,
10242
11383
  uncommittedChanges,
11384
+ uncommittedChangesWithStatus,
10243
11385
  gitHead,
10244
11386
  isBranchAlreadyIntegrated,
10245
11387
  selectResumeRecoveryTarget,