claude-code-session-manager 0.79.0 → 0.81.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/dist/assets/{AgentLibrary-COtVRqBR.js → AgentLibrary-psZYVM2w.js} +1 -1
  2. package/dist/assets/{DataModel-CSEKw_OR.js → DataModel-BMied5pg.js} +1 -1
  3. package/dist/assets/{History-CHHovrAO.js → History-BFC0oaKc.js} +1 -1
  4. package/dist/assets/{Hooks-BZU6C3x6.js → Hooks-CTLfO9G8.js} +1 -1
  5. package/dist/assets/{HostBilko-CqTUoq37.js → HostBilko-D4I0Cpwn.js} +1 -1
  6. package/dist/assets/{Library-BtxdyTLz.js → Library-BTzS8KsS.js} +1 -1
  7. package/dist/assets/{ListDetail-qZc7Zm-6.js → ListDetail-CuRmT008.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-BHe_4fJR.js → MarkdownEditor-BgQGtvWo.js} +1 -1
  9. package/dist/assets/{McpServers-7Z98HLNo.js → McpServers-DiVQUK57.js} +1 -1
  10. package/dist/assets/{Memory-CR72KoyP.js → Memory-DSu15JmL.js} +1 -1
  11. package/dist/assets/{Panel-pL6H3dpQ.js → Panel-CD5wxGSR.js} +1 -1
  12. package/dist/assets/{Permissions-CWSWjyXM.js → Permissions-C_tAE6yJ.js} +1 -1
  13. package/dist/assets/{Plugins-CN6lX2lt.js → Plugins-kwI7W-eK.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-BXSXwIsk.js → ProvenanceBadge-CQceOgsH.js} +1 -1
  15. package/dist/assets/{SaveBar-BlB5TGpR.js → SaveBar-BDk5e3Pp.js} +1 -1
  16. package/dist/assets/{Scheduler-DRciWUmR.js → Scheduler-DRnvEYzv.js} +7 -7
  17. package/dist/assets/{ScopeSwitcher-kFrXtjpr.js → ScopeSwitcher-CeifaOlq.js} +1 -1
  18. package/dist/assets/{Settings-BXuyf4lJ.js → Settings-BWQ1Utop.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-Dfacb0PE.js → SkillReferenceGraph-CW6e1SW1.js} +1 -1
  20. package/dist/assets/{Skills-CHqcpiyt.js → Skills-BLwFB0E4.js} +1 -1
  21. package/dist/assets/{SystemPrompt-fxXm0BZr.js → SystemPrompt-DC4ZTArJ.js} +1 -1
  22. package/dist/assets/{TagLibrary-DOz65ZTz.js → TagLibrary-pMeGfjUb.js} +1 -1
  23. package/dist/assets/{TiptapBody-D0bWx_9o.js → TiptapBody-BHFid2pZ.js} +1 -1
  24. package/dist/assets/{Toggle-C9jBwGSx.js → Toggle-D9eYoZh4.js} +1 -1
  25. package/dist/assets/{index-DPYa6jbM.js → index-CuyM9vAP.js} +5 -5
  26. package/dist/assets/{settingsSchema-BTPw1bR3.js → settingsSchema-DrxC67uZ.js} +1 -1
  27. package/dist/index.html +1 -1
  28. package/package.json +1 -1
  29. package/plugins/session-manager-dev/skills/builder/3-publish/SKILL.md +10 -0
  30. package/plugins/session-manager-dev/skills/develop/standards.md +1 -0
  31. package/src/main/__tests__/computeDepHistorySatisfaction.test.cjs +66 -0
  32. package/src/main/__tests__/prdCreate.test.cjs +133 -8
  33. package/src/main/__tests__/prdFrontmatterDependsOn.test.cjs +136 -0
  34. package/src/main/__tests__/prdUpdateDependsOn.test.cjs +160 -0
  35. package/src/main/__tests__/queueHistory.test.cjs +33 -0
  36. package/src/main/__tests__/rcaReport.test.cjs +24 -0
  37. package/src/main/__tests__/runVerify-blocked-by-foreign-wip.test.cjs +58 -0
  38. package/src/main/__tests__/runVerify-policy-denial.test.cjs +89 -0
  39. package/src/main/__tests__/scheduleJobTransitions.test.cjs +1 -0
  40. package/src/main/__tests__/scheduler-already-satisfied-on-main.test.cjs +105 -0
  41. package/src/main/__tests__/scheduler-autofix-outcome.test.cjs +73 -1
  42. package/src/main/__tests__/scheduler-autofix-select.test.cjs +17 -0
  43. package/src/main/__tests__/scheduler-blocked-by-foreign-wip.test.cjs +107 -0
  44. package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +20 -0
  45. package/src/main/__tests__/scheduler-finalize-dispatch-guards.test.cjs +229 -0
  46. package/src/main/__tests__/scheduler-leftover-quarantine.test.cjs +199 -0
  47. package/src/main/__tests__/scheduler-looks-done.test.cjs +141 -1
  48. package/src/main/__tests__/scheduler-mechanical-recovery.test.cjs +222 -0
  49. package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +81 -0
  50. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +205 -2
  51. package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +51 -0
  52. package/src/main/__tests__/scheduler-resume-recovery.test.cjs +254 -0
  53. package/src/main/__tests__/schedulerBatchRootBlocker.test.cjs +117 -0
  54. package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -2
  55. package/src/main/ipcSchemas.cjs +15 -1
  56. package/src/main/lib/__tests__/branchSweep.test.cjs +164 -0
  57. package/src/main/lib/__tests__/fixtures/204-mercury-steam-horse.log.txt +13 -0
  58. package/src/main/lib/__tests__/gitWorktree.test.cjs +129 -10
  59. package/src/main/lib/__tests__/landedSinceRun.test.cjs +61 -1
  60. package/src/main/lib/__tests__/rateLimitWindow.test.cjs +88 -0
  61. package/src/main/lib/__tests__/reaperHelpers.test.cjs +120 -1
  62. package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +59 -7
  63. package/src/main/lib/branchSweep.cjs +127 -0
  64. package/src/main/lib/depSlugResolve.cjs +72 -0
  65. package/src/main/lib/epicWorktreeMerge.cjs +3 -3
  66. package/src/main/lib/epicWorktreeMint.cjs +17 -5
  67. package/src/main/lib/fixPlanSlug.cjs +62 -0
  68. package/src/main/lib/gitWorktree.cjs +117 -12
  69. package/src/main/lib/landedSinceRun.cjs +41 -1
  70. package/src/main/lib/mcpToolCatalog.cjs +4 -1
  71. package/src/main/lib/prdCreate.cjs +84 -5
  72. package/src/main/lib/prdFrontmatter.cjs +56 -8
  73. package/src/main/lib/queueHistory.cjs +50 -5
  74. package/src/main/lib/rateLimitWindow.cjs +62 -0
  75. package/src/main/lib/rcaReport.cjs +18 -3
  76. package/src/main/lib/reaperHelpers.cjs +169 -3
  77. package/src/main/lib/scheduleJobTransitions.cjs +28 -5
  78. package/src/main/lib/schedulerBatch.cjs +181 -23
  79. package/src/main/runVerify.cjs +71 -3
  80. package/src/main/scheduler/prdParser.cjs +7 -0
  81. package/src/main/scheduler.cjs +1554 -71
  82. package/src/preload/api.d.ts +8 -0
@@ -57,9 +57,14 @@ const { resolveClaudeBin, probeClaudeVersion } = require('./lib/claudeBin.cjs');
57
57
  const launchFailure = require('./lib/launchFailure.cjs');
58
58
  const { appendError } = require('./lib/opsErrorLog.cjs');
59
59
  const { readTail } = require('./lib/fileTail.cjs');
60
- const { claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
60
+ const {
61
+ claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs,
62
+ findLiveProcessForJob, logHasOutput, resolvePidlessGateOutcome, resolveCommitGuardOutcome,
63
+ } = require('./lib/reaperHelpers.cjs');
64
+ const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
61
65
  const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
62
66
  const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
67
+ const { resolveBindingRateLimitReset } = require('./lib/rateLimitWindow.cjs');
63
68
  const { computeQueueHealth } = require('./lib/queueHealth.cjs');
64
69
  const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
65
70
  const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
@@ -71,9 +76,10 @@ const { enqueueExternalPrompt } = require('./chatRunner.cjs');
71
76
  const { appendResponseEventIfKnown } = require('./promptSessionEvents.cjs');
72
77
  const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs');
73
78
  const promptSessionTranscript = require('./promptSessionTranscript.cjs');
74
- const { verifyRun } = require('./runVerify.cjs');
79
+ const { verifyRun, parseLog, scanSentinel, scanForeignWipPathsClaim } = require('./runVerify.cjs');
75
80
  const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
76
- const { landedSinceRun } = require('./lib/landedSinceRun.cjs');
81
+ const { isFixPlanSlug, classifyDiscoveredFixPlan, resolveIsFixPlan } = require('./lib/fixPlanSlug.cjs');
82
+ const { landedSinceRun, landedOnMainSince } = require('./lib/landedSinceRun.cjs');
77
83
  const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
78
84
  const logs = require('./logs.cjs');
79
85
  const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
@@ -99,7 +105,7 @@ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
99
105
  const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
100
106
  ? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
101
107
  : JOB_OVERRUN_FLOOR_MS_DEFAULT;
102
- const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
108
+ const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD, DEP_HISTORY_FAIL_OPEN } = require('./lib/schedulerBatch.cjs');
103
109
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
104
110
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
105
111
  const queueHistory = require('./lib/queueHistory.cjs');
@@ -140,6 +146,7 @@ const jobWorktree = require('./lib/jobWorktree.cjs');
140
146
  const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
141
147
  const queueStore = require('./lib/queueStore.cjs');
142
148
  const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
149
+ const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
143
150
  const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
144
151
  const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
145
152
 
@@ -266,9 +273,10 @@ deferring it.
266
273
  Your own work must still never be left uncommitted — this only changes
267
274
  which paths get staged, never whether you commit.
268
275
  5. VERDICT SENTINEL — as the LAST LINE of your final result text, emit exactly
269
- one of these two lines (no trailing text after it):
276
+ one of these lines (no trailing text after it):
270
277
  SCHEDULER_VERDICT: PASS
271
278
  SCHEDULER_VERDICT: FAIL <one-line reason>
279
+ SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP
272
280
  Print PASS only when the AC gate is green AND the commit from step 4 landed.
273
281
  Print FAIL (and exit 1) if the AC gate was red or the commit could not land.
274
282
  NEVER print PASS on a red AC gate — a lying PASS turns the verifier from a
@@ -276,6 +284,22 @@ deferring it.
276
284
  landed commit lets the verifier override incidental transcript noise (grep
277
285
  results containing "Error", a TDD red-test run early in the session, debug
278
286
  Tracebacks) so those do not false-trip a needs_review downgrade.
287
+ Print BLOCKED_BY_FOREIGN_WIP (and exit 1) ONLY when your own AC gate failed
288
+ because it ran against a SIBLING job's in-flight, uncommitted file — never
289
+ because of your own regression — AND every failing path is one this prompt
290
+ already disclosed to you as foreign (see the "FOREIGN WORKING-TREE STATE"
291
+ section above, if present). It MUST be accompanied by a second line naming
292
+ every such path:
293
+ FOREIGN_WIP_PATHS: <path1>, <path2>, ...
294
+ The scheduler independently validates every listed path against the exact
295
+ foreign-WIP manifest it disclosed to you. Only list a path that (a) this
296
+ prompt already told you is foreign WIP, not yours, AND (b) is why your own
297
+ AC gate failed — never a path you own, and never a path that failed for
298
+ some other reason of your own making. Any listed path NOT in that manifest
299
+ downgrades this whole verdict back to FAIL, with your claim rejected. This
300
+ is not an escape hatch for your own broken code: claiming it for a
301
+ regression you introduced, or for a path you never received as foreign
302
+ WIP, is a lying verdict exactly like a false PASS.
279
303
 
280
304
  A job that exits with uncommitted changes is treated as INCOMPLETE and flagged
281
305
  for review. Do NOT add work beyond the acceptance criteria — this protocol is the
@@ -569,6 +593,52 @@ async function computeCommittedDuringRun(cwd, headBefore, headAfter, startedAt,
569
593
  return module.exports.committedInWindow(cwd, startedAt, untilIso);
570
594
  }
571
595
 
596
+ /**
597
+ * Read-only check: is a job's `sm-job/<slug>` worktree branch already fully
598
+ * integrated into `cwd`'s current HEAD — i.e. every commit on the branch is
599
+ * an ancestor of (or equal to) HEAD? Mirrors gitWorktree.cjs's own
600
+ * integrateBranch() no-op detection (`mergeBase === branchHead`) exactly, but
601
+ * deliberately never calls integrateBranch itself: this is used by
602
+ * reapDeadRunningJobs to PROVE a dead job's work already landed, never to
603
+ * perform the landing — attempting a real merge from an audit check is a
604
+ * mutating action with its own conflict/timing risk, and "land the still-
605
+ * stranded branch" is explicitly a separate, later decision (mechanical
606
+ * recovery, or a human), not something a reap pass should do on its own
607
+ * initiative. See reapDeadRunningJobs's own header comment for why the
608
+ * conservative failure direction here is needs_review, never a silent merge.
609
+ *
610
+ * Returns:
611
+ * - true — branch exists and is fully contained in HEAD (already
612
+ * integrated by a prior pass, or a genuine no-op branch that never
613
+ * diverged from its base — both cases legitimately "landed").
614
+ * - false — branch exists and still holds commits not in HEAD (a real
615
+ * stranded deliverable, e.g. the PRD 1118 shape).
616
+ * - null — branch does not exist at all (never a worktree run — an
617
+ * in-place run, or SM_JOB_WORKTREE_DISABLE), or the git calls errored.
618
+ * Never throws.
619
+ */
620
+ function isBranchAlreadyIntegrated(cwd, branch) {
621
+ return new Promise((resolve) => {
622
+ if (!cwd || !branch) { resolve(null); return; }
623
+ execFile(
624
+ 'git', ['-C', cwd, 'rev-parse', '--verify', branch],
625
+ { timeout: 10_000, windowsHide: true },
626
+ (err, branchHeadOut) => {
627
+ if (err) { resolve(null); return; } // branch doesn't exist — not a worktree run
628
+ const branchHead = String(branchHeadOut || '').trim();
629
+ execFile(
630
+ 'git', ['-C', cwd, 'merge-base', 'HEAD', branch],
631
+ { timeout: 10_000, windowsHide: true },
632
+ (mbErr, mbOut) => {
633
+ const mergeBase = mbErr ? '' : String(mbOut || '').trim();
634
+ resolve(mergeBase !== '' && mergeBase === branchHead);
635
+ },
636
+ );
637
+ },
638
+ );
639
+ });
640
+ }
641
+
572
642
  /**
573
643
  * Override for a SIGTERM'd (143) run when a commit landed in its window.
574
644
  * Exit 143 alone doesn't prove the deliverable is missing — the 776/779
@@ -590,6 +660,74 @@ function classifySigtermWithCommit(exitCode, commitFoundInWindow) {
590
660
  };
591
661
  }
592
662
 
663
+ /**
664
+ * Pure decision for spawnJob's finalize mutate (2026-09-06 incident): given
665
+ * the row this run is trying to finalize (or its absence) and this run's own
666
+ * identity, decide whether the finalize must be DROPPED instead of applied.
667
+ * Never mutates anything itself — callers apply `stampLandedCommit` and emit
668
+ * the audit event/log line.
669
+ *
670
+ * - row missing entirely (i2 === -1 upstream): always drop, nothing to
671
+ * stamp — there is no row left to carry the fact.
672
+ * - row present but not 'running': drop the STATUS change (a cancelled row,
673
+ * PRD 1024, must never be re-legalized to completed/needs_review by a
674
+ * stale exit handler) but still let a genuinely-landed commit be stamped
675
+ * as a fact — never a transition — when this run owns the row's current
676
+ * runId, or the row doesn't already carry a newer one of its own.
677
+ * - row present and 'running': not a drop; caller proceeds to finalize.
678
+ */
679
+ function evaluateFinalizeDrop({ rowExists, rowStatus, rowRunId, rowLandedCommit, runId, landedCommit }) {
680
+ if (!rowExists) {
681
+ return { drop: true, reason: 'row-missing', stampLandedCommit: null };
682
+ }
683
+ if (rowStatus !== 'running') {
684
+ const stampLandedCommit = landedCommit && (rowRunId === runId || !rowLandedCommit) ? landedCommit : null;
685
+ return { drop: true, reason: 'row-not-running', stampLandedCommit };
686
+ }
687
+ return { drop: false, reason: null, stampLandedCommit: null };
688
+ }
689
+
690
+ /**
691
+ * Pure decision for spawnJob's dispatch mutate: given a PENDING row about to
692
+ * be dispatched and the newest terminalRunOutcome sidecar result for its
693
+ * slug, decide whether this dispatch should be SKIPPED because a prior run
694
+ * already completed this exact slug — the gap left by reconcile()'s own
695
+ * anti-resurrection guard, which only ever sees a slug BEFORE it first lands
696
+ * in s.jobs (2026-09-06 incident: an existing pending row was re-dispatched
697
+ * three times against already-shipped work).
698
+ *
699
+ * Compares the sidecar's finishedAt against THIS row's own last pending
700
+ * transition (or queuedAt, when the row has never been reset) rather than
701
+ * merely "a completed sidecar exists somewhere" — so a deliberate human
702
+ * re-queue of the same slug in a LATER episode (whose queuedAt/pending
703
+ * transition postdates the earlier completion) still runs.
704
+ */
705
+ function evaluateDispatchSidecarReconcile({ rowStatus, rowRunId, statusHistory, queuedAt, outcome }) {
706
+ if (rowStatus !== 'pending') return { skip: false };
707
+ if (!outcome || outcome.status !== 'completed') return { skip: false };
708
+ if (outcome.runId === rowRunId) return { skip: false };
709
+ const lastPendingEntry = Array.isArray(statusHistory)
710
+ ? [...statusHistory].reverse().find((h) => h.to === 'pending')
711
+ : null;
712
+ const lastPendingAt = lastPendingEntry?.at ?? queuedAt ?? null;
713
+ if (lastPendingAt && (!outcome.finishedAt || outcome.finishedAt < lastPendingAt)) {
714
+ return { skip: false };
715
+ }
716
+ return { skip: true, runId: outcome.runId, finishedAt: outcome.finishedAt };
717
+ }
718
+
719
+ /**
720
+ * Pure detector for mutate()'s write-path regression check: transitionJob
721
+ * only ever APPENDS to statusHistory (capped at STATUS_HISTORY_CAP, never
722
+ * shrunk below it once reached) — so a running->pending change whose
723
+ * statusHistory got SHORTER is not a real transition, it's evidence this
724
+ * row's prior state was lost (e.g. a finalize mutate working off a stale
725
+ * in-memory snapshot).
726
+ */
727
+ function isQueueRowRegression({ statusBefore, statusAfter, historyLenBefore, historyLenAfter }) {
728
+ return statusBefore === 'running' && statusAfter === 'pending' && historyLenAfter < historyLenBefore;
729
+ }
730
+
593
731
  const ROOT = path.join(os.homedir(), '.claude', 'session-manager', 'scheduled-plans');
594
732
  const PRDS_DIR = path.join(ROOT, 'prds');
595
733
  const RUNS_DIR = path.join(ROOT, 'runs');
@@ -1563,7 +1701,42 @@ function mutate(fn) {
1563
1701
  if (state.unreadable) {
1564
1702
  throw new Error(`queue mutation skipped: queue.json unreadable (${state.unreadable})`);
1565
1703
  }
1704
+ const beforeBySlug = new Map(
1705
+ (state.jobs || []).map((j) => [
1706
+ j.slug,
1707
+ { status: j.status, historyLen: Array.isArray(j.statusHistory) ? j.statusHistory.length : 0 },
1708
+ ]),
1709
+ );
1566
1710
  const ret = await fn(state);
1711
+ // Detection-only regression check: transitionJob only ever APPENDS to
1712
+ // statusHistory (capped at STATUS_HISTORY_CAP, never shrunk below it) —
1713
+ // so a running->pending change whose statusHistory got SHORTER is not a
1714
+ // real transition, it's evidence this row's prior state was lost (e.g. a
1715
+ // finalize mutate working off a stale in-memory snapshot). Never
1716
+ // repaired here — mutate()'s job is to persist, not reconcile.
1717
+ for (const j of state.jobs || []) {
1718
+ const prior = beforeBySlug.get(j.slug);
1719
+ if (!prior) continue;
1720
+ const afterLen = Array.isArray(j.statusHistory) ? j.statusHistory.length : 0;
1721
+ if (isQueueRowRegression({
1722
+ statusBefore: prior.status,
1723
+ statusAfter: j.status,
1724
+ historyLenBefore: prior.historyLen,
1725
+ historyLenAfter: afterLen,
1726
+ })) {
1727
+ console.error(
1728
+ `[scheduler] queue row regressed: slug=${j.slug} running->pending with `
1729
+ + `statusHistory ${prior.historyLen} -> ${afterLen}`,
1730
+ );
1731
+ appendAuditEvent('queue_row_regressed', {
1732
+ slug: j.slug,
1733
+ from: 'running',
1734
+ to: 'pending',
1735
+ historyBefore: prior.historyLen,
1736
+ historyAfter: afterLen,
1737
+ });
1738
+ }
1739
+ }
1567
1740
  await writeQueue(state);
1568
1741
  return ret;
1569
1742
  });
@@ -2072,11 +2245,24 @@ async function reconcile(state) {
2072
2245
  exitCode: null,
2073
2246
  error: null,
2074
2247
  };
2075
- // Newly-discovered fix-plan PRD: stamp its investigationDepth relative to
2076
- // the original job it heals, so selectAutoFixTargets/spawnInvestigation
2077
- // can bound the fix-of-a-fix recursion (see MAX_INVESTIGATION_DEPTH).
2078
- // Non-fix-plan jobs get no explicit field — they read as depth 1 via `?? 1`.
2079
- if (isFixPlanSlug(slug)) {
2248
+ // Fix-plan classification (PRD 1131): a freshly-discovered PRD is a
2249
+ // genuine scheduler-authored fix plan only when its OWN provenance says
2250
+ // so — an explicit isFixPlan:true stamp (spawnInvestigation's prompt
2251
+ // template) or the absence of any createdVia stamp at all (legacy
2252
+ // fallback, matching the "no provenance = trust the name" rule the
2253
+ // quarantine gate below already applies) — never merely because the
2254
+ // slug looks like one. See lib/fixPlanSlug.cjs's header for why (PRD
2255
+ // 1126: a scheduler_create_prd-authored PRD whose slug happened to start
2256
+ // with "fix-" was wrongly stamped investigationDepth before it ever ran).
2257
+ // Persisted onto the queue row so every later consumer
2258
+ // (commitGuardVerdict, isFixPlanBeyondDepthCap, the fix-plan-completion
2259
+ // checks) reads this stamp instead of re-deriving it from the name.
2260
+ entry.isFixPlan = classifyDiscoveredFixPlan(p, slug);
2261
+ // Stamp investigationDepth relative to the original job it heals, so
2262
+ // selectAutoFixTargets/spawnInvestigation can bound the fix-of-a-fix
2263
+ // recursion (see MAX_INVESTIGATION_DEPTH). Non-fix-plan jobs get no
2264
+ // explicit field — they read as depth 1 via `?? 1`.
2265
+ if (entry.isFixPlan) {
2080
2266
  const parent = healTargetForFix(slug, state.jobs);
2081
2267
  entry.investigationDepth = parent ? (parent.investigationDepth ?? 1) + 1 : 2;
2082
2268
  }
@@ -2088,8 +2274,9 @@ async function reconcile(state) {
2088
2274
  // guard-prd-writes.cjs PreToolUse hook should have denied. Fix-plan PRDs
2089
2275
  // are exempt: spawnInvestigation's own probe writes them directly by
2090
2276
  // design (a trusted, scheduler-spawned internal loop, not an
2091
- // agent/human authoring a PRD), matching the isFixPlanSlug convention
2092
- // used everywhere else this distinction matters.
2277
+ // agent/human authoring a PRD) — entry.isFixPlan (just classified above)
2278
+ // is the provenance-aware verdict for that exemption now, not a raw
2279
+ // isFixPlanSlug name check.
2093
2280
  //
2094
2281
  // Quarantine is loud and reversible, never a silent skip (see the
2095
2282
  // 2026-08-01 23-PRD outage this file's header references for what a
@@ -2098,7 +2285,7 @@ async function reconcile(state) {
2098
2285
  // (schedule:adopt-prd) that stamps the file via the same update-prd API
2099
2286
  // route the MCP tool uses — reconcile()'s adopt path above promotes it
2100
2287
  // to 'pending' on the very next pass, within one tick of being stamped.
2101
- if (!p.createdVia && !isFixPlanSlug(slug)) {
2288
+ if (!p.createdVia && !entry.isFixPlan) {
2102
2289
  entry.status = 'quarantined';
2103
2290
  // Stamped at creation (not via transitionJob, since this is a
2104
2291
  // brand-new row minted directly at 'quarantined' rather than
@@ -2150,6 +2337,12 @@ async function reconcile(state) {
2150
2337
  state.jobs = sorted;
2151
2338
  }
2152
2339
 
2340
+ // Auto-requeue any job parked 'skipped' over a validated BLOCKED_BY_FOREIGN_WIP
2341
+ // verdict once the paths that blocked it are no longer dirty in its own cwd —
2342
+ // the only place this check runs, so a sibling landing its commit resumes the
2343
+ // blocked job with no human action, on the very next reconcile() pass.
2344
+ await requeueForeignWipBlockedJobs(state.jobs);
2345
+
2153
2346
  // Auto-archive completed PRDs' .md files out of the live prds/ dir. Runs
2154
2347
  // AFTER the history append above (which is awaited) so a job's queue row
2155
2348
  // is always durably in history.jsonl before its file can be moved — a
@@ -2192,6 +2385,20 @@ function getNextResetCached() {
2192
2385
  return cachedNextReset;
2193
2386
  }
2194
2387
 
2388
+ /**
2389
+ * Pure: picks the reset to pause against for a rate-limited run (PRD 1118).
2390
+ * Prefers the BINDING window read off the run's own log — refreshNextReset()
2391
+ * only ever reports five_hour, which is the wrong clock when a
2392
+ * seven_day/seven_day_overage_included window is what actually 429'd
2393
+ * (five_hour can read 0% utilization at the very same moment). Falls back
2394
+ * to the billing-endpoint-derived reset only when the log yields nothing.
2395
+ */
2396
+ function resolveRateLimitPauseReset(logPath, billingResetIso) {
2397
+ const logReset = resolveBindingRateLimitReset(logPath);
2398
+ if (logReset != null) return new Date(logReset * 1000).toISOString();
2399
+ return billingResetIso ?? null;
2400
+ }
2401
+
2195
2402
  // ---------- health / poll state ----------
2196
2403
 
2197
2404
  let bootedAt = Date.now();
@@ -2475,6 +2682,33 @@ function nextRapidRateLimitCount(prevCount, { rateLimited, durationMs }) {
2475
2682
  return prevCount || 0;
2476
2683
  }
2477
2684
 
2685
+ /**
2686
+ * Pure: decides the resumeAt actually armed for a pause. 'network' and
2687
+ * 'rate_limit' (PRD 1118) both get a bounded 30-minute fallback when no
2688
+ * explicit resumeAt is supplied — the live rate_limit failure mode is the
2689
+ * billing usage endpoint itself 429ing while the log yields no binding
2690
+ * window either, which used to leave an indefinite pause with no resume
2691
+ * timer at all (a queue that never comes back on its own).
2692
+ */
2693
+ function computeEffectiveResumeAt(reason, resumeAtIso, nowMs = Date.now()) {
2694
+ if (resumeAtIso) return resumeAtIso;
2695
+ if (reason === 'network' || reason === 'rate_limit') {
2696
+ return new Date(nowMs + 30 * 60_000).toISOString();
2697
+ }
2698
+ return null;
2699
+ }
2700
+
2701
+ /**
2702
+ * Pure: the setTimeout delay for a resume timer, plus whether it overflows
2703
+ * setTimeout's signed-32-bit max (~24.8 days) and must not be armed.
2704
+ * Resume fires 30s after the reset to give the auth/billing endpoint time
2705
+ * to flip.
2706
+ */
2707
+ function computeResumeDelay(effectiveResumeAtIso, nowMs = Date.now()) {
2708
+ const delayMs = Math.max(30_000, new Date(effectiveResumeAtIso).getTime() - nowMs + 30_000);
2709
+ return { delayMs, tooFar: delayMs > 0x7fffffff };
2710
+ }
2711
+
2478
2712
  async function setPaused(reason, resumeAtIso, opts = {}) {
2479
2713
  const { observedAt = null, force = false } = opts;
2480
2714
  // Honor manual-override cooldown: if the user cleared a pause within the
@@ -2491,11 +2725,7 @@ async function setPaused(reason, resumeAtIso, opts = {}) {
2491
2725
  console.log(`[scheduler] setPaused(${reason}) engaging despite manual override cooldown — triggering run started after the manual clear`);
2492
2726
  }
2493
2727
 
2494
- // For 'network' with no explicit resumeAt, auto-resume after 30 minutes.
2495
- let effectiveResumeAt = resumeAtIso;
2496
- if (reason === 'network' && !resumeAtIso) {
2497
- effectiveResumeAt = new Date(Date.now() + 30 * 60_000).toISOString();
2498
- }
2728
+ const effectiveResumeAt = computeEffectiveResumeAt(reason, resumeAtIso);
2499
2729
 
2500
2730
  await mutate((s) => {
2501
2731
  if (s.paused && s.paused.reason === reason) {
@@ -2509,9 +2739,8 @@ async function setPaused(reason, resumeAtIso, opts = {}) {
2509
2739
  if (resumeTimer) { clearTimeout(resumeTimer); resumeTimer = null; }
2510
2740
  if (!effectiveResumeAt) return;
2511
2741
 
2512
- // Resume 30s after the reset to give the auth/billing endpoint time to flip.
2513
- const delay = Math.max(30_000, new Date(effectiveResumeAt).getTime() - Date.now() + 30_000);
2514
- if (delay > 0x7fffffff) {
2742
+ const { delayMs: delay, tooFar } = computeResumeDelay(effectiveResumeAt);
2743
+ if (tooFar) {
2515
2744
  console.warn(`[scheduler] paused (${reason}); resumeAt too far for setTimeout (${delay}ms)`);
2516
2745
  return;
2517
2746
  }
@@ -2584,6 +2813,17 @@ function resetJobFields(job, errorMsg, opts = {}) {
2584
2813
  delete job.verifierVerdict;
2585
2814
  delete job.uncommittedPaths;
2586
2815
  delete job.resumeRecoveryAttempted;
2816
+ // Same one-attempt-per-episode category as resumeRecoveryAttempted above —
2817
+ // a re-fired row must be able to earn a fresh mechanical-recovery attempt
2818
+ // if it parks needs_review again (PRD 1130).
2819
+ delete job.mechanicalRecoveryAttempted;
2820
+ // Quarantine (PRD 1128) is scoped to THIS run's episode exactly like
2821
+ // resumeRecoveryAttempted above — a re-fired row must be able to earn a
2822
+ // fresh quarantine attempt if it parks needs_review again.
2823
+ delete job.leftoverQuarantineAttempted;
2824
+ delete job.quarantinedTo;
2825
+ delete job.quarantinedCommit;
2826
+ delete job.quarantinedPaths;
2587
2827
  // Same "this run's outcome, not durable across a reset" category as the
2588
2828
  // fields above — a stale 'archive' recoveryAction from a prior life of this
2589
2829
  // slug must never survive a reset and silently exclude a genuinely-new
@@ -2592,6 +2832,10 @@ function resetJobFields(job, errorMsg, opts = {}) {
2592
2832
  // can otherwise linger forever when RCA is disabled or errors).
2593
2833
  delete job.rcaFailureClass;
2594
2834
  delete job.rcaRecoveryAction;
2835
+ // Same "this run's outcome, not durable across a reset" category — a
2836
+ // human-driven reset must genuinely start the auto-fix budget over,
2837
+ // including the one-time dead-fix-plan-child reopen (PRD 1129).
2838
+ delete job.autoFixReopened;
2595
2839
  // Like exitCode: this run's outcome, not durable across a reset — a stale
2596
2840
  // leak badge from a prior attempt must not linger once the job re-fires.
2597
2841
  delete job.leakedDescendants;
@@ -2605,6 +2849,13 @@ function resetJobFields(job, errorMsg, opts = {}) {
2605
2849
  delete job.leftoverCount;
2606
2850
  delete job.leftoverPathsTruncated;
2607
2851
  delete job.preRunDirtyPaths;
2852
+ // A manual/force reset (e.g. a human clearing a 3-strikes needs_review
2853
+ // park) must not leave the row permanently excluded from the auto-fix
2854
+ // chain (selectAutoFixTargets checks job.blockedByForeignWip) — this run's
2855
+ // BLOCKED_BY_FOREIGN_WIP history is done, the human is taking over.
2856
+ delete job.blockedByForeignWip;
2857
+ delete job.foreignWipBlockedPaths;
2858
+ delete job.foreignWipBlockCount;
2608
2859
  // Deliberately NOT deleting job.landedCommit: it must outlive a reset so a
2609
2860
  // re-fired run of this same slug can pass it to verifyRun as
2610
2861
  // priorLandedCommit (pass_no_commit_prior_run_verified exemption).
@@ -3071,6 +3322,44 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
3071
3322
  return { action: 'retry', transientKind, retries };
3072
3323
  }
3073
3324
 
3325
+ /**
3326
+ * Evidence gathering for the commit-guard's already-satisfied-on-main
3327
+ * exemption (PRD 1136): does a commit reachable from `main`, landed AFTER
3328
+ * this job's `queuedAt`, touch a path this PRD itself declares? Scoped to
3329
+ * the PRD's own declared paths via declaredPathsForPrd — the same
3330
+ * path-extraction computeLooksDone already uses — so an unrelated commit
3331
+ * elsewhere in the repo is never credited to this job. Returns `[]` (never
3332
+ * fabricates evidence) when the PRD names no paths or has no `queuedAt`.
3333
+ *
3334
+ * Requires EVERY declared path to have landed, not just one — declaredPathsForPrd's
3335
+ * regex matches any backtick-quoted path in the PRD's Implementation notes or
3336
+ * Acceptance criteria, including a path cited only as context (e.g.
3337
+ * `` `src/main/scheduler.cjs:1234` `` pointing at a call site, not a file this
3338
+ * PRD's own work touches). Unlike computeLooksDone's identical path-overlap
3339
+ * heuristic — which only ever annotates a still-needs_review row for a human
3340
+ * to confirm — this check drives an UNATTENDED transition straight to
3341
+ * 'completed', so a single incidental hot-file citation must never be
3342
+ * sufficient evidence on its own (code-review finding, 2026-09-07: a PRD
3343
+ * that cites both its real target file and one hot file purely as context
3344
+ * would auto-complete off any unrelated commit touching that hot file).
3345
+ * Requiring full coverage of the declared-path set trades recall for safety
3346
+ * exactly as this PRD's own constraint demands — a PRD that fails this
3347
+ * stricter check still falls back to the existing needs_review park, never
3348
+ * a false 'completed'.
3349
+ *
3350
+ * @returns {Promise<string[]>} full commit SHAs reachable from main, newest first
3351
+ */
3352
+ async function findSatisfyingCommitOnMain(job) {
3353
+ if (!job?.queuedAt) return [];
3354
+ const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
3355
+ const paths = declaredPathsForPrd(prdPath);
3356
+ if (!paths.length) return [];
3357
+ await fetchAllRefs(job.cwd);
3358
+ const perPathCommits = await Promise.all(paths.map((p) => landedOnMainSince(job.cwd, job.queuedAt, [p])));
3359
+ if (perPathCommits.some((commits) => commits.length === 0)) return [];
3360
+ return landedOnMainSince(job.cwd, job.queuedAt, paths);
3361
+ }
3362
+
3074
3363
  /**
3075
3364
  * Commit-guard verdict decision. Pure/no I/O so the false-positive defenses
3076
3365
  * can be unit-tested directly rather than only through a live spawnJob run.
@@ -3241,6 +3530,73 @@ function buildForeignWipSection({ preRunDirtyPaths, carriedPaths } = {}) {
3241
3530
  return '';
3242
3531
  }
3243
3532
 
3533
+ // A job whose validated BLOCKED_BY_FOREIGN_WIP claim keeps recurring against
3534
+ // the same sibling WIP is not making progress by re-firing forever — cap the
3535
+ // auto-requeue cycle and hand it to a human instead, naming the paths that
3536
+ // never went clean. Matches TRANSIENT_RETRY_CAP's "bounded self-heal, then
3537
+ // escalate" shape.
3538
+ const FOREIGN_WIP_BLOCK_STREAK_LIMIT = 3;
3539
+
3540
+ /**
3541
+ * Pure: validate an executor's BLOCKED_BY_FOREIGN_WIP claim against the exact
3542
+ * foreign-WIP manifest THIS job was disclosed at dispatch time
3543
+ * (preRunDirtyPaths / carriedPaths — PRD 1105). Every claimed path must be a
3544
+ * member of that manifest; a path the job was never told was foreign cannot
3545
+ * be laundered into a block, whether that's a genuine regression the
3546
+ * executor is trying to dodge or an honest mistake. `job` is any object
3547
+ * carrying those two array fields (the live queue row at finalize time).
3548
+ *
3549
+ * Returns `ok: false` for an empty claim too — a BLOCKED_BY_FOREIGN_WIP
3550
+ * verdict with no FOREIGN_WIP_PATHS evidence at all is exactly as
3551
+ * unsubstantiated as one naming an unlisted path.
3552
+ */
3553
+ function validateForeignWipBlockClaim(claimedPaths, job) {
3554
+ const manifest = new Set([
3555
+ ...(Array.isArray(job?.preRunDirtyPaths) ? job.preRunDirtyPaths : []),
3556
+ ...(Array.isArray(job?.carriedPaths) ? job.carriedPaths : []),
3557
+ ]);
3558
+ const claimed = Array.isArray(claimedPaths) ? claimedPaths.filter(Boolean) : [];
3559
+ const validPaths = claimed.filter((p) => manifest.has(p));
3560
+ const invalidPaths = claimed.filter((p) => !manifest.has(p));
3561
+ return { ok: claimed.length > 0 && invalidPaths.length === 0, validPaths, invalidPaths };
3562
+ }
3563
+
3564
+ /**
3565
+ * Auto-requeue jobs parked 'skipped' with a validated BLOCKED_BY_FOREIGN_WIP
3566
+ * verdict once none of the paths that blocked them are still dirty in their
3567
+ * own `cwd` — so a sibling job landing its commit (or a human resolving their
3568
+ * own WIP) resumes the blocked job with no human action needed, on the very
3569
+ * next reconcile() pass. `getDirtyPaths` is injectable (defaults to the real
3570
+ * `uncommittedChanges` git-status wrapper) so this is unit-testable without a
3571
+ * real git repo. Mutates eligible jobs in place via transitionJob; returns
3572
+ * nothing.
3573
+ */
3574
+ async function requeueForeignWipBlockedJobs(jobs, { getDirtyPaths = uncommittedChanges } = {}) {
3575
+ const candidates = (Array.isArray(jobs) ? jobs : []).filter(
3576
+ (j) => j && j.status === 'skipped' && j.blockedByForeignWip === true,
3577
+ );
3578
+ for (const job of candidates) {
3579
+ const blockedPaths = Array.isArray(job.foreignWipBlockedPaths) ? job.foreignWipBlockedPaths : [];
3580
+ if (!blockedPaths.length) continue;
3581
+ const dirty = await getDirtyPaths(job.cwd);
3582
+ if (dirty === null) continue; // non-git cwd or git error — best-effort, leave parked
3583
+ const dirtySet = new Set(dirty);
3584
+ const stillDirty = blockedPaths.filter((p) => dirtySet.has(p));
3585
+ if (stillDirty.length === 0) {
3586
+ // resetJobFields (not a bare transitionJob) so the row re-enters
3587
+ // 'pending' with none of the blocked run's stale exitCode/error/
3588
+ // verifierVerdict/finishedAt/leftoverPaths left on it for a viewer to
3589
+ // misread against the new 'pending' status — same reasoning as every
3590
+ // other terminal->pending path in this file. force:true is required
3591
+ // here: resetJobFields refuses skipped->pending without it.
3592
+ resetJobFields(job, `foreign WIP cleared (was blocked on: ${blockedPaths.join(', ')}) — auto-requeued`, {
3593
+ source: 'reconcile-foreign-wip-clear',
3594
+ force: true,
3595
+ });
3596
+ }
3597
+ }
3598
+ }
3599
+
3244
3600
  /**
3245
3601
  * Resume-first recovery (PRD 1111). A job parked in needs_review with verdict
3246
3602
  * 'uncommitted_changes' has a live claude session (job.sessionId, minted by
@@ -3303,6 +3659,302 @@ As the LAST LINE of your final result text, emit exactly one of:
3303
3659
  Print PASS only once the commit above has actually landed.`;
3304
3660
  }
3305
3661
 
3662
+ /**
3663
+ * Mechanical recovery (PRD 1130). isFixPlanBeyondDepthCap (below) is the
3664
+ * ONLY gate on re-investigating a fix-plan job at investigationDepth >= 2 —
3665
+ * correct for open-ended "author another plan" recursion, but it also
3666
+ * strands a depth-capped job whose failure was fully mechanical (no
3667
+ * judgement required) with no other ladder rung, since resume-first recovery
3668
+ * (selectResumeRecoveryTarget above) is hard-gated on verdict
3669
+ * 'uncommitted_changes'. This rung is evaluated INDEPENDENTLY of
3670
+ * isFixPlanBeyondDepthCap — depth never disqualifies it, because unlike
3671
+ * auto-fix it authors no plan and spawns no model; it is pure git.
3672
+ *
3673
+ * The closed set of mechanically-resolvable verdicts starts at exactly
3674
+ * 'worktree_integration_failed': PRD 1125 already taught integrateBranch to
3675
+ * parse git's "would be overwritten by merge" stderr, verify the blocking
3676
+ * paths are byte-identical to the branch, discard the proven duplicates, and
3677
+ * retry the merge once. A job parked with this verdict has its `sm-job/
3678
+ * <slug>` branch preserved (integrateJobBranch never deletes the branch on
3679
+ * failure — see cleanupJobWorktree's `keepBranch: !integration.ok`), so a
3680
+ * plain re-call of integrateBranch against that same branch inherits PRD
3681
+ * 1125's auto-resolution for free — no re-implementation needed here.
3682
+ *
3683
+ * Bounded to exactly one attempt via job.mechanicalRecoveryAttempted,
3684
+ * stamped in the SAME mutate as the outcome (performMechanicalRecovery,
3685
+ * below) — never here — so this selector alone can be unit-tested exactly
3686
+ * like selectResumeRecoveryTarget/selectLeftoverQuarantineTarget.
3687
+ *
3688
+ * Kill-switch: SM_MECHANICAL_RECOVERY_DISABLE=1 restores today's behaviour
3689
+ * exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
3690
+ */
3691
+ const MECHANICALLY_RESOLVABLE_VERDICTS = new Set(['worktree_integration_failed']);
3692
+
3693
+ function selectMechanicalRecoveryTarget(job) {
3694
+ if (process.env.SM_MECHANICAL_RECOVERY_DISABLE === '1') return null;
3695
+ if (!job || job.status !== 'needs_review') return null;
3696
+ if (!MECHANICALLY_RESOLVABLE_VERDICTS.has(job.verifierVerdict)) return null;
3697
+ if (job.mechanicalRecoveryAttempted === true) return null;
3698
+ const cwd = job.cwd || DEFAULT_PROJECT_CWD;
3699
+ return { slug: job.slug, cwd, branch: jobWorktree.branchNameFor(job.slug), carriedPaths: job.carriedPaths || [] };
3700
+ }
3701
+
3702
+ /**
3703
+ * Perform an already-selected mechanical recovery (selectMechanicalRecoveryTarget
3704
+ * above) — a direct re-attempt of integrateBranch against the job's preserved
3705
+ * branch, never a fresh `claude -p` dispatch. On success the job transitions
3706
+ * needs_review -> completed and its verifierVerdict is cleared; the branch,
3707
+ * now merged, is deleted like any other successfully-integrated job branch.
3708
+ * On failure (including a branch that no longer exists — already deleted or
3709
+ * already merged) the job stays needs_review, mechanicalRecoveryAttempted is
3710
+ * stamped, and the retry's own failure text is appended to `error`. Either
3711
+ * way mechanicalRecoveryAttempted is stamped in this SAME mutate, so a crash
3712
+ * between the git call returning and this mutate landing simply repeats an
3713
+ * idempotent git operation on the next pass rather than leaving the job
3714
+ * re-eligible forever.
3715
+ */
3716
+ async function performMechanicalRecovery(job, target) {
3717
+ const integration = await jobWorktree.integrateJobBranch({
3718
+ cwd: target.cwd, branch: target.branch, slug: target.slug, carriedPaths: target.carriedPaths,
3719
+ });
3720
+ if (integration.ok) {
3721
+ await jobWorktree.cleanupJobWorktree({ cwd: target.cwd, dir: undefined, branch: target.branch, keepBranch: false });
3722
+ }
3723
+ let becameCompleted = false;
3724
+ await mutate((s) => {
3725
+ const j = s.jobs.find((x) => x.slug === job.slug);
3726
+ if (!j) return;
3727
+ j.mechanicalRecoveryAttempted = true;
3728
+ if (integration.ok) {
3729
+ if (transitionJob(j, 'completed', {
3730
+ reason: `mechanical recovery: ${target.branch} re-integrated successfully`,
3731
+ source: 'scheduler:mechanicalRecovery',
3732
+ })) {
3733
+ delete j.verifierVerdict;
3734
+ j.exitCode = 0;
3735
+ j.error = null;
3736
+ becameCompleted = true;
3737
+ }
3738
+ } else {
3739
+ const pointer = `Mechanical recovery retry failed: ${integration.reason}`;
3740
+ j.error = j.error ? `${j.error}\n${pointer}` : pointer;
3741
+ }
3742
+ });
3743
+ if (integration.ok) {
3744
+ console.log(`[scheduler] mechanical-recovery: ${job.slug} → completed (branch ${target.branch} re-integrated)`);
3745
+ if (becameCompleted) await archiveCompletedPrd(job.slug, job.cwd);
3746
+ } else {
3747
+ console.error(`[scheduler] mechanical-recovery: ${job.slug} → retry failed: ${integration.reason}`);
3748
+ }
3749
+ }
3750
+
3751
+ /**
3752
+ * Leftover quarantine (PRD 1128). Resume-first recovery gets exactly one
3753
+ * `--resume` attempt (selectResumeRecoveryTarget above); when that attempt
3754
+ * ALSO parks needs_review with 'uncommitted_changes', the leftovers are
3755
+ * about to sit dirty in the SHARED tree forever — git then refuses any later
3756
+ * worktree merge for this cwd that would overwrite them, turning one parked
3757
+ * job into a project-wide stall (216-jupiter-sand-kazekage, 2026-09-06).
3758
+ * Pure/no I/O, mirroring selectResumeRecoveryTarget so the eligibility rule
3759
+ * is unit-testable directly.
3760
+ *
3761
+ * Bounded to exactly one attempt via job.leftoverQuarantineAttempted, stamped
3762
+ * synchronously by the caller in the SAME mutate as this decision (never
3763
+ * here) — see spawnJob's finalize and reverifyNeedsReview's periodic pass.
3764
+ *
3765
+ * Kill-switch: SM_LEFTOVER_QUARANTINE_DISABLE=1 restores today's behaviour
3766
+ * exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
3767
+ */
3768
+ function selectLeftoverQuarantineTarget(job) {
3769
+ if (process.env.SM_LEFTOVER_QUARANTINE_DISABLE === '1') return null;
3770
+ if (!job || job.status !== 'needs_review') return null;
3771
+ if (job.verifierVerdict !== 'uncommitted_changes') return null;
3772
+ if (job.resumeRecoveryAttempted !== true) return null;
3773
+ if (job.leftoverQuarantineAttempted === true) return null;
3774
+ const uncommittedPaths = Array.isArray(job.uncommittedPaths)
3775
+ ? job.uncommittedPaths.filter((p) => typeof p === 'string' && p.length > 0)
3776
+ : [];
3777
+ if (!uncommittedPaths.length) return null;
3778
+ // The single most important constraint: never touch a path that was
3779
+ // ALREADY dirty at this run's own dispatch time (preRunDirtyPaths) — that
3780
+ // is foreign WIP (a human's or a sibling's), not this job's own leftover.
3781
+ const preRunDirty = new Set(Array.isArray(job.preRunDirtyPaths) ? job.preRunDirtyPaths : []);
3782
+ const paths = uncommittedPaths.filter((p) => !preRunDirty.has(p));
3783
+ if (!paths.length) return null;
3784
+ return { slug: job.slug, cwd: job.cwd, paths };
3785
+ }
3786
+
3787
+ function execGitAt(cwd, args, { env, timeout = 20_000 } = {}) {
3788
+ return new Promise((resolve, reject) => {
3789
+ execFile(
3790
+ 'git',
3791
+ ['-C', cwd, ...args],
3792
+ { timeout, windowsHide: true, encoding: 'utf8', env: env ? { ...process.env, ...env } : process.env },
3793
+ (err, stdout, stderr) => {
3794
+ if (err) {
3795
+ err.stderrText = stderr;
3796
+ reject(err);
3797
+ return;
3798
+ }
3799
+ resolve(stdout || '');
3800
+ },
3801
+ );
3802
+ });
3803
+ }
3804
+
3805
+ async function pathExistsInTree(cwd, treeish, p) {
3806
+ try {
3807
+ await execGitAt(cwd, ['cat-file', '-e', `${treeish}:${p}`]);
3808
+ return true;
3809
+ } catch {
3810
+ return false;
3811
+ }
3812
+ }
3813
+
3814
+ /**
3815
+ * Commit exactly `paths` (must already be dirty on disk) onto a dedicated
3816
+ * `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
3817
+ * unavailable) via a THROWAWAY `GIT_INDEX_FILE` — never touches the live
3818
+ * index, never moves the checked-out branch — then restores those paths to
3819
+ * match that baseline commit's tree, so the shared working tree returns to
3820
+ * its pre-run state. This is deliberately NOT `git stash` (the destructive-
3821
+ * git guard blocks stash on a shared tree, and a stash nobody restores
3822
+ * strands the work invisibly — see standards.md).
3823
+ *
3824
+ * Never throws: any git failure, or a non-git cwd, aborts the WHOLE attempt
3825
+ * with the tree untouched (no partial restore) — restore only ever runs
3826
+ * after the salvage ref/commit has safely landed, so a failure there leaves
3827
+ * the data recoverable from the ref even though the tree stayed dirty.
3828
+ * A path no longer dirty on disk (already committed, or reverted since) is
3829
+ * skipped, never force-restored.
3830
+ */
3831
+ async function quarantineLeftovers({ cwd, slug, paths, headBefore }) {
3832
+ if (!cwd || !slug || !Array.isArray(paths) || paths.length === 0) {
3833
+ return { ok: false, reason: 'no cwd/slug/paths given' };
3834
+ }
3835
+ let baseline = headBefore || null;
3836
+ try {
3837
+ if (!baseline) {
3838
+ baseline = (await execGitAt(cwd, ['rev-parse', 'HEAD'])).trim();
3839
+ }
3840
+ if (!baseline) return { ok: false, reason: 'could not resolve a baseline commit (non-git cwd?)' };
3841
+
3842
+ const dirtyNowRaw = await execGitAt(cwd, ['status', '--porcelain', '--', ...paths]);
3843
+ const dirtyNow = new Set(parsePorcelain(dirtyNowRaw));
3844
+ const toQuarantine = paths.filter((p) => dirtyNow.has(p));
3845
+ const skippedPaths = paths.filter((p) => !dirtyNow.has(p));
3846
+ if (!toQuarantine.length) {
3847
+ return { ok: true, ref: null, commit: null, quarantinedPaths: [], skippedPaths };
3848
+ }
3849
+
3850
+ const tmpIndex = path.join(os.tmpdir(), `sm-salvage-index-${slug}-${process.pid}-${Date.now()}`);
3851
+ const env = { GIT_INDEX_FILE: tmpIndex };
3852
+ let treeSha;
3853
+ let commitSha;
3854
+ try {
3855
+ await execGitAt(cwd, ['read-tree', baseline], { env });
3856
+ for (const p of toQuarantine) {
3857
+ if (fs.existsSync(path.join(cwd, p))) {
3858
+ await execGitAt(cwd, ['add', '--', p], { env });
3859
+ } else {
3860
+ await execGitAt(cwd, ['rm', '--cached', '--ignore-unmatch', '--', p], { env });
3861
+ }
3862
+ }
3863
+ treeSha = (await execGitAt(cwd, ['write-tree'], { env })).trim();
3864
+ commitSha = (await execGitAt(cwd, ['commit-tree', treeSha, '-p', baseline, '-m', `salvage: leftover changes from ${slug}`], { env })).trim();
3865
+ } catch (e) {
3866
+ return { ok: false, reason: `git command failed while building the salvage commit: ${(e && (e.stderrText || e.message)) || e}` };
3867
+ } finally {
3868
+ await fsp.rm(tmpIndex, { force: true }).catch(() => {});
3869
+ }
3870
+
3871
+ const ref = `sm-salvage/${slug}`;
3872
+ try {
3873
+ await execGitAt(cwd, ['update-ref', `refs/heads/${ref}`, commitSha]);
3874
+ } catch (e) {
3875
+ return { ok: false, reason: `git command failed updating ${ref}: ${(e && (e.stderrText || e.message)) || e}` };
3876
+ }
3877
+
3878
+ // The salvage commit is safely landed at this point — a failure from here
3879
+ // on is reported with the ref/commit still attached so nothing looks lost
3880
+ // even if the tree itself couldn't be fully restored.
3881
+ try {
3882
+ const inBaseline = [];
3883
+ const notInBaseline = [];
3884
+ for (const p of toQuarantine) {
3885
+ // eslint-disable-next-line no-await-in-loop
3886
+ if (await pathExistsInTree(cwd, baseline, p)) inBaseline.push(p); else notInBaseline.push(p);
3887
+ }
3888
+ if (inBaseline.length) {
3889
+ await execGitAt(cwd, ['checkout', baseline, '--', ...inBaseline]);
3890
+ }
3891
+ if (notInBaseline.length) {
3892
+ await execGitAt(cwd, ['reset', '--', ...notInBaseline]).catch(() => {});
3893
+ for (const p of notInBaseline) {
3894
+ // eslint-disable-next-line no-await-in-loop
3895
+ await fsp.rm(path.join(cwd, p), { force: true });
3896
+ }
3897
+ }
3898
+ } catch (e) {
3899
+ return {
3900
+ ok: false,
3901
+ ref,
3902
+ commit: commitSha,
3903
+ reason: `salvage commit landed at ${ref} (${commitSha}) but restoring the working tree failed: ${(e && (e.stderrText || e.message)) || e}`,
3904
+ };
3905
+ }
3906
+
3907
+ return { ok: true, ref, commit: commitSha, quarantinedPaths: toQuarantine, skippedPaths };
3908
+ } catch (e) {
3909
+ return { ok: false, reason: `git command failed: ${(e && (e.stderrText || e.message)) || e}` };
3910
+ }
3911
+ }
3912
+
3913
+ /**
3914
+ * Perform an already-selected quarantine (job.leftoverQuarantineAttempted
3915
+ * must already be true, stamped by the caller) and persist the outcome onto
3916
+ * the job row: `quarantinedTo`/`quarantinedCommit`/`quarantinedPaths` on
3917
+ * success, plus a one-line pointer appended to `error` naming the ref so a
3918
+ * human can recover with a single named command
3919
+ * (`git show sm-salvage/<slug>`). The belt-and-braces `salvagePatch` (when
3920
+ * present) is referenced alongside it, never removed. On failure, only a
3921
+ * diagnostic is appended — the job row's dirt-describing fields are left as
3922
+ * they were, since the tree itself was left untouched (or, for a
3923
+ * restore-only failure, the salvage ref is still named in the note).
3924
+ *
3925
+ * `headBefore`, when the caller has it fresh (spawnJob's own finalize still
3926
+ * has the local `guardHeadBefore` in scope for the run that just parked —
3927
+ * the same value is deleted off the job ROW earlier in that same finalize),
3928
+ * is used as the salvage ref's baseline commit; otherwise (the periodic
3929
+ * reverifyNeedsReview pass, re-discovering an already-parked row) this falls
3930
+ * back to the current HEAD inside quarantineLeftovers itself.
3931
+ */
3932
+ async function performLeftoverQuarantine(job, paths, headBefore = null) {
3933
+ const result = await quarantineLeftovers({
3934
+ cwd: job.cwd || DEFAULT_PROJECT_CWD,
3935
+ slug: job.slug,
3936
+ paths,
3937
+ headBefore: headBefore || job.guardHeadBefore || null,
3938
+ });
3939
+ await mutate((s) => {
3940
+ const j = s.jobs.find((x) => x.slug === job.slug);
3941
+ if (!j) return;
3942
+ if (result.ok && Array.isArray(result.quarantinedPaths) && result.quarantinedPaths.length) {
3943
+ j.quarantinedTo = result.ref;
3944
+ j.quarantinedCommit = result.commit;
3945
+ j.quarantinedPaths = capDirtyPaths(result.quarantinedPaths);
3946
+ const salvageNote = j.salvagePatch ? `; salvage patch also at ${j.salvagePatch}` : '';
3947
+ const pointer = `Leftovers quarantined to ${result.ref} (commit ${result.commit}) — recover via \`git show ${result.ref}\`${salvageNote}`;
3948
+ j.error = j.error ? `${j.error}\n${pointer}` : pointer;
3949
+ console.log(`[scheduler] ${job.slug}: quarantined ${result.quarantinedPaths.length} leftover path(s) to ${result.ref} (${result.commit})`);
3950
+ } else if (!result.ok) {
3951
+ const pointer = `Leftover quarantine failed: ${result.reason}`;
3952
+ j.error = j.error ? `${j.error}\n${pointer}` : pointer;
3953
+ console.error(`[scheduler] ${job.slug}: leftover quarantine failed: ${result.reason}`);
3954
+ }
3955
+ });
3956
+ }
3957
+
3306
3958
  /**
3307
3959
  * Pure argv builder for a `claude -p` child spawn, shared so the
3308
3960
  * resume-vs-fresh-session choice is made in exactly one place. `resume`
@@ -3782,13 +4434,9 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
3782
4434
  // project (keyed by cwd) so jobs in different repos run concurrently up to
3783
4435
  // the cap; within one project, sequential-group semantics are preserved.
3784
4436
 
3785
- /**
3786
- * Recognize fix-plan slugs (NN-fix-...) so we don't recurse on a fix-plan that
3787
- * itself failed. The pattern matches the slug we generate in spawnInvestigation.
3788
- */
3789
- function isFixPlanSlug(slug) {
3790
- return /^\d+-fix-/.test(slug);
3791
- }
4437
+ // isFixPlanSlug/classifyDiscoveredFixPlan/resolveIsFixPlan now live in
4438
+ // lib/fixPlanSlug.cjs (PRD 1131) — see that module's header for why slug
4439
+ // shape alone is no longer sufficient to classify a fix plan.
3792
4440
 
3793
4441
  /**
3794
4442
  * The fix-plan slug spawnInvestigation authors for a given failed job —
@@ -3827,7 +4475,25 @@ function healTargetForFix(fixSlug, jobs) {
3827
4475
  * unit-tested (no spawn, no fs). Inputs are the already-resolved values that
3828
4476
  * spawnInvestigation computes.
3829
4477
  */
3830
- function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group }) {
4478
+ function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild = null }) {
4479
+ const deadFixChildNote = deadChild ? `
4480
+
4481
+ # This is a REOPENED investigation — your own prior fix plan died
4482
+ You already investigated this job once and produced a fix-plan PRD, \`${deadChild.slug}\`, which was
4483
+ supposed to heal it. That fix-plan job itself reached a terminal, non-completed status
4484
+ (\`${deadChild.status}\`) without ever fixing the original failure — so the parent job you are now
4485
+ investigating is stuck again with nothing left to retry it automatically. This is the ONE reopen
4486
+ this parent gets; do not cold-read the log and re-derive the plan that already failed.
4487
+
4488
+ Dead fix-plan child's own outcome:
4489
+ - Slug: ${deadChild.slug}
4490
+ - Status: ${deadChild.status}
4491
+ - Verifier verdict: ${deadChild.verifierVerdict ?? '(none recorded)'}
4492
+ - Error: ${deadChild.error ?? '(none recorded)'}
4493
+
4494
+ Read why THAT job died (its own run log, if any, under the runs directory) before writing a new
4495
+ fix-plan PRD, and make sure your new plan is genuinely different from — not a repeat of — whatever
4496
+ that dead child attempted.` : '';
3831
4497
  const abandonedBackgroundTaskNote = failedJob.verifierVerdict === 'abandoned_background_task' ? `
3832
4498
 
3833
4499
  # Known failure class: abandoned background task
@@ -3848,7 +4514,7 @@ The fix-plan PRD you write for this MUST instruct its executor to, in order:
3848
4514
  1. Check for a salvage patch (named \`<slug>.uncommitted.patch\` in the run directory${failedJob.salvagePatch ? `, e.g. \`${failedJob.salvagePatch}\`` : ''}) and, if found, apply it to the working tree BEFORE inspecting \`git status\`/\`git diff\` in ${cwd} for uncommitted changes matching the original PRD's acceptance criteria.
3849
4515
  2. If the work is present (via the applied patch or already in the tree) and satisfies the acceptance criteria, run the project's verify commands and COMMIT it — do not re-implement or re-plan the PRD from scratch.
3850
4516
  3. Only fall back to re-implementing whatever acceptance criteria are genuinely missing after applying any salvage patch, not the whole PRD.` : '';
3851
- return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${abandonedBackgroundTaskNote}
4517
+ return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${deadFixChildNote}${abandonedBackgroundTaskNote}
3852
4518
 
3853
4519
  # Failed job
3854
4520
  - Slug: ${failedJob.slug}
@@ -3893,8 +4559,13 @@ ${logTail}
3893
4559
  cwd: ${cwd}
3894
4560
  parallelGroup: ${group}
3895
4561
  estimateMinutes: <your time estimate>
4562
+ isFixPlan: true
3896
4563
  ---
3897
4564
  \`\`\`
4565
+ \`isFixPlan: true\` is REQUIRED — it is the scheduler's provenance signal that this PRD is a
4566
+ genuine auto-authored fix plan (not a human/agent PRD whose slug merely happens to start with
4567
+ "fix-"); omitting it means this fix plan will not get its depth-cap/zero-edit-commit-guard
4568
+ exemptions.
3898
4569
  \`cwd\` must be the git repo root where the fix will actually land. If the failed job's cwd is
3899
4570
  not that repo (e.g. a scratch dir like \`/tmp\`), set \`cwd:\` to the correct repo root instead —
3900
4571
  the scheduler's commit guard and post-run verifier read git state from this path, and a
@@ -3951,6 +4622,11 @@ function readRunOutcomeSidecars(runDir, slug) {
3951
4622
  return {
3952
4623
  meta: readJson(path.join(runDir, `${slug}.meta.json`)),
3953
4624
  verdicts: readJson(path.join(runDir, `${slug}.verdicts.json`)),
4625
+ // outcome.json (launchFailure.writeOutcomeSidecar) is the one sidecar
4626
+ // that carries landedCommit — reused by the dispatch-time sidecar-
4627
+ // reconcile guard and the pre-dispatch landedCommit backfill below,
4628
+ // rather than growing a second reader for the same directory.
4629
+ outcome: readJson(path.join(runDir, `${slug}.outcome.json`)),
3954
4630
  };
3955
4631
  }
3956
4632
 
@@ -3968,7 +4644,7 @@ function readRunOutcomeSidecars(runDir, slug) {
3968
4644
  */
3969
4645
  const INVESTIGATION_LAUNCH_KEY = 'investigation';
3970
4646
 
3971
- async function spawnInvestigation(failedJob, runDir) {
4647
+ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {}) {
3972
4648
  // The probe launches with the same CLI as the job it diagnoses. While
3973
4649
  // that CLI cannot launch at all (launch circuit breaker, issue #11 list
3974
4650
  // B1: probes e4f82da2/d374e6bf died on the same HTTP 400 as the runs
@@ -3997,7 +4673,14 @@ async function spawnInvestigation(failedJob, runDir) {
3997
4673
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is resume-recovery eligible`);
3998
4674
  return { deferred: false };
3999
4675
  }
4000
- if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth)) {
4676
+ // Mechanical recovery (PRD 1130): same first-refusal treatment — a job
4677
+ // eligible for a pure-git retry must never also get a cold-read fix-plan
4678
+ // PRD authored in the same pass.
4679
+ if (selectMechanicalRecoveryTarget(failedJob)) {
4680
+ console.log(`[scheduler] skip investigation: ${failedJob.slug} is mechanical-recovery eligible`);
4681
+ return { deferred: false };
4682
+ }
4683
+ if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth, failedJob.isFixPlan)) {
4001
4684
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
4002
4685
  return { deferred: false };
4003
4686
  }
@@ -4054,7 +4737,13 @@ async function spawnInvestigation(failedJob, runDir) {
4054
4737
 
4055
4738
  const logTail = readTail(failedLogPath, 16 * 1024) || '(failed to read log)';
4056
4739
 
4057
- if (fs.existsSync(fixPath)) {
4740
+ // A dead-fix-plan reopen (PRD 1129) targets the SAME fixPath its dead
4741
+ // child was originally authored at, by construction (fixSlugFor is a pure
4742
+ // function of the parent) — the file existing is not staleness here, it's
4743
+ // the whole reason a reopen was offered. Skip the guard in that one case
4744
+ // so the second investigation can overwrite the dead plan; every other
4745
+ // caller keeps the original protection against clobbering a live sibling.
4746
+ if (fs.existsSync(fixPath) && !deadChild) {
4058
4747
  console.log(`[scheduler] skip investigation: fix plan already exists at ${fixPath}`);
4059
4748
  releaseSlot();
4060
4749
  return { deferred: false };
@@ -4080,7 +4769,7 @@ async function spawnInvestigation(failedJob, runDir) {
4080
4769
  console.warn(`[scheduler] investigation cwd is not a git repo (${cwd}); falling back to ${DEFAULT_PROJECT_CWD}`);
4081
4770
  cwd = DEFAULT_PROJECT_CWD;
4082
4771
  }
4083
- const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group });
4772
+ const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild });
4084
4773
 
4085
4774
  // Phase 1: open log fd for pre-spawn diagnostics.
4086
4775
  const { fd, safeLog, closeFd } = openLog(investigationLogPath);
@@ -4209,6 +4898,22 @@ async function spawnInvestigation(failedJob, runDir) {
4209
4898
  mutate((s) => {
4210
4899
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
4211
4900
  if (j) j.autoFixOutcome = 'plan';
4901
+ // Dead-fix-plan reopen (PRD 1129): fixSlugFor is a pure function of
4902
+ // the parent, so the freshly-authored plan landed at the SAME slug
4903
+ // as the dead child — reconcile() sees an already-known slug and
4904
+ // will never re-mint a pending row for it. Explicitly reset the
4905
+ // dead child's own row here so the overwritten plan actually gets
4906
+ // a chance to run, rather than sitting inert behind a permanently
4907
+ // terminal queue row. force:true because 'skipped' (a valid dead
4908
+ // status here) is otherwise reset-refused by design.
4909
+ if (deadChild) {
4910
+ const child = s.jobs.find((x) => x.slug === deadChild.slug);
4911
+ if (child) {
4912
+ resetJobFields(child, 'reset by dead-fix-plan reopen: parent investigation authored a new plan', {
4913
+ force: true, source: 'spawnInvestigation:dead-fix-plan-reopen',
4914
+ });
4915
+ }
4916
+ }
4212
4917
  }).catch(() => {});
4213
4918
  } else {
4214
4919
  console.log(`[scheduler] investigation finished WITHOUT producing fix plan (slug=${failedJob.slug}, code=${exitCode})`);
@@ -4291,6 +4996,58 @@ async function computeLaunchHolds(state, { now = Date.now(), claudeVersion } = {
4291
4996
  return held;
4292
4997
  }
4293
4998
 
4999
+ /**
5000
+ * computeDepHistorySatisfaction(state) → Map<cwd, Set<string>|symbol>
5001
+ *
5002
+ * PRD 1122's once-per-tick dependsOn history/archive lookup: for every
5003
+ * distinct project cwd with jobs this tick, builds the set of dep slugs that
5004
+ * have no live queue row but are nonetheless known-satisfied — a completed
5005
+ * record in that project's own `state/history.jsonl` shard
5006
+ * (queueHistory.completedSlugsForCwd, scoped per-project so a same-named PRD
5007
+ * in an unrelated project can never satisfy a dep here), or a `.md` file
5008
+ * under any of that project's `prds-archived/` dirs (listArchivedPrdDirs —
5009
+ * covers both the retired flat layout and every Epic's own sibling archive).
5010
+ * findBlockingDep (schedulerBatch.cjs) treats a dep slug as blocking
5011
+ * whenever it has no live row AND is absent from this set, so a typo or a
5012
+ * double-prefixed slug (the exact 2026-09-06 starry-night-ships incident)
5013
+ * HOLDS its dependent instead of silently dispatching it.
5014
+ *
5015
+ * Fails OPEN per project, never queue-wide: a history-shard or archive-scan
5016
+ * read error for one cwd degrades that cwd's value to
5017
+ * `DEP_HISTORY_FAIL_OPEN` (findBlockingDep then treats every rowless dep in
5018
+ * that project as satisfied, exactly today's pre-1122 behaviour) with a
5019
+ * logged warning — it never throws out of this function and never blocks
5020
+ * every OTHER project's dispatch for one project's bad fs state.
5021
+ *
5022
+ * Computed ONCE here, before pickNextBatch runs, and threaded down as pure
5023
+ * data (quietOpts.satisfiedSlugsByCwd) — schedulerBatch.cjs itself does no
5024
+ * I/O, so this is the only fs read this gate costs per tick, not one per job
5025
+ * per dep.
5026
+ */
5027
+ async function computeDepHistorySatisfaction(state) {
5028
+ const byCwd = new Map();
5029
+ const cwds = new Set((state?.jobs || []).map((j) => j.cwd || DEFAULT_PROJECT_CWD));
5030
+ for (const cwd of cwds) {
5031
+ const satisfied = new Set();
5032
+ try {
5033
+ for (const slug of await queueHistory.completedSlugsForCwd(cwd)) satisfied.add(slug);
5034
+ for (const dir of listArchivedPrdDirs(cwd)) {
5035
+ let entries;
5036
+ try { entries = await fsp.readdir(dir); } catch { continue; }
5037
+ for (const name of entries) {
5038
+ if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
5039
+ }
5040
+ }
5041
+ } catch (e) {
5042
+ console.warn(`[scheduler] depHistorySatisfaction: history/archive lookup failed for ${cwd} (${e?.message}) — falling back to fail-open dep resolution for this project this tick`);
5043
+ byCwd.set(cwd, DEP_HISTORY_FAIL_OPEN);
5044
+ continue;
5045
+ }
5046
+ byCwd.set(cwd, satisfied);
5047
+ }
5048
+ return byCwd;
5049
+ }
5050
+
4294
5051
  /**
4295
5052
  * A run that never got a turn (res.launchFailure — see executeJob's onExit)
4296
5053
  * is routed here instead of the failed/investigation path (issue #11 lists
@@ -4468,9 +5225,64 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4468
5225
  // started after the manual clear, not whatever startedAt this row
4469
5226
  // carried from a prior run.
4470
5227
  let dispatchStartedAtMs = null;
5228
+ let dispatchSkippedAlreadyCompleted = false;
4471
5229
  await mutate((s) => {
4472
5230
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4473
5231
  if (idx >= 0) {
5232
+ // Anti-resurrection guard for an EXISTING pending row (the gap left
5233
+ // by reconcile()'s own guard, which only ever sees a slug BEFORE it
5234
+ // first lands in s.jobs — 2026-09-06 incident: a finalize dropped
5235
+ // silently three times over left this same slug 'pending' and the
5236
+ // dispatcher re-fired it three more times against work that had
5237
+ // already shipped). Skipped for a resume-recovery dispatch (that
5238
+ // targets a specific prior session on purpose) and for anything not
5239
+ // currently 'pending' (e.g. a needs_review->running recovery row).
5240
+ if (!resumeTarget && s.jobs[idx].status === 'pending') {
5241
+ const outcome = latestTerminalOutcomeForSlug(job.slug, { runsDir: RUNS_DIR });
5242
+ const reconcileDecision = evaluateDispatchSidecarReconcile({
5243
+ rowStatus: s.jobs[idx].status,
5244
+ rowRunId: s.jobs[idx].runId ?? null,
5245
+ statusHistory: s.jobs[idx].statusHistory,
5246
+ queuedAt: s.jobs[idx].queuedAt ?? null,
5247
+ outcome,
5248
+ });
5249
+ if (reconcileDecision.skip) {
5250
+ const sidecar = readRunOutcomeSidecars(path.join(RUNS_DIR, reconcileDecision.runId), job.slug);
5251
+ transitionJob(s.jobs[idx], 'completed', {
5252
+ reason: `prior run ${reconcileDecision.runId} already completed this slug (sidecar-reconciled)`,
5253
+ source: 'spawnJob:dispatch-sidecar-reconcile',
5254
+ });
5255
+ s.jobs[idx].runId = reconcileDecision.runId;
5256
+ s.jobs[idx].finishedAt = reconcileDecision.finishedAt;
5257
+ s.jobs[idx].exitCode = 0;
5258
+ if (sidecar.outcome?.landedCommit) {
5259
+ s.jobs[idx].landedCommit = sidecar.outcome.landedCommit;
5260
+ }
5261
+ appendAuditEvent('job_dispatch_skipped_already_completed', {
5262
+ slug: job.slug,
5263
+ priorRunId: reconcileDecision.runId,
5264
+ cwd: job.cwd || defaultCwd,
5265
+ });
5266
+ console.warn(
5267
+ `[scheduler] ${job.slug}: dispatch skipped — prior run ${reconcileDecision.runId} already `
5268
+ + 'completed this slug (sidecar-reconciled)',
5269
+ );
5270
+ dispatchSkippedAlreadyCompleted = true;
5271
+ return;
5272
+ }
5273
+ // Belt-and-braces (PRD fix-plan step 3): a row about to dispatch
5274
+ // with no landedCommit of its own, whose newest sidecar for this
5275
+ // slug DOES record one, gets it backfilled before spawn so
5276
+ // verifyRun receives a real priorLandedCommit and the
5277
+ // pass_no_commit_prior_run_verified exemption can fire on this
5278
+ // run if it turns out to be another no-op re-verification.
5279
+ if (!s.jobs[idx].landedCommit && outcome?.runId) {
5280
+ const sidecar = readRunOutcomeSidecars(path.join(RUNS_DIR, outcome.runId), job.slug);
5281
+ if (sidecar.outcome?.landedCommit) {
5282
+ s.jobs[idx].landedCommit = sidecar.outcome.landedCommit;
5283
+ }
5284
+ }
5285
+ }
4474
5286
  transitionJob(s.jobs[idx], 'running', {
4475
5287
  reason: resumeTarget ? 'dispatched for resume-recovery' : 'dispatched for execution',
4476
5288
  source: 'spawnJob:dispatch',
@@ -4494,6 +5306,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4494
5306
  }
4495
5307
  });
4496
5308
  await broadcast({ flush: true });
5309
+ if (dispatchSkippedAlreadyCompleted) return;
4497
5310
 
4498
5311
  // Commit-guard baseline: snapshot the working tree BEFORE the run so the
4499
5312
  // post-run check flags only paths THIS job left dirty, not pre-existing WIP.
@@ -4586,6 +5399,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4586
5399
  let res;
4587
5400
  let worktreeLeftoverDirty = [];
4588
5401
  let worktreeIntegrationFailure = null;
5402
+ // Set only when integrateJobBranch's stderr-parsing auto-resolve fired
5403
+ // (PRD 1125) — surfaced on the job row so the Queue UI can say the merge
5404
+ // self-healed rather than silently looking like an ordinary merge.
5405
+ let mergeAutoResolved = null;
5406
+ let mergeAutoResolvedPaths = null;
4589
5407
  // A job's uncommitted-work patch, whichever isolation mode produced it —
4590
5408
  // set by EITHER branch below, never both (worktree.ok picks exactly one
4591
5409
  // shape for the whole run). Named generically (not "worktree...") because
@@ -4633,6 +5451,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4633
5451
  console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
4634
5452
  } else if (integration.integrated) {
4635
5453
  console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
5454
+ if (integration.autoResolved) {
5455
+ mergeAutoResolved = integration.autoResolved;
5456
+ mergeAutoResolvedPaths = integration.resolvedPaths || [];
5457
+ console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
5458
+ }
4636
5459
  }
4637
5460
  await jobWorktree.cleanupJobWorktree({
4638
5461
  cwd: guardCwd,
@@ -4727,7 +5550,9 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4727
5550
  }
4728
5551
 
4729
5552
  if (res.rateLimited) {
4730
- const resetIso = await refreshNextReset().catch(() => cachedNextReset);
5553
+ const logPath = path.join(runDir, `${job.slug}.log`);
5554
+ const billingResetIso = await refreshNextReset().catch(() => cachedNextReset);
5555
+ const resetIso = resolveRateLimitPauseReset(logPath, billingResetIso);
4731
5556
  const observedAt = dispatchStartedAtMs;
4732
5557
  const prevCount = consecutiveRapidRateLimitsBySlug.get(job.slug) || 0;
4733
5558
  const nextCount = nextRapidRateLimitCount(prevCount, { rateLimited: true, durationMs: res.durationMs });
@@ -4868,14 +5693,27 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4868
5693
  ranInWorktree: worktree.ok,
4869
5694
  jobSelfCommitted,
4870
5695
  legitimateNoOp: guardIsLegitimateNoOp,
4871
- isFixPlanJob: isFixPlanSlug(job.slug),
5696
+ isFixPlanJob: resolveIsFixPlan(job.slug, job.isFixPlan),
4872
5697
  verifyResult,
4873
5698
  salvagePatch,
4874
5699
  });
4875
5700
  if (guardVerdict) {
4876
- verifyResult = guardVerdict;
4877
- const what = newlyDirty.length > 0 ? `left ${newlyDirty.length} files uncommitted` : 'made no commit on an already-clean tree';
4878
- console.log(`[scheduler] commit-guard: ${job.slug} ${what} → needs_review`);
5701
+ // Already-satisfied-on-main exemption (PRD 1136): resolveCommitGuardOutcome
5702
+ // only ever touches the clean-tree/no-commit shape ('silent_no_op') —
5703
+ // a job that left dirty files behind is a genuine finish-protocol
5704
+ // violation regardless of what already landed on main, so it is
5705
+ // never routed through this check (see that function's own doc).
5706
+ const satisfyingCommits = guardVerdict.verdict === 'silent_no_op'
5707
+ ? await findSatisfyingCommitOnMain(job)
5708
+ : [];
5709
+ const finalVerdict = resolveCommitGuardOutcome(guardVerdict, satisfyingCommits);
5710
+ verifyResult = finalVerdict;
5711
+ if (finalVerdict.verdict === 'already_satisfied_on_main') {
5712
+ console.log(`[scheduler] already-satisfied-on-main: ${job.slug} → completed (${finalVerdict.satisfyingSha})`);
5713
+ } else {
5714
+ const what = newlyDirty.length > 0 ? `left ${newlyDirty.length} files uncommitted` : 'made no commit on an already-clean tree';
5715
+ console.log(`[scheduler] commit-guard: ${job.slug} ${what} → needs_review`);
5716
+ }
4879
5717
  }
4880
5718
  }
4881
5719
  }
@@ -4941,13 +5779,37 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4941
5779
  );
4942
5780
  }
4943
5781
 
5782
+ // BLOCKED_BY_FOREIGN_WIP claim scan: the executor exits non-zero for this
5783
+ // outcome (same as FAIL — see FINISH_PROTOCOL), so it is never seen by the
5784
+ // exit=0-only verifyRun call above; scanned here, against THIS run's own
5785
+ // log, before the generic non-zero-exit -> 'failed' classification below.
5786
+ // Read outside mutate() (I/O); actual manifest validation happens inside
5787
+ // mutate(), against the LIVE row's preRunDirtyPaths/carriedPaths, so a
5788
+ // stale local `job` snapshot can never be the source of truth for it.
5789
+ let foreignWipClaimedPaths = null;
5790
+ if (res.exitCode !== 0 && !res.rateLimited) {
5791
+ try {
5792
+ const { resultEvent, events } = parseLog(path.join(runDir, `${job.slug}.log`));
5793
+ if (scanSentinel(resultEvent, events) === 'blocked_by_foreign_wip') {
5794
+ foreignWipClaimedPaths = scanForeignWipPathsClaim(resultEvent, events);
5795
+ }
5796
+ } catch (e) {
5797
+ console.warn(`[scheduler] ${job.slug}: foreign-WIP verdict scan failed, falling through to ordinary failed classification`, e?.message);
5798
+ }
5799
+ }
5800
+
4944
5801
  let actuallyFailed = false;
4945
5802
  let failedJobSnapshot = null;
4946
5803
  let needsInvestigationNow = false;
4947
5804
  let investigationJobSnapshot = null;
5805
+ let investigationDeadChildSnapshot = null;
4948
5806
  let needsReviewRcaSnapshot = null;
4949
5807
  let resumeRecoveryJob = null;
4950
5808
  let resumeRecoveryTarget = null;
5809
+ let quarantineJob = null;
5810
+ let quarantinePaths = null;
5811
+ let mechanicalRecoveryJob = null;
5812
+ let mechanicalRecoveryTarget = null;
4951
5813
  let terminalNotifySnapshot = null;
4952
5814
  const newlyCompletedPrds = [];
4953
5815
  await mutate((s) => {
@@ -4960,7 +5822,44 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4960
5822
  // scheduleJobTransitions.cjs's LEGAL_TRANSITIONS) and silently
4961
5823
  // undo the cancellation. Skip — the row already reflects its real
4962
5824
  // terminal state.
4963
- if (i2 >= 0 && s.jobs[i2].status !== 'running') return;
5825
+ const finalizeDrop = evaluateFinalizeDrop({
5826
+ rowExists: i2 >= 0,
5827
+ rowStatus: i2 >= 0 ? s.jobs[i2].status : null,
5828
+ rowRunId: i2 >= 0 ? (s.jobs[i2].runId ?? null) : null,
5829
+ rowLandedCommit: i2 >= 0 ? (s.jobs[i2].landedCommit ?? null) : null,
5830
+ runId,
5831
+ landedCommit: jobLandedCommitThisRun ?? null,
5832
+ });
5833
+ if (finalizeDrop.drop) {
5834
+ // Never silent (2026-09-06 incident: a bare early-return here dropped
5835
+ // three legitimate no-op verifications of an already-shipped PRD with
5836
+ // no trace at all, leaving the row stuck 'pending' so the dispatcher
5837
+ // re-fired it three more times). 'row-not-running' covers a job
5838
+ // already moved off 'running' by someone else (namely
5839
+ // remote.cancelJob, PRD 1024) — re-finalizing anyway could
5840
+ // re-legalize the row via a legal failed->completed/needs_review edge
5841
+ // and silently undo the cancellation, so the STATUS change is still
5842
+ // skipped; only a genuinely-landed commit is stamped as a fact.
5843
+ const logFn = finalizeDrop.reason === 'row-missing' ? console.error : console.warn;
5844
+ logFn(
5845
+ `[scheduler] finalize dropped: slug=${job.slug} runId=${runId} reason=${finalizeDrop.reason} `
5846
+ + `actualStatus=${i2 >= 0 ? s.jobs[i2].status : '(row-missing)'} `
5847
+ + `rowRunId=${i2 >= 0 ? (s.jobs[i2].runId ?? '(none)') : '(none)'}`,
5848
+ );
5849
+ appendAuditEvent('job_finalize_dropped', {
5850
+ slug: job.slug,
5851
+ runId,
5852
+ reason: finalizeDrop.reason,
5853
+ actualStatus: i2 >= 0 ? s.jobs[i2].status : null,
5854
+ rowRunId: i2 >= 0 ? (s.jobs[i2].runId ?? null) : null,
5855
+ landedCommit: jobLandedCommitThisRun ?? null,
5856
+ exitCode: res.exitCode,
5857
+ });
5858
+ if (finalizeDrop.stampLandedCommit) {
5859
+ s.jobs[i2].landedCommit = finalizeDrop.stampLandedCommit;
5860
+ }
5861
+ return;
5862
+ }
4964
5863
  if (i2 >= 0) {
4965
5864
  const treatAsPending = res.rateLimited || (s.paused && s.paused.reason === 'rate_limit');
4966
5865
  if (treatAsPending) {
@@ -4972,14 +5871,66 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4972
5871
  const sigtermOverride = res.exitCode !== 0
4973
5872
  ? classifySigtermWithCommit(res.exitCode, sigtermCommitFound)
4974
5873
  : null;
5874
+ // Validated against the LIVE row's own foreign-WIP manifest — never
5875
+ // the stale outer `job` snapshot — so an unlisted path cannot
5876
+ // launder a real regression into a block (PRD: give the executor a
5877
+ // first-class verdict for "the gate failed on a sibling's in-flight
5878
+ // file", but VALIDATE the claim rather than trust it).
5879
+ const foreignWipValidation = (!sigtermOverride && foreignWipClaimedPaths !== null)
5880
+ ? validateForeignWipBlockClaim(foreignWipClaimedPaths, s.jobs[i2])
5881
+ : null;
5882
+ // Consecutive-block streak: cleared by default on every outcome and
5883
+ // only re-established below inside the validated-block branch —
5884
+ // so a completed run, an ordinary failure, or any other outcome
5885
+ // between two blocks always resets "in a row" back to zero.
5886
+ const priorForeignWipBlockCount = s.jobs[i2].foreignWipBlockCount ?? 0;
5887
+ delete s.jobs[i2].foreignWipBlockCount;
4975
5888
  if (sigtermOverride) {
4976
5889
  effectiveStatus = sigtermOverride.status;
4977
5890
  sigtermOverrideReason = sigtermOverride.reason;
5891
+ } else if (foreignWipValidation && foreignWipValidation.ok) {
5892
+ const blockCount = priorForeignWipBlockCount + 1;
5893
+ s.jobs[i2].foreignWipBlockCount = blockCount;
5894
+ s.jobs[i2].blockedByForeignWip = true;
5895
+ s.jobs[i2].foreignWipBlockedPaths = foreignWipValidation.validPaths;
5896
+ if (blockCount >= FOREIGN_WIP_BLOCK_STREAK_LIMIT) {
5897
+ // Blocked FOREIGN_WIP_BLOCK_STREAK_LIMIT times in a row on the
5898
+ // same tree: auto-requeuing again would spin forever against
5899
+ // paths that never go clean. Park for a human instead — never
5900
+ // routed through the auto-fix chain (selectAutoFixTargets
5901
+ // excludes any job.blockedByForeignWip row), since there is no
5902
+ // fix-plan to author against another job's WIP.
5903
+ effectiveStatus = 'needs_review';
5904
+ s.jobs[i2].verifierVerdict = 'blocked_by_foreign_wip_streak';
5905
+ sigtermOverrideReason = `blocked by foreign WIP ${blockCount} times in a row on persistently-dirty path(s): ${foreignWipValidation.validPaths.join(', ')} — auto-requeue exhausted, parked for human review`;
5906
+ } else {
5907
+ // Terminal-but-retryable, same shape as the existing 'skipped'
5908
+ // status (never counts as failed, never enters the auto-fix
5909
+ // chain, reconcile()'s requeueForeignWipBlockedJobs promotes it
5910
+ // straight back to 'pending' once these exact paths go clean).
5911
+ effectiveStatus = 'skipped';
5912
+ sigtermOverrideReason = `blocked by foreign WIP (attempt ${blockCount}/${FOREIGN_WIP_BLOCK_STREAK_LIMIT}): ${foreignWipValidation.validPaths.join(', ')} — will auto-requeue once these path(s) are no longer dirty`;
5913
+ }
5914
+ } else if (foreignWipValidation && !foreignWipValidation.ok) {
5915
+ // Claimed BLOCKED_BY_FOREIGN_WIP but named a path outside this
5916
+ // job's own disclosed manifest (or named none at all) — downgrade
5917
+ // to an ordinary failure and log the offending paths loudly so the
5918
+ // rejection is never silent.
5919
+ console.warn(`[scheduler] ${job.slug}: SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP downgraded to FAIL — claimed path(s) not in this job's foreign-WIP manifest: ${foreignWipValidation.invalidPaths.join(', ') || '(no FOREIGN_WIP_PATHS line)'}`);
5920
+ effectiveStatus = 'failed';
5921
+ sigtermOverrideReason = `SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP rejected — unlisted path(s) not in the disclosed foreign-WIP manifest: ${foreignWipValidation.invalidPaths.join(', ') || '(no FOREIGN_WIP_PATHS line)'}`;
4978
5922
  } else if (res.exitCode !== 0) {
4979
5923
  effectiveStatus = 'failed';
4980
5924
  } else if (
4981
5925
  !verifyResult
4982
5926
  || COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict)
5927
+ // Already-satisfied-on-main (PRD 1136): a second, independently-
5928
+ // evidenced route to 'completed' alongside COMPLETED_EQUIVALENT_
5929
+ // VERDICTS above — kept as its own explicit check rather than
5930
+ // folded into that shared Set so it can never leak into
5931
+ // reverifyNeedsReview's or runVerify.cjs's unrelated healing
5932
+ // decisions, which consult that Set for a different purpose.
5933
+ || verifyResult.verdict === 'already_satisfied_on_main'
4983
5934
  ) {
4984
5935
  effectiveStatus = 'completed';
4985
5936
  } else if (verifyResult.downgradeTo === 'pending') {
@@ -4991,7 +5942,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4991
5942
  effectiveStatus = 'needs_review';
4992
5943
  }
4993
5944
 
4994
- transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
5945
+ // already_satisfied_on_main names the satisfying sha in its own
5946
+ // reason — surface that on the completed row's statusHistory
5947
+ // instead of the generic "run finished with exit 0" every other
5948
+ // completed run gets.
5949
+ const finalizeReason = (effectiveStatus === 'completed' && verifyResult?.verdict === 'already_satisfied_on_main')
5950
+ ? verifyResult.reason
5951
+ : (sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`);
5952
+ transitionJob(s.jobs[i2], effectiveStatus, { reason: finalizeReason, source: 'spawnJob:finalize' });
4995
5953
  s.jobs[i2].finishedAt = new Date().toISOString();
4996
5954
  s.jobs[i2].exitCode = res.exitCode;
4997
5955
  s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
@@ -5000,7 +5958,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5000
5958
  } else {
5001
5959
  delete s.jobs[i2].salvagePatch;
5002
5960
  }
5003
- s.jobs[i2].error = effectiveStatus === 'needs_review'
5961
+ s.jobs[i2].error = (effectiveStatus === 'needs_review' || s.jobs[i2].blockedByForeignWip === true)
5004
5962
  ? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
5005
5963
  // A failed job (non-zero exit) never consults verifyResult above,
5006
5964
  // but a worktree integration failure is still worth surfacing on
@@ -5017,9 +5975,12 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5017
5975
  s.jobs[i2].landedCommit = jobLandedCommitThisRun;
5018
5976
  }
5019
5977
  // Persist the verifier's verdict string so the renderer can show it.
5978
+ // 'blocked_by_foreign_wip_streak' is set above from the sigterm/
5979
+ // exit-code path, never from verifyResult (which stays null on a
5980
+ // non-zero exit) — never clobber it here.
5020
5981
  if (verifyResult?.verdict && verifyResult.verdict !== 'clean') {
5021
5982
  s.jobs[i2].verifierVerdict = verifyResult.verdict;
5022
- } else {
5983
+ } else if (s.jobs[i2].verifierVerdict !== 'blocked_by_foreign_wip_streak') {
5023
5984
  delete s.jobs[i2].verifierVerdict;
5024
5985
  }
5025
5986
  // Closed-set outcome taxonomy (issue #11 list A2) so a queue row
@@ -5042,6 +6003,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5042
6003
  } else {
5043
6004
  delete s.jobs[i2].uncommittedPaths;
5044
6005
  }
6006
+ // Worktree merge self-healed (PRD 1125) — every blocking path was
6007
+ // proven byte-identical to the branch, so the duplicate was
6008
+ // discarded and the merge retried once, successfully. Surfaced so
6009
+ // the Queue UI shows a self-heal instead of an ordinary merge.
6010
+ if (mergeAutoResolved) {
6011
+ s.jobs[i2].mergeAutoResolved = mergeAutoResolved;
6012
+ s.jobs[i2].mergeAutoResolvedPaths = capDirtyPaths(mergeAutoResolvedPaths);
6013
+ } else {
6014
+ delete s.jobs[i2].mergeAutoResolved;
6015
+ delete s.jobs[i2].mergeAutoResolvedPaths;
6016
+ }
5045
6017
  // Non-blocking notes (e.g. a recovered missing-dependency probe, or a
5046
6018
  // pattern hit demoted because a materially-checkable verdict outranked
5047
6019
  // it) — surfaced even on completed jobs so the signal isn't lost.
@@ -5092,6 +6064,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5092
6064
  // takes the treatAsPending branch above and never reaches here).
5093
6065
  needsReviewRcaSnapshot = { ...s.jobs[i2] };
5094
6066
 
6067
+ // Mechanical recovery (PRD 1130): evaluated FIRST, ahead of both
6068
+ // resume-first recovery and auto-fix — a job parked with a
6069
+ // mechanically-resolvable verdict (see
6070
+ // selectMechanicalRecoveryTarget) needs no model, no plan, and no
6071
+ // depth-cap check, so it must never fall through to either.
6072
+ // Snapshot only (no I/O inside mutate()); the actual git retry
6073
+ // happens outside mutate(), below.
6074
+ const mTarget = selectMechanicalRecoveryTarget(s.jobs[i2]);
6075
+ if (mTarget) {
6076
+ mechanicalRecoveryJob = { ...s.jobs[i2] };
6077
+ mechanicalRecoveryTarget = mTarget;
6078
+ } else {
5095
6079
  // Resume-first recovery (PRD 1111): evaluated BEFORE the auto-fix
5096
6080
  // eligibility check below — a job whose verdict is
5097
6081
  // 'uncommitted_changes' with a live sessionId gets one bounded
@@ -5105,6 +6089,22 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5105
6089
  resumeRecoveryJob = { ...s.jobs[i2] };
5106
6090
  resumeRecoveryTarget = target;
5107
6091
  } else {
6092
+ // Leftover quarantine (PRD 1128): resume recovery is spent
6093
+ // (resumeRecoveryAttempted already true) and this run STILL parked
6094
+ // needs_review with uncommitted_changes — the leftovers are about
6095
+ // to sit dirty in the shared tree forever, poisoning every later
6096
+ // worktree merge for this cwd. Stamp the one-attempt marker HERE,
6097
+ // synchronously in the same mutate as this decision (mirrors
6098
+ // resumeRecoveryAttempted's own stamp-before-acting rule above),
6099
+ // so a concurrent reverifyNeedsReview pass can never double-fire
6100
+ // this. The actual git work is async and runs outside mutate(),
6101
+ // below (performLeftoverQuarantine).
6102
+ const quarantineTarget = selectLeftoverQuarantineTarget(s.jobs[i2]);
6103
+ if (quarantineTarget) {
6104
+ s.jobs[i2].leftoverQuarantineAttempted = true;
6105
+ quarantineJob = { ...s.jobs[i2] };
6106
+ quarantinePaths = quarantineTarget.paths;
6107
+ }
5108
6108
  // Same-tick auto-fix (feedback 2026-07-12): rather than waiting up to
5109
6109
  // 10 min for reverifyNeedsReview()'s periodic pass, check right here
5110
6110
  // whether this job qualifies for auto-fix (same eligibility rule
@@ -5119,8 +6119,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5119
6119
  isEligibleForImmediateAutoFix(s.jobs[i2], s.jobs, fixSlugExists)
5120
6120
  ) {
5121
6121
  const isRetryAttempt = s.jobs[i2].autoFixAttempted === true;
6122
+ const isDeadFixPlanReopen = isFixPlanDead(s.jobs[i2], s.jobs);
6123
+ if (isDeadFixPlanReopen) {
6124
+ investigationDeadChildSnapshot = s.jobs.find((x) => x.slug === fixSlugFor(s.jobs[i2])) || null;
6125
+ }
5122
6126
  s.jobs[i2].autoFixAttempted = true;
5123
6127
  if (!s.jobs[i2].runId) s.jobs[i2].runId = runId;
6128
+ if (isDeadFixPlanReopen) s.jobs[i2].autoFixReopened = true;
5124
6129
  if (isRetryAttempt) {
5125
6130
  s.jobs[i2].autoFixRetries = (s.jobs[i2].autoFixRetries ?? 0) + 1;
5126
6131
  delete s.jobs[i2].autoFixOutcome;
@@ -5129,13 +6134,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5129
6134
  investigationJobSnapshot = { ...s.jobs[i2] };
5130
6135
  }
5131
6136
  }
6137
+ }
5132
6138
  }
5133
6139
  // Auto-promote: when a fix-* PRD completes successfully, the original
5134
6140
  // failed PRD's work is logically done. Flip its status to 'completed'
5135
6141
  // so the cross-group failure gate in pickNextBatch releases. Without
5136
6142
  // this, the queue stalls indefinitely behind a stale failure even
5137
6143
  // though the auto-recovery did its job.
5138
- if (effectiveStatus === 'completed' && isFixPlanSlug(job.slug)) {
6144
+ if (effectiveStatus === 'completed' && resolveIsFixPlan(job.slug, job.isFixPlan)) {
5139
6145
  const orig = healTargetForFix(job.slug, s.jobs);
5140
6146
  if (orig) {
5141
6147
  const priorStatus = orig.status;
@@ -5207,6 +6213,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5207
6213
  });
5208
6214
  }
5209
6215
 
6216
+ if (mechanicalRecoveryJob && mechanicalRecoveryTarget) {
6217
+ console.log(`[scheduler] needs_review ${job.slug} → mechanical-recovery (re-integrating ${mechanicalRecoveryTarget.branch})`);
6218
+ performMechanicalRecovery(mechanicalRecoveryJob, mechanicalRecoveryTarget).catch((e) => {
6219
+ console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
6220
+ });
6221
+ }
6222
+
5210
6223
  if (resumeRecoveryJob && resumeRecoveryTarget) {
5211
6224
  console.log(`[scheduler] needs_review ${job.slug} → resume-recovery (session ${resumeRecoveryTarget.sessionId}, ${resumeRecoveryTarget.dirtyPaths.length} dirty path(s))`);
5212
6225
  spawnResumeRecovery(resumeRecoveryJob, resumeRecoveryTarget).catch((e) => {
@@ -5214,6 +6227,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5214
6227
  });
5215
6228
  }
5216
6229
 
6230
+ if (quarantineJob && quarantinePaths) {
6231
+ console.log(`[scheduler] needs_review ${job.slug} → quarantining ${quarantinePaths.length} leftover path(s) (resume recovery already spent)`);
6232
+ performLeftoverQuarantine(quarantineJob, quarantinePaths, guardHeadBefore).catch((e) => {
6233
+ console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
6234
+ });
6235
+ }
6236
+
5217
6237
  if (actuallyFailed && failedJobSnapshot) {
5218
6238
  // Transient-failure detector. A 143/137 exit is ALWAYS a signal kill — the
5219
6239
  // agent never self-exits with those — so the only question is WHO killed it.
@@ -5286,7 +6306,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5286
6306
  }
5287
6307
  } else if (needsInvestigationNow && investigationJobSnapshot) {
5288
6308
  console.log(`[scheduler] needs_review ${job.slug} → immediate auto-fix investigation (not waiting for periodic reverify)`);
5289
- spawnInvestigation(investigationJobSnapshot, runDir).catch((e) => {
6309
+ spawnInvestigation(investigationJobSnapshot, runDir, { deadChild: investigationDeadChildSnapshot }).catch((e) => {
5290
6310
  console.error('[scheduler] spawnInvestigation error', job.slug, e);
5291
6311
  });
5292
6312
  }
@@ -5356,11 +6376,13 @@ function tickQueue({ bypassLoadGate = false } = {}) {
5356
6376
  // ceilinged the queue at 3 while the pool the user configured said 5.
5357
6377
  const freeSlots = sessionSlots.available();
5358
6378
  const heldSlugs = await computeLaunchHolds(state);
6379
+ const satisfiedSlugsByCwd = await computeDepHistorySatisfaction(state);
5359
6380
  const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
5360
6381
  leaseHeld: quietMachineLease.isHeld(),
5361
6382
  machineInUse: sessionSlots.inUse(),
5362
6383
  now: Date.now(),
5363
6384
  heldSlugs,
6385
+ satisfiedSlugsByCwd,
5364
6386
  });
5365
6387
  if (batch.length === 0 && freeSlots === 0) {
5366
6388
  const snap = sessionSlots.snapshot();
@@ -5663,6 +6685,55 @@ function runQueueHealthSweep(jobs) {
5663
6685
  }
5664
6686
  }
5665
6687
 
6688
+ // Cross-cycle "already attempted" memory for the branch sweep (PRD 1135) —
6689
+ // in-memory only, keyed `"<cwd>::<branch>"`. Resets on restart, same as
6690
+ // consecutiveRapidRateLimitsBySlug above; the goal is "don't retry a
6691
+ // permanently-conflicting branch every cycle forever" while the process
6692
+ // stays up, not durability across a restart.
6693
+ const branchSweepAttempted = new Set();
6694
+
6695
+ /**
6696
+ * Recover any `sm-job/*` branch left stranded (unmerged, owning row terminal
6697
+ * or gone) across every known project — the other half of PRD 1135's
6698
+ * invariant alongside the reaper's liveness check above: that check stops
6699
+ * NEW stranding, this sweep recovers anything stranded before it (or by a
6700
+ * crash the liveness check can't cover). Read-only reporting plus AT MOST one
6701
+ * bounded integrateBranch attempt per branch per process lifetime (see
6702
+ * branchSweepAttempted) — never forces anything, never deletes a branch.
6703
+ * Hangs off the same cadence as runQueueHealthSweep. Never throws.
6704
+ */
6705
+ async function runBranchSweep(jobs) {
6706
+ if (process.env.SM_BRANCH_SWEEP_DISABLE === '1') return;
6707
+ try {
6708
+ for (const cwd of allProjectCwds()) {
6709
+ let sweep;
6710
+ try {
6711
+ sweep = await sweepStrandedJobBranches({ cwd, jobs, attemptedBranches: branchSweepAttempted });
6712
+ } catch (e) {
6713
+ console.warn(`[scheduler] branch sweep error for ${cwd}`, e?.message);
6714
+ continue;
6715
+ }
6716
+ for (const r of sweep.results) {
6717
+ if (r.action === 'integrated') {
6718
+ console.log(`[scheduler] branch-sweep: merged stranded branch ${r.branch} into ${cwd} HEAD`);
6719
+ appendAuditEvent('branch_sweep_integrated', { cwd, branch: r.branch, slug: r.slug });
6720
+ } else if (r.action === 'conflict') {
6721
+ console.warn(`[scheduler] branch-sweep: ${r.branch} could not be integrated: ${r.integration?.reason}`);
6722
+ appendAuditEvent('branch_sweep_conflict', { cwd, branch: r.branch, slug: r.slug, reason: r.integration?.reason });
6723
+ if (r.slug) {
6724
+ await mutate((s) => {
6725
+ const j = s.jobs.find((x) => x.slug === r.slug);
6726
+ if (j) j.strandedBranch = r.branch;
6727
+ });
6728
+ }
6729
+ }
6730
+ }
6731
+ }
6732
+ } catch (e) {
6733
+ console.warn('[scheduler] branch sweep error', e?.message);
6734
+ }
6735
+ }
6736
+
5666
6737
  /**
5667
6738
  * Scan running jobs, identify those whose claude process is provably dead OR
5668
6739
  * whose spawn never got far enough to record a runtime.pid in the first
@@ -5680,14 +6751,35 @@ async function reapDeadRunningJobs() {
5680
6751
  // status:"running" with no slug left in runningSet to trigger reconciliation.
5681
6752
  // queue.json is the source of truth for which jobs are actually running.
5682
6753
  const state = await readQueue();
5683
- const { reapable, warnings } = selectReapableJobs(state.jobs, Date.now(), {
6754
+ const { reapable, warnings, recovered } = selectReapableJobs(state.jobs, Date.now(), {
5684
6755
  pidAlive: claudePidAlive,
5685
6756
  grace: PIDLESS_SPAWN_GRACE_MS,
6757
+ findLiveProcess: (j) => findLiveProcessForJob(j, {
6758
+ worktreeDir: jobWorktree.worktreeDirFor(j.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD, j.slug),
6759
+ }),
5686
6760
  });
5687
6761
  for (const w of warnings) {
5688
6762
  console.warn(`[scheduler] reapDeadRunningJobs: ${w.reason} slug=${w.slug} — leaving row alone`);
5689
6763
  }
5690
6764
 
6765
+ // A pidless row whose process was proven alive by the /proc liveness
6766
+ // scan must never be terminalized (2026-09-06 incident — see
6767
+ // findLiveProcessForJob's header). Re-stamp the recovered pid so future
6768
+ // cycles see it as an ordinary live-pid row, and stop here for it.
6769
+ if (recovered.length) {
6770
+ await mutate(async (s) => {
6771
+ for (const r of recovered) {
6772
+ const idx = s.jobs.findIndex((x) => x.slug === r.slug);
6773
+ if (idx < 0 || s.jobs[idx].status !== 'running') continue;
6774
+ s.jobs[idx].runtime = { ...(s.jobs[idx].runtime || {}), pid: r.pid };
6775
+ }
6776
+ });
6777
+ for (const r of recovered) {
6778
+ console.log(`[scheduler] reapDeadRunningJobs: pid=${r.pid} recovered by /proc liveness scan for slug=${r.slug} — row stays running`);
6779
+ appendAuditEvent('job_pid_recovered_by_liveness_scan', { slug: r.slug, pid: r.pid });
6780
+ }
6781
+ }
6782
+
5691
6783
  const dead = [];
5692
6784
  for (const { slug, pid, pidless, reason } of reapable) {
5693
6785
  const j = state.jobs.find((x) => x.slug === slug);
@@ -5698,15 +6790,18 @@ async function reapDeadRunningJobs() {
5698
6790
  // 'no_result' → non-success below → filed as failed, never completed.
5699
6791
  const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
5700
6792
  // A pidless reap means the spawn never got far enough to record a
5701
- // pid — the gate could not possibly have run, regardless of what
5702
- // classifyRunOutcome makes of an absent/empty log.
5703
- const gateOutcome = pidless ? 'never_ran' : mapOutcomeToGateOutcome(outcome);
5704
- dead.push({ slug, pid, outcome, gateOutcome, pidless, reason });
6793
+ // pid — the gate could not possibly have run. But `never_ran` is only
6794
+ // a true claim when the run dir produced no log output at all; a log
6795
+ // with real content proves the job DID run (see
6796
+ // resolvePidlessGateOutcome's header).
6797
+ const gateOutcome = pidless ? resolvePidlessGateOutcome(outcome, logHasOutput(logPath)) : mapOutcomeToGateOutcome(outcome);
6798
+ dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath });
5705
6799
  }
5706
6800
 
5707
6801
  queueHealthSweepCycle += 1;
5708
6802
  if (queueHealthSweepCycle % QUEUE_HEALTH_SWEEP_EVERY_N_CYCLES === 0) {
5709
6803
  runQueueHealthSweep(state.jobs);
6804
+ await runBranchSweep(state.jobs);
5710
6805
  }
5711
6806
 
5712
6807
  if (dead.length === 0) return;
@@ -5718,8 +6813,9 @@ async function reapDeadRunningJobs() {
5718
6813
  // the same still-active rate limit — the spin loop this PRD exists to
5719
6814
  // stop. Done once, outside mutate(), before finalizing any row below.
5720
6815
  if (dead.some((d) => d.outcome === 'rate_limited')) {
5721
- const resetIso = await refreshNextReset().catch(() => cachedNextReset);
5722
6816
  const triggering = dead.find((d) => d.outcome === 'rate_limited');
6817
+ const billingResetIso = await refreshNextReset().catch(() => cachedNextReset);
6818
+ const resetIso = resolveRateLimitPauseReset(triggering.logPath, billingResetIso);
5723
6819
  const triggeringRow = triggering ? state.jobs.find((x) => x.slug === triggering.slug) : null;
5724
6820
  const observedAtMs = triggeringRow?.startedAt ? Date.parse(triggeringRow.startedAt) : null;
5725
6821
  // Same rapid-repeat circuit breaker spawnJob's own res.rateLimited
@@ -5739,6 +6835,64 @@ async function reapDeadRunningJobs() {
5739
6835
  await setPaused('rate_limit', resetIso, { observedAt: observedAtMs, force: forceHardPause });
5740
6836
  }
5741
6837
 
6838
+ // Prove integration BEFORE entering mutate() (PRD 1133): the check below
6839
+ // shells out to git — including a full `git fetch --all --prune` (up to
6840
+ // 20s) via computeCommittedDuringRun's committedInWindow, plus up to two
6841
+ // 2s retry sleeps when nothing landed — for every dead job that needs the
6842
+ // in-place fallback. mutate() serializes through ONE global mutateTail
6843
+ // promise chain shared by every project's dispatch/admin/cancel mutation,
6844
+ // same reason the rate-limit handling above already runs outside it — so
6845
+ // doing this git work inside the mutate() callback would stall the whole
6846
+ // scheduler's queue writes for the sum of these calls across every dead
6847
+ // job in the batch (e.g. an app-restart reap that dead-letters several
6848
+ // running jobs at once).
6849
+ const integrationResults = new Map();
6850
+ for (const d of dead) {
6851
+ if (d.outcome !== 'success') continue;
6852
+ const row = state.jobs.find((x) => x.slug === d.slug);
6853
+ if (!row) continue;
6854
+ const rowCwd = row.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD;
6855
+ if (!isGitRepoSync(rowCwd)) continue;
6856
+ const branch = jobWorktree.branchNameFor(d.slug);
6857
+ const integrated = await isBranchAlreadyIntegrated(rowCwd, branch);
6858
+ const headNow = await gitHead(rowCwd);
6859
+ const guardHeadBeforeVal = row.guardHeadBefore || null;
6860
+ let landedCommit = null;
6861
+ let notLandedInfo = null;
6862
+ let effectiveSuccess = true;
6863
+ if (integrated === null) {
6864
+ // No sm-job/<slug> branch — an in-place run (or worktree isolation
6865
+ // disabled). Prove landing via the exact same HEAD-advance evidence
6866
+ // the live commit-guard already uses — reused, not re-derived.
6867
+ const committed = await computeCommittedDuringRun(
6868
+ rowCwd, guardHeadBeforeVal, headNow, row.startedAt, new Date().toISOString(),
6869
+ );
6870
+ if (committed) {
6871
+ if (guardHeadBeforeVal && headNow && headNow !== guardHeadBeforeVal) landedCommit = headNow;
6872
+ } else {
6873
+ effectiveSuccess = false;
6874
+ notLandedInfo = {
6875
+ verdict: 'reaped_without_integration',
6876
+ reason: 'no commit landed during the run window — HEAD never advanced',
6877
+ };
6878
+ }
6879
+ } else if (integrated === true) {
6880
+ if (guardHeadBeforeVal && headNow && headNow !== guardHeadBeforeVal) landedCommit = headNow;
6881
+ } else {
6882
+ // Branch exists and still holds commits never merged into rowCwd's
6883
+ // HEAD — the exact PRD 1118 shape. Never merge it here (see
6884
+ // isBranchAlreadyIntegrated's header comment); park for review
6885
+ // instead, naming the branch so the work is easy to find and land by
6886
+ // hand or via mechanical recovery.
6887
+ effectiveSuccess = false;
6888
+ notLandedInfo = {
6889
+ verdict: 'reaped_without_integration',
6890
+ reason: `worktree branch ${branch} still holds unintegrated work — never merged into ${rowCwd}`,
6891
+ };
6892
+ }
6893
+ integrationResults.set(d.slug, { effectiveSuccess, landedCommit, notLandedInfo });
6894
+ }
6895
+
5742
6896
  await mutate(async (s) => {
5743
6897
  for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
5744
6898
  const idx = s.jobs.findIndex((x) => x.slug === slug);
@@ -5777,12 +6931,38 @@ async function reapDeadRunningJobs() {
5777
6931
  console.error(`[scheduler] reapDeadRunningJobs: in-place salvage failed for ${slug}`, e);
5778
6932
  }
5779
6933
  }
6934
+ // Prove integration before this reap is allowed to say 'completed'
6935
+ // (PRD 1133): a reaped job's owning process vanished before
6936
+ // spawnJob's own post-run integration/commit-guard ever ran, so
6937
+ // 'outcome=success' alone (a clean result event in the log) is not
6938
+ // proof anything actually landed — the two live incidents this PRD
6939
+ // exists for (1118, starry-night-ships 224) both had exactly that
6940
+ // shape. Only evaluated for a genuinely successful, non-rate-limited
6941
+ // outcome; a non-git cwd (isGitRepoSync false) skips this entirely,
6942
+ // preserving today's behaviour exactly. The actual git work already
6943
+ // ran ABOVE, before this mutate() call, into integrationResults — see
6944
+ // that block's own header comment for why it must not run in here.
6945
+ let effectiveSuccess = success;
6946
+ let landedCommit = null;
6947
+ let notLandedInfo = null;
6948
+ if (success && !rateLimited) {
6949
+ const ir = integrationResults.get(slug);
6950
+ if (ir) {
6951
+ effectiveSuccess = ir.effectiveSuccess;
6952
+ landedCommit = ir.landedCommit;
6953
+ notLandedInfo = ir.notLandedInfo;
6954
+ }
6955
+ }
6956
+
5780
6957
  const leftoverSuffix = deltaPaths && deltaPaths.length
5781
6958
  ? ` — left ${deltaPaths.length} files uncommitted`
5782
6959
  : '';
6960
+ const baseReason = notLandedInfo
6961
+ ? `reaped: ${notLandedInfo.reason}`
6962
+ : (pidless ? reason : `reaped: process gone (outcome=${outcome})`);
5783
6963
  const transitionReason = rateLimited
5784
6964
  ? `reaped: rate limit detected — reset to pending, not failed (outcome=${outcome})${leftoverSuffix}`
5785
- : (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
6965
+ : baseReason + leftoverSuffix;
5786
6966
 
5787
6967
  if (rateLimited) {
5788
6968
  // Retryable, never terminal (PRD 1117) — same resetJobFields path
@@ -5791,11 +6971,18 @@ async function reapDeadRunningJobs() {
5791
6971
  // paused-for-rate-limit reset: fresh runId/startedAt/exitCode.
5792
6972
  resetJobFields(s.jobs[idx], transitionReason, { source: 'reapDeadRunningJobs:rate-limit' });
5793
6973
  } else {
5794
- transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
5795
- s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
6974
+ const targetStatus = effectiveSuccess ? 'completed' : (notLandedInfo ? 'needs_review' : 'failed');
6975
+ transitionJob(s.jobs[idx], targetStatus, { reason: transitionReason, source: 'reapDeadRunningJobs' });
6976
+ s.jobs[idx].exitCode = effectiveSuccess ? 0 : (s.jobs[idx].exitCode ?? 1);
5796
6977
  s.jobs[idx].finishedAt = new Date().toISOString();
5797
- s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
6978
+ s.jobs[idx].error = effectiveSuccess ? null : `${transitionReason} (outcome=${outcome})`;
5798
6979
  s.jobs[idx].gateOutcome = gateOutcome;
6980
+ if (notLandedInfo) {
6981
+ s.jobs[idx].verifierVerdict = notLandedInfo.verdict;
6982
+ } else {
6983
+ delete s.jobs[idx].verifierVerdict;
6984
+ }
6985
+ if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
5799
6986
  }
5800
6987
  delete s.jobs[idx].runtime;
5801
6988
  delete s.jobs[idx].guardBaseline;
@@ -6025,10 +7212,16 @@ const MAX_INVESTIGATION_DEPTH = 1;
6025
7212
  * no recorded investigationDepth (a job already in the queue before this
6026
7213
  * depth tracking shipped) is treated as excluded too, preserving the
6027
7214
  * pre-existing blanket-exclusion behavior for legacy jobs — no retroactive
6028
- * migration. Non-fix-plan slugs are never capped here. Exported for tests.
7215
+ * migration. Non-fix-plan jobs are never capped here.
7216
+ *
7217
+ * `isFixPlan` (PRD 1131) is the job's own persisted classification stamp
7218
+ * (see lib/fixPlanSlug.cjs's resolveIsFixPlan) — an explicit true/false wins
7219
+ * over the slug; only a row with the field entirely absent (persisted
7220
+ * before this change shipped) falls back to the legacy slug-only heuristic.
7221
+ * Exported for tests.
6029
7222
  */
6030
- function isFixPlanBeyondDepthCap(slug, investigationDepth) {
6031
- if (!isFixPlanSlug(slug)) return false;
7223
+ function isFixPlanBeyondDepthCap(slug, investigationDepth, isFixPlan) {
7224
+ if (!resolveIsFixPlan(slug, isFixPlan)) return false;
6032
7225
  if (investigationDepth == null) return true;
6033
7226
  return investigationDepth >= MAX_INVESTIGATION_DEPTH + 1;
6034
7227
  }
@@ -6081,12 +7274,18 @@ function isUnresolvableNeedsReview(job, { hasRunDir }) {
6081
7274
  * ('no-plan', 'error', and unstamped/undefined) — mirrors the retry
6082
7275
  * eligibility rule in selectAutoFixTargets so a job can never be retry-
6083
7276
  * eligible there and simultaneously un-annotatable here.
7277
+ *
7278
+ * A parent stamped `autoFixReopened: true` (its dead fix-plan child earned
7279
+ * it exactly one further attempt — see isFixPlanDead) is a separate
7280
+ * exhaustion path: it is spent as soon as that second investigation
7281
+ * concludes with ANY outcome, including another 'plan' — a reopened parent
7282
+ * never gets a third attempt, so unlike the fresh case a 'plan' outcome does
7283
+ * not exempt it here.
6084
7284
  */
6085
7285
  function isExhaustedAutoFix(job) {
6086
- return !!job && job.status === 'needs_review'
6087
- && job.autoFixAttempted === true
6088
- && job.autoFixOutcome !== 'plan'
6089
- && (job.autoFixRetries ?? 0) >= 1;
7286
+ if (!job || job.status !== 'needs_review' || job.autoFixAttempted !== true) return false;
7287
+ if (job.autoFixReopened === true) return job.autoFixOutcome != null;
7288
+ return job.autoFixOutcome !== 'plan' && (job.autoFixRetries ?? 0) >= 1;
6090
7289
  }
6091
7290
 
6092
7291
  /**
@@ -6102,6 +7301,31 @@ function isPlanUnqueued(job, queuedSlugs) {
6102
7301
  return !queuedSlugs.has(fixSlugFor(job));
6103
7302
  }
6104
7303
 
7304
+ // Terminal-and-not-completed statuses a fix-plan child can die in — see
7305
+ // isFixPlanDead.
7306
+ const DEAD_FIX_CHILD_STATUSES = new Set(['needs_review', 'failed', 'quarantined', 'skipped']);
7307
+
7308
+ /**
7309
+ * Pure predicate: a parent stuck at outcome 'plan' whose own fix-plan child
7310
+ * (fixSlugFor(job)) has ITSELF died — reached a terminal non-completed
7311
+ * status — with nothing left in the ladder that will ever revisit either
7312
+ * row again (selectAutoFixTargets skips a 'plan' outcome outright, and
7313
+ * isPlanUnqueued only fires when the child never reached the queue at all,
7314
+ * which isn't true once a dead child row exists). `job.autoFixReopened`
7315
+ * gates this to exactly once per parent — once stamped, this always returns
7316
+ * false so the parent can never be reopened a second time. Exported for
7317
+ * tests.
7318
+ */
7319
+ function isFixPlanDead(job, jobsInProject) {
7320
+ if (!job || job.status !== 'needs_review') return false;
7321
+ if (job.autoFixOutcome !== 'plan') return false;
7322
+ if (job.autoFixReopened === true) return false;
7323
+ const fixSlug = fixSlugFor(job);
7324
+ const child = (jobsInProject || []).find((j) => j.slug === fixSlug);
7325
+ if (!child) return false;
7326
+ return DEAD_FIX_CHILD_STATUSES.has(child.status);
7327
+ }
7328
+
6105
7329
  /**
6106
7330
  * Pure predicate: is this job eligible for the boot re-verify self-heal? Only
6107
7331
  * needs_review jobs with a run log (own or backfilled via resolveRunId) AND a
@@ -6238,6 +7462,11 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
6238
7462
  const slugsInQueue = new Set(jobs.map((j) => j.slug));
6239
7463
  return jobs.filter((job) => {
6240
7464
  if (job.status !== 'needs_review') return false;
7465
+ // A job parked here because it was blocked by a sibling's foreign WIP
7466
+ // three times in a row (never its own regression) has nothing for a
7467
+ // fix-plan investigation to diagnose — there is no code defect to
7468
+ // author a PRD against, only another job's still-uncommitted tree.
7469
+ if (job.blockedByForeignWip === true) return false;
6241
7470
  // A stale re-run whose work already shipped (rcaReport's 'already-shipped'
6242
7471
  // class) must never buy a fix-plan PRD — there is nothing to fix, and the
6243
7472
  // correct recovery (archiving the PRD) is a human/reconcile action, not
@@ -6247,16 +7476,30 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
6247
7476
  // bounded `--resume` attempt must never also become a fix-plan target
6248
7477
  // in the same pass — see spawnInvestigation's own identical guard.
6249
7478
  if (selectResumeRecoveryTarget(job)) return false;
7479
+ // Mechanical recovery (PRD 1130): a job eligible for a pure-git retry
7480
+ // must never also become a fix-plan target — it needs no plan and no
7481
+ // model. Defensive: today's single mechanically-resolvable verdict
7482
+ // (worktree_integration_failed) is already excluded below via the depth
7483
+ // cap, but this must hold even if that stops being true.
7484
+ if (selectMechanicalRecoveryTarget(job)) return false;
6250
7485
  const runId = job.runId || resolveJobRunId(job);
6251
7486
  if (!runId) return false;
6252
- if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth)) return false;
7487
+ if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth, job.isFixPlan)) return false;
7488
+ // A dead fix-plan child (PRD 1129) earns its parent exactly one further
7489
+ // attempt, bypassing the normal 'plan' exclusion and the fix-slug/queue
7490
+ // membership checks below — those checks exist to stop a FRESH
7491
+ // investigation from clobbering a live sibling, but here the sibling is
7492
+ // dead and reusing its slug is the whole point of the reopen.
7493
+ const dead = isFixPlanDead(job, jobs);
6253
7494
  if (job.autoFixAttempted) {
6254
- const retryEligible = job.autoFixOutcome === 'no-plan'
7495
+ const retryEligible = dead
7496
+ || job.autoFixOutcome === 'no-plan'
6255
7497
  || job.autoFixOutcome === 'error'
6256
7498
  || job.autoFixOutcome == null;
6257
7499
  if (!retryEligible) return false;
6258
- if ((job.autoFixRetries ?? 0) >= 1) return false;
7500
+ if (!dead && (job.autoFixRetries ?? 0) >= 1) return false;
6259
7501
  }
7502
+ if (dead) return true;
6260
7503
  const fixSlug = fixSlugFor(job);
6261
7504
  if (fixSlugExists(fixSlug)) return false;
6262
7505
  if (slugsInQueue.has(fixSlug)) return false;
@@ -6366,7 +7609,15 @@ async function reverifyNeedsReview() {
6366
7609
  // Still needs_review after the existing heal pass — widen the evidence
6367
7610
  // window before giving up on it entirely (unchanged heal semantics for
6368
7611
  // rows that already qualified above; this only adds an annotation).
6369
- if (stillOpen) {
7612
+ // Skipped when a fix-plan investigation was already minted for this row
7613
+ // (job.autoFixAttempted) — PRD 1136: 'looks done, confirm before
7614
+ // archiving' and 'a -fix- child is already investigating this' are two
7615
+ // different claims about the SAME evidence, and stamping both leaves a
7616
+ // human reading two contradictory signals off one row. autoFixAttempted
7617
+ // is stamped synchronously in spawnJob's same-tick auto-fix branch,
7618
+ // always before this periodic/boot pass can run against the same row, so
7619
+ // this check reliably catches the only order that can occur.
7620
+ if (stillOpen && job.autoFixAttempted !== true) {
6370
7621
  const looksDone = await computeLooksDone(job);
6371
7622
  if (looksDone) {
6372
7623
  looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
@@ -6435,7 +7686,7 @@ async function reverifyNeedsReview() {
6435
7686
  const promotedPrds = [];
6436
7687
  await mutate((s) => {
6437
7688
  for (const job of s.jobs) {
6438
- if (job.status !== 'completed' || !isFixPlanSlug(job.slug)) continue;
7689
+ if (job.status !== 'completed' || !resolveIsFixPlan(job.slug, job.isFixPlan)) continue;
6439
7690
  const orig = healTargetForFix(job.slug, s.jobs);
6440
7691
  if (!orig) continue;
6441
7692
  const priorStatus = orig.status;
@@ -6519,6 +7770,23 @@ async function reverifyNeedsReview() {
6519
7770
  ? await readQueue()
6520
7771
  : afterHealForAnnotate;
6521
7772
 
7773
+ // Mechanical recovery (PRD 1130): evaluated first, ahead of both
7774
+ // resume-first recovery and auto-fix below — catches a job whose
7775
+ // mechanically-resolvable verdict this periodic pass finds still eligible
7776
+ // (e.g. one already parked before this rung shipped, or one the same-tick
7777
+ // check in spawnJob missed because the app restarted in between). Depth
7778
+ // never disqualifies it, so it runs regardless of investigationDepth.
7779
+ {
7780
+ for (const job of queueForResumeAndAutofix.jobs) {
7781
+ const target = selectMechanicalRecoveryTarget(job);
7782
+ if (!target) continue;
7783
+ console.log(`[scheduler] mechanical-recovery: needs_review ${job.slug} → re-integrating ${target.branch}`);
7784
+ performMechanicalRecovery(job, target).catch((e) => {
7785
+ console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
7786
+ });
7787
+ }
7788
+ }
7789
+
6522
7790
  // Resume-first recovery (PRD 1111): before any fix-plan investigation is
6523
7791
  // authored below, offer the bounded one-attempt `--resume` dispatch to any
6524
7792
  // needs_review job this periodic pass finds still eligible — e.g. one the
@@ -6537,6 +7805,26 @@ async function reverifyNeedsReview() {
6537
7805
  }
6538
7806
  }
6539
7807
 
7808
+ // Leftover quarantine (PRD 1128), periodic pass: catches a job parked
7809
+ // needs_review with resume recovery already spent BEFORE this feature
7810
+ // shipped, or one the same-tick check in spawnJob missed because the app
7811
+ // restarted in between. Stamps the one-attempt marker in its own mutate
7812
+ // BEFORE the async git work starts (same race-closing rule as the resume
7813
+ // loop above and spawnJob's own dispatch stamp).
7814
+ {
7815
+ for (const job of queueForResumeAndAutofix.jobs) {
7816
+ const quarantineTarget = selectLeftoverQuarantineTarget(job);
7817
+ if (!quarantineTarget) continue;
7818
+ console.log(`[scheduler] leftover-quarantine: needs_review ${job.slug} → quarantining ${quarantineTarget.paths.length} leftover path(s)`);
7819
+ mutate((s) => {
7820
+ const j = s.jobs.find((x) => x.slug === job.slug);
7821
+ if (j) j.leftoverQuarantineAttempted = true;
7822
+ }).then(() => performLeftoverQuarantine(job, quarantineTarget.paths)).catch((e) => {
7823
+ console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
7824
+ });
7825
+ }
7826
+ }
7827
+
6540
7828
  // Auto-fix: spawn a fix-plan investigation for each job still in
6541
7829
  // needs_review after the heal pass (kill-switch: SM_AUTOFIX_DISABLE=1).
6542
7830
  // spawnInvestigation early-returns once investigationsInFlight reaches
@@ -6550,23 +7838,30 @@ async function reverifyNeedsReview() {
6550
7838
  const runId = job.runId || resolveRunId(job);
6551
7839
  const runDir = path.join(RUNS_DIR, runId);
6552
7840
  const isRetryAttempt = job.autoFixAttempted === true;
7841
+ const isDeadFixPlanReopen = isFixPlanDead(job, queueForResumeAndAutofix.jobs);
7842
+ const deadChild = isDeadFixPlanReopen
7843
+ ? queueForResumeAndAutofix.jobs.find((j) => j.slug === fixSlugFor(job))
7844
+ : null;
6553
7845
  // Persist the attempt BEFORE spawning — a crash mid-investigation still
6554
7846
  // counts it (mirrors orphanRetries). Safe even when the slot is busy: the
6555
7847
  // investigation is queued and drained as slots free, so it is genuinely
6556
- // attempted rather than silently dropped.
7848
+ // attempted rather than silently dropped. autoFixReopened is stamped in
7849
+ // this SAME mutate so a crash between selection and dispatch can never
7850
+ // leave the parent re-eligible for a second reopen (PRD 1129).
6557
7851
  await mutate((s) => {
6558
7852
  const j = s.jobs.find((x) => x.slug === job.slug);
6559
7853
  if (j) {
6560
7854
  j.autoFixAttempted = true;
6561
7855
  if (!j.runId && runId) j.runId = runId;
7856
+ if (isDeadFixPlanReopen) j.autoFixReopened = true;
6562
7857
  if (isRetryAttempt) {
6563
7858
  j.autoFixRetries = (j.autoFixRetries ?? 0) + 1;
6564
7859
  delete j.autoFixOutcome;
6565
7860
  }
6566
7861
  }
6567
7862
  });
6568
- console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'})`);
6569
- spawnInvestigation(job, runDir).catch((e) => {
7863
+ console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'}${isDeadFixPlanReopen ? ', dead fix-plan child reopen' : ''})`);
7864
+ spawnInvestigation(job, runDir, { deadChild }).catch((e) => {
6570
7865
  console.error('[scheduler] auto-fix spawnInvestigation error', job.slug, e);
6571
7866
  });
6572
7867
  }
@@ -7657,6 +8952,36 @@ const remote = {
7657
8952
  return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
7658
8953
  }
7659
8954
 
8955
+ // Write-time FK check for a patched dependsOn (PRD 1124), reusing the
8956
+ // SAME resolution rule scheduler_create_prd's prdCreate.cjs applies (exact
8957
+ // slug, else bare-name after stripping one leading `NN-`) so update and
8958
+ // create can never disagree about what a dependsOn entry resolves to. An
8959
+ // explicit empty array CLEARS the dependency and skips validation — there
8960
+ // is nothing to resolve. A listPrds() read failure is skipped-with-a-
8961
+ // warning, matching createPrd's tolerance for an I/O hiccup.
8962
+ if (frontmatter && Array.isArray(frontmatter.dependsOn) && frontmatter.dependsOn.length) {
8963
+ let listing;
8964
+ try {
8965
+ listing = await this.listPrds({ cwd, limit: Number.MAX_SAFE_INTEGER });
8966
+ } catch (e) {
8967
+ console.warn(`[scheduler] updatePrd: dependsOn validation skipped (listPrds failed): ${e?.message ?? e}`);
8968
+ listing = null;
8969
+ }
8970
+ if (listing) {
8971
+ const candidateSlugs = (listing.prds ?? []).map((p) => p.slug);
8972
+ for (const dep of frontmatter.dependsOn) {
8973
+ if (resolveDepSlug(dep, candidateSlugs).length > 0) continue;
8974
+ const near = findNearMatches(dep, candidateSlugs);
8975
+ const suggestion = near.length ? ` Closest existing slug(s): ${near.join(', ')}.` : '';
8976
+ return {
8977
+ ok: false,
8978
+ error: `dependsOn entry "${dep}" does not resolve to any existing PRD in this project.${suggestion} ` +
8979
+ 'Pass the bare name (preferred) or the exact NN-prefixed slug of an existing PRD.',
8980
+ };
8981
+ }
8982
+ }
8983
+ }
8984
+
7660
8985
  let dir = null;
7661
8986
  let filePath = null;
7662
8987
  if (cwd) {
@@ -7787,4 +9112,162 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
7787
9112
  });
7788
9113
  }
7789
9114
 
7790
- module.exports = { classifyQueueStarvation, runQueueStarvationWatchdog, QUEUE_STARVATION_MS, computeBlockedChains, stripAppOwnedChurn, findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure, setPaused, clearPause, tickQueue, runDueJobs, isCooldownSuppressed, nextRapidRateLimitCount, CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD, RAPID_RATE_LIMIT_WINDOW_MS, MANUAL_PAUSE_COOLDOWN_MS, RUNS_DIR, pickRunDir };
9115
+ module.exports = {
9116
+ classifyQueueStarvation,
9117
+ runQueueStarvationWatchdog,
9118
+ QUEUE_STARVATION_MS,
9119
+ computeBlockedChains,
9120
+ stripAppOwnedChurn,
9121
+ findOverrunningJobs,
9122
+ JOB_OVERRUN_FACTOR,
9123
+ JOB_OVERRUN_FLOOR_MS,
9124
+ registerScheduleHandlers,
9125
+ attachWindow,
9126
+ init,
9127
+ ROOT,
9128
+ PRDS_DIR,
9129
+ healRefusalReason,
9130
+ writeQueue,
9131
+ reconcile,
9132
+ reconcileSourcePromptId,
9133
+ allocateParallelGroup,
9134
+ selectHistoryJobs,
9135
+ parsePorcelain,
9136
+ FINISH_PROTOCOL,
9137
+ IDLE_OUTPUT_KILL_MS,
9138
+ BASH_DEFAULT_TIMEOUT_MS,
9139
+ BASH_MAX_TIMEOUT_MS,
9140
+ remote,
9141
+ pickNextBatch,
9142
+ pickForProject,
9143
+ reapDeadRunningJobs,
9144
+ runBranchSweep,
9145
+ pollRecoveryClearSource,
9146
+ memoryLimitedBatchSize,
9147
+ availableForJobs,
9148
+ reverifyNeedsReview,
9149
+ isRescanCandidate,
9150
+ isFailedUnverifiedShaped,
9151
+ computeLooksDone,
9152
+ isPromotableOriginal,
9153
+ selectAutoFixTargets,
9154
+ applyRcaClassification,
9155
+ isEligibleForImmediateAutoFix,
9156
+ resolveRunId,
9157
+ isUnresolvableNeedsReview,
9158
+ isExhaustedAutoFix,
9159
+ isPlanUnqueued,
9160
+ isFixPlanDead,
9161
+ fixSlugFor,
9162
+ healTargetForFix,
9163
+ buildInvestigationPrompt,
9164
+ isGitRepoSync,
9165
+ committedInWindow,
9166
+ computeCommittedDuringRun,
9167
+ classifySigtermWithCommit,
9168
+ evaluateFinalizeDrop,
9169
+ evaluateDispatchSidecarReconcile,
9170
+ isQueueRowRegression,
9171
+ readRunOutcomeSidecars,
9172
+ isFixPlanSlug,
9173
+ classifyDiscoveredFixPlan,
9174
+ resolveIsFixPlan,
9175
+ isFixPlanBeyondDepthCap,
9176
+ MAX_INVESTIGATION_DEPTH,
9177
+ forceTickOutcome,
9178
+ applyPauseCleared,
9179
+ detectNetworkErrorInLog,
9180
+ detectRateLimitInLog,
9181
+ classifyFailureOutcome,
9182
+ commitGuardVerdict,
9183
+ findSatisfyingCommitOnMain,
9184
+ leftoverFieldsFrom,
9185
+ applyLeftoverFields,
9186
+ LEFTOVER_PATHS_CAP,
9187
+ capDirtyPaths,
9188
+ buildForeignWipSection,
9189
+ PRE_RUN_DIRTY_PATHS_CAP,
9190
+ FOREIGN_WIP_DELIMITER,
9191
+ FOREIGN_WIP_END_DELIMITER,
9192
+ TRANSIENT_RETRY_CAP,
9193
+ buildScheduleStatePayload,
9194
+ partitionBootOrphans,
9195
+ applyOrphanOutcome,
9196
+ BOOT_ORPHAN_KILL_GRACE_MS,
9197
+ registerAdminRoutes,
9198
+ notifyOriginatingTab,
9199
+ notifyNeedsReview,
9200
+ isNotifiableTerminalStatus,
9201
+ extractResultTextFromLog,
9202
+ candidatePrdsDirs,
9203
+ candidateArchivedPrdsDirs,
9204
+ resolveArchivedPrdStatus,
9205
+ prdDirForCwd,
9206
+ prdPathForJob,
9207
+ archivedPrdPathForJob,
9208
+ archivedTwinExists,
9209
+ findPrdDir,
9210
+ resolveVerifyPrdPath,
9211
+ resolveFixPlanPath,
9212
+ resolveNotifyPrd,
9213
+ runPrdMigration,
9214
+ consolidateAllFlatPrds,
9215
+ shouldSkipInvestigationForCleanRun,
9216
+ archiveCompletedPrd,
9217
+ retireCompletedSlugs,
9218
+ SCHEDULER_BOOTED_AT,
9219
+ SCHEDULER_CODE_SHA,
9220
+ resetJobFields,
9221
+ executeJob,
9222
+ prdArchivedSkipResult,
9223
+ spawnJob,
9224
+ listPrdsInternal,
9225
+ computeStallSummary,
9226
+ findStaleQuarantinedJobs,
9227
+ QUARANTINE_ESCALATE_MS,
9228
+ applyClearQueueVictims,
9229
+ PIDLESS_SPAWN_GRACE_MS,
9230
+ findStrandedInvestigations,
9231
+ INVESTIGATION_MAX_MS,
9232
+ stashList,
9233
+ parseStashLine,
9234
+ pathsChangedSince,
9235
+ restoreSpecificStash,
9236
+ evaluateSharedTreeGuard,
9237
+ checkSharedTreeGuard,
9238
+ uncommittedChanges,
9239
+ gitHead,
9240
+ isBranchAlreadyIntegrated,
9241
+ selectResumeRecoveryTarget,
9242
+ buildResumeRecoveryPreamble,
9243
+ buildClaudeSpawnArgs,
9244
+ spawnResumeRecovery,
9245
+ selectMechanicalRecoveryTarget,
9246
+ performMechanicalRecovery,
9247
+ MECHANICALLY_RESOLVABLE_VERDICTS,
9248
+ selectLeftoverQuarantineTarget,
9249
+ quarantineLeftovers,
9250
+ performLeftoverQuarantine,
9251
+ spawnInvestigation,
9252
+ computeLaunchHolds,
9253
+ computeDepHistorySatisfaction,
9254
+ handleLaunchFailure,
9255
+ applyLaunchFailure,
9256
+ setPaused,
9257
+ clearPause,
9258
+ tickQueue,
9259
+ runDueJobs,
9260
+ isCooldownSuppressed,
9261
+ nextRapidRateLimitCount,
9262
+ CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
9263
+ RAPID_RATE_LIMIT_WINDOW_MS,
9264
+ MANUAL_PAUSE_COOLDOWN_MS,
9265
+ RUNS_DIR,
9266
+ pickRunDir,
9267
+ resolveRateLimitPauseReset,
9268
+ computeEffectiveResumeAt,
9269
+ computeResumeDelay,
9270
+ FOREIGN_WIP_BLOCK_STREAK_LIMIT,
9271
+ validateForeignWipBlockClaim,
9272
+ requeueForeignWipBlockedJobs,
9273
+ };