claude-code-session-manager 0.75.3 → 0.76.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. package/dist/assets/{AgentLibrary-CzQqcObq.js → AgentLibrary-CBx9l4zN.js} +1 -1
  2. package/dist/assets/{DataModel-Bj_WlLz8.js → DataModel-Bf0EIE_t.js} +1 -1
  3. package/dist/assets/{History-DnSi_OHm.js → History-CpdtWhC8.js} +1 -1
  4. package/dist/assets/{Hooks-0BB0dp3S.js → Hooks-DyUbMDmg.js} +1 -1
  5. package/dist/assets/{HostBilko-DHpwwsLQ.js → HostBilko-By-wIpry.js} +1 -1
  6. package/dist/assets/{Library-CaJVqVvi.js → Library-CQmo4QVC.js} +1 -1
  7. package/dist/assets/{ListDetail-C1W2HmC2.js → ListDetail-BQMd6NOm.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-5Ob9FW3z.js → MarkdownEditor-DEp43FXX.js} +1 -1
  9. package/dist/assets/{McpServers-JxCSfm1S.js → McpServers-CLarzwqA.js} +1 -1
  10. package/dist/assets/{Memory-BDeqlqwH.js → Memory-B0sCdIy1.js} +1 -1
  11. package/dist/assets/{Panel-Dh9ZHuEj.js → Panel-BhWPVOCD.js} +1 -1
  12. package/dist/assets/{Permissions-DXy-CbEY.js → Permissions-Ddlq8T_O.js} +1 -1
  13. package/dist/assets/{Plugins-_n1Iuc8T.js → Plugins-D2oA_2Jl.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-BP_evfxE.js → ProvenanceBadge-DgAgavUM.js} +1 -1
  15. package/dist/assets/{SaveBar-D-gCUx4n.js → SaveBar-Qvc4Ek-H.js} +1 -1
  16. package/dist/assets/{Scheduler-Bpd4OGju.js → Scheduler-BmYJvNzK.js} +1 -1
  17. package/dist/assets/{ScopeSwitcher-CAWzM6RI.js → ScopeSwitcher-C_zWEtIl.js} +1 -1
  18. package/dist/assets/{Settings-DRRozLyT.js → Settings-2Vx3X5SI.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-DGHDWlz4.js → SkillReferenceGraph-BDEUjlTQ.js} +1 -1
  20. package/dist/assets/{Skills-D8L66eiX.js → Skills-Cmrz_LeN.js} +1 -1
  21. package/dist/assets/{SystemPrompt-CYtUsonD.js → SystemPrompt-DVA1eYDP.js} +1 -1
  22. package/dist/assets/{TagLibrary-E5CLeuVk.js → TagLibrary-DYJGAKZu.js} +1 -1
  23. package/dist/assets/{TiptapBody-B2hRgbPE.js → TiptapBody-DmPc3amD.js} +1 -1
  24. package/dist/assets/{Toggle-BTwsbxam.js → Toggle-zfd5LJkK.js} +1 -1
  25. package/dist/assets/{index-DijufvkJ.js → index-B_4PNh9T.js} +676 -676
  26. package/dist/assets/{index-CMLnzdZC.css → index-DIjnPkRN.css} +1 -1
  27. package/dist/assets/{settingsSchema-D6wzxAi6.js → settingsSchema-B9es6fdA.js} +1 -1
  28. package/dist/index.html +2 -2
  29. package/package.json +1 -1
  30. package/scripts/lib/activeSessions.cjs +116 -6
  31. package/scripts/scheduler-mcp-server.cjs +154 -95
  32. package/src/main/__tests__/epicStatusMirror.test.cjs +110 -0
  33. package/src/main/__tests__/health-delegation-chain.test.cjs +105 -0
  34. package/src/main/__tests__/prdAdminRoutes.test.cjs +295 -0
  35. package/src/main/__tests__/prdCreate.test.cjs +109 -0
  36. package/src/main/__tests__/scheduler-autofix-select.test.cjs +15 -3
  37. package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +41 -0
  38. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +60 -1
  39. package/src/main/__tests__/scheduler-stranded-investigation.test.cjs +185 -0
  40. package/src/main/__tests__/seedSchedulerMcp.test.cjs +66 -0
  41. package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -5
  42. package/src/main/bilkoHost.cjs +4 -3
  43. package/src/main/chatRunner.cjs +6 -1
  44. package/src/main/config.cjs +22 -33
  45. package/src/main/health.cjs +153 -2
  46. package/src/main/index.cjs +56 -4
  47. package/src/main/ipcSchemas.cjs +18 -1
  48. package/src/main/lib/__tests__/activeIndexRebuild.test.cjs +179 -0
  49. package/src/main/lib/__tests__/childWithLog.test.cjs +63 -0
  50. package/src/main/lib/__tests__/delegationReadiness.test.cjs +241 -42
  51. package/src/main/lib/__tests__/ephemeralCwd.test.cjs +91 -0
  52. package/src/main/lib/__tests__/epicWorktreeMint.test.cjs +1 -1
  53. package/src/main/lib/__tests__/gitWorktree.test.cjs +14 -2
  54. package/src/main/lib/__tests__/gitWorktreeSalvage.test.cjs +107 -0
  55. package/src/main/lib/__tests__/jobWorktree.test.cjs +2 -2
  56. package/src/main/lib/__tests__/loadGate.test.cjs +159 -0
  57. package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +101 -0
  58. package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +151 -0
  59. package/src/main/lib/__tests__/opsRootResolve.test.cjs +149 -0
  60. package/src/main/lib/__tests__/projectRootResolve.test.cjs +148 -0
  61. package/src/main/lib/__tests__/reaperHelpers.test.cjs +112 -0
  62. package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +19 -9
  63. package/src/main/lib/__tests__/schedulerBatchFairness.test.cjs +213 -0
  64. package/src/main/lib/__tests__/schedulerBatchProjectCap.test.cjs +127 -0
  65. package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +217 -0
  66. package/src/main/lib/activeIndexMerge.cjs +15 -0
  67. package/src/main/lib/activeIndexRebuild.cjs +133 -0
  68. package/src/main/lib/buildTarget.cjs +3 -2
  69. package/src/main/lib/childWithLog.cjs +33 -1
  70. package/src/main/lib/crossProjectFeedback.cjs +8 -1
  71. package/src/main/lib/delegationReadiness.cjs +408 -26
  72. package/src/main/lib/ephemeralCwd.cjs +78 -0
  73. package/src/main/lib/epicDelegationStats.cjs +2 -1
  74. package/src/main/lib/epicMint.cjs +17 -1
  75. package/src/main/lib/epicStatusMirror.cjs +95 -0
  76. package/src/main/lib/epicValidationHook.cjs +2 -1
  77. package/src/main/lib/gitWorktree.cjs +56 -2
  78. package/src/main/lib/jobWorktree.cjs +1 -0
  79. package/src/main/lib/loadGate.cjs +134 -0
  80. package/src/main/lib/mcpToolCatalog.cjs +285 -0
  81. package/src/main/lib/opsErrorLog.cjs +12 -1
  82. package/src/main/lib/opsOwnership.cjs +94 -0
  83. package/src/main/lib/prdAdminRoutes.cjs +43 -3
  84. package/src/main/lib/prdCreate.cjs +46 -14
  85. package/src/main/lib/prdLocations.cjs +13 -6
  86. package/src/main/lib/projectRootResolve.cjs +134 -0
  87. package/src/main/lib/promptSessionSchema.cjs +7 -0
  88. package/src/main/lib/queueStore.cjs +31 -5
  89. package/src/main/lib/rcaReport.cjs +1 -1
  90. package/src/main/lib/reaperHelpers.cjs +47 -1
  91. package/src/main/lib/schedulerBatch.cjs +171 -29
  92. package/src/main/lib/schedulerConfig.cjs +80 -0
  93. package/src/main/projectBrief.cjs +3 -2
  94. package/src/main/projectPages.cjs +2 -1
  95. package/src/main/promptSessionTranscript.cjs +0 -0
  96. package/src/main/pty.cjs +5 -0
  97. package/src/main/queueOps.cjs +15 -8
  98. package/src/main/scheduler.cjs +346 -49
  99. package/src/main/seedSchedulerMcp.cjs +58 -4
  100. package/src/preload/api.d.ts +69 -1
  101. package/src/preload/index.cjs +2 -0
@@ -55,7 +55,8 @@ const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
55
55
  const supervisor = require('./supervisor.cjs');
56
56
  const { resolveClaudeBin } = require('./lib/claudeBin.cjs');
57
57
  const { readTail } = require('./lib/fileTail.cjs');
58
- const { claudePidAlive, classifyRunOutcome, ORPHAN_REQUEUE_CAP } = require('./lib/reaperHelpers.cjs');
58
+ const { claudePidAlive, classifyRunOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
59
+ const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
59
60
  const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
60
61
  const { sendIfAlive } = require('./lib/sendToRenderer.cjs');
61
62
  const { createBroadcastCoalescer } = require('./lib/broadcastCoalescer.cjs');
@@ -68,7 +69,7 @@ const promptSessionTranscript = require('./promptSessionTranscript.cjs');
68
69
  const { verifyRun } = require('./runVerify.cjs');
69
70
  const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
70
71
  const logs = require('./logs.cjs');
71
- const { schemas, validated } = require('./ipcSchemas.cjs');
72
+ const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
72
73
  const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
73
74
  const {
74
75
  POLL_INTERVAL_MS,
@@ -78,6 +79,9 @@ const {
78
79
  QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
79
80
  JOB_OVERRUN_FACTOR: JOB_OVERRUN_FACTOR_DEFAULT,
80
81
  JOB_OVERRUN_FLOOR_MS: JOB_OVERRUN_FLOOR_MS_DEFAULT,
82
+ PIDLESS_SPAWN_GRACE_MS,
83
+ INVESTIGATION_MAX_MS,
84
+ STARVATION_ESCALATE_MS,
81
85
  } = require('./lib/schedulerConfig.cjs');
82
86
  const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
83
87
  ? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
@@ -88,7 +92,7 @@ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
88
92
  const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
89
93
  ? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
90
94
  : JOB_OVERRUN_FLOOR_MS_DEFAULT;
91
- const { pickForProject, pickNextBatch, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
95
+ const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
92
96
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
93
97
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
94
98
  const queueHistory = require('./lib/queueHistory.cjs');
@@ -100,7 +104,7 @@ const queueOps = require('./queueOps.cjs');
100
104
  // home-dir layout.
101
105
  const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
102
106
  const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
103
- const { transitionJob, STATUS_HISTORY_CAP } = require('./lib/scheduleJobTransitions.cjs');
107
+ const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
104
108
  const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
105
109
  const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
106
110
  const { appendAuditEvent } = require('./lib/auditLog.cjs');
@@ -205,6 +209,18 @@ const FINISH_PROTOCOL = `
205
209
  Once every acceptance-criteria line above is satisfied, finish in this EXACT
206
210
  sequence. Do not stop before the commit lands; committing is part of the job.
207
211
 
212
+ RUN VERIFICATION IN THE FOREGROUND — this applies to the whole run, not just
213
+ step 3 below: every test/typecheck/lint/build command you run, whether while
214
+ implementing the AC or during VERIFY, must run SYNCHRONOUSLY and you must wait
215
+ for it to return. Never start a verification command as a background task
216
+ (no background Bash) and then call Monitor, TaskOutput, or ScheduleWakeup to
217
+ pick up its result later — a headless \`claude -p\` run has no later turn, so
218
+ nothing ever delivers that notification and the run dies mid-verification with
219
+ no commit and no verdict. For a long-running command, bound it yourself with
220
+ the shell (e.g. \`timeout 300 npm test\`) and a matching foreground tool
221
+ timeout; if it still cannot finish inside budget, stop and emit
222
+ SCHEDULER_VERDICT: FAIL with the reason instead of deferring it.
223
+
208
224
  1. CODE REVIEW — run \`/code-review --fix\` on your changes and apply the fixes it
209
225
  surfaces (correctness first). For any finding you judge a false positive, say
210
226
  why in your result; do not silently skip it. If \`/code-review\` is not
@@ -690,6 +706,43 @@ async function safeSlugPath(slug) {
690
706
  return safeSlugPathIn(dir, slug);
691
707
  }
692
708
 
709
+ /**
710
+ * The two distinct failure modes safeSlugPath collapses into one nullable
711
+ * return (the defect this fixes — see the PRD that added this helper's
712
+ * Goal): a slug that fails SCHEDULE_SLUG_RE is a caller mistake ("invalid
713
+ * slug"), while a well-formed slug that exists in no candidate PRD dir is a
714
+ * lookup miss ("unknown slug") — an agent retrying the first as if it were
715
+ * the second (or vice versa) burns a turn on the wrong fix. Returns
716
+ * `{ ok: true, path }` or `{ ok: false, reason: 'invalid-slug' | 'not-found' }`.
717
+ * `cwd`, if given, narrows the search to that one project's own PRD dirs
718
+ * (prdDirForCwd + its Epic-scoped dirs — same pattern as getPrdParsed);
719
+ * omitted, it searches every candidate dir machine-wide via findPrdDir.
720
+ */
721
+ async function resolveSlugOrReason(slug, cwd) {
722
+ if (!SCHEDULE_SLUG_RE.test(slug)) return { ok: false, reason: 'invalid-slug' };
723
+ if (cwd) {
724
+ for (const dir of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
725
+ const p = safeSlugPathIn(dir, slug);
726
+ if (!p) continue;
727
+ try {
728
+ await fsp.access(p);
729
+ return { ok: true, path: p };
730
+ } catch { /* not in this dir — try the next candidate */ }
731
+ }
732
+ return { ok: false, reason: 'not-found' };
733
+ }
734
+ const dir = await findPrdDir(slug);
735
+ if (!dir) return { ok: false, reason: 'not-found' };
736
+ const p = safeSlugPathIn(dir, slug);
737
+ if (!p) return { ok: false, reason: 'not-found' };
738
+ return { ok: true, path: p };
739
+ }
740
+
741
+ /** Actionable message for `resolveSlugOrReason`'s 'not-found' reason. */
742
+ function unknownSlugMessage(slug) {
743
+ return `unknown slug "${slug}": no PRD file with that name in any known project — call scheduler_list_prds (optionally with cwd) to see what exists`;
744
+ }
745
+
693
746
  /**
694
747
  * Move a completed job's `<slug>.md` out of its PRD dir into that dir's
695
748
  * sibling `prds-archived/`, so a finished slug can't be re-fired by the
@@ -1098,6 +1151,73 @@ function findOverrunningJobs(jobs, now, { factor, floorMs } = {}) {
1098
1151
  return out;
1099
1152
  }
1100
1153
 
1154
+ /**
1155
+ * findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
1156
+ * → [{ slug, cwd, ageMs, restoreStatus }]
1157
+ *
1158
+ * Pure (besides the warn-log side effect on the two unprovable-age cases
1159
+ * below), no other IO. spawnInvestigation's own restore of a job's
1160
+ * pre-investigation status runs entirely inside the process that spawned the
1161
+ * probe (its withChildAndLog onExit handler, or the synchronous-throw catch
1162
+ * path) — so a job left 'investigating' when the app itself dies or restarts
1163
+ * has NOTHING left to restore it. The comment at spawnInvestigation's onExit
1164
+ * asserts "'investigating' must never be the job's resting state"; this is
1165
+ * the sweep that makes that true across a restart, not just within one.
1166
+ *
1167
+ * A row qualifies only when ALL of:
1168
+ * - status is 'investigating'
1169
+ * - its most recent transition INTO 'investigating' (statusHistory's last
1170
+ * `to === 'investigating'` entry — a job can be investigated more than
1171
+ * once across its life, e.g. a retried auto-fix) is older than `maxMs`
1172
+ * - it has no live probe process behind it (checked via runtime.pid, set by
1173
+ * spawnInvestigation once its child spawns and cleared on every restore
1174
+ * path, the same shape reapDeadRunningJobs already uses for 'running' rows)
1175
+ *
1176
+ * `restoreStatus` is that transition entry's `from` — the exact value
1177
+ * spawnInvestigation itself would have restored to (`failedJob.status ||
1178
+ * 'failed'`), which for a row that already finished and recorded
1179
+ * finishedAt+exitCode (the burrow-834 shape) is whatever terminal status was
1180
+ * computed for that outcome BEFORE the probe was spawned — this sweep never
1181
+ * re-derives it from exitCode, only replays the already-recorded decision.
1182
+ *
1183
+ * A row with no recoverable transition timestamp cannot have its age proven,
1184
+ * so it is warn-logged and left alone rather than guessed at — same posture
1185
+ * as findStaleQuarantinedJobs/findOverrunningJobs.
1186
+ *
1187
+ * `restoreStatus` is validated against LEGAL_TRANSITIONS['investigating']
1188
+ * before being returned — `statusHistory`'s `from` should only ever be
1189
+ * 'failed' or 'needs_review' (the only two states LEGAL_TRANSITIONS allows
1190
+ * into 'investigating'), but a corrupted/unexpected value must not be handed
1191
+ * straight to transitionJob: an illegal target is refused outright (row stays
1192
+ * stuck at 'investigating', re-detected as stranded every sweep with no path
1193
+ * out), so an out-of-set `from` falls back to 'failed' here instead.
1194
+ */
1195
+ const INVESTIGATING_RESTORE_TARGETS = new Set(LEGAL_TRANSITIONS.investigating);
1196
+ function findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive) {
1197
+ const out = [];
1198
+ for (const j of jobs ?? []) {
1199
+ if (j.status !== 'investigating') continue;
1200
+ const entries = (j.statusHistory || []).filter((h) => h.to === 'investigating');
1201
+ const entry = entries[entries.length - 1];
1202
+ if (!entry) {
1203
+ console.warn(`[scheduler] findStrandedInvestigations: ${j.slug} is 'investigating' with no statusHistory entry recording the transition — cannot prove age, leaving alone`);
1204
+ continue;
1205
+ }
1206
+ const since = Date.parse(entry.at ?? '');
1207
+ if (Number.isNaN(since)) {
1208
+ console.warn(`[scheduler] findStrandedInvestigations: ${j.slug} has an unparseable investigating-transition timestamp (${entry.at}) — cannot prove age, leaving alone`);
1209
+ continue;
1210
+ }
1211
+ const ageMs = now - since;
1212
+ if (ageMs < maxMs) continue; // a live probe must not be yanked out from under itself
1213
+ const pid = j.runtime?.pid;
1214
+ if (pid && isAlive(pid)) continue; // probe genuinely still running — not stranded
1215
+ const restoreStatus = INVESTIGATING_RESTORE_TARGETS.has(entry.from) ? entry.from : 'failed';
1216
+ out.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, restoreStatus });
1217
+ }
1218
+ return out;
1219
+ }
1220
+
1101
1221
  // An empty queue and an unreadable queue are NOT the same thing, and
1102
1222
  // conflating them is destructive: reconcile() treats every PRD .md with no
1103
1223
  // matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
@@ -1677,6 +1797,10 @@ async function reconcile(state) {
1677
1797
  originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
1678
1798
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1679
1799
  status: 'pending',
1800
+ // Enqueue time (PRD 1086/1087): the cross-project fairness tiebreak and
1801
+ // the starvation escalation both need a provable age for a pending row;
1802
+ // before this stamp a freshly minted row carried no timestamp at all.
1803
+ queuedAt: new Date().toISOString(),
1680
1804
  runId: null,
1681
1805
  startedAt: null,
1682
1806
  finishedAt: null,
@@ -1864,6 +1988,9 @@ function drainDeferredInvestigation() {
1864
1988
  let cancelToken = { cancelled: false };
1865
1989
  // Last memory-gate observation; included in snapshot for renderer visibility.
1866
1990
  let lastMemGate = null;
1991
+ // CPU-load launch gate (PRD 1085, lib/loadGate.cjs) — innermost launch
1992
+ // predicate after pool → project cap → memory. Withholds launches only.
1993
+ const loadGate = createLoadGate();
1867
1994
 
1868
1995
  // Last tickQueue outcome, kept for the UI. tickQueue already computes a precise
1869
1996
  // reason for every way a batch can come back empty (dependency holds, slot
@@ -1932,6 +2059,8 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
1932
2059
  lastFailureKind,
1933
2060
  },
1934
2061
  memGate: lastMemGate,
2062
+ // Why nothing is launching when the box is CPU-saturated (PRD 1085).
2063
+ loadGate: loadGate.snapshot(),
1935
2064
  lastTick,
1936
2065
  // The machine-wide slot pool IS the concurrency limit — there is no
1937
2066
  // separate scheduler cap any more. `source` distinguishes the
@@ -2609,9 +2738,15 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
2609
2738
  * tree was dirty — closed by widening that call site's condition, not by
2610
2739
  * changing this function's four defenses below, which still apply to both
2611
2740
  * shapes identically:
2612
- * - siblingRunning: a concurrent job in the same cwd makes working-tree
2613
- * evidence unreliable in both directions (extra dirt OR a clean tree
2614
- * that isn't this job's doing).
2741
+ * - siblingRunning: on a SHARED tree only — a concurrent job in the same
2742
+ * cwd makes working-tree evidence unreliable in both directions (extra
2743
+ * dirt OR a clean tree that isn't this job's doing). Suppressed by
2744
+ * ranInWorktree: when this job ran in its own git worktree, the
2745
+ * newly-dirty set and the integrated HEAD are attributable to this job
2746
+ * alone regardless of what siblings were doing concurrently in their own
2747
+ * worktrees, so the excuse does not apply (PRD 109 shipped 'completed'
2748
+ * with nothing committed specifically because this carve-out fired
2749
+ * unconditionally during a high-concurrency run).
2615
2750
  * - jobSelfCommitted: HEAD moved during the run, so the job's deliverable
2616
2751
  * landed even if dirt (from a concurrent actor) remains.
2617
2752
  * - legitimateNoOp (COMPLETED_EQUIVALENT_VERDICTS): runVerify.cjs's own
@@ -2631,8 +2766,8 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
2631
2766
  * still a genuine finish-protocol violation (incident:
2632
2767
  * 523-fix-bounded-fix-plan-retry, 2026-07-12).
2633
2768
  */
2634
- function commitGuardVerdict({ newlyDirty, siblingRunning, jobSelfCommitted, legitimateNoOp, isFixPlanJob, verifyResult }) {
2635
- if (siblingRunning || jobSelfCommitted || legitimateNoOp) return null;
2769
+ function commitGuardVerdict({ newlyDirty, siblingRunning, ranInWorktree, jobSelfCommitted, legitimateNoOp, isFixPlanJob, verifyResult, salvagePatch }) {
2770
+ if ((siblingRunning && !ranInWorktree) || jobSelfCommitted || legitimateNoOp) return null;
2636
2771
  const dirty = newlyDirty || [];
2637
2772
  if (dirty.length === 0 && isFixPlanJob) return null;
2638
2773
 
@@ -2651,9 +2786,10 @@ function commitGuardVerdict({ newlyDirty, siblingRunning, jobSelfCommitted, legi
2651
2786
  }
2652
2787
 
2653
2788
  const sample = dirty.slice(0, 3).join(', ');
2789
+ const salvageNote = salvagePatch ? ` — recoverable from salvage patch ${salvagePatch}` : '';
2654
2790
  return {
2655
2791
  verdict: 'uncommitted_changes',
2656
- reason: `finish protocol incomplete: ${dirty.length} uncommitted file(s) left in working tree (e.g. ${sample})`,
2792
+ reason: `finish protocol incomplete: ${dirty.length} uncommitted file(s) left in working tree (e.g. ${sample})${salvageNote}`,
2657
2793
  downgradeTo: 'needs_review',
2658
2794
  annotations: carried.length ? carried : undefined,
2659
2795
  };
@@ -2813,7 +2949,12 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
2813
2949
  // overrides `--model sonnet`, so scheduled jobs burn Opus credits silently.
2814
2950
  // PATH must include Homebrew/user bins or the job's node/git children ENOENT
2815
2951
  // when Electron was launched from Finder/Dock on macOS (stripped PATH).
2816
- const childEnv = cleanChildEnv({ PATH: pathWithUserBins() });
2952
+ // SM_PROJECT_ROOT is the main-tree cwd (never spawnCwd, which may be a
2953
+ // job/epic worktree) — forwarded by scheduler-mcp-server.cjs as
2954
+ // originProjectRoot so a job running inside its own worktree can still
2955
+ // resolve the real project for create-prd/open-session/readiness. See
2956
+ // projectRootResolve.cjs.
2957
+ const childEnv = cleanChildEnv({ PATH: pathWithUserBins(), SM_PROJECT_ROOT: cwd });
2817
2958
 
2818
2959
  // Track whether the agent has emitted a `result` event in its JSONL stream.
2819
2960
  // null until seen; then one of "success" | "error_max_turns" | … per the
@@ -3317,7 +3458,10 @@ async function spawnInvestigation(failedJob, runDir) {
3317
3458
  // 'investigating' must never be the job's resting state.
3318
3459
  mutate((s) => {
3319
3460
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
3320
- if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation probe exited — restoring prior status', source: 'spawnInvestigation:onExit' });
3461
+ if (j && j.status === 'investigating') {
3462
+ transitionJob(j, failedJob.status || 'failed', { reason: 'investigation probe exited — restoring prior status', source: 'spawnInvestigation:onExit' });
3463
+ delete j.runtime;
3464
+ }
3321
3465
  })
3322
3466
  .then(() => broadcast({ flush: true }))
3323
3467
  .catch(() => {});
@@ -3374,6 +3518,15 @@ async function spawnInvestigation(failedJob, runDir) {
3374
3518
 
3375
3519
  if (child) {
3376
3520
  safeLog(`[scheduler] investigation pid=${child.pid}\n\n`);
3521
+ // Recorded so findStrandedInvestigations (a post-restart maintenance
3522
+ // sweep — the live process has no other way to know a probe is still
3523
+ // running) can tell a live probe apart from one whose owning process is
3524
+ // long gone, the same way reapDeadRunningJobs checks a running job's
3525
+ // runtime.pid.
3526
+ mutate((s) => {
3527
+ const j = s.jobs.find((x) => x.slug === failedJob.slug);
3528
+ if (j && j.status === 'investigating') j.runtime = { pid: child.pid };
3529
+ }).catch(() => {});
3377
3530
  }
3378
3531
  return { deferred: false };
3379
3532
  } catch (e) {
@@ -3383,7 +3536,10 @@ async function spawnInvestigation(failedJob, runDir) {
3383
3536
  releaseSlot();
3384
3537
  mutate((s) => {
3385
3538
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
3386
- if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation spawn threw before exiting — restoring prior status', source: 'spawnInvestigation:catch' });
3539
+ if (j && j.status === 'investigating') {
3540
+ transitionJob(j, failedJob.status || 'failed', { reason: 'investigation spawn threw before exiting — restoring prior status', source: 'spawnInvestigation:catch' });
3541
+ delete j.runtime;
3542
+ }
3387
3543
  })
3388
3544
  .then(() => broadcast({ flush: true }))
3389
3545
  .catch(() => {});
@@ -3430,6 +3586,16 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3430
3586
  console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
3431
3587
  } else {
3432
3588
  console.log(`[scheduler] ${job.slug}: running in main tree (worktree not used: ${worktree.reason})`);
3589
+ // Surface any degraded-isolation fallback on the job row itself so it's
3590
+ // queryable from the queue instead of console-only — except the
3591
+ // deliberate env-disable flag, which is an intentional opt-out, not a
3592
+ // degradation worth flagging.
3593
+ if (!jobWorktree.isWorktreeDisabled()) {
3594
+ await mutate((s) => {
3595
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
3596
+ if (idx >= 0) s.jobs[idx].worktreeFallbackReason = worktree.reason;
3597
+ });
3598
+ }
3433
3599
  }
3434
3600
 
3435
3601
  // Integrate the job's branch back into guardCwd's own HEAD, THEN tear the
@@ -3448,6 +3614,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3448
3614
  let res;
3449
3615
  let worktreeLeftoverDirty = [];
3450
3616
  let worktreeIntegrationFailure = null;
3617
+ let worktreeSalvagePatch = null;
3451
3618
  try {
3452
3619
  res = await executeJob(job, runDir, defaultCwd, async (pid, sessionId, cwd) => {
3453
3620
  await mutate((s) => {
@@ -3462,6 +3629,19 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3462
3629
  } finally {
3463
3630
  if (worktree.ok) {
3464
3631
  worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
3632
+ // Salvage the worktree's full diff (tracked + untracked) to the run
3633
+ // dir BEFORE the checkout is removed below — otherwise a job killed
3634
+ // before its finish-protocol commit loses that work outright, with
3635
+ // no branch, no stash, no patch anywhere. Best-effort: never blocks
3636
+ // integration/cleanup and never changes the job's verdict.
3637
+ if (worktreeLeftoverDirty.length) {
3638
+ const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
3639
+ const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
3640
+ if (salvage && salvage.ok) {
3641
+ worktreeSalvagePatch = salvagePath;
3642
+ console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
3643
+ }
3644
+ }
3465
3645
  const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug });
3466
3646
  if (!integration.ok) {
3467
3647
  worktreeIntegrationFailure = integration.reason;
@@ -3614,10 +3794,12 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3614
3794
  const guardVerdict = commitGuardVerdict({
3615
3795
  newlyDirty,
3616
3796
  siblingRunning,
3797
+ ranInWorktree: worktree.ok,
3617
3798
  jobSelfCommitted,
3618
3799
  legitimateNoOp: guardIsLegitimateNoOp,
3619
3800
  isFixPlanJob: isFixPlanSlug(job.slug),
3620
3801
  verifyResult,
3802
+ salvagePatch: worktreeSalvagePatch,
3621
3803
  });
3622
3804
  if (guardVerdict) {
3623
3805
  verifyResult = guardVerdict;
@@ -3709,6 +3891,11 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3709
3891
  transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
3710
3892
  s.jobs[i2].finishedAt = new Date().toISOString();
3711
3893
  s.jobs[i2].exitCode = res.exitCode;
3894
+ if (worktreeSalvagePatch) {
3895
+ s.jobs[i2].worktreeSalvagePatch = worktreeSalvagePatch;
3896
+ } else {
3897
+ delete s.jobs[i2].worktreeSalvagePatch;
3898
+ }
3712
3899
  s.jobs[i2].error = effectiveStatus === 'needs_review'
3713
3900
  ? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
3714
3901
  // A failed job (non-zero exit) never consults verifyResult above,
@@ -3887,12 +4074,13 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3887
4074
  });
3888
4075
  await broadcast({ flush: true });
3889
4076
  } else if (decision.action === 'fail-dirty') {
3890
- console.log(`[scheduler] transient failure (${decision.transientKind}) for ${job.slug} left ${newlyDirtyCount} uncommitted file(s) (e.g. ${dirtySample}) — not auto-requeuing`);
4077
+ const salvageNote = worktreeSalvagePatch ? ` — recoverable from salvage patch ${worktreeSalvagePatch}` : '';
4078
+ console.log(`[scheduler] transient failure (${decision.transientKind}) for ${job.slug} left ${newlyDirtyCount} uncommitted file(s) (e.g. ${dirtySample})${salvageNote} — not auto-requeuing`);
3891
4079
  await mutate((s) => {
3892
4080
  const i = s.jobs.findIndex((x) => x.slug === job.slug);
3893
4081
  if (i >= 0) {
3894
4082
  transitionJob(s.jobs[i], 'failed', { reason: `transient failure (${decision.transientKind}) left uncommitted work — not auto-requeued`, source: 'spawnJob:fail-dirty' });
3895
- s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample}) — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
4083
+ s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample})${salvageNote} — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
3896
4084
  }
3897
4085
  });
3898
4086
  await broadcast({ flush: true });
@@ -3937,7 +4125,10 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3937
4125
  // is synchronous and spawnJob is fire-and-forget.
3938
4126
  let tickTail = Promise.resolve();
3939
4127
 
3940
- function tickQueue() {
4128
+ // `bypassLoadGate` is set only by the explicit human run-now / force-tick
4129
+ // paths (via runDueJobs): the human is asking, so the CPU-load gate yields
4130
+ // and logs that it did. Every automatic caller leaves it false.
4131
+ function tickQueue({ bypassLoadGate = false } = {}) {
3941
4132
  const next = tickTail.then(async () => {
3942
4133
  const state = await readQueue();
3943
4134
  // Never reconcile against an unreadable queue: reconcile() would see zero
@@ -4025,6 +4216,35 @@ function tickQueue() {
4025
4216
  lastMemGate = null;
4026
4217
  }
4027
4218
 
4219
+ // Load gate (PRD 1085) — the INNERMOST launch predicate, evaluated only
4220
+ // once every outer gate (sessionSlots pool → per-project cap inside
4221
+ // pickNextBatch → memory above) has already admitted `gatedBatch`. It
4222
+ // never touches running jobs and never becomes a second pool: it only
4223
+ // withholds this tick's launches while the 1-minute loadavg per core is
4224
+ // over LOAD_GATE_PER_CORE. An explicit human Run now bypasses it.
4225
+ const load = loadGate.evaluate({ bypass: bypassLoadGate });
4226
+ if (load.bypassed) {
4227
+ console.log(`[scheduler] load gate: BYPASSED by run-now (loadavg1=${load.loadavg1} cores=${load.cores} ratio=${load.ratio} > ${load.threshold})`);
4228
+ } else if (load.gated) {
4229
+ const line = `[scheduler] load gate: loadavg1=${load.loadavg1} cores=${load.cores} ratio=${load.ratio} > ${load.threshold} — holding ${gatedBatch.length} eligible job(s)`;
4230
+ if (load.escalate) {
4231
+ const top = topCpuConsumers(3);
4232
+ console.warn(`${line} for ${Math.round(load.gatedSinceMs / 60_000)}m; top CPU: ${top.length ? top.join(' | ') : 'n/a'}`);
4233
+ } else {
4234
+ console.log(line);
4235
+ }
4236
+ if (load.shouldAudit) {
4237
+ appendAuditEvent('launch_load_gated', {
4238
+ loadavg1: load.loadavg1, cores: load.cores, ratio: load.ratio, threshold: load.threshold,
4239
+ held: gatedBatch.map((j) => j.slug), gatedSinceMs: load.gatedSinceMs,
4240
+ });
4241
+ }
4242
+ return recordTick(
4243
+ { fired: false, reason: 'load-deferred', deferredCount: gatedBatch.length, ratio: load.ratio, threshold: load.threshold },
4244
+ { detail: `load gate: ${load.loadavg1} / ${load.cores} cores = ${load.ratio} > ${load.threshold}`, holds },
4245
+ );
4246
+ }
4247
+
4028
4248
  await mutate((s) => { s.lastRunAt = new Date().toISOString(); });
4029
4249
  await broadcast();
4030
4250
 
@@ -4072,7 +4292,7 @@ function forceTickOutcome(result) {
4072
4292
  }
4073
4293
  }
4074
4294
 
4075
- async function runDueJobs() {
4295
+ async function runDueJobs({ bypassLoadGate = false } = {}) {
4076
4296
  const state = await readQueue();
4077
4297
  if (state.unreadable) {
4078
4298
  console.error('[scheduler] runDueJobs skipped: queue.json unreadable');
@@ -4083,7 +4303,7 @@ async function runDueJobs() {
4083
4303
  return { fired: false, reason: 'paused' };
4084
4304
  }
4085
4305
  cancelToken = { cancelled: false };
4086
- const result = await tickQueue();
4306
+ const result = await tickQueue({ bypassLoadGate });
4087
4307
  // Clear the one-shot scheduledFor without waiting for jobs to settle.
4088
4308
  await mutate((s) => { s.scheduledFor = null; });
4089
4309
  await broadcast();
@@ -4109,11 +4329,13 @@ async function maybeLaunchWhenAvailable(state) {
4109
4329
  // ---------- dead-process reaper ----------
4110
4330
 
4111
4331
  /**
4112
- * Scan running jobs, identify those whose claude process is provably dead, and
4113
- * finalize them to completed/failed by reading the run log. Called once per
4114
- * poll cycle. Conservative: a job with no runtime.pid yet (spawn mid-flight)
4115
- * is always skipped. A job whose pid is alive (claudePidAlive) is always skipped.
4116
- * Exported so unit tests can invoke it directly.
4332
+ * Scan running jobs, identify those whose claude process is provably dead OR
4333
+ * whose spawn never got far enough to record a runtime.pid in the first
4334
+ * place, and finalize them to completed/failed by reading the run log.
4335
+ * Called once per poll cycle. A job whose pid is alive (claudePidAlive) is
4336
+ * always skipped. A pidless job younger than PIDLESS_SPAWN_GRACE_MS is
4337
+ * skipped too (spawn may still be mid-flight) — see selectReapableJobs for
4338
+ * the full predicate. Exported so unit tests can invoke it directly.
4117
4339
  */
4118
4340
  async function reapDeadRunningJobs() {
4119
4341
  try {
@@ -4123,32 +4345,45 @@ async function reapDeadRunningJobs() {
4123
4345
  // status:"running" with no slug left in runningSet to trigger reconciliation.
4124
4346
  // queue.json is the source of truth for which jobs are actually running.
4125
4347
  const state = await readQueue();
4348
+ const { reapable, warnings } = selectReapableJobs(state.jobs, Date.now(), {
4349
+ pidAlive: claudePidAlive,
4350
+ grace: PIDLESS_SPAWN_GRACE_MS,
4351
+ });
4352
+ for (const w of warnings) {
4353
+ console.warn(`[scheduler] reapDeadRunningJobs: ${w.reason} slug=${w.slug} — leaving row alone`);
4354
+ }
4355
+
4126
4356
  const dead = [];
4127
- for (const j of state.jobs) {
4128
- if (j.status !== 'running') continue;
4129
- const pid = j.runtime?.pid;
4130
- if (!pid) continue; // spawn may be mid-flight; give it a cycle
4131
- if (claudePidAlive(pid)) continue;
4132
- const logPath = j.runId
4357
+ for (const { slug, pid, pidless, reason } of reapable) {
4358
+ const j = state.jobs.find((x) => x.slug === slug);
4359
+ const logPath = j?.runId
4133
4360
  ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`)
4134
4361
  : null;
4362
+ // Absent/empty run dir → classifyRunOutcome finds no result event →
4363
+ // 'no_result' → non-success below → filed as failed, never completed.
4135
4364
  const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
4136
- dead.push({ slug: j.slug, pid, outcome });
4365
+ dead.push({ slug, pid, outcome, pidless, reason });
4137
4366
  }
4138
4367
  if (dead.length === 0) return;
4139
4368
 
4140
4369
  await mutate((s) => {
4141
- for (const { slug, pid, outcome } of dead) {
4370
+ for (const { slug, pid, outcome, pidless, reason } of dead) {
4142
4371
  const idx = s.jobs.findIndex((x) => x.slug === slug);
4143
4372
  if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
4144
4373
  const success = outcome === 'success';
4145
- transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: `reaped: process gone (outcome=${outcome})`, source: 'reapDeadRunningJobs' });
4374
+ const transitionReason = pidless ? reason : `reaped: process gone (outcome=${outcome})`;
4375
+ transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
4146
4376
  s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
4147
4377
  s.jobs[idx].finishedAt = new Date().toISOString();
4148
- s.jobs[idx].error = success ? null : `reaped: process gone, no success result in log (${outcome})`;
4378
+ s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
4149
4379
  delete s.jobs[idx].runtime;
4150
4380
  runningSet.delete(slug);
4151
- console.log(`[scheduler] reaped dead job slug=${slug} pid=${pid} outcome=${outcome}`);
4381
+ if (pidless) {
4382
+ console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
4383
+ appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
4384
+ } else {
4385
+ console.log(`[scheduler] reaped dead job slug=${slug} pid=${pid} outcome=${outcome}`);
4386
+ }
4152
4387
  }
4153
4388
  });
4154
4389
 
@@ -4337,12 +4572,17 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
4337
4572
  // their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
4338
4573
  const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'pass_no_commit', 'pass_no_commit_already_shipped']);
4339
4574
 
4340
- // Bounds fix-plan recursion: depth 1 = the original job, depth 2 = its fix
4341
- // (gets exactly one follow-up investigation if it also lands in
4342
- // needs_review), depth 3+ (a fix-of-a-fix-of-a-fix) is excluded. Shared by
4343
- // selectAutoFixTargets and spawnInvestigation so both call sites agree on
4344
- // one threshold.
4345
- const MAX_INVESTIGATION_DEPTH = 2;
4575
+ // Bounds fix-plan recursion: cap N permits at most N+1 fix jobs per original
4576
+ // slug (depth 1 = the original job, depth 2 = its `-fix`, depth 3+ is
4577
+ // excluded). With N=1 that's `<slug>-fix` and `<slug>-fix-fix`, never a third
4578
+ // `-fix-fix-fix`. Lowered from 2 to 1 on 2026-08-31 (starry-night-ships):
4579
+ // three concurrent chains (115-fix-fix, 113-fix-fix, 111-fix-fix-fix) were
4580
+ // riding the old cap, and 115-fix-fix's own root-cause section read "The
4581
+ // code was already CORRECT. Only verification and commit failed." — a third
4582
+ // auto-retry re-runs an entire PRD and test battery to redo a `git commit`,
4583
+ // at near-zero marginal success probability. Shared by selectAutoFixTargets
4584
+ // and spawnInvestigation so both call sites agree on one threshold.
4585
+ const MAX_INVESTIGATION_DEPTH = 1;
4346
4586
 
4347
4587
  /**
4348
4588
  * True when a fix-plan job's investigationDepth is at or past the recursion
@@ -4841,7 +5081,7 @@ function registerScheduleHandlers() {
4841
5081
  // Clears any existing pause first (same semantics as run-now).
4842
5082
  await clearPause('run-now');
4843
5083
  try {
4844
- const result = await runDueJobs();
5084
+ const result = await runDueJobs({ bypassLoadGate: true });
4845
5085
  return forceTickOutcome(result);
4846
5086
  } catch (e) {
4847
5087
  logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (force-tick)', meta: { error: e?.message } });
@@ -4917,7 +5157,7 @@ function registerScheduleHandlers() {
4917
5157
  ipcMain.handle('schedule:run-now', async () => {
4918
5158
  // Manual run-now overrides any auto-pause. Clear it first.
4919
5159
  await clearPause('run-now');
4920
- runDueJobs().catch((e) => logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (run-now)', meta: { error: e?.message } }));
5160
+ runDueJobs({ bypassLoadGate: true }).catch((e) => logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (run-now)', meta: { error: e?.message } }));
4921
5161
  return { ok: true };
4922
5162
  });
4923
5163
 
@@ -5286,6 +5526,49 @@ async function init() {
5286
5526
  slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
5287
5527
  });
5288
5528
  }
5529
+
5530
+ // Stranded-investigation restore. Unlike the two escalations above, this
5531
+ // one ACTS: 'investigating' is a transient status whose restore
5532
+ // (spawnInvestigation's onExit/catch) only runs inside the process that
5533
+ // spawned the probe, so an app restart mid-probe leaves the row frozen
5534
+ // there forever (see findStrandedInvestigations' header, and the
5535
+ // "'investigating' must never be the job's resting state" comment at
5536
+ // spawnInvestigation's onExit). This restores each stranded row to the
5537
+ // exact terminal status it already carried before the probe was
5538
+ // spawned — it never re-runs or re-investigates anything.
5539
+ const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
5540
+ if (stranded.length > 0) {
5541
+ mutate((ms) => {
5542
+ for (const st of stranded) {
5543
+ const j = ms.jobs.find((x) => x.slug === st.slug);
5544
+ if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
5545
+ transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
5546
+ delete j.runtime;
5547
+ console.warn(
5548
+ `[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
5549
+ + `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
5550
+ + `restored to '${st.restoreStatus}'`,
5551
+ );
5552
+ appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
5553
+ }
5554
+ })
5555
+ .then(() => broadcast({ flush: true }))
5556
+ .catch(() => {});
5557
+ }
5558
+
5559
+ // Per-project starvation (PRD 1087): a project with pending work that has
5560
+ // been passed over on every tick while OTHER projects dispatch. Nothing
5561
+ // else distinguishes "no pending work" from "pending work, never
5562
+ // started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
5563
+ // Escalation only, same shape as the quarantine/overrun warnings above.
5564
+ for (const sp of findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS)) {
5565
+ console.warn(
5566
+ `[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
5567
+ + `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
5568
+ + `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
5569
+ );
5570
+ appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
5571
+ }
5289
5572
  }, 10 * 60_000);
5290
5573
 
5291
5574
  // Self-rescheduling poll loop with exponential backoff. Replaces the
@@ -5638,9 +5921,16 @@ const remote = {
5638
5921
  },
5639
5922
 
5640
5923
  async resetJob(slug, opts = {}) {
5641
- if (!(await safeSlugPath(slug))) return { ok: false, error: 'invalid slug' };
5924
+ const resolved = await resolveSlugOrReason(slug, opts.cwd);
5925
+ if (!resolved.ok) {
5926
+ return { ok: false, error: resolved.reason === 'invalid-slug' ? 'invalid slug' : unknownSlugMessage(slug) };
5927
+ }
5642
5928
  const outcome = await mutate((state) => {
5643
- const idx = state.jobs.findIndex((j) => j.slug === slug);
5929
+ // Same cwd filter as resolveSlugOrReason's file lookup above — slugs are
5930
+ // derived from title text with no cwd salt, so two different projects
5931
+ // can independently produce the identical slug; an opts.cwd caller must
5932
+ // reset THAT project's job, not just any queue row matching the string.
5933
+ const idx = state.jobs.findIndex((j) => j.slug === slug && (!opts.cwd || j.cwd === opts.cwd));
5644
5934
  if (idx < 0) return { kind: 'not-found' };
5645
5935
  // Terminal-status guard lives in resetJobFields itself; force:true
5646
5936
  // threads through to override it.
@@ -5828,10 +6118,16 @@ const remote = {
5828
6118
  // cancelled job lands in 'failed' with an error naming the cause,
5829
6119
  // consistent with every other non-success terminal outcome. Refuses a
5830
6120
  // slug that's already terminal — nothing left to cancel.
5831
- async cancelJob(slug) {
6121
+ async cancelJob(slug, opts = {}) {
6122
+ if (!SCHEDULE_SLUG_RE.test(slug)) return { ok: false, error: 'invalid slug' };
5832
6123
  const state = await readQueue();
5833
- const job = state.jobs.find((j) => j.slug === slug);
5834
- if (!job) return { ok: false, error: 'not found' };
6124
+ const job = state.jobs.find((j) => j.slug === slug && (!opts.cwd || j.cwd === opts.cwd));
6125
+ if (!job) {
6126
+ return {
6127
+ ok: false,
6128
+ error: `unknown slug "${slug}": no queued job with that name${opts.cwd ? ` in cwd ${opts.cwd}` : ''} — call scheduler_list_jobs to see what exists`,
6129
+ };
6130
+ }
5835
6131
  if (job.status === 'completed' || job.status === 'failed' || job.status === 'needs_review' || job.status === 'skipped') {
5836
6132
  return { ok: false, error: `job already terminal (status: "${job.status}") — nothing to cancel` };
5837
6133
  }
@@ -5889,9 +6185,10 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
5889
6185
  return;
5890
6186
  }
5891
6187
  const force = parsed.force === true;
5892
- const result = await remoteObj.resetJob(slug, { force });
6188
+ const cwd = typeof parsed.cwd === 'string' ? parsed.cwd : undefined;
6189
+ const result = await remoteObj.resetJob(slug, { force, cwd });
5893
6190
  sendJson(res, 200, result);
5894
6191
  });
5895
6192
  }
5896
6193
 
5897
- module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims };
6194
+ module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS };