claude-code-session-manager 0.82.0 → 0.84.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/dist/assets/{AgentLibrary-pwlkAFb3.js → AgentLibrary-CCVIpoSz.js} +1 -1
  2. package/dist/assets/{DataModel-BZK9PXFD.js → DataModel-BltREYde.js} +1 -1
  3. package/dist/assets/{History-BVRjxJjS.js → History-BZxFkOp6.js} +2 -2
  4. package/dist/assets/{Hooks-CNuwHeGx.js → Hooks-Bznlfaaa.js} +1 -1
  5. package/dist/assets/{HostBilko-cjwNodhV.js → HostBilko-DUG5YHA_.js} +1 -1
  6. package/dist/assets/{Library-YPNm9W92.js → Library-Bi2Fn3w9.js} +1 -1
  7. package/dist/assets/{ListDetail-CY4GM1Om.js → ListDetail-4VBvXKrz.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-BF4y2Jiz.js → MarkdownEditor-D2ft_v5j.js} +1 -1
  9. package/dist/assets/{McpServers-CmBdWtX_.js → McpServers-FDekKkyE.js} +1 -1
  10. package/dist/assets/{Memory-CvkIXNl1.js → Memory-Dkl80uj_.js} +6 -6
  11. package/dist/assets/{Panel-D93o-sxe.js → Panel-LurG5VfD.js} +1 -1
  12. package/dist/assets/{Permissions-DQipg16I.js → Permissions-D2wHBCFA.js} +1 -1
  13. package/dist/assets/{Plugins-B6NwzPfK.js → Plugins-CTw_wwbI.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-DACVJhrB.js → ProvenanceBadge-CW6HNUv6.js} +1 -1
  15. package/dist/assets/{SaveBar-Cg4lbChb.js → SaveBar-alDHG6cP.js} +1 -1
  16. package/dist/assets/{Scheduler-Dr5ZcBLe.js → Scheduler-DxiPcaiW.js} +7 -7
  17. package/dist/assets/{ScopeSwitcher-C-locvy0.js → ScopeSwitcher-DzFXUKLZ.js} +1 -1
  18. package/dist/assets/Settings-B6H3v2am.js +3 -0
  19. package/dist/assets/{SkillReferenceGraph-DDzuYgSK.js → SkillReferenceGraph-B8EZalpN.js} +1 -1
  20. package/dist/assets/{Skills-DFvhiOAQ.js → Skills-DwuHuA08.js} +2 -2
  21. package/dist/assets/SystemPrompt-CMMqpGYn.js +1 -0
  22. package/dist/assets/{TagLibrary-DruUYaAc.js → TagLibrary-C2N5C1n9.js} +1 -1
  23. package/dist/assets/{TiptapBody-Dr4a--42.js → TiptapBody-btlID-dQ.js} +1 -1
  24. package/dist/assets/{Toggle-B_EH2TFb.js → Toggle-BGOn3DZj.js} +1 -1
  25. package/dist/assets/{index-DApB4DHS.js → index-QLRf0epp.js} +316 -314
  26. package/dist/assets/{index-CYhdtisq.css → index-mnjNDpb1.css} +1 -1
  27. package/dist/assets/{settingsSchema-BKa-xk8g.js → settingsSchema-DA3N2Up3.js} +1 -1
  28. package/dist/index.html +2 -2
  29. package/package.json +4 -1
  30. package/scripts/hooks/guard-destructive-git.cjs +514 -0
  31. package/scripts/hooks/guard-inline-implementation.cjs +219 -0
  32. package/scripts/hooks/guard-prd-writes.cjs +200 -0
  33. package/src/main/__tests__/epicMintTelemetryTap.test.cjs +64 -0
  34. package/src/main/__tests__/health-delegation-chain.test.cjs +2 -1
  35. package/src/main/__tests__/health-queue-dispatch.test.cjs +84 -0
  36. package/src/main/__tests__/health-usage-poller.test.cjs +97 -0
  37. package/src/main/__tests__/health-worktree-cap-blocked.test.cjs +65 -0
  38. package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +143 -0
  39. package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +120 -0
  40. package/src/main/__tests__/promptSessionTranscript.test.cjs +0 -0
  41. package/src/main/__tests__/queue-starvation-dispatch-driver.test.cjs +143 -0
  42. package/src/main/__tests__/rateLimitPollerStreak.test.cjs +79 -0
  43. package/src/main/__tests__/scheduleJobTransitionsTelemetryTap.test.cjs +72 -0
  44. package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +74 -0
  45. package/src/main/__tests__/scheduler-job-overrun.test.cjs +58 -0
  46. package/src/main/__tests__/scheduler-notify-originating-tab-transcript.test.cjs +1 -0
  47. package/src/main/__tests__/scheduler-periodic-reverify-guard.test.cjs +134 -2
  48. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +33 -0
  49. package/src/main/__tests__/scheduler-stuck-failed-escalation.test.cjs +136 -0
  50. package/src/main/__tests__/telemetryClient.test.cjs +810 -0
  51. package/src/main/__tests__/telemetryContract.test.cjs +883 -0
  52. package/src/main/crashDiagnostics.cjs +29 -1
  53. package/src/main/health.cjs +197 -3
  54. package/src/main/index.cjs +65 -6
  55. package/src/main/ipcSchemas.cjs +19 -2
  56. package/src/main/lib/__tests__/crashTelemetry.test.cjs +97 -0
  57. package/src/main/lib/__tests__/delegationReadiness.test.cjs +302 -4
  58. package/src/main/lib/__tests__/fixtures/scheduler-machine.json.corrupt-1789147548 +34 -0
  59. package/src/main/lib/__tests__/gitWorktree.test.cjs +413 -1
  60. package/src/main/lib/__tests__/jobWorktreeBootLive.test.cjs +71 -0
  61. package/src/main/lib/__tests__/queueStoreAtomicWrite.test.cjs +88 -0
  62. package/src/main/lib/__tests__/queueStoreMachineStateRecovery.test.cjs +123 -0
  63. package/src/main/lib/__tests__/reaperHelpers.test.cjs +58 -0
  64. package/src/main/lib/__tests__/telemetryBacklog.test.cjs +620 -0
  65. package/src/main/lib/__tests__/telemetryBoot.test.cjs +125 -0
  66. package/src/main/lib/__tests__/telemetryConsent.test.cjs +130 -0
  67. package/src/main/lib/__tests__/telemetryCounters.test.cjs +57 -0
  68. package/src/main/lib/__tests__/telemetryCountersMetadataColumn.test.cjs +89 -0
  69. package/src/main/lib/crashTelemetry.cjs +37 -0
  70. package/src/main/lib/delegationReadiness.cjs +119 -1
  71. package/src/main/lib/epicMint.cjs +2 -0
  72. package/src/main/lib/gitWorktree.cjs +427 -17
  73. package/src/main/lib/jobWorktree.cjs +2 -1
  74. package/src/main/lib/jobWorktreeBootLive.cjs +51 -0
  75. package/src/main/lib/jobWorktreeTerminalOrphanLive.cjs +68 -0
  76. package/src/main/lib/opsErrorLog.cjs +78 -25
  77. package/src/main/lib/queueStore.cjs +233 -20
  78. package/src/main/lib/reaperHelpers.cjs +23 -1
  79. package/src/main/lib/scheduleJobSchema.cjs +7 -0
  80. package/src/main/lib/scheduleJobTransitions.cjs +12 -0
  81. package/src/main/lib/telemetryBacklog.cjs +601 -0
  82. package/src/main/lib/telemetryBoot.cjs +71 -0
  83. package/src/main/lib/telemetryClient.cjs +653 -0
  84. package/src/main/lib/telemetryConsent.cjs +34 -0
  85. package/src/main/lib/telemetryCounters.cjs +45 -0
  86. package/src/main/promptSessionTranscript.cjs +0 -0
  87. package/src/main/pty.cjs +2 -0
  88. package/src/main/scheduler.cjs +481 -33
  89. package/src/preload/api.d.ts +84 -4
  90. package/src/preload/index.cjs +9 -0
  91. package/dist/assets/Settings-BVrAle90.js +0 -3
  92. package/dist/assets/SystemPrompt-8PiTyUyL.js +0 -1
@@ -143,6 +143,9 @@ function resolveOriginSessionId(cwd, epicId) {
143
143
  const sessionSlots = require('./lib/sessionSlots.cjs');
144
144
  const quietMachineLease = require('./lib/quietMachineLease.cjs');
145
145
  const jobWorktree = require('./lib/jobWorktree.cjs');
146
+ const gitWorktree = require('./lib/gitWorktree.cjs');
147
+ const { buildJobWorktreeIsLive } = require('./lib/jobWorktreeBootLive.cjs');
148
+ const { buildTerminalOrphanIsLive } = require('./lib/jobWorktreeTerminalOrphanLive.cjs');
146
149
  const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
147
150
  const queueStore = require('./lib/queueStore.cjs');
148
151
  const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
@@ -1294,6 +1297,7 @@ function loadSchedulerState() {
1294
1297
  if (typeof s.backoffMs === 'number') backoffMs = s.backoffMs;
1295
1298
  if (typeof s.pauseClearedManuallyAt === 'number') pauseClearedManuallyAt = s.pauseClearedManuallyAt;
1296
1299
  if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
1300
+ if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
1297
1301
  } catch { /* first boot or corrupt — start fresh */ }
1298
1302
  }
1299
1303
 
@@ -1313,6 +1317,7 @@ function persistSchedulerState() {
1313
1317
  pausedReason: null,
1314
1318
  pausedSince: null,
1315
1319
  pauseClearedManuallyAt,
1320
+ failureStreakWarned,
1316
1321
  });
1317
1322
  } catch (e) {
1318
1323
  console.warn('[scheduler] failed to persist scheduler state', e?.message);
@@ -1629,14 +1634,16 @@ function findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
1629
1634
  // (The single-file EMPTY_QUEUE/shapeQueue readers were retired with the
1630
1635
  // global queue.json — queueStore.cjs's merged readers own the shape now.)
1631
1636
 
1632
- // Quarantine a corrupt shard alongside itself (once per process — the first
1633
- // copy is the one that matters; later ticks would just overwrite it with the
1634
- // same bytes) so a human can diff it against the .bak-* snapshots.
1635
- let quarantined = false;
1637
+ // Quarantine a corrupt shard alongside itself (once per PATH, not once per
1638
+ // process — a module-level boolean let one file's tear consume the latch and
1639
+ // silently swallow every other file's `.corrupt-*` copy for the rest of the
1640
+ // process's life, observed live 2026-09-11) so a human can diff it against
1641
+ // the .bak-* snapshots.
1642
+ const quarantinedPaths = new Set();
1636
1643
  function flagUnreadable(state) {
1637
1644
  if (!state.unreadable) return state;
1638
- if (!quarantined && state.unreadablePath) {
1639
- quarantined = true;
1645
+ if (state.unreadablePath && !quarantinedPaths.has(state.unreadablePath)) {
1646
+ quarantinedPaths.add(state.unreadablePath);
1640
1647
  try {
1641
1648
  fs.copyFileSync(state.unreadablePath, `${state.unreadablePath}.corrupt-${Date.now()}`);
1642
1649
  } catch { /* best-effort: the read already failed, the copy may too */ }
@@ -2411,6 +2418,48 @@ let firstFailureAt = null;
2411
2418
  let firstNon429FailureAt = null; // tracks only transient/config failures; 429s don't count toward network-pause threshold
2412
2419
  let lastFailureKind = null; // 'transient' | 'meter_rate_limited' | 'auth' | null
2413
2420
  let pauseClearedManuallyAt = null;
2421
+ // PRD: the usage-poller silent-failure-streak WARN is emitted once per streak,
2422
+ // not once per failure (57 failures must produce ONE opsErrorLog line, not 57).
2423
+ // Reset alongside consecutiveFailures everywhere that resets to 0.
2424
+ let failureStreakWarned = false;
2425
+ // Ceiling on pollLoop's exponential poll backoff (both the 'transient'/'config'
2426
+ // branch and the 'meter_rate_limited' branch below share this cap — a single
2427
+ // constant so the two never drift to different ceilings).
2428
+ const BACKOFF_MAX_MS = 480_000; // 8 minutes
2429
+ // Threshold past which a growing consecutiveFailures streak stops being normal
2430
+ // jitter and becomes worth a human's attention. health.cjs imports this so the
2431
+ // WARN and the `npm run health` non-GREEN trip at the exact same count.
2432
+ const FAILURE_STREAK_WARN_THRESHOLD = 5;
2433
+
2434
+ /** Pure: exponential backoff with a cap, shared by every pollLoop failure branch. Exported for unit testing. */
2435
+ function nextBackoffMs(prevBackoffMs) {
2436
+ return prevBackoffMs ? Math.min(prevBackoffMs * 2, BACKOFF_MAX_MS) : 30_000;
2437
+ }
2438
+
2439
+ /**
2440
+ * Pure: should this failure count trigger the one-time streak WARN? Exported
2441
+ * for unit testing. `alreadyWarned` is the current streak's warned flag —
2442
+ * true for every failure after the threshold-crossing one, so this returns
2443
+ * true exactly once per streak.
2444
+ */
2445
+ function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold = FAILURE_STREAK_WARN_THRESHOLD) {
2446
+ return consecutiveFailures >= threshold && !alreadyWarned;
2447
+ }
2448
+
2449
+ /** Emits the one-time opsErrorLog WARN for a failure streak crossing the threshold, if not already warned this streak. */
2450
+ function warnFailureStreakIfNeeded() {
2451
+ if (!shouldWarnFailureStreak(consecutiveFailures, failureStreakWarned)) return;
2452
+ failureStreakWarned = true;
2453
+ try {
2454
+ appendError({
2455
+ cwd: DEFAULT_PROJECT_CWD,
2456
+ scope: 'scheduler',
2457
+ level: 'warn',
2458
+ message: `usage/rate-limit poller has failed ${consecutiveFailures} consecutive times (kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${SCHEDULER_STATE_PATH}`,
2459
+ meta: { consecutiveFailures, backoffMs, lastFailureKind },
2460
+ });
2461
+ } catch { /* durable logging must never break the poll loop */ }
2462
+ }
2414
2463
  // PRD 1119: consecutive-rapid-rate-limit hard-pause tracking, keyed per slug.
2415
2464
  // See isCooldownSuppressed/nextRapidRateLimitCount below for the pure rules.
2416
2465
  const consecutiveRapidRateLimitsBySlug = new Map();
@@ -2783,6 +2832,7 @@ async function clearPause(source) {
2783
2832
  firstFailureAt = null;
2784
2833
  firstNon429FailureAt = null;
2785
2834
  lastFailureKind = null;
2835
+ failureStreakWarned = false;
2786
2836
  persistSchedulerState();
2787
2837
  }
2788
2838
  if (wasPaused) await broadcast({ flush: true });
@@ -2836,6 +2886,9 @@ function resetJobFields(job, errorMsg, opts = {}) {
2836
2886
  // human-driven reset must genuinely start the auto-fix budget over,
2837
2887
  // including the one-time dead-fix-plan-child reopen (PRD 1129).
2838
2888
  delete job.autoFixReopened;
2889
+ // Same category — an overrun badge is this run's outcome (see
2890
+ // findOverrunningJobs's header), never durable across a reset.
2891
+ delete job.overrun;
2839
2892
  // Like exitCode: this run's outcome, not durable across a reset — a stale
2840
2893
  // leak badge from a prior attempt must not linger once the job re-fires.
2841
2894
  delete job.leakedDescendants;
@@ -3140,6 +3193,7 @@ async function notifyOriginatingTab(job, {
3140
3193
  await appendTranscriptTurn(job.cwd, epicIdForTranscript, {
3141
3194
  role: 'assistant',
3142
3195
  text: resultText || message,
3196
+ eventId: `prd-result:${job.slug}:${job.runId || ''}`,
3143
3197
  });
3144
3198
  } catch (e) {
3145
3199
  console.error('[scheduler] notifyOriginatingTab transcript append error', job?.slug, e);
@@ -3997,13 +4051,20 @@ function pickRunDir() {
3997
4051
  /**
3998
4052
  * Execute a single PRD job. Writes stdout/stderr to a log file and a meta
3999
4053
  * JSON sidecar. Accepts an optional onPid(pid) callback called synchronously
4000
- * after spawn so callers can persist the pid before the job finishes.
4054
+ * after spawn so callers can persist the pid before the job finishes, and an
4055
+ * optional onPhase(phase) callback invoked at the top of this function (see
4056
+ * spawnJob's dispatchPhase breadcrumb) — this function has no access to
4057
+ * `mutate`, so the caller injects the stamping side effect instead.
4001
4058
  *
4002
4059
  * Uses withChildAndLog for the child lifecycle (fd open/close, watchdog timers).
4003
4060
  * Watchdogs are declared as an array; the result-tailer's exit-code mapping
4004
4061
  * (success+killedBySignal → 0) is scheduler-specific and lives in onExit.
4005
4062
  */
4006
- async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget = null, foreignWip = null, launchEnv = null) {
4063
+ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget = null, foreignWip = null, launchEnv = null, onPhase = null) {
4064
+ // First statement — stamps the 'exec-entered' dispatch-phase breadcrumb
4065
+ // before openLog below, so a hang inside openLog/spawn itself still shows
4066
+ // execution reached this function (see spawnJob's dispatchPhase comment).
4067
+ if (onPhase) await onPhase('exec-entered');
4007
4068
  const logPath = path.join(runDir, `${job.slug}.log`);
4008
4069
  const metaPath = path.join(runDir, `${job.slug}.meta.json`);
4009
4070
  // `cwd` stays the MAIN tree throughout — PRD lookup (findPrdDir/prdPathForJob)
@@ -4049,6 +4110,12 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4049
4110
 
4050
4111
  let prompt;
4051
4112
  let prdPath = null;
4113
+ // Captured on a successful parsePrd() so the raw, untruncated PRD body can
4114
+ // be recorded as a 'user' turn in the durable per-Epic transcript below —
4115
+ // the durable store must see the actual prompt the executor ran, not just
4116
+ // the scheduler's own short status chip (that's notifyOriginatingTab's job
4117
+ // for the assistant side).
4118
+ let parsedPrdMeta = null;
4052
4119
  if (resumeTarget) {
4053
4120
  // Resume mode (PRD 1111): a short deterministic preamble naming the
4054
4121
  // recorded dirty paths, NEVER the original PRD body — the resumed
@@ -4071,6 +4138,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4071
4138
  // resolved — never concatenated here, so it can't end up ahead of the
4072
4139
  // digest in the composed prompt (PRD 992).
4073
4140
  prompt = parsed.body;
4141
+ parsedPrdMeta = parsed;
4074
4142
  } catch (e) {
4075
4143
  // The project-scoped dir isn't the only place a PRD source can live — a
4076
4144
  // writer that hasn't migrated to prdLocations.cjs yet (or a not-yet-run
@@ -4084,6 +4152,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4084
4152
  const parsed = await parsePrd(fallbackPath);
4085
4153
  prompt = parsed.body;
4086
4154
  prdPath = fallbackPath;
4155
+ parsedPrdMeta = parsed;
4087
4156
  } catch (e2) {
4088
4157
  // Found the dir a moment ago but the read still failed. Case A: the
4089
4158
  // source has since been archived — stale-skip as usual. Case B: it
@@ -4121,6 +4190,26 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
4121
4190
  }
4122
4191
  } // end resumeTarget ? preamble : normal-PRD-read
4123
4192
 
4193
+ // Record the raw, untruncated PRD body as a 'user' turn before any digest/
4194
+ // finish-protocol boilerplate is concatenated onto `prompt` below — mirrors
4195
+ // notifyOriginatingTab's epicId resolution order (sourcePromptId, then
4196
+ // sourceTabId, then job.epicId) so the same Epic's user/assistant turns
4197
+ // land in the same transcript file. Best-effort: never blocks a spawn.
4198
+ if (!resumeTarget && parsedPrdMeta) {
4199
+ const epicIdForPromptTurn = parsedPrdMeta.sourcePromptId || parsedPrdMeta.sourceTabId || job.epicId || null;
4200
+ if (epicIdForPromptTurn && cwd) {
4201
+ try {
4202
+ await promptSessionTranscript.appendTurn(cwd, epicIdForPromptTurn, {
4203
+ role: 'user',
4204
+ text: prompt,
4205
+ eventId: `prd:${job.slug}`,
4206
+ });
4207
+ } catch (e) {
4208
+ safeLog(`[scheduler] transcript append (PRD prompt) failed: ${e?.message ?? e}\n`);
4209
+ }
4210
+ }
4211
+ }
4212
+
4124
4213
  let contextDigestApplied = false;
4125
4214
  let originSessionId = null;
4126
4215
  if (!resumeTarget) {
@@ -5165,10 +5254,30 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5165
5254
  : await jobWorktree.createJobWorktree({ cwd: job.cwd || defaultCwd, slug: job.slug });
5166
5255
  if (!preflightWorktree.ok && /^worktree cap reached\b/.test(preflightWorktree.reason || '')) {
5167
5256
  console.log(`[scheduler] ${job.slug}: deferring — ${preflightWorktree.reason}`);
5168
- await mutate((s) => {
5257
+ // A leaked worktree cap count used to strand a job silently — nothing
5258
+ // but this console.log + a heldReason field the human had to go
5259
+ // looking for (see gitWorktree.cjs's reserveWorktreeSlot: the cap
5260
+ // itself now self-heals, but a GENUINE block, while it lasts, must
5261
+ // still reach a durable, visible channel). Fired only the first time
5262
+ // THIS slug gets held on this exact reason — a job stuck for hours
5263
+ // must not flood opsErrorLog with one line per dispatch attempt.
5264
+ const wasAlreadyFlaggedForThisReason = await mutate((s) => {
5169
5265
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
5266
+ const already = idx >= 0 && s.jobs[idx].heldReason === preflightWorktree.reason;
5170
5267
  if (idx >= 0) s.jobs[idx].heldReason = preflightWorktree.reason;
5268
+ return already;
5171
5269
  });
5270
+ if (!wasAlreadyFlaggedForThisReason) {
5271
+ try {
5272
+ appendError({
5273
+ cwd: job.cwd || defaultCwd,
5274
+ scope: 'scheduler',
5275
+ level: 'error',
5276
+ message: `${job.slug}: dispatch blocked — ${preflightWorktree.reason}`,
5277
+ meta: { slug: job.slug, reason: preflightWorktree.reason },
5278
+ });
5279
+ } catch { /* durable logging must never break the queue */ }
5280
+ }
5172
5281
  await broadcast({ flush: true });
5173
5282
  return;
5174
5283
  }
@@ -5290,6 +5399,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5290
5399
  delete s.jobs[idx].heldReason;
5291
5400
  s.jobs[idx].runId = runId;
5292
5401
  s.jobs[idx].startedAt = new Date().toISOString();
5402
+ // Dispatch-phase breadcrumb (PRD: dispatch-region diagnostic
5403
+ // breadcrumb) — a transient marker of how far THIS dispatch got,
5404
+ // exactly like heldReason: stamped at each step of the
5405
+ // running-transition → executeJob → onPid region and deleted at
5406
+ // finalize (see the two finalize mutates below) so a
5407
+ // completed/failed row never carries a stale one.
5408
+ s.jobs[idx].dispatchPhase = 'running-stamped';
5409
+ s.jobs[idx].dispatchPhaseAt = s.jobs[idx].startedAt;
5293
5410
  dispatchStartedAtMs = Date.parse(s.jobs[idx].startedAt);
5294
5411
  if (job.quietMachine === true) {
5295
5412
  s.jobs[idx].quietMachine = true;
@@ -5308,6 +5425,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5308
5425
  await broadcast({ flush: true });
5309
5426
  if (dispatchSkippedAlreadyCompleted) return;
5310
5427
 
5428
+ await mutate((s) => {
5429
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
5430
+ if (idx >= 0) {
5431
+ s.jobs[idx].dispatchPhase = 'pre-run-git-snapshot';
5432
+ s.jobs[idx].dispatchPhaseAt = new Date().toISOString();
5433
+ }
5434
+ });
5435
+
5311
5436
  // Commit-guard baseline: snapshot the working tree BEFORE the run so the
5312
5437
  // post-run check flags only paths THIS job left dirty, not pre-existing WIP.
5313
5438
  const guardCwd = job.cwd || defaultCwd;
@@ -5340,6 +5465,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5340
5465
  s.jobs[idx].guardHeadBefore = guardHeadBefore || null;
5341
5466
  if (preRunDirtyPaths.length) s.jobs[idx].preRunDirtyPaths = preRunDirtyPaths;
5342
5467
  else delete s.jobs[idx].preRunDirtyPaths;
5468
+ s.jobs[idx].dispatchPhase = 'baseline-persisted';
5469
+ s.jobs[idx].dispatchPhaseAt = new Date().toISOString();
5343
5470
  }
5344
5471
  });
5345
5472
 
@@ -5361,17 +5488,21 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5361
5488
  console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
5362
5489
  } else {
5363
5490
  console.log(`[scheduler] ${job.slug}: running in main tree (worktree not used: ${worktree.reason})`);
5364
- // Surface any degraded-isolation fallback on the job row itself so it's
5365
- // queryable from the queue instead of console-only — except the
5366
- // deliberate env-disable flag, which is an intentional opt-out, not a
5367
- // degradation worth flagging.
5368
- if (!jobWorktree.isWorktreeDisabled()) {
5369
- await mutate((s) => {
5370
- const idx = s.jobs.findIndex((x) => x.slug === job.slug);
5371
- if (idx >= 0) s.jobs[idx].worktreeFallbackReason = worktree.reason;
5372
- });
5373
- }
5374
5491
  }
5492
+ // dispatchPhase stamp folded into a single unconditional mutate covering
5493
+ // both branches above — the degraded-isolation fallback flag (skipped
5494
+ // for the deliberate env-disable opt-out) rides along in the SAME
5495
+ // mutate rather than a second one.
5496
+ await mutate((s) => {
5497
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
5498
+ if (idx >= 0) {
5499
+ s.jobs[idx].dispatchPhase = 'worktree-resolved';
5500
+ s.jobs[idx].dispatchPhaseAt = new Date().toISOString();
5501
+ if (!worktree.ok && !jobWorktree.isWorktreeDisabled()) {
5502
+ s.jobs[idx].worktreeFallbackReason = worktree.reason;
5503
+ }
5504
+ }
5505
+ });
5375
5506
  // Base-tree WIP carried into the worktree (createWorktree, PRD 1094) —
5376
5507
  // recorded on the job row so integration can exclude these paths from
5377
5508
  // the branch diff below, and so it's queryable from the queue.
@@ -5422,10 +5553,20 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5422
5553
  if (idx >= 0) {
5423
5554
  s.jobs[idx].sessionId = sessionId;
5424
5555
  s.jobs[idx].runtime = { pid, runId, startedAt: s.jobs[idx].startedAt, sessionId, cwd };
5556
+ s.jobs[idx].dispatchPhase = 'spawned';
5557
+ s.jobs[idx].dispatchPhaseAt = new Date().toISOString();
5425
5558
  }
5426
5559
  });
5427
5560
  await broadcast({ flush: true });
5428
- }, worktree.ok ? worktree.dir : undefined, resumeTarget, foreignWip, launchEnv);
5561
+ }, worktree.ok ? worktree.dir : undefined, resumeTarget, foreignWip, launchEnv, async (phase) => {
5562
+ await mutate((s) => {
5563
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
5564
+ if (idx >= 0) {
5565
+ s.jobs[idx].dispatchPhase = phase;
5566
+ s.jobs[idx].dispatchPhaseAt = new Date().toISOString();
5567
+ }
5568
+ });
5569
+ });
5429
5570
  } finally {
5430
5571
  if (worktree.ok) {
5431
5572
  worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
@@ -6033,6 +6174,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6033
6174
  delete s.jobs[i2].sharedTreeGuard;
6034
6175
  }
6035
6176
  delete s.jobs[i2].runtime;
6177
+ // Dispatch-phase breadcrumb was only ever meant to cover THIS
6178
+ // dispatch's pre-spawn/spawn region — a terminal row must not
6179
+ // carry a stale one into its next dispatch (unlike guardBaseline's
6180
+ // preRunDirtyPaths sibling, which deliberately survives to
6181
+ // history.jsonl — this has no such carry-forward use).
6182
+ delete s.jobs[i2].dispatchPhase;
6183
+ delete s.jobs[i2].dispatchPhaseAt;
6184
+ // A completed/failed/needs_review row is no longer running — its
6185
+ // overrun badge (if any) was this run's outcome and must not
6186
+ // linger onto whatever the next dispatch of this slug does.
6187
+ delete s.jobs[i2].overrun;
6036
6188
  // Pre-run baseline no longer needed once this run has finalized —
6037
6189
  // its whole purpose (letting THIS finalize compute a truthful
6038
6190
  // delta) is done; a fresh one is captured at the next dispatch.
@@ -6326,6 +6478,78 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6326
6478
  }
6327
6479
  }
6328
6480
 
6481
+ // Throttle for reclaimTerminalJobOrphansThrottled below — a `git worktree
6482
+ // list` + per-candidate `status`/`merge-base` call per known project cwd is
6483
+ // cheap once, but tickQueue can fire every few seconds; there is no value in
6484
+ // re-running this on every single tick.
6485
+ const TERMINAL_ORPHAN_RECLAIM_INTERVAL_MS = 10 * 60_000; // 10 minutes
6486
+ let lastTerminalOrphanReclaimAt = 0;
6487
+
6488
+ /**
6489
+ * Fire-and-forget, throttled sweep for job-kind worktrees whose owning queue
6490
+ * row has already resolved (completed/failed/skipped) but whose checkout is
6491
+ * still on disk — see gitWorktree.cjs's reclaimTerminalJobOrphans header for
6492
+ * the two independent safety proofs it requires before deleting anything,
6493
+ * and jobWorktreeTerminalOrphanLive.cjs for the liveness gate checked BEFORE
6494
+ * those proofs (PRD 1163: neither proof is itself a liveness check, so a
6495
+ * still-running executor behind a prematurely-terminal row could otherwise
6496
+ * satisfy both).
6497
+ *
6498
+ * Runs once per TERMINAL_ORPHAN_RECLAIM_INTERVAL_MS across every project cwd
6499
+ * this tick's queue state knows about. The `/proc` holder scan
6500
+ * (`listCwdHolders`) is computed ONCE per sweep here and reused for every
6501
+ * candidate across every cwd, rather than re-scanning `/proc` per candidate.
6502
+ * Never throws (each per-cwd call is itself never-throws; this wrapper is
6503
+ * just the throttle + fan-out + queue-row stamping).
6504
+ */
6505
+ async function reclaimTerminalJobOrphansThrottled(state) {
6506
+ const now = Date.now();
6507
+ if (now - lastTerminalOrphanReclaimAt < TERMINAL_ORPHAN_RECLAIM_INTERVAL_MS) return;
6508
+ lastTerminalOrphanReclaimAt = now;
6509
+
6510
+ const terminalJobsByCwd = new Map();
6511
+ for (const j of state.jobs || []) {
6512
+ if (j.status !== 'completed' && j.status !== 'failed' && j.status !== 'skipped') continue;
6513
+ const cwd = j.cwd || DEFAULT_PROJECT_CWD;
6514
+ if (!terminalJobsByCwd.has(cwd)) terminalJobsByCwd.set(cwd, []);
6515
+ terminalJobsByCwd.get(cwd).push(j);
6516
+ }
6517
+
6518
+ const cwdHolders = gitWorktree.listCwdHolders();
6519
+ // { slug -> reason }, collected across every cwd this sweep touches, so a
6520
+ // single mutate() at the end can stamp every blocked-live row at once.
6521
+ const blockedLive = new Map();
6522
+
6523
+ for (const [cwd, terminalJobs] of terminalJobsByCwd) {
6524
+ const terminalSlugs = new Set(terminalJobs.map((j) => j.slug));
6525
+ const isLive = buildTerminalOrphanIsLive({
6526
+ terminalJobs,
6527
+ claudePidAlive,
6528
+ hasLiveHolder: gitWorktree.hasLiveHolder,
6529
+ cwdHolders,
6530
+ onLive: (slug, reason) => blockedLive.set(slug, reason),
6531
+ });
6532
+ const reclaimed = await jobWorktree.reclaimTerminalJobOrphans({ cwd, terminalSlugs, isLive });
6533
+ if (reclaimed.length) {
6534
+ console.log(`[scheduler] reclaimed ${reclaimed.length} terminal job-worktree orphan(s) in ${cwd}: ${reclaimed.join(', ')}`);
6535
+ }
6536
+ }
6537
+
6538
+ if (blockedLive.size > 0) {
6539
+ const stampedAt = new Date().toISOString();
6540
+ await mutate((s) => {
6541
+ for (const j of s.jobs || []) {
6542
+ if (!blockedLive.has(j.slug)) continue;
6543
+ j.worktreeReclaimBlockedLive = true;
6544
+ j.worktreeReclaimBlockedLiveAt = stampedAt;
6545
+ j.worktreeReclaimBlockedLiveReason = blockedLive.get(j.slug);
6546
+ }
6547
+ }).catch((e) => {
6548
+ console.warn('[scheduler] failed to stamp worktreeReclaimBlockedLive', e?.message);
6549
+ });
6550
+ }
6551
+ }
6552
+
6329
6553
  /**
6330
6554
  * Dispatch a resume-recovery attempt (PRD 1111) for a job already found
6331
6555
  * eligible by selectResumeRecoveryTarget. Thin wrapper around spawnJob —
@@ -6365,10 +6589,29 @@ function tickQueue({ bypassLoadGate = false } = {}) {
6365
6589
  }
6366
6590
  if (cancelToken.cancelled) return { fired: false, reason: 'cancelled' };
6367
6591
 
6592
+ // Stamped here — the moment tickQueue actually reaches the picker,
6593
+ // regardless of whether this pass ends in a launch — so
6594
+ // classifyQueueStarvation can tell "the engine keeps evaluating the
6595
+ // queue" apart from "nothing has invoked tickQueue in a long time".
6596
+ // Distinct from `lastRunAt` below, which stays true to its existing
6597
+ // meaning (a batch actually launched) since other readers depend on that.
6598
+ await mutate((s) => { s.lastDispatchAttemptAt = new Date().toISOString(); });
6599
+
6368
6600
  // The retired-flat-dir sweep now lives inside reconcile() itself (see its
6369
6601
  // own comment) so every caller of reconcile — not just this tick — gets
6370
6602
  // the guarantee.
6371
6603
  await reconcile(state);
6604
+ // Reclaim any job-kind worktree whose owning row already resolved
6605
+ // (completed/failed/skipped) without the run ever reaching
6606
+ // cleanupWorktree — a leaked checkout that would otherwise sit until the
6607
+ // stale-age sweep (default 24h) or the next app restart
6608
+ // (reconcileWorktreesOnBoot only runs once, at boot). Throttled to once
6609
+ // per interval (not every tick) since it's a `git worktree list` +
6610
+ // per-candidate git call per known project cwd. Best-effort and
6611
+ // fire-and-forget: must never hold up dispatch.
6612
+ reclaimTerminalJobOrphansThrottled(state).catch((e) => {
6613
+ console.warn('[scheduler] terminal job-worktree orphan reclaim failed', e?.message);
6614
+ });
6372
6615
  // Session-Manager's machine-wide slot pool is the ONLY concurrency limit
6373
6616
  // the picker answers to (plus the memory gate below). The scheduler used
6374
6617
  // to also carry a private `concurrencyCap` of 3 — the exact per-consumer
@@ -6624,11 +6867,15 @@ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now,
6624
6867
  * with ready work idle indefinitely.
6625
6868
  */
6626
6869
  async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
6870
+ // lastDispatchAttemptAt, not lastRunAt: the latter only advances when a
6871
+ // batch actually launches, so a poll that keeps succeeding while dispatch
6872
+ // itself never gets invoked would otherwise mask a stall behind a fresh-
6873
+ // looking timestamp that was never actually tracking dispatch liveness.
6627
6874
  const verdict = classifyQueueStarvation({
6628
6875
  jobs: state?.jobs,
6629
6876
  paused: state?.paused,
6630
6877
  runningCount: runningSet.size,
6631
- lastRunAtMs: Date.parse(state?.lastRunAt ?? ''),
6878
+ lastRunAtMs: Date.parse(state?.lastDispatchAttemptAt ?? ''),
6632
6879
  now,
6633
6880
  thresholdMs,
6634
6881
  });
@@ -6655,6 +6902,15 @@ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs
6655
6902
  // returns early on null). Treat unknown as safe here, exactly as the
6656
6903
  // billing meter's own 429 fallback already does.
6657
6904
  if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
6905
+ // The in-process cancelToken is only ever reset by runDueJobs() (force-tick
6906
+ // / run-now / resume-timer) — every other path that clears a pause
6907
+ // (clearPause(), the poll loop's own auto-recovery) leaves it untouched
6908
+ // (see applyPauseCleared's header for the 2026-07-14 incident this caused).
6909
+ // A watchdog that fires but doesn't clear a stuck cancelToken would forever
6910
+ // hit tickQueue's very first guard and report a forced tick that never
6911
+ // actually ticked. The watchdog is the last line of defence against a
6912
+ // wedged dispatcher, so it must be able to un-wedge this too.
6913
+ cancelToken.cancelled = false;
6658
6914
  await tickQueue({ bypassLoadGate: false }).catch((e) => console.error('[scheduler] starvation tick error', e));
6659
6915
  return verdict;
6660
6916
  }
@@ -6794,8 +7050,17 @@ async function reapDeadRunningJobs() {
6794
7050
  // a true claim when the run dir produced no log output at all; a log
6795
7051
  // with real content proves the job DID run (see
6796
7052
  // resolvePidlessGateOutcome's header).
6797
- const gateOutcome = pidless ? resolvePidlessGateOutcome(outcome, logHasOutput(logPath)) : mapOutcomeToGateOutcome(outcome);
6798
- dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath });
7053
+ // A pidless row was stamped with the runId of whatever batch dir
7054
+ // tickQueue handed its dispatch (pickRunDir's header: "tickQueue hands
7055
+ // ONE shared batch dir to every spawnJob in the batch") BEFORE the
7056
+ // spawn that never completed — so that dir may hold nothing of this
7057
+ // slug's own, or only a sibling's `<other-slug>.log` from the same
7058
+ // batch. logHasOutput's own existsSync+size check is exactly "does
7059
+ // THIS slug have any artifact in there" — reused rather than
7060
+ // re-derived so a phantom link never survives the reap.
7061
+ const hasOwnArtifact = pidless ? logHasOutput(logPath) : true;
7062
+ const gateOutcome = pidless ? resolvePidlessGateOutcome(outcome, hasOwnArtifact) : mapOutcomeToGateOutcome(outcome);
7063
+ dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath, noOwnArtifact: pidless && !hasOwnArtifact });
6799
7064
  }
6800
7065
 
6801
7066
  queueHealthSweepCycle += 1;
@@ -6894,7 +7159,7 @@ async function reapDeadRunningJobs() {
6894
7159
  }
6895
7160
 
6896
7161
  await mutate(async (s) => {
6897
- for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
7162
+ for (const { slug, pid, outcome, gateOutcome, pidless, reason, noOwnArtifact } of dead) {
6898
7163
  const idx = s.jobs.findIndex((x) => x.slug === slug);
6899
7164
  if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
6900
7165
  const rateLimited = outcome === 'rate_limited';
@@ -6984,7 +7249,22 @@ async function reapDeadRunningJobs() {
6984
7249
  }
6985
7250
  if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
6986
7251
  }
7252
+ // A pidless spawn that never wrote its own '<slug>.log' into the
7253
+ // batch runId dir it was stamped with must not keep that runId —
7254
+ // it points at a directory with no artifacts for THIS slug (at best
7255
+ // empty, at worst only a sibling's log from the same batch — see the
7256
+ // 'dead' push above). resolveRunId's own backfill scan
7257
+ // (existsSync-per-dir on '<slug>.log') would find nothing here
7258
+ // either, so clearing this makes job.runId and resolveRunId(job)
7259
+ // agree: both null, rather than one falsely pointing at a run this
7260
+ // slug never produced.
7261
+ if (!rateLimited && noOwnArtifact) {
7262
+ s.jobs[idx].runId = null;
7263
+ }
6987
7264
  delete s.jobs[idx].runtime;
7265
+ delete s.jobs[idx].dispatchPhase;
7266
+ delete s.jobs[idx].dispatchPhaseAt;
7267
+ delete s.jobs[idx].overrun;
6988
7268
  delete s.jobs[idx].guardBaseline;
6989
7269
  delete s.jobs[idx].guardHeadBefore;
6990
7270
  applyLeftoverFields(s.jobs[idx], deltaPaths);
@@ -7046,6 +7326,7 @@ async function pollLoop() {
7046
7326
  firstFailureAt = null;
7047
7327
  firstNon429FailureAt = null;
7048
7328
  lastFailureKind = null;
7329
+ failureStreakWarned = false;
7049
7330
  lastPollAt = Date.now();
7050
7331
  lastPollOk = true;
7051
7332
  persistSchedulerState();
@@ -7072,6 +7353,7 @@ async function pollLoop() {
7072
7353
  firstFailureAt = null;
7073
7354
  firstNon429FailureAt = null;
7074
7355
  lastFailureKind = null;
7356
+ failureStreakWarned = false;
7075
7357
  lastPollAt = Date.now();
7076
7358
  lastPollOk = true;
7077
7359
  persistSchedulerState();
@@ -7089,13 +7371,23 @@ async function pollLoop() {
7089
7371
  } else if (r.kind === 'meter_rate_limited') {
7090
7372
  // Billing meter is itself being rate-limited. Treat as "utilization unknown but safe":
7091
7373
  // fire available jobs anyway at utilization=0 rather than pausing the queue.
7374
+ // Still back off the POLL cadence itself (same curve/cap as the transient
7375
+ // branch) and persist state every cycle — without this, a sustained 429
7376
+ // streak hammered the already-rate-limited endpoint every POLL_INTERVAL_MS
7377
+ // forever AND never wrote lastPollAt/consecutiveFailures back to
7378
+ // scheduler-state.json, so the sidecar froze stale while the loop kept
7379
+ // failing silently underneath it (the 57-consecutive-failure incident).
7092
7380
  lastPollAt = Date.now();
7093
7381
  lastPollOk = false;
7094
7382
  consecutiveFailures++;
7095
7383
  lastFailureKind = 'meter_rate_limited';
7096
7384
  // Don't update firstNon429FailureAt — 429s don't count toward the 30-min network-pause threshold.
7385
+ backoffMs = nextBackoffMs(backoffMs);
7386
+ backoffNextAt = Date.now() + backoffMs;
7097
7387
  cachedUtilization = 0; // assume safe; fire any pending work
7098
- console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on heuristic (failure #${consecutiveFailures})`);
7388
+ console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on heuristic (failure #${consecutiveFailures}); retry in ${backoffMs / 1000}s`);
7389
+ warnFailureStreakIfNeeded();
7390
+ persistSchedulerState();
7099
7391
  const cur = await readQueue();
7100
7392
  await maybeLaunchWhenAvailable(cur);
7101
7393
  await broadcast();
@@ -7113,7 +7405,7 @@ async function pollLoop() {
7113
7405
  // transient or config — apply exponential backoff and count toward 30-min threshold.
7114
7406
  lastFailureKind = 'transient';
7115
7407
  if (!firstNon429FailureAt) firstNon429FailureAt = Date.now();
7116
- backoffMs = backoffMs ? Math.min(backoffMs * 2, 480_000) : 30_000;
7408
+ backoffMs = nextBackoffMs(backoffMs);
7117
7409
  const totalNon429FailureMs = Date.now() - firstNon429FailureAt;
7118
7410
  console.log(`[scheduler] transient failure #${consecutiveFailures}: ${r.kind} ${r.message ?? ''}; retry in ${backoffMs / 1000}s`);
7119
7411
 
@@ -7127,7 +7419,21 @@ async function pollLoop() {
7127
7419
  }
7128
7420
 
7129
7421
  backoffNextAt = Date.now() + backoffMs;
7422
+ warnFailureStreakIfNeeded();
7130
7423
  persistSchedulerState();
7424
+ // A failed billing poll must not silently stop dispatch — only the
7425
+ // 'ok' and 'meter_rate_limited' branches used to reach
7426
+ // maybeLaunchWhenAvailable, so auth/transient failures left ready
7427
+ // pending work untouched until either the queue-starvation watchdog's
7428
+ // 10-minute safety net fired or the poll itself recovered. Utilization
7429
+ // is unknown during a failed poll, not unsafe — treated the same way
7430
+ // the meter_rate_limited branch above already treats a 429 as safe to
7431
+ // fire through. maybeLaunchWhenAvailable itself still honors an
7432
+ // 'auth'/'network' pause (state.paused), so this is a no-op whenever
7433
+ // setPaused() above actually engaged one.
7434
+ if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
7435
+ await maybeLaunchWhenAvailable(await readQueue());
7436
+ await broadcast();
7131
7437
  }
7132
7438
  } catch (e) {
7133
7439
  // Unexpected error (e.g., IPC transport failure)
@@ -7137,9 +7443,17 @@ async function pollLoop() {
7137
7443
  lastFailureKind = 'transient';
7138
7444
  if (!firstFailureAt) firstFailureAt = Date.now();
7139
7445
  if (!firstNon429FailureAt) firstNon429FailureAt = Date.now();
7140
- backoffMs = backoffMs ? Math.min(backoffMs * 2, 480_000) : 30_000;
7446
+ backoffMs = nextBackoffMs(backoffMs);
7141
7447
  backoffNextAt = Date.now() + backoffMs;
7448
+ warnFailureStreakIfNeeded();
7142
7449
  persistSchedulerState();
7450
+ // Same rationale as the auth/transient branch above: the outer catch
7451
+ // must not be a silent dispatch dead-end either.
7452
+ try {
7453
+ if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
7454
+ await maybeLaunchWhenAvailable(await readQueue());
7455
+ await broadcast();
7456
+ } catch { /* best-effort — the poll loop must still re-arm below */ }
7143
7457
  } finally {
7144
7458
  const delay = backoffMs || POLL_INTERVAL_MS;
7145
7459
  pollLoopTimer = setTimeout(() => { pollLoop().catch(() => {}); }, delay);
@@ -7422,13 +7736,90 @@ function isRescanCandidate(job) {
7422
7736
  * the unguarded boot call, could reach it). Reported 2026-09-10 from
7423
7737
  * social-signals-trader.
7424
7738
  *
7425
- * Cost: one extra classifyRunOutcome() log stat per `failed` row, and only
7426
- * for rows that reach isFailedUnverifiedShaped's log check — bounded by the
7427
- * failed-row count, and strictly cheaper than firing the whole pass (which
7428
- * does git fetches via computeLooksDone) on every tick.
7739
+ * Reopened through a different door 2026-09-12 (starry-night-ships
7740
+ * 231-saturn-record-and-docs / 243-neptune-kurama-mode): the guard above only
7741
+ * covered isRescanCandidate (re-verification), but reverifyNeedsReview's auto-
7742
+ * fix loop ALSO lives behind this same guard, and a `needs_review` row whose
7743
+ * mechanical recovery already ran and failed (mechanicalRecoveryAttempted:
7744
+ * true, verdict still 'worktree_integration_failed', not itself a
7745
+ * RESCANNABLE_VERDICTS member) is invisible to isRescanCandidate — so the
7746
+ * next rung (a fix-plan investigation via selectAutoFixTargets) never got a
7747
+ * chance to fire either. Widened to OR in every live target of the recovery
7748
+ * ladder the periodic pass actually drives (selectMechanicalRecoveryTarget /
7749
+ * selectResumeRecoveryTarget / selectAutoFixTargets) so the guard can never
7750
+ * again be narrower than the work reverifyNeedsReview performs.
7751
+ *
7752
+ * Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget are pure
7753
+ * (no I/O). selectAutoFixTargets is called with an injected fixSlugExists
7754
+ * that always returns false — cheap and deliberately over-inclusive (a false
7755
+ * positive here just means one extra periodic pass, never a missed one) so
7756
+ * this guard never pays selectAutoFixTargets's production fs.existsSync scan
7757
+ * per tick. resolveRunId's IO only fires for rows missing job.runId, same as
7758
+ * isRescanCandidate already incurs above.
7429
7759
  */
7430
7760
  function shouldRunPeriodicReverify(jobs) {
7431
- return Array.isArray(jobs) && jobs.some((j) => isRescanCandidate(j));
7761
+ if (!Array.isArray(jobs)) return false;
7762
+ if (jobs.some((j) => isRescanCandidate(j))) return true;
7763
+ if (jobs.some((j) => selectMechanicalRecoveryTarget(j) || selectResumeRecoveryTarget(j))) return true;
7764
+ return selectAutoFixTargets(jobs, { fixSlugExists: () => false }).length > 0;
7765
+ }
7766
+
7767
+ // Default 24h, overridable via SM_STUCK_FAILED_ESCALATE_HOURS — same
7768
+ // env-override shape as QUARANTINE_ESCALATE_MS above.
7769
+ const STUCK_FAILED_ESCALATE_MS = process.env.SM_STUCK_FAILED_ESCALATE_HOURS
7770
+ ? Number(process.env.SM_STUCK_FAILED_ESCALATE_HOURS) * 60 * 60_000
7771
+ : 24 * 60 * 60_000;
7772
+
7773
+ /**
7774
+ * Kill-switch gate for the stuck-failed escalation below
7775
+ * (SM_STUCK_FAILED_ESCALATE_DISABLE=1), mirroring the
7776
+ * SM_REVERIFY_PERIODIC_DISABLE / SM_RCA_DISABLE convention. A tiny wrapper
7777
+ * so the disable path is unit-testable without invoking the 10-minute
7778
+ * setInterval body directly.
7779
+ */
7780
+ function stuckFailedEscalationDisabled() {
7781
+ return process.env.SM_STUCK_FAILED_ESCALATE_DISABLE === '1';
7782
+ }
7783
+
7784
+ /**
7785
+ * findStuckFailedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
7786
+ *
7787
+ * Pure predicate (isRescanCandidate's own resolveRunId/classifyRunOutcome log
7788
+ * read is the only IO, gated per-job exactly like shouldRunPeriodicReverify
7789
+ * above). `failed` is a fully terminal state for every automated recovery
7790
+ * path — selectResumeRecoveryTarget/selectAutoFixTargets both require
7791
+ * needs_review, reapDeadRunningJobs only ever writes running → failed, and
7792
+ * reconcile-repair's to-pending is for structurally invalid rows. Only a
7793
+ * human's scheduler_reset_job ever takes failed → pending (LEGAL_TRANSITIONS).
7794
+ * A rescan candidate (isRescanCandidate) that has sat failed longer than
7795
+ * `thresholdMs` can therefore go silently stuck forever — job
7796
+ * 4056-outcome-stats sat `failed` for five days with no operator signal
7797
+ * (reported 2026-09-10, social-signals-trader) even though the periodic
7798
+ * reverify pass (once shouldRunPeriodicReverify's guard was fixed) WAS firing
7799
+ * on it — reverifyNeedsReview's failed branch can annotate looksDone but can
7800
+ * never resolve a failed row itself (see its own header). This is the
7801
+ * visibility half that guard fix was missing: escalate once, never requeue.
7802
+ *
7803
+ * `stuckFailedNotified` gates this to exactly once per row — once the caller
7804
+ * stamps it, this always excludes that row so a human is never re-paged on
7805
+ * every 10-minute tick for the same stuck job. A row with no recoverable
7806
+ * 'failed' timestamp is skipped rather than guessed at (mirrors
7807
+ * findStaleQuarantinedJobs above).
7808
+ */
7809
+ function findStuckFailedJobs(jobs, now, thresholdMs) {
7810
+ const stuck = [];
7811
+ for (const j of jobs ?? []) {
7812
+ if (j.status !== 'failed') continue;
7813
+ if (j.stuckFailedNotified === true) continue;
7814
+ if (!isRescanCandidate(j)) continue;
7815
+ const entry = (j.statusHistory || []).find((h) => h.to === 'failed');
7816
+ if (!entry) continue;
7817
+ const since = Date.parse(entry.at);
7818
+ if (Number.isNaN(since)) continue;
7819
+ const ageMs = now - since;
7820
+ if (ageMs >= thresholdMs) stuck.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs });
7821
+ }
7822
+ return stuck;
7432
7823
  }
7433
7824
 
7434
7825
  /**
@@ -8264,7 +8655,18 @@ async function init() {
8264
8655
  try {
8265
8656
  const worktreeCwds = new Set(bootSnap.jobs.map((j) => j.cwd).filter(Boolean));
8266
8657
  worktreeCwds.add(DEFAULT_PROJECT_CWD);
8267
- await jobWorktree.reconcileWorktreesOnBoot([...worktreeCwds]);
8658
+ // A job is spawned `detached: true`, so its `claude -p` executor can
8659
+ // survive this very app restart — a worktree found at boot is NOT, by
8660
+ // itself, proof its run already died. isLive checks the already-read
8661
+ // bootSnap (no extra queue read) for a live running-row pid, OR a live
8662
+ // /proc cwd holder under the checkout itself. See jobWorktreeBootLive.cjs.
8663
+ const isLive = buildJobWorktreeIsLive({
8664
+ bootJobs: bootSnap.jobs,
8665
+ claudePidAlive,
8666
+ hasLiveHolder: gitWorktree.hasLiveHolder,
8667
+ cwdHolders: gitWorktree.listCwdHolders(),
8668
+ });
8669
+ await jobWorktree.reconcileWorktreesOnBoot([...worktreeCwds], { isLive });
8268
8670
  } catch (e) {
8269
8671
  console.error('[scheduler] boot worktree reconciliation failed', e?.message);
8270
8672
  }
@@ -8430,6 +8832,17 @@ async function init() {
8430
8832
  appendAuditEvent('job_overrunning_estimate', {
8431
8833
  slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
8432
8834
  });
8835
+ // Durable stamp so schedule:state (and therefore the renderer) can see
8836
+ // this without re-deriving it — the console.warn/audit event above are
8837
+ // visible only in the log, never on the row itself. Display-only
8838
+ // advisory field; re-stamped in place every sweep, never appended.
8839
+ mutate((state) => {
8840
+ const j = state.jobs.find((x) => x.slug === over.slug);
8841
+ if (!j) return;
8842
+ j.overrun = {
8843
+ ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
8844
+ };
8845
+ }).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
8433
8846
  }
8434
8847
 
8435
8848
  // Stranded-investigation restore. Unlike the two escalations above, this
@@ -8474,6 +8887,31 @@ async function init() {
8474
8887
  );
8475
8888
  appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
8476
8889
  }
8890
+
8891
+ // Stuck-failed escalation (2026-09-10, social-signals-trader): see
8892
+ // findStuckFailedJobs' header for why `failed` has no automated way
8893
+ // back to pending. Escalation only, same shape as the three warnings
8894
+ // above — never an automatic failed → pending requeue (that could
8895
+ // discard uncommitted work left by the failed run; see
8896
+ // spawnJob:fail-dirty). Kill-switch: SM_STUCK_FAILED_ESCALATE_DISABLE=1.
8897
+ if (!stuckFailedEscalationDisabled()) {
8898
+ const stuckFailed = findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
8899
+ if (stuckFailed.length > 0) {
8900
+ mutate((ms) => {
8901
+ for (const stuck of stuckFailed) {
8902
+ const j = ms.jobs.find((x) => x.slug === stuck.slug);
8903
+ if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
8904
+ j.stuckFailedNotified = true;
8905
+ console.warn(
8906
+ `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
8907
+ + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
8908
+ + `no automated recovery reaches a failed row; reset it by hand via scheduler_reset_job`,
8909
+ );
8910
+ appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
8911
+ }
8912
+ }).catch(() => {});
8913
+ }
8914
+ }
8477
8915
  }, 10 * 60_000);
8478
8916
 
8479
8917
  // Self-rescheduling poll loop with exponential backoff. Replaces the
@@ -9149,6 +9587,11 @@ module.exports = {
9149
9587
  init,
9150
9588
  ROOT,
9151
9589
  PRDS_DIR,
9590
+ SCHEDULER_STATE_PATH,
9591
+ BACKOFF_MAX_MS,
9592
+ FAILURE_STREAK_WARN_THRESHOLD,
9593
+ nextBackoffMs,
9594
+ shouldWarnFailureStreak,
9152
9595
  healRefusalReason,
9153
9596
  writeQueue,
9154
9597
  reconcile,
@@ -9170,6 +9613,9 @@ module.exports = {
9170
9613
  availableForJobs,
9171
9614
  reverifyNeedsReview,
9172
9615
  shouldRunPeriodicReverify,
9616
+ findStuckFailedJobs,
9617
+ STUCK_FAILED_ESCALATE_MS,
9618
+ stuckFailedEscalationDisabled,
9173
9619
  isRescanCandidate,
9174
9620
  isFailedUnverifiedShaped,
9175
9621
  computeLooksDone,
@@ -9281,6 +9727,8 @@ module.exports = {
9281
9727
  clearPause,
9282
9728
  tickQueue,
9283
9729
  runDueJobs,
9730
+ pollLoop,
9731
+ maybeLaunchWhenAvailable,
9284
9732
  isCooldownSuppressed,
9285
9733
  nextRapidRateLimitCount,
9286
9734
  CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,