claude-code-session-manager 0.77.0 → 0.79.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/dist/assets/{AgentLibrary-B2ie8bbw.js → AgentLibrary-COtVRqBR.js} +1 -1
  2. package/dist/assets/{DataModel-BIJPYw32.js → DataModel-CSEKw_OR.js} +1 -1
  3. package/dist/assets/{History-CeY6dk9S.js → History-CHHovrAO.js} +1 -1
  4. package/dist/assets/{Hooks-BFH2ocKg.js → Hooks-BZU6C3x6.js} +1 -1
  5. package/dist/assets/{HostBilko-36gj9wLz.js → HostBilko-CqTUoq37.js} +1 -1
  6. package/dist/assets/{Library-C-hBct39.js → Library-BtxdyTLz.js} +1 -1
  7. package/dist/assets/{ListDetail-CNq64VWV.js → ListDetail-qZc7Zm-6.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-Bh3qt5-1.js → MarkdownEditor-BHe_4fJR.js} +1 -1
  9. package/dist/assets/{McpServers-DpGN0oyz.js → McpServers-7Z98HLNo.js} +1 -1
  10. package/dist/assets/{Memory-D59hUjC4.js → Memory-CR72KoyP.js} +1 -1
  11. package/dist/assets/{Panel-DCgbaoci.js → Panel-pL6H3dpQ.js} +1 -1
  12. package/dist/assets/{Permissions-DAmQ0DYV.js → Permissions-CWSWjyXM.js} +1 -1
  13. package/dist/assets/{Plugins-Dyfgn6Is.js → Plugins-CN6lX2lt.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-BiYhPO1U.js → ProvenanceBadge-BXSXwIsk.js} +1 -1
  15. package/dist/assets/{SaveBar-RV7B6sOh.js → SaveBar-BlB5TGpR.js} +1 -1
  16. package/dist/assets/{Scheduler-BPaNqx1b.js → Scheduler-DRciWUmR.js} +1 -1
  17. package/dist/assets/{ScopeSwitcher-P4mdLGNU.js → ScopeSwitcher-kFrXtjpr.js} +1 -1
  18. package/dist/assets/{Settings-BL4vf5aX.js → Settings-BXuyf4lJ.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-BRBDyi1_.js → SkillReferenceGraph-Dfacb0PE.js} +1 -1
  20. package/dist/assets/{Skills-BV08gDUH.js → Skills-CHqcpiyt.js} +1 -1
  21. package/dist/assets/{SystemPrompt-CLftSsDw.js → SystemPrompt-fxXm0BZr.js} +1 -1
  22. package/dist/assets/{TagLibrary-Bp8jGsd5.js → TagLibrary-DOz65ZTz.js} +1 -1
  23. package/dist/assets/{TiptapBody-jCpuB6E5.js → TiptapBody-D0bWx_9o.js} +1 -1
  24. package/dist/assets/{Toggle-D2paA1xf.js → Toggle-C9jBwGSx.js} +1 -1
  25. package/dist/assets/{index-BDRSqBl3.js → index-DPYa6jbM.js} +419 -419
  26. package/dist/assets/{settingsSchema-6IOLjZZN.js → settingsSchema-BTPw1bR3.js} +1 -1
  27. package/dist/index.html +1 -1
  28. package/package.json +1 -1
  29. package/src/main/__tests__/runLogRetention.test.cjs +59 -0
  30. package/src/main/__tests__/scheduler-never-stop.test.cjs +157 -0
  31. package/src/main/__tests__/scheduler-no-orphan-run-dir.test.cjs +81 -0
  32. package/src/main/__tests__/scheduler-rate-limit-cooldown-freshness.test.cjs +123 -0
  33. package/src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs +158 -0
  34. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +30 -0
  35. package/src/main/index.cjs +8 -3
  36. package/src/main/ipcSchemas.cjs +2 -0
  37. package/src/main/lib/__tests__/delegationReadiness.test.cjs +45 -0
  38. package/src/main/lib/__tests__/reaperHelpers.test.cjs +90 -1
  39. package/src/main/lib/jobDirtFilter.cjs +54 -0
  40. package/src/main/lib/rateLimitDetect.cjs +35 -0
  41. package/src/main/lib/reaperHelpers.cjs +31 -6
  42. package/src/main/lib/runLogRetention.cjs +82 -4
  43. package/src/main/scheduler.cjs +353 -31
  44. package/src/preload/api.d.ts +8 -4
  45. package/src/preload/index.cjs +1 -0
@@ -58,6 +58,8 @@ const launchFailure = require('./lib/launchFailure.cjs');
58
58
  const { appendError } = require('./lib/opsErrorLog.cjs');
59
59
  const { readTail } = require('./lib/fileTail.cjs');
60
60
  const { claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
61
+ const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
62
+ const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
61
63
  const { computeQueueHealth } = require('./lib/queueHealth.cjs');
62
64
  const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
63
65
  const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
@@ -1252,6 +1254,89 @@ function computeStallSummary(state) {
1252
1254
  return { stalled, total, running, pending, byProject };
1253
1255
  }
1254
1256
 
1257
+ /**
1258
+ * computeBlockedChains(jobs) → [{ cwd, blockedBy, blocked }]
1259
+ *
1260
+ * Pure, no IO. The gap computeStallSummary above cannot see.
1261
+ *
1262
+ * `stalled` is defined as `running === 0 && pending === 0`, which encodes an
1263
+ * assumption that a PENDING row is healthy in-progress work. It is not: a
1264
+ * pending row whose `dependsOn` chain terminates in a TERMINAL non-completed
1265
+ * status (`failed`/`skipped`) can never be dispatched by pickForProject, and
1266
+ * never will be, but it still counts toward `pending` and so reads as a
1267
+ * healthy queue to every monitor in the app.
1268
+ *
1269
+ * On 2026-09-05 starry-night-ships held 42 such rows behind one `failed`
1270
+ * job for three hours. Machine-wide `stalled` was false (42 pending),
1271
+ * per-project `stalled` was false (42 pending), the queue-health sweep
1272
+ * doesn't count `failed` at all, and the supervisor only probes `running` —
1273
+ * so nothing anywhere reported a problem while nothing could ever run.
1274
+ *
1275
+ * Reported per project as { blockedBy: [terminal slugs], blocked: count }.
1276
+ * Transitive by construction: a row blocked by a row that is itself blocked
1277
+ * resolves through the same walk.
1278
+ */
1279
+ function computeBlockedChains(jobs) {
1280
+ const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
1281
+ const byCwd = new Map();
1282
+ for (const j of rows) {
1283
+ const key = j.cwd || '(unknown)';
1284
+ if (!byCwd.has(key)) byCwd.set(key, []);
1285
+ byCwd.get(key).push(j);
1286
+ }
1287
+
1288
+ const out = [];
1289
+ for (const [cwd, projectJobs] of byCwd) {
1290
+ // Reuse the picker's OWN dep resolution so this can never disagree with
1291
+ // what the scheduler will actually dispatch (bare-name fallback included).
1292
+ const rowBySlug = new Map(projectJobs.map((j) => [j.slug, j]));
1293
+ const rowsByBareSlug = new Map();
1294
+ for (const j of projectJobs) {
1295
+ const bare = String(j.slug ?? '').replace(/^\d+-/, '');
1296
+ if (!rowsByBareSlug.has(bare)) rowsByBareSlug.set(bare, []);
1297
+ rowsByBareSlug.get(bare).push(j);
1298
+ }
1299
+ const rowsForDep = (slug) => {
1300
+ const exact = rowBySlug.get(slug);
1301
+ if (exact) return [exact];
1302
+ return rowsByBareSlug.get(String(slug ?? '').replace(/^\d+-/, '')) ?? [];
1303
+ };
1304
+
1305
+ // Memoised walk: does this row's dep closure hit a terminally-stuck row?
1306
+ const TERMINAL_STUCK = new Set(['failed', 'skipped']);
1307
+ const verdicts = new Map(); // slug -> Set of terminal blocker slugs
1308
+ const visiting = new Set();
1309
+ const blockersFor = (job) => {
1310
+ if (!job) return new Set();
1311
+ if (verdicts.has(job.slug)) return verdicts.get(job.slug);
1312
+ if (visiting.has(job.slug)) return new Set(); // dependsOn cycle — not our problem here
1313
+ visiting.add(job.slug);
1314
+ const found = new Set();
1315
+ for (const depSlug of job.dependsOn ?? []) {
1316
+ for (const dep of rowsForDep(depSlug)) {
1317
+ if (TERMINAL_STUCK.has(dep.status)) found.add(dep.slug);
1318
+ else if (dep.status !== 'completed') for (const b of blockersFor(dep)) found.add(b);
1319
+ }
1320
+ }
1321
+ visiting.delete(job.slug);
1322
+ verdicts.set(job.slug, found);
1323
+ return found;
1324
+ };
1325
+
1326
+ const blockedBy = new Set();
1327
+ let blocked = 0;
1328
+ for (const j of projectJobs) {
1329
+ if (j.status !== 'pending') continue;
1330
+ const bs = blockersFor(j);
1331
+ if (bs.size === 0) continue;
1332
+ blocked += 1;
1333
+ for (const b of bs) blockedBy.add(b);
1334
+ }
1335
+ if (blocked > 0) out.push({ cwd, blockedBy: [...blockedBy].sort(), blocked });
1336
+ }
1337
+ return out;
1338
+ }
1339
+
1255
1340
  /**
1256
1341
  * findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
1257
1342
  *
@@ -2119,6 +2204,9 @@ let firstFailureAt = null;
2119
2204
  let firstNon429FailureAt = null; // tracks only transient/config failures; 429s don't count toward network-pause threshold
2120
2205
  let lastFailureKind = null; // 'transient' | 'meter_rate_limited' | 'auth' | null
2121
2206
  let pauseClearedManuallyAt = null;
2207
+ // PRD 1119: consecutive-rapid-rate-limit hard-pause tracking, keyed per slug.
2208
+ // See isCooldownSuppressed/nextRapidRateLimitCount below for the pure rules.
2209
+ const consecutiveRapidRateLimitsBySlug = new Map();
2122
2210
 
2123
2211
  // ---------- timer ----------
2124
2212
 
@@ -2337,13 +2425,71 @@ async function rescheduleTimer() {
2337
2425
 
2338
2426
  // ---------- pause / resume ----------
2339
2427
 
2340
- async function setPaused(reason, resumeAtIso) {
2428
+ const MANUAL_PAUSE_COOLDOWN_MS = 300_000;
2429
+ // PRD 1119: after this many consecutive rate-limited dispatches of the SAME
2430
+ // slug that EACH also finished in under RAPID_RATE_LIMIT_WINDOW_MS, the rate
2431
+ // limit is not a stale/flaky auto-detection any more — it's real and
2432
+ // persistent for this job. Engage a hard pause the manual-clear cooldown
2433
+ // cannot suppress at all. This exists because the freshness check alone
2434
+ // (isCooldownSuppressed) is not sufficient: if the computed resumeAt is
2435
+ // itself wrong or stale (e.g. a failed usage-API fetch), the resume timer
2436
+ // can keep re-clearing the pause every ~30s, and every SUBSEQUENT dispatch
2437
+ // is genuinely "fresh" (it started after that re-clear) — so freshness alone
2438
+ // would let the spin continue indefinitely within the same 5-minute cooldown
2439
+ // window. The rapid-repeat count is an independent circuit breaker of last
2440
+ // resort for exactly that case.
2441
+ const CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD = 3;
2442
+ const RAPID_RATE_LIMIT_WINDOW_MS = 30_000;
2443
+
2444
+ /**
2445
+ * Pure: should setPaused()'s manual-override cooldown suppress WRITING this
2446
+ * pause? `force` (the rapid-repeat hard pause) always answers no — that path
2447
+ * exists precisely to bypass the cooldown. Otherwise, suppress only while
2448
+ * inside the cooldown window AND the triggering observation is stale, i.e.
2449
+ * it was NOT produced by a run that started after the human's manual clear.
2450
+ * A run that started after the clear is fresh evidence the human's fix (if
2451
+ * any) did not hold, and must be allowed to re-engage the pause regardless
2452
+ * of the cooldown — the cooldown's job is to ignore STALE auto-detections,
2453
+ * never to ignore new evidence.
2454
+ */
2455
+ function isCooldownSuppressed({ pauseClearedManuallyAt: clearedAt, now, observedAt, force }) {
2456
+ if (force) return false;
2457
+ if (!clearedAt) return false;
2458
+ if (now - clearedAt >= MANUAL_PAUSE_COOLDOWN_MS) return false;
2459
+ const isFresh = typeof observedAt === 'number' && observedAt > clearedAt;
2460
+ return !isFresh;
2461
+ }
2462
+
2463
+ /**
2464
+ * Pure: the next consecutive-rapid-rate-limit count for a slug, given its
2465
+ * previous count and this run's outcome. Increments only on a rate-limited
2466
+ * run that ALSO ran under RAPID_RATE_LIMIT_WINDOW_MS (a genuine "dispatch,
2467
+ * 429, die" cycle — not a job that ran for a while before hitting the
2468
+ * limit). Resets to 0 on any non-rate-limited outcome. A rate-limited-but-
2469
+ * slow run leaves the count unchanged: still a rate limit, just not the
2470
+ * rapid-spin shape this cap exists to catch.
2471
+ */
2472
+ function nextRapidRateLimitCount(prevCount, { rateLimited, durationMs }) {
2473
+ if (!rateLimited) return 0;
2474
+ if (durationMs < RAPID_RATE_LIMIT_WINDOW_MS) return (prevCount || 0) + 1;
2475
+ return prevCount || 0;
2476
+ }
2477
+
2478
+ async function setPaused(reason, resumeAtIso, opts = {}) {
2479
+ const { observedAt = null, force = false } = opts;
2341
2480
  // Honor manual-override cooldown: if the user cleared a pause within the
2342
- // last 5 minutes, suppress auto-pause re-engagement on the same condition.
2343
- if (pauseClearedManuallyAt && Date.now() - pauseClearedManuallyAt < 300_000) {
2481
+ // last 5 minutes, suppress auto-pause re-engagement UNLESS this pause is
2482
+ // backed by a fresh observation (a run that started after the clear) or is
2483
+ // forced (the rapid-repeat hard pause, which the cooldown cannot suppress).
2484
+ if (isCooldownSuppressed({ pauseClearedManuallyAt, now: Date.now(), observedAt, force })) {
2344
2485
  console.log(`[scheduler] setPaused(${reason}) suppressed by manual override cooldown`);
2345
2486
  return;
2346
2487
  }
2488
+ if (force) {
2489
+ console.log(`[scheduler] setPaused(${reason}) forced past manual override cooldown — rapid-repeat rate-limit cap engaged`);
2490
+ } else if (pauseClearedManuallyAt && Date.now() - pauseClearedManuallyAt < MANUAL_PAUSE_COOLDOWN_MS) {
2491
+ console.log(`[scheduler] setPaused(${reason}) engaging despite manual override cooldown — triggering run started after the manual clear`);
2492
+ }
2347
2493
 
2348
2494
  // For 'network' with no explicit resumeAt, auto-resume after 30 minutes.
2349
2495
  let effectiveResumeAt = resumeAtIso;
@@ -2862,21 +3008,6 @@ async function notifyNeedsReview(job, report, {
2862
3008
  }
2863
3009
  }
2864
3010
 
2865
- /** Scan the tail of a job's log for the canonical rate-limit signal. We look
2866
- * at the last 16 KB — final result event always lands at the end.
2867
- * Uses readTail() so no raw fd lifecycle is needed here. */
2868
- function detectRateLimitInLog(logPath) {
2869
- try {
2870
- const text = readTail(logPath, 16384);
2871
- if (!text) return false;
2872
- return /"rateLimitType":"five_hour"/.test(text)
2873
- || /"api_error_status":429/.test(text)
2874
- || /You'?ve hit your limit/.test(text);
2875
- } catch {
2876
- return false;
2877
- }
2878
- }
2879
-
2880
3011
  /** Scan the tail of a job's log for a network-outage signal: the structured
2881
3012
  * `terminal_reason":"api_error"` field alongside a network-class error
2882
3013
  * string. This is NOT a real code defect — spawning an auto-fix
@@ -3197,10 +3328,17 @@ function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }
3197
3328
 
3198
3329
  // ---------- execution ----------
3199
3330
 
3331
+ // Allocates the runId/dir pair at dispatch time WITHOUT creating the
3332
+ // directory — a dispatch that aborts inside spawnJob before executeJob's
3333
+ // openLog() call (slot-acquire miss, worktree-cap deferral, launch-gate
3334
+ // block, ...) must leave no trace on disk. The directory is materialised
3335
+ // lazily, the first time something actually needs to write into it (see
3336
+ // openLog's mkdirSync in executeJob below). Because tickQueue hands ONE
3337
+ // shared batch dir to every spawnJob in the batch, several jobs may race to
3338
+ // create it — `recursive: true` makes that race safe.
3200
3339
  function pickRunDir() {
3201
3340
  const ts = new Date().toISOString().replace(/[:.]/g, '-');
3202
3341
  const dir = path.join(RUNS_DIR, ts);
3203
- fs.mkdirSync(dir, { recursive: true });
3204
3342
  return { runId: ts, dir };
3205
3343
  }
3206
3344
 
@@ -3228,6 +3366,12 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
3228
3366
  // fresh one via `--session-id` is the entire point of the recovery.
3229
3367
  const sessionId = resumeTarget ? resumeTarget.sessionId : randomUUID();
3230
3368
 
3369
+ // Materialise the (possibly shared-batch) run dir lazily, right before the
3370
+ // first write into it — see pickRunDir's comment for why this is deferred
3371
+ // this far. recursive:true makes it safe if a sibling job in the same
3372
+ // batch dir already created it.
3373
+ fs.mkdirSync(runDir, { recursive: true });
3374
+
3231
3375
  // Phase 1: open log fd so we can emit pre-spawn diagnostics (early-exit
3232
3376
  // error paths) before the child is created. withChildAndLog takes ownership
3233
3377
  // of fd/safeLog/closeFd from the point it is called.
@@ -4317,6 +4461,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4317
4461
  console.log(`[scheduler] ${job.slug}: dispatching as launch probe for '${launchKey}'${launchEnv ? ` with mitigation ${JSON.stringify(launchEnv)}` : ''}`);
4318
4462
  }
4319
4463
 
4464
+ // Captured here (not read back off `job`, a pre-dispatch snapshot that
4465
+ // mutate()'s fresh-from-disk read never touches) so the rate-limited
4466
+ // branch below has this run's OWN start time — the freshness check
4467
+ // (isCooldownSuppressed) needs to know whether this specific dispatch
4468
+ // started after the manual clear, not whatever startedAt this row
4469
+ // carried from a prior run.
4470
+ let dispatchStartedAtMs = null;
4320
4471
  await mutate((s) => {
4321
4472
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4322
4473
  if (idx >= 0) {
@@ -4327,6 +4478,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4327
4478
  delete s.jobs[idx].heldReason;
4328
4479
  s.jobs[idx].runId = runId;
4329
4480
  s.jobs[idx].startedAt = new Date().toISOString();
4481
+ dispatchStartedAtMs = Date.parse(s.jobs[idx].startedAt);
4330
4482
  if (job.quietMachine === true) {
4331
4483
  s.jobs[idx].quietMachine = true;
4332
4484
  s.jobs[idx].quietLeaseDegraded = job.quietLeaseDegraded === true;
@@ -4531,12 +4683,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4531
4683
  // (non-git cwd / git errored) — NEVER treated as "left nothing", exactly
4532
4684
  // like every other best-effort git-state check in this function.
4533
4685
  const afterGuardCwd = await uncommittedChanges(guardCwd);
4686
+ // stripAppOwnedChurn: the app writes session-manager-operations/ (queue.json,
4687
+ // history.jsonl, active-index.json, transcripts) DURING this job's own guard
4688
+ // window, so those land in the delta and get blamed on the job. A job can
4689
+ // never be responsible for them — see jobDirtFilter.cjs. Applied here, at
4690
+ // the single place the delta is computed, so the commit guard, the
4691
+ // transient-retry dirty check and leftoverPaths all agree.
4534
4692
  const newlyDirtyAll = afterGuardCwd === null
4535
4693
  ? null
4536
- : [...new Set([
4694
+ : stripAppOwnedChurn([...new Set([
4537
4695
  ...afterGuardCwd.filter((p) => !new Set(guardBaseline || []).has(p)),
4538
4696
  ...worktreeLeftoverDirty,
4539
- ])];
4697
+ ])]);
4540
4698
 
4541
4699
  if (res.launchFailure) {
4542
4700
  await handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion: claudeVersionNow });
@@ -4570,7 +4728,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4570
4728
 
4571
4729
  if (res.rateLimited) {
4572
4730
  const resetIso = await refreshNextReset().catch(() => cachedNextReset);
4573
- await setPaused('rate_limit', resetIso);
4731
+ const observedAt = dispatchStartedAtMs;
4732
+ const prevCount = consecutiveRapidRateLimitsBySlug.get(job.slug) || 0;
4733
+ const nextCount = nextRapidRateLimitCount(prevCount, { rateLimited: true, durationMs: res.durationMs });
4734
+ consecutiveRapidRateLimitsBySlug.set(job.slug, nextCount);
4735
+ const forceHardPause = nextCount >= CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD;
4736
+ if (forceHardPause) {
4737
+ console.log(`[scheduler] ${job.slug}: ${nextCount} consecutive rate-limited dispatches under ${RAPID_RATE_LIMIT_WINDOW_MS / 1000}s each — engaging hard pause`);
4738
+ }
4739
+ await setPaused('rate_limit', resetIso, { observedAt, force: forceHardPause });
4740
+ } else {
4741
+ consecutiveRapidRateLimitsBySlug.delete(job.slug);
4574
4742
  }
4575
4743
 
4576
4744
  // Stale queue entry: the PRD was archived (already shipped) or is gone
@@ -5366,6 +5534,109 @@ async function maybeLaunchWhenAvailable(state) {
5366
5534
  tickQueue().catch((e) => console.error('[scheduler] tickQueue error', e));
5367
5535
  }
5368
5536
 
5537
+ /** How long a queue may hold ready work with nothing running before the
5538
+ * starvation watchdog forces a tick. Deliberately longer than the poll
5539
+ * loop's own cadence + backoff, so this only ever fires when the normal
5540
+ * path has genuinely stopped driving the queue — it is a safety net, not a
5541
+ * second scheduler. */
5542
+ const QUEUE_STARVATION_MS = 10 * 60_000;
5543
+
5544
+ /**
5545
+ * classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs })
5546
+ * → null | { kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }
5547
+ *
5548
+ * Pure, no IO. Answers the one question the user's invariant reduces to:
5549
+ * "there are PRDs in a queue — is anything actually going to run them?"
5550
+ *
5551
+ * Every stall this codebase has seen was a DIFFERENT cause with the SAME
5552
+ * shape: ready rows, nothing running, nobody ticking. A rate-limited exit
5553
+ * stamped terminal `failed` (2026-09-05, 42 rows); a spin loop past the
5554
+ * manual-clear cooldown; a worktree merge-back that left the project on a
5555
+ * job branch; a job parked `needs_review` with no fix plan; app churn
5556
+ * counted as unfinished work. Guarding each cause individually will always
5557
+ * lag the next one, so this guards the SHAPE instead.
5558
+ *
5559
+ * Two outcomes, deliberately distinguished — they need opposite responses:
5560
+ * 'starved' — at least one pending row is dispatchable RIGHT NOW and
5561
+ * nothing is running. Whatever should have ticked, didn't.
5562
+ * Forcing a tick is safe and fixes it.
5563
+ * 'blocked' — every pending row is behind a terminal/parked dependency.
5564
+ * A tick cannot help; this needs a human (or a heal pass) to
5565
+ * resolve the blocker, and must be reported as such rather
5566
+ * than silently re-ticking forever.
5567
+ *
5568
+ * Returns null when the queue is healthy (work running, nothing pending,
5569
+ * paused on purpose, or simply not idle long enough yet).
5570
+ */
5571
+ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
5572
+ if (paused) return null; // paused is a DECISION, not a stall
5573
+ if (runningCount > 0) return null; // work is flowing
5574
+ const rows = Array.isArray(jobs) ? jobs : [];
5575
+ const pending = rows.filter((j) => j && j.status === 'pending');
5576
+ if (pending.length === 0) return null; // nothing to run — not a stall
5577
+
5578
+ const idleMs = Number.isFinite(lastRunAtMs) ? now - lastRunAtMs : Infinity;
5579
+ if (idleMs < thresholdMs) return null; // give the normal path its chance first
5580
+
5581
+ // Which pending rows could actually dispatch? Anything NOT named by a
5582
+ // blocked chain. computeBlockedChains already walks dependsOn with the
5583
+ // picker's own resolution, so the two can never disagree.
5584
+ const blockedChains = computeBlockedChains(rows);
5585
+ const blockedTotal = blockedChains.reduce((n, c) => n + c.blocked, 0);
5586
+ const dispatchable = pending.length - blockedTotal;
5587
+
5588
+ return {
5589
+ kind: dispatchable > 0 ? 'starved' : 'blocked',
5590
+ pending: pending.length,
5591
+ dispatchable,
5592
+ blockedChains,
5593
+ idleMs,
5594
+ };
5595
+ }
5596
+
5597
+ /**
5598
+ * The watchdog half: acts on classifyQueueStarvation. Called from the
5599
+ * heartbeat, which already runs on its own timer independent of the billing
5600
+ * poll loop — so a wedged or never-succeeding poll (the /api/oauth/usage
5601
+ * endpoint was itself 429ing all of 2026-09-05) can no longer leave a queue
5602
+ * with ready work idle indefinitely.
5603
+ */
5604
+ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
5605
+ const verdict = classifyQueueStarvation({
5606
+ jobs: state?.jobs,
5607
+ paused: state?.paused,
5608
+ runningCount: runningSet.size,
5609
+ lastRunAtMs: Date.parse(state?.lastRunAt ?? ''),
5610
+ now,
5611
+ thresholdMs,
5612
+ });
5613
+ if (!verdict) return null;
5614
+
5615
+ const mins = Math.round(verdict.idleMs / 60_000);
5616
+ if (verdict.kind === 'blocked') {
5617
+ console.warn(
5618
+ `[scheduler] QUEUE BLOCKED: ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
5619
+ + `terminal or parked dependency, so ticking cannot help. Blockers: `
5620
+ + verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
5621
+ );
5622
+ appendAuditEvent('queue_blocked_stall', { pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
5623
+ return verdict;
5624
+ }
5625
+
5626
+ console.warn(
5627
+ `[scheduler] QUEUE STARVED: ${verdict.dispatchable} dispatchable job(s) of ${verdict.pending} pending, 0 running, `
5628
+ + `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
5629
+ );
5630
+ appendAuditEvent('queue_starvation_forced_tick', { pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
5631
+ // A never-populated utilization reading is itself one of the ways the
5632
+ // when-available path silently never fires (maybeLaunchWhenAvailable
5633
+ // returns early on null). Treat unknown as safe here, exactly as the
5634
+ // billing meter's own 429 fallback already does.
5635
+ if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
5636
+ await tickQueue({ bypassLoadGate: false }).catch((e) => console.error('[scheduler] starvation tick error', e));
5637
+ return verdict;
5638
+ }
5639
+
5369
5640
  // ---------- dead-process reaper ----------
5370
5641
 
5371
5642
  // Queue-health sweep cadence: hangs off reapDeadRunningJobs's own cycle
@@ -5440,11 +5711,41 @@ async function reapDeadRunningJobs() {
5440
5711
 
5441
5712
  if (dead.length === 0) return;
5442
5713
 
5714
+ // A rate-limited death is retryable, not terminal — mirror spawnJob's own
5715
+ // live-process handling (PRD 1117) exactly: engage the SAME setPaused
5716
+ // pause here too. Skipping this would reset the row to 'pending' but
5717
+ // leave dispatch unpaused, so the next tick immediately re-fires it into
5718
+ // the same still-active rate limit — the spin loop this PRD exists to
5719
+ // stop. Done once, outside mutate(), before finalizing any row below.
5720
+ if (dead.some((d) => d.outcome === 'rate_limited')) {
5721
+ const resetIso = await refreshNextReset().catch(() => cachedNextReset);
5722
+ const triggering = dead.find((d) => d.outcome === 'rate_limited');
5723
+ const triggeringRow = triggering ? state.jobs.find((x) => x.slug === triggering.slug) : null;
5724
+ const observedAtMs = triggeringRow?.startedAt ? Date.parse(triggeringRow.startedAt) : null;
5725
+ // Same rapid-repeat circuit breaker spawnJob's own res.rateLimited
5726
+ // branch drives (see consecutiveRapidRateLimitsBySlug above) — a
5727
+ // process that gets rate-limited and then dies without spawnJob's own
5728
+ // branch ever running is reconciled HERE instead, and must feed the
5729
+ // same counter or a stale/wrong resumeAt could keep re-clearing this
5730
+ // path's "fresh" pause every reap cycle with no hard cap ever engaging.
5731
+ const durationMs = Number.isFinite(observedAtMs) ? Date.now() - observedAtMs : Infinity;
5732
+ const prevCount = consecutiveRapidRateLimitsBySlug.get(triggering.slug) || 0;
5733
+ const nextCount = nextRapidRateLimitCount(prevCount, { rateLimited: true, durationMs });
5734
+ consecutiveRapidRateLimitsBySlug.set(triggering.slug, nextCount);
5735
+ const forceHardPause = nextCount >= CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD;
5736
+ if (forceHardPause) {
5737
+ console.log(`[scheduler] ${triggering.slug}: ${nextCount} consecutive rate-limited dispatches under ${RAPID_RATE_LIMIT_WINDOW_MS / 1000}s each (reaped) — engaging hard pause`);
5738
+ }
5739
+ await setPaused('rate_limit', resetIso, { observedAt: observedAtMs, force: forceHardPause });
5740
+ }
5741
+
5443
5742
  await mutate(async (s) => {
5444
5743
  for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
5445
5744
  const idx = s.jobs.findIndex((x) => x.slug === slug);
5446
5745
  if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
5746
+ const rateLimited = outcome === 'rate_limited';
5447
5747
  const success = outcome === 'success';
5748
+ if (!rateLimited) consecutiveRapidRateLimitsBySlug.delete(slug);
5448
5749
 
5449
5750
  // Best-effort in-place leftover computation: a job whose owning
5450
5751
  // process vanished without spawnJob()'s own finally block ever
@@ -5479,13 +5780,23 @@ async function reapDeadRunningJobs() {
5479
5780
  const leftoverSuffix = deltaPaths && deltaPaths.length
5480
5781
  ? ` — left ${deltaPaths.length} files uncommitted`
5481
5782
  : '';
5482
- const transitionReason = (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
5483
-
5484
- transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
5485
- s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
5486
- s.jobs[idx].finishedAt = new Date().toISOString();
5487
- s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
5488
- s.jobs[idx].gateOutcome = gateOutcome;
5783
+ const transitionReason = rateLimited
5784
+ ? `reaped: rate limit detected — reset to pending, not failed (outcome=${outcome})${leftoverSuffix}`
5785
+ : (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
5786
+
5787
+ if (rateLimited) {
5788
+ // Retryable, never terminal (PRD 1117) — same resetJobFields path
5789
+ // spawnJob's own rateLimited branch uses (see ~4797's
5790
+ // treatAsPending), so the row comes back exactly like any other
5791
+ // paused-for-rate-limit reset: fresh runId/startedAt/exitCode.
5792
+ resetJobFields(s.jobs[idx], transitionReason, { source: 'reapDeadRunningJobs:rate-limit' });
5793
+ } else {
5794
+ transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
5795
+ s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
5796
+ s.jobs[idx].finishedAt = new Date().toISOString();
5797
+ s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
5798
+ s.jobs[idx].gateOutcome = gateOutcome;
5799
+ }
5489
5800
  delete s.jobs[idx].runtime;
5490
5801
  delete s.jobs[idx].guardBaseline;
5491
5802
  delete s.jobs[idx].guardHeadBefore;
@@ -5499,7 +5810,10 @@ async function reapDeadRunningJobs() {
5499
5810
  // with no exit event) wedges the lease held forever and stalls
5500
5811
  // dispatch for every project until the app restarts.
5501
5812
  if (s.jobs[idx].quietMachine === true) quietMachineLease.release(slug);
5502
- if (pidless) {
5813
+ if (rateLimited) {
5814
+ console.log(`[scheduler] reaped rate-limited job slug=${slug} — reset to pending, pause engaged`);
5815
+ appendAuditEvent('job_reaped_rate_limited', { slug, cwd: s.jobs[idx].cwd ?? null });
5816
+ } else if (pidless) {
5503
5817
  console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
5504
5818
  appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
5505
5819
  } else {
@@ -6865,6 +7179,14 @@ async function init() {
6865
7179
  if (heartbeatInterval) clearInterval(heartbeatInterval);
6866
7180
  heartbeatInterval = setInterval(() => {
6867
7181
  const s = readQueueSync();
7182
+ // NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
7183
+ // running, something must drive it. This is the only driver that does
7184
+ // not depend on the billing poll loop, a pause timer, or a completing
7185
+ // job to schedule the next tick — every one of which has failed at
7186
+ // least once. See classifyQueueStarvation.
7187
+ if (!s.unreadable) {
7188
+ runQueueStarvationWatchdog(s).catch((e) => console.error('[scheduler] starvation watchdog error', e));
7189
+ }
6868
7190
  // Initialise from the real status union (scheduleJobSchema.cjs) rather
6869
7191
  // than a hand-maintained subset — the old `{ pending, running, completed,
6870
7192
  // failed }` literal silently minted a NEW key for any other value
@@ -7465,4 +7787,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
7465
7787
  });
7466
7788
  }
7467
7789
 
7468
- module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure };
7790
+ module.exports = { classifyQueueStarvation, runQueueStarvationWatchdog, QUEUE_STARVATION_MS, computeBlockedChains, stripAppOwnedChurn, findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure, setPaused, clearPause, tickQueue, runDueJobs, isCooldownSuppressed, nextRapidRateLimitCount, CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD, RAPID_RATE_LIMIT_WINDOW_MS, MANUAL_PAUSE_COOLDOWN_MS, RUNS_DIR, pickRunDir };
@@ -354,20 +354,23 @@ export interface DelegationReadinessCheck {
354
354
  | 'scheduler-mcp-project-duplicate'
355
355
  | 'dev-plugin'
356
356
  | 'agent-personas'
357
- | 'prd-write-guard';
357
+ | 'prd-write-guard'
358
+ | 'destructive-git-guard';
358
359
  label: string;
359
360
  ok: boolean;
360
361
  detail: string;
361
362
  fix: string | null;
362
363
  /** Non-null when Session Manager can install this fix itself, one press. */
363
- fixAction: 'install-prd-write-guard' | null;
364
+ fixAction: 'install-prd-write-guard' | 'install-destructive-git-guard' | null;
364
365
  /** True when ok:true is a WARNING (still passing, but worth a human's attention) — today only scheduler-mcp-project-duplicate. */
365
366
  warn?: boolean;
366
367
  /** True when this check didn't run because a precondition (another check) already failed — reported ok:true, not a failure. */
367
368
  skipped?: boolean;
368
369
  }
369
370
 
370
- export interface InstallPrdWriteGuardResult {
371
+ /** Result of installPrdWriteGuard and installDestructiveGitGuard alike
372
+ * (delegationReadiness.cjs) — the two installers share one contract. */
373
+ export interface InstallGuardResult {
371
374
  ok: boolean;
372
375
  action: 'installed' | 'repaired' | 'already-installed' | 'error';
373
376
  settingsPath?: string;
@@ -1472,7 +1475,8 @@ export interface SessionManagerAPI {
1472
1475
  /** "Can this project actually delegate?" — the 4 preconditions for
1473
1476
  * scheduler_create_prd being in an agent's tool list at all. */
1474
1477
  delegationReadiness: (cwd: string) => Promise<DelegationReadiness>;
1475
- installPrdWriteGuard: (cwd: string) => Promise<InstallPrdWriteGuardResult>;
1478
+ installPrdWriteGuard: (cwd: string) => Promise<InstallGuardResult>;
1479
+ installDestructiveGitGuard: (cwd: string) => Promise<InstallGuardResult>;
1476
1480
  onNewSession: (handler: () => void) => () => void;
1477
1481
  onRebootSession: (handler: () => void) => () => void;
1478
1482
  archiveProject: (encoded: string) => Promise<{ ok: boolean; error?: string }>;
@@ -24,6 +24,7 @@ contextBridge.exposeInMainWorld('api', {
24
24
  seedStatus: () => ipcRenderer.invoke('app:seed-status'),
25
25
  delegationReadiness: (cwd) => ipcRenderer.invoke('app:delegation-readiness', { cwd }),
26
26
  installPrdWriteGuard: (cwd) => ipcRenderer.invoke('app:install-prd-write-guard', { cwd }),
27
+ installDestructiveGitGuard: (cwd) => ipcRenderer.invoke('app:install-destructive-git-guard', { cwd }),
27
28
  onNewSession: (handler) => {
28
29
  const listener = () => handler();
29
30
  ipcRenderer.on('app:new-session', listener);