claude-code-session-manager 0.85.0 → 0.86.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/dist/assets/{AgentLibrary-Bkv-HcP1.js → AgentLibrary-DTFL7y8G.js} +1 -1
  2. package/dist/assets/{DataModel-DRH-Ty20.js → DataModel-Q4jhl24R.js} +1 -1
  3. package/dist/assets/{History-CfRhT1Im.js → History-Cj2FejEo.js} +1 -1
  4. package/dist/assets/{Hooks-wEmh_U6c.js → Hooks-CaelQI6t.js} +1 -1
  5. package/dist/assets/{HostBilko-D_t7Rbi7.js → HostBilko--v7cMR8I.js} +1 -1
  6. package/dist/assets/{Library-CpArQ-OJ.js → Library-DgI9oCCZ.js} +1 -1
  7. package/dist/assets/{ListDetail-pjaKYs84.js → ListDetail-DYUZN-x-.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-Xc141kjj.js → MarkdownEditor-DCIubYWf.js} +1 -1
  9. package/dist/assets/{McpServers-ftqaV3kn.js → McpServers-ypCYURh3.js} +1 -1
  10. package/dist/assets/{Memory-ChMWkNd0.js → Memory-C2qYp-3M.js} +1 -1
  11. package/dist/assets/{Panel-D9Kr40Ai.js → Panel-Cj2kw-Zv.js} +1 -1
  12. package/dist/assets/{Permissions-DKoNVgzj.js → Permissions-BiYZNGYW.js} +1 -1
  13. package/dist/assets/{Plugins-BtChISho.js → Plugins-C1Vj8_dU.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-DBA5EcYy.js → ProvenanceBadge-DczPNM5U.js} +1 -1
  15. package/dist/assets/{SaveBar-I0_dWNTX.js → SaveBar-Cd_7U6Gb.js} +1 -1
  16. package/dist/assets/{Scheduler-CbES7MC8.js → Scheduler-DcLBiJBq.js} +1 -1
  17. package/dist/assets/{ScopeSwitcher-5GTEveb2.js → ScopeSwitcher-DVSyI44-.js} +1 -1
  18. package/dist/assets/{Settings-BX3FElXk.js → Settings-Cv-pRyms.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-DNBFGrYE.js → SkillReferenceGraph-CUv1_Q2c.js} +1 -1
  20. package/dist/assets/{Skills-DJB6-bBM.js → Skills-C_YHkAy-.js} +1 -1
  21. package/dist/assets/{SystemPrompt-BiDDrJUA.js → SystemPrompt-B8R7T9xn.js} +1 -1
  22. package/dist/assets/{TagLibrary-_Wrevtop.js → TagLibrary-dj9YHWyy.js} +1 -1
  23. package/dist/assets/{TiptapBody-OWWXdLRy.js → TiptapBody-DnSBUjHE.js} +1 -1
  24. package/dist/assets/{Toggle-B122N0HL.js → Toggle-CjV_BJn6.js} +1 -1
  25. package/dist/assets/{index-DhvuQL4C.js → index-CXFQIPhO.js} +3 -3
  26. package/dist/assets/{settingsSchema-sGoCTd7J.js → settingsSchema-BJVciriw.js} +1 -1
  27. package/dist/index.html +1 -1
  28. package/package.json +3 -2
  29. package/src/main/__tests__/health-starve-escalation.test.cjs +94 -0
  30. package/src/main/__tests__/loadGateDetailTick.test.cjs +31 -0
  31. package/src/main/__tests__/machineProfile.test.cjs +19 -1
  32. package/src/main/__tests__/pty-session-open-telemetry.test.cjs +96 -0
  33. package/src/main/__tests__/queue-starvation-per-project.test.cjs +135 -0
  34. package/src/main/__tests__/scheduler-failed-autoreset.test.cjs +121 -0
  35. package/src/main/__tests__/scheduler-needs-review-autoresolve.test.cjs +189 -0
  36. package/src/main/__tests__/scheduler-no-dead-end-status.test.cjs +152 -0
  37. package/src/main/__tests__/scheduler-quarantine-autoresolve.test.cjs +165 -0
  38. package/src/main/__tests__/scheduler-starve-escalation.test.cjs +144 -0
  39. package/src/main/__tests__/telemetryClient.test.cjs +75 -5
  40. package/src/main/__tests__/telemetrySettings.test.cjs +32 -0
  41. package/src/main/health.cjs +75 -2
  42. package/src/main/lib/__tests__/loadGate.test.cjs +103 -2
  43. package/src/main/lib/__tests__/telemetryBoot.test.cjs +11 -0
  44. package/src/main/lib/loadGate.cjs +23 -1
  45. package/src/main/lib/machineProfile.cjs +15 -0
  46. package/src/main/lib/schedulerBatch.cjs +12 -1
  47. package/src/main/lib/schedulerConfig.cjs +13 -0
  48. package/src/main/lib/telemetryBoot.cjs +11 -8
  49. package/src/main/lib/telemetryClient.cjs +44 -2
  50. package/src/main/lib/telemetrySettings.cjs +20 -3
  51. package/src/main/pty.cjs +9 -0
  52. package/src/main/scheduler.cjs +600 -66
@@ -95,6 +95,7 @@ const {
95
95
  PIDLESS_SPAWN_GRACE_MS,
96
96
  INVESTIGATION_MAX_MS,
97
97
  STARVATION_ESCALATE_MS,
98
+ STARVE_ESCALATION_MS,
98
99
  } = require('./lib/schedulerConfig.cjs');
99
100
  const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
100
101
  ? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
@@ -1484,14 +1485,15 @@ function computeBlockedChains(jobs) {
1484
1485
  * findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
1485
1486
  *
1486
1487
  * Pure, no IO. A 'quarantined' row (no createdVia provenance) can otherwise
1487
- * sit forever with nothing looking at it — quarantine only ever clears via a
1488
- * human adopting or archiving it. This is the escalation half of that gate:
1489
- * any quarantined row whose recorded quarantine timestamp (statusHistory's
1490
- * `to === 'quarantined'` entry — stamped at creation, or backfilled from the
1491
- * PRD file's mtime by reconcile() for rows quarantined before that stamp
1492
- * existed) is older than `thresholdMs` is reported so the caller can
1493
- * warn-log and surface it distinctly. A row with no recoverable timestamp is
1494
- * skipped rather than guessed at.
1488
+ * sit forever with nothing looking at it — quarantine used to clear only via
1489
+ * a human adopting or archiving it; autoResolveQuarantine below now gives it
1490
+ * a bounded automatic exit too. This function stays the escalation/warn half
1491
+ * of that gate: any quarantined row whose recorded quarantine timestamp
1492
+ * (statusHistory's `to === 'quarantined'` entry — stamped at creation, or
1493
+ * backfilled from the PRD file's mtime by reconcile() for rows quarantined
1494
+ * before that stamp existed) is older than `thresholdMs` is reported so the
1495
+ * caller can warn-log and surface it distinctly. A row with no recoverable
1496
+ * timestamp is skipped rather than guessed at.
1495
1497
  */
1496
1498
  function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
1497
1499
  const stale = [];
@@ -1507,6 +1509,106 @@ function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
1507
1509
  return stale;
1508
1510
  }
1509
1511
 
1512
+ // Bounded automatic exit for a quarantined row (this PRD): up to
1513
+ // QUARANTINE_RESOLVE_CAP auto-resolve attempts, each gated on having sat
1514
+ // `quarantined` for QUARANTINE_ESCALATE_MS, before autoResolveQuarantine
1515
+ // below settles the row to 'skipped' rather than leaving it as a dead end
1516
+ // only a human `scheduler_reset_job`/adopt action could ever clear. A single
1517
+ // attempt is enough in practice — the outcome is terminal — but the counter
1518
+ // still guards against two overlapping ticks both trying to resolve the
1519
+ // same row.
1520
+ const QUARANTINE_RESOLVE_CAP = 1;
1521
+
1522
+ /**
1523
+ * Kill-switch gate for the quarantine auto-resolve pass below
1524
+ * (SM_QUARANTINE_AUTORESOLVE_DISABLE=1), same shape as
1525
+ * failedAutoResetDisabled/needsReviewAutoResolveDisabled.
1526
+ */
1527
+ function quarantineAutoResolveDisabled() {
1528
+ return process.env.SM_QUARANTINE_AUTORESOLVE_DISABLE === '1';
1529
+ }
1530
+
1531
+ /**
1532
+ * selectQuarantineAutoResolveTargets(jobs, now, thresholdMs) →
1533
+ * [{ slug, cwd, ageMs }]
1534
+ *
1535
+ * Pure selector — no IO. Same age computation as findStaleQuarantinedJobs
1536
+ * above, bounded additionally by quarantineResolveAttempts so a row already
1537
+ * auto-resolved (or mid-resolve on a race) is never re-selected. Deliberately
1538
+ * does NOT check createdVia here — that requires a disk read of the PRD
1539
+ * file, and doing it at selection time would let this pass act on a
1540
+ * snapshot that's gone stale by the time the mutate() pass actually runs.
1541
+ * autoResolveQuarantine below re-reads createdVia fresh, immediately before
1542
+ * transitioning, inside the same mutate() callback that applies this
1543
+ * selector's targets — see that function's own header for why.
1544
+ */
1545
+ function selectQuarantineAutoResolveTargets(jobs, now, thresholdMs) {
1546
+ const targets = [];
1547
+ for (const j of jobs ?? []) {
1548
+ if (j.status !== 'quarantined') continue;
1549
+ if ((j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue;
1550
+ const entry = (j.statusHistory || []).find((h) => h.to === 'quarantined');
1551
+ if (!entry) continue;
1552
+ const since = Date.parse(entry.at);
1553
+ if (Number.isNaN(since)) continue;
1554
+ const ageMs = now - since;
1555
+ if (ageMs < thresholdMs) continue;
1556
+ targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs });
1557
+ }
1558
+ return targets;
1559
+ }
1560
+
1561
+ /**
1562
+ * autoResolveQuarantine(job, ageMs) → Promise<'skipped'|null>
1563
+ *
1564
+ * Applies the bounded automatic exit to a single quarantined row (mutates in
1565
+ * place; calls transitionJob + appendAuditEvent) — extracted so it's
1566
+ * unit-testable without going through mutate()/queue.json IO, same shape as
1567
+ * applyNeedsReviewAutoResolve above.
1568
+ *
1569
+ * Re-validates status + the attempts cap itself (race guard, mirrors the
1570
+ * other auto-resolve loops in the 10-minute interval body), THEN re-reads the
1571
+ * PRD file's createdVia frontmatter fresh from disk before doing anything
1572
+ * else. That ordering is load-bearing: reconcile()'s adopt path (the only
1573
+ * OTHER route off 'quarantined') promotes a row to 'pending' the instant it
1574
+ * observes a createdVia stamp, on its own independent pass — if this
1575
+ * function trusted a snapshot taken before its own turn to run, it could
1576
+ * transition a row to 'skipped' the same tick reconcile() already adopted it
1577
+ * to 'pending', silently discarding a PRD a human just fixed. Checking here,
1578
+ * immediately before the transition, inside the caller's mutate() callback,
1579
+ * closes that window.
1580
+ *
1581
+ * A PRD file that cannot be found or parsed at all is treated as still
1582
+ * lacking provenance — there is no proof it has one, and stalling forever on
1583
+ * an unreadable file would defeat the point of a bounded exit (same
1584
+ * can't-prove-it/don't-guess-but-don't-stall posture as the rest of this
1585
+ * file's stale-row detectors).
1586
+ */
1587
+ async function autoResolveQuarantine(job, ageMs) {
1588
+ if (!job || job.status !== 'quarantined') return null;
1589
+ if ((job.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) return null;
1590
+
1591
+ let createdVia = null;
1592
+ try {
1593
+ const resolvedDir = await findPrdDir(job.slug);
1594
+ const prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
1595
+ const parsed = await parsePrd(prdPath);
1596
+ createdVia = parsed.createdVia ?? null;
1597
+ } catch { /* unreadable/gone — no provenance found, so it stays "lacking" */ }
1598
+ if (createdVia) return null; // reconcile()'s own adopt path owns this row now
1599
+
1600
+ const attempt = (job.quarantineResolveAttempts ?? 0) + 1;
1601
+ job.quarantineResolveAttempts = attempt;
1602
+ job.error = `quarantined without createdVia provenance past the ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h `
1603
+ + 'escalation window — auto-resolved to skipped';
1604
+ transitionJob(job, 'skipped', {
1605
+ reason: 'quarantined without createdVia provenance past escalation window',
1606
+ source: 'autoResolveQuarantine',
1607
+ });
1608
+ appendAuditEvent('quarantine_auto_resolved', { slug: job.slug, cwd: job.cwd ?? null, ageMs: ageMs ?? null, attempt });
1609
+ return 'skipped';
1610
+ }
1611
+
1510
1612
  /**
1511
1613
  * findOverrunningJobs(jobs, now, { factor, floorMs }) → [{ slug, cwd, estimateMinutes, ranMs, ratio }]
1512
1614
  *
@@ -2559,6 +2661,19 @@ function applyPauseCleared(wasPaused, token) {
2559
2661
  return token;
2560
2662
  }
2561
2663
 
2664
+ /**
2665
+ * Human-readable explanation for a `reason: 'load-deferred'` tick, surfaced
2666
+ * to the renderer via lastTick.detail. Names the gate, the measured ratio,
2667
+ * the threshold and how long the stretch has been held — the box could sit
2668
+ * gated for 80+ minutes with nothing in the UI naming why (PRD: load gate
2669
+ * hysteresis). Pure so it's unit-testable without driving tickQueue's full
2670
+ * fs/worktree machinery.
2671
+ */
2672
+ function formatLoadGateDetail(load) {
2673
+ const heldMinutes = Math.round(load.gatedSinceMs / 60_000);
2674
+ return `CPU load gate: loadavg1 ${load.loadavg1} / ${load.cores} cores = ${load.ratio} > threshold ${load.threshold}, held for ${heldMinutes}m`;
2675
+ }
2676
+
2562
2677
  function attachWindow(w) { mainWindow = w; }
2563
2678
 
2564
2679
  /**
@@ -6714,7 +6829,7 @@ function tickQueue({ bypassLoadGate = false } = {}) {
6714
6829
  }
6715
6830
  return recordTick(
6716
6831
  { fired: false, reason: 'load-deferred', deferredCount: gatedBatch.length, ratio: load.ratio, threshold: load.threshold },
6717
- { detail: `load gate: ${load.loadavg1} / ${load.cores} cores = ${load.ratio} > ${load.threshold}`, holds },
6832
+ { detail: formatLoadGateDetail(load), holds },
6718
6833
  );
6719
6834
  }
6720
6835
 
@@ -6860,43 +6975,107 @@ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now,
6860
6975
  }
6861
6976
 
6862
6977
  /**
6863
- * The watchdog half: acts on classifyQueueStarvation. Called from the
6864
- * heartbeat, which already runs on its own timer independent of the billing
6865
- * poll loop — so a wedged or never-succeeding poll (the /api/oauth/usage
6866
- * endpoint was itself 429ing all of 2026-09-05) can no longer leave a queue
6867
- * with ready work idle indefinitely.
6978
+ * classifyQueueStarvationByProject({ jobs, paused, runningSet, lastRunAtMs, now, thresholdMs })
6979
+ * → [{ cwd, kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }]
6980
+ *
6981
+ * Per-project driver around classifyQueueStarvation's pure single-project
6982
+ * core. `runningCount > 0` inside that core used to be fed the MACHINE-WIDE
6983
+ * `runningSet.size`, which meant one long-lived job in ANY project disarmed
6984
+ * the watchdog for EVERY other project on the box — observed live
6985
+ * 2026-09-12: a job in starry-night-ships ran 80+ minutes while two other
6986
+ * projects sat starved/blocked for hours, and the watchdog never fired once
6987
+ * because "work is flowing" was true somewhere else. Partitioning by cwd
6988
+ * (the same grouping computeBlockedChains already does) fixes DETECTION only
6989
+ * — the idle clock (`lastRunAtMs`) stays machine-wide, since
6990
+ * `lastDispatchAttemptAt` is machine-level state, and only one tick is ever
6991
+ * forced per watchdog pass regardless of how many cwds are starved.
6992
+ *
6993
+ * Pure, no IO. Returns [] when paused (a DECISION, not a stall) or when no
6994
+ * project has a verdict.
6995
+ */
6996
+ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
6997
+ if (paused) return [];
6998
+ const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
6999
+ const byCwd = new Map();
7000
+ for (const j of rows) {
7001
+ const key = j.cwd || '(unknown)';
7002
+ if (!byCwd.has(key)) byCwd.set(key, []);
7003
+ byCwd.get(key).push(j);
7004
+ }
7005
+
7006
+ const verdicts = [];
7007
+ for (const [cwd, projectJobs] of byCwd) {
7008
+ // Same source of truth tickQueue itself uses for "is anything running":
7009
+ // the in-process runningSet OR a row already stamped status:'running'.
7010
+ const projRunningCount = projectJobs.filter(
7011
+ (j) => j.status === 'running' || runningSlugs?.has?.(j.slug),
7012
+ ).length;
7013
+ const verdict = classifyQueueStarvation({
7014
+ jobs: projectJobs,
7015
+ paused: false,
7016
+ runningCount: projRunningCount,
7017
+ lastRunAtMs,
7018
+ now,
7019
+ thresholdMs,
7020
+ });
7021
+ if (verdict) verdicts.push({ cwd, ...verdict });
7022
+ }
7023
+ return verdicts;
7024
+ }
7025
+
7026
+ /**
7027
+ * The watchdog half: acts on classifyQueueStarvationByProject. Called from
7028
+ * the heartbeat, which already runs on its own timer independent of the
7029
+ * billing poll loop — so a wedged or never-succeeding poll (the
7030
+ * /api/oauth/usage endpoint was itself 429ing all of 2026-09-05) can no
7031
+ * longer leave a queue with ready work idle indefinitely.
7032
+ *
7033
+ * Logs and audits one event PER starved/blocked cwd (each carrying that
7034
+ * cwd), but still forces at most one machine-wide tickQueue() per pass —
7035
+ * the tick itself is machine-wide (it drives whatever the picker finds
7036
+ * across every project), only the DETECTION is per-project.
6868
7037
  */
6869
7038
  async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
6870
7039
  // lastDispatchAttemptAt, not lastRunAt: the latter only advances when a
6871
7040
  // batch actually launches, so a poll that keeps succeeding while dispatch
6872
7041
  // itself never gets invoked would otherwise mask a stall behind a fresh-
6873
7042
  // looking timestamp that was never actually tracking dispatch liveness.
6874
- const verdict = classifyQueueStarvation({
7043
+ const verdicts = classifyQueueStarvationByProject({
6875
7044
  jobs: state?.jobs,
6876
7045
  paused: state?.paused,
6877
- runningCount: runningSet.size,
7046
+ runningSet,
6878
7047
  lastRunAtMs: Date.parse(state?.lastDispatchAttemptAt ?? ''),
6879
7048
  now,
6880
7049
  thresholdMs,
6881
7050
  });
6882
- if (!verdict) return null;
7051
+ if (verdicts.length === 0) return null;
7052
+
7053
+ let anyStarved = false;
7054
+ let primary = null;
7055
+ for (const verdict of verdicts) {
7056
+ const mins = Math.round(verdict.idleMs / 60_000);
7057
+ if (verdict.kind === 'blocked') {
7058
+ console.warn(
7059
+ `[scheduler] QUEUE BLOCKED (${verdict.cwd}): ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
7060
+ + `terminal or parked dependency, so ticking cannot help. Blockers: `
7061
+ + verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
7062
+ );
7063
+ appendAuditEvent('queue_blocked_stall', { cwd: verdict.cwd, pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
7064
+ if (!primary) primary = verdict;
7065
+ continue;
7066
+ }
6883
7067
 
6884
- const mins = Math.round(verdict.idleMs / 60_000);
6885
- if (verdict.kind === 'blocked') {
6886
7068
  console.warn(
6887
- `[scheduler] QUEUE BLOCKED: ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
6888
- + `terminal or parked dependency, so ticking cannot help. Blockers: `
6889
- + verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
7069
+ `[scheduler] QUEUE STARVED (${verdict.cwd}): ${verdict.dispatchable} dispatchable job(s) of ${verdict.pending} pending, 0 running, `
7070
+ + `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
6890
7071
  );
6891
- appendAuditEvent('queue_blocked_stall', { pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
6892
- return verdict;
7072
+ appendAuditEvent('queue_starvation_forced_tick', { cwd: verdict.cwd, pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
7073
+ anyStarved = true;
7074
+ primary = verdict;
6893
7075
  }
6894
7076
 
6895
- console.warn(
6896
- `[scheduler] QUEUE STARVED: ${verdict.dispatchable} dispatchable job(s) of ${verdict.pending} pending, 0 running, `
6897
- + `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
6898
- );
6899
- appendAuditEvent('queue_starvation_forced_tick', { pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
7077
+ if (!anyStarved) return primary;
7078
+
6900
7079
  // A never-populated utilization reading is itself one of the ways the
6901
7080
  // when-available path silently never fires (maybeLaunchWhenAvailable
6902
7081
  // returns early on null). Treat unknown as safe here, exactly as the
@@ -6911,8 +7090,76 @@ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs
6911
7090
  // actually ticked. The watchdog is the last line of defence against a
6912
7091
  // wedged dispatcher, so it must be able to un-wedge this too.
6913
7092
  cancelToken.cancelled = false;
7093
+ // A forced tick is machine-wide by construction (the picker considers
7094
+ // every project's rows) — one call here services every starved cwd found
7095
+ // this pass, not one call per cwd.
6914
7096
  await tickQueue({ bypassLoadGate: false }).catch((e) => console.error('[scheduler] starvation tick error', e));
6915
- return verdict;
7097
+ return primary;
7098
+ }
7099
+
7100
+ // One-shot latch, keyed per cwd, for the starve-escalation consequence below —
7101
+ // never escalate the same starve stretch twice. Cleared the moment that cwd
7102
+ // stops appearing in findStarvedProjects at all (dispatched, or the machine
7103
+ // went idle/paused), mirroring the heartbeat's stallSince/stallToasted pair.
7104
+ const starveEscalated = new Set();
7105
+
7106
+ /**
7107
+ * selectStarveEscalations(starvedProjects, escalatedCwds, thresholdMs)
7108
+ * → { toEscalate: [...sp], toClear: [cwd, ...] }
7109
+ *
7110
+ * Pure. `starvedProjects` is this sweep's findStarvedProjects() output (the
7111
+ * per-cwd STARVED verdict — cwd, pendingCount, oldestPendingSlug, ageMs);
7112
+ * `escalatedCwds` is the Set already latched from a prior sweep.
7113
+ *
7114
+ * toEscalate: rows crossing thresholdMs for the FIRST time this stretch —
7115
+ * i.e. old enough AND not already latched.
7116
+ * toClear: previously-latched cwds no longer reported as starved at all this
7117
+ * sweep, so a LATER starve on that project escalates again instead of being
7118
+ * silently suppressed forever by a stale latch.
7119
+ */
7120
+ function selectStarveEscalations(starvedProjects, escalatedCwds, thresholdMs = STARVE_ESCALATION_MS) {
7121
+ const stillStarved = new Set(starvedProjects.map((sp) => sp.cwd));
7122
+ const toClear = [...escalatedCwds].filter((cwd) => !stillStarved.has(cwd));
7123
+ const toEscalate = starvedProjects.filter((sp) => sp.ageMs >= thresholdMs && !escalatedCwds.has(sp.cwd));
7124
+ return { toEscalate, toClear };
7125
+ }
7126
+
7127
+ /**
7128
+ * runStarveEscalationSweep(starvedProjects) — acts on selectStarveEscalations'
7129
+ * verdict: audits a DISTINCT 'project_starve_escalated' event (once per starve
7130
+ * stretch, per cwd) and pushes the same toast-channel error the heartbeat's
7131
+ * stall detector already uses ('schedule:stall' → renderer toast.error), so a
7132
+ * starve that has gone on long enough to matter is visible without grepping
7133
+ * the audit log. The hold reason is read from `lastTick` (recordTick's own
7134
+ * last-computed outcome) — never re-evaluated here, so this can never
7135
+ * disagree with what actually happened on the last tick.
7136
+ *
7137
+ * Escalation only: never mutates a job, never dispatches, never bypasses a
7138
+ * gate. Exported for direct unit testing (attach a fake window via
7139
+ * attachWindow() first to assert the toast send).
7140
+ */
7141
+ function runStarveEscalationSweep(starvedProjects) {
7142
+ const { toEscalate, toClear } = selectStarveEscalations(starvedProjects, starveEscalated, STARVE_ESCALATION_MS);
7143
+ for (const cwd of toClear) starveEscalated.delete(cwd);
7144
+ for (const sp of toEscalate) {
7145
+ starveEscalated.add(sp.cwd);
7146
+ const holdReason = lastTick?.reason ?? 'unknown';
7147
+ const mins = Math.round(sp.ageMs / 60_000);
7148
+ console.error(
7149
+ `[scheduler] PROJECT STARVE ESCALATED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
7150
+ + `waiting=${mins}m (>= ${Math.round(STARVE_ESCALATION_MS / 60_000)}m escalation threshold), hold reason=${holdReason} — `
7151
+ + 'a bounded escalation only; nothing was auto-reset, cancelled, or dispatched',
7152
+ );
7153
+ appendAuditEvent('project_starve_escalated', {
7154
+ cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs, holdReason,
7155
+ });
7156
+ sendIfAlive(mainWindow, 'schedule:stall', {
7157
+ message: `Project starved: ${sp.cwd} has ${sp.pendingCount} pending PRD(s), oldest waiting ~${mins}m `
7158
+ + `(hold reason: ${holdReason}) — check the Scheduler tab.`,
7159
+ total: sp.pendingCount,
7160
+ byProject: { [sp.cwd]: { starved: sp.pendingCount } },
7161
+ });
7162
+ }
6916
7163
  }
6917
7164
 
6918
7165
  // ---------- dead-process reaper ----------
@@ -7781,24 +8028,90 @@ function stuckFailedEscalationDisabled() {
7781
8028
  return process.env.SM_STUCK_FAILED_ESCALATE_DISABLE === '1';
7782
8029
  }
7783
8030
 
8031
+ // Bounded automatic failed -> pending recovery (PRD 1151): a failed row gets
8032
+ // up to FAILED_AUTORESET_CAP auto-reset attempts, each gated on having sat
8033
+ // `failed` for FAILED_AUTORESET_MS, before the stuck-failed escalation below
8034
+ // is allowed to page a human. Same env-override shape as
8035
+ // STUCK_FAILED_ESCALATE_MS/QUARANTINE_ESCALATE_MS above.
8036
+ const FAILED_AUTORESET_CAP = 3;
8037
+ const FAILED_AUTORESET_MS = process.env.SM_FAILED_AUTORESET_MINUTES
8038
+ ? Number(process.env.SM_FAILED_AUTORESET_MINUTES) * 60_000
8039
+ : 10 * 60_000;
8040
+
8041
+ /**
8042
+ * Kill-switch gate for the failed-autoreset pass below
8043
+ * (SM_FAILED_AUTORESET_DISABLE=1), same shape as stuckFailedEscalationDisabled
8044
+ * above.
8045
+ */
8046
+ function failedAutoResetDisabled() {
8047
+ return process.env.SM_FAILED_AUTORESET_DISABLE === '1';
8048
+ }
8049
+
8050
+ /**
8051
+ * selectFailedAutoResetTargets(jobs, now, thresholdMs) →
8052
+ * [{ slug, cwd, ageMs, attempts }]
8053
+ *
8054
+ * Pure selector — no IO, no `require` inside the function. Selects `failed`
8055
+ * rows whose newest statusHistory entry with `to === 'failed'` is older than
8056
+ * `thresholdMs` and whose failedAutoResetAttempts counter hasn't yet spent
8057
+ * FAILED_AUTORESET_CAP attempts. "Newest" (not first) matters because a row
8058
+ * can have failed more than once across its lifetime (an earlier auto-reset
8059
+ * attempt that itself failed again) — only the most recent failed-since
8060
+ * timestamp should gate the next attempt.
8061
+ *
8062
+ * Excludes a row whose newest failed-entry came from spawnJob:fail-dirty —
8063
+ * that source means a transient failure left genuinely uncommitted work in
8064
+ * the job's worktree and the system already decided once, deliberately, not
8065
+ * to auto-requeue it (see that call site's own comment: "could discard
8066
+ * uncommitted work left by the failed run"). This bounded auto-reset is a
8067
+ * different, slower mechanism and must not quietly override that decision
8068
+ * 10 minutes later — a human should look at a dirty worktree before it gets
8069
+ * re-driven.
8070
+ */
8071
+ function selectFailedAutoResetTargets(jobs, now, thresholdMs) {
8072
+ const targets = [];
8073
+ for (const j of jobs ?? []) {
8074
+ if (j.status !== 'failed') continue;
8075
+ const attempts = j.failedAutoResetAttempts ?? 0;
8076
+ if (attempts >= FAILED_AUTORESET_CAP) continue;
8077
+ const history = j.statusHistory || [];
8078
+ let entry = null;
8079
+ for (let i = history.length - 1; i >= 0; i--) {
8080
+ if (history[i].to === 'failed') { entry = history[i]; break; }
8081
+ }
8082
+ if (!entry) continue;
8083
+ if (entry.source === 'spawnJob:fail-dirty') continue;
8084
+ const since = Date.parse(entry.at);
8085
+ if (Number.isNaN(since)) continue;
8086
+ const ageMs = now - since;
8087
+ if (ageMs < thresholdMs) continue;
8088
+ targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, attempts });
8089
+ }
8090
+ return targets;
8091
+ }
8092
+
7784
8093
  /**
7785
8094
  * findStuckFailedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
7786
8095
  *
7787
8096
  * Pure predicate (isRescanCandidate's own resolveRunId/classifyRunOutcome log
7788
8097
  * read is the only IO, gated per-job exactly like shouldRunPeriodicReverify
7789
- * above). `failed` is a fully terminal state for every automated recovery
7790
- * path — selectResumeRecoveryTarget/selectAutoFixTargets both require
7791
- * needs_review, reapDeadRunningJobs only ever writes running → failed, and
7792
- * reconcile-repair's to-pending is for structurally invalid rows. Only a
7793
- * human's scheduler_reset_job ever takes failed → pending (LEGAL_TRANSITIONS).
7794
- * A rescan candidate (isRescanCandidate) that has sat failed longer than
7795
- * `thresholdMs` can therefore go silently stuck forever — job
7796
- * 4056-outcome-stats sat `failed` for five days with no operator signal
7797
- * (reported 2026-09-10, social-signals-trader) even though the periodic
7798
- * reverify pass (once shouldRunPeriodicReverify's guard was fixed) WAS firing
7799
- * on it — reverifyNeedsReview's failed branch can annotate looksDone but can
7800
- * never resolve a failed row itself (see its own header). This is the
7801
- * visibility half that guard fix was missing: escalate once, never requeue.
8098
+ * above). `failed` used to be a fully terminal state for every automated
8099
+ * recovery path — selectResumeRecoveryTarget/selectAutoFixTargets both
8100
+ * require needs_review, reapDeadRunningJobs only ever writes running →
8101
+ * failed, and reconcile-repair's to-pending is for structurally invalid rows.
8102
+ * That is no longer true: selectFailedAutoResetTargets above now drives a
8103
+ * bounded failed → pending auto-reset (LEGAL_TRANSITIONS already allowed the
8104
+ * edge). This escalation now only fires once that auto-reset budget is
8105
+ * genuinely spent (see the interval body's filter on failedAutoResetAttempts)
8106
+ * — a rescan candidate (isRescanCandidate) that has sat failed longer than
8107
+ * `thresholdMs` AND exhausted its auto-reset attempts can therefore go
8108
+ * silently stuck forever — job 4056-outcome-stats sat `failed` for five days
8109
+ * with no operator signal (reported 2026-09-10, social-signals-trader) even
8110
+ * though the periodic reverify pass (once shouldRunPeriodicReverify's guard
8111
+ * was fixed) WAS firing on it — reverifyNeedsReview's failed branch can
8112
+ * annotate looksDone but can never resolve a failed row itself (see its own
8113
+ * header). This is the visibility half that guard fix was missing: escalate
8114
+ * once per exhausted row, never requeue from here.
7802
8115
  *
7803
8116
  * `stuckFailedNotified` gates this to exactly once per row — once the caller
7804
8117
  * stamps it, this always excludes that row so a human is never re-paged on
@@ -7812,7 +8125,18 @@ function findStuckFailedJobs(jobs, now, thresholdMs) {
7812
8125
  if (j.status !== 'failed') continue;
7813
8126
  if (j.stuckFailedNotified === true) continue;
7814
8127
  if (!isRescanCandidate(j)) continue;
7815
- const entry = (j.statusHistory || []).find((h) => h.to === 'failed');
8128
+ // Newest (not first) to === 'failed' entry — same rationale as
8129
+ // selectFailedAutoResetTargets above: a row can have failed more than
8130
+ // once across its lifetime (an earlier auto-reset attempt that itself
8131
+ // failed again), and only the CURRENT failure episode's age should gate
8132
+ // escalation. Using the first/oldest entry would report a stale age
8133
+ // (and become instantly escalation-eligible) for a row that failed
8134
+ // months ago, recovered, and has only just failed again.
8135
+ const history = j.statusHistory || [];
8136
+ let entry = null;
8137
+ for (let i = history.length - 1; i >= 0; i--) {
8138
+ if (history[i].to === 'failed') { entry = history[i]; break; }
8139
+ }
7816
8140
  if (!entry) continue;
7817
8141
  const since = Date.parse(entry.at);
7818
8142
  if (Number.isNaN(since)) continue;
@@ -7822,6 +8146,123 @@ function findStuckFailedJobs(jobs, now, thresholdMs) {
7822
8146
  return stuck;
7823
8147
  }
7824
8148
 
8149
+ // Bounded automatic terminal decision for an EXHAUSTED needs_review row
8150
+ // (isExhaustedAutoFix === true — auto-fix attempted, no plan produced,
8151
+ // retries spent): up to NEEDS_REVIEW_RESOLVE_CAP requeue attempts (a
8152
+ // needs_review -> pending -> ... -> needs_review round trip counts as one
8153
+ // spent attempt), each gated on having sat exhausted-needs_review for
8154
+ // NEEDS_REVIEW_RESOLVE_MS, before the row is auto-skipped so a `dependsOn`
8155
+ // chain behind it always drains without an operator. Same env-override
8156
+ // shape as FAILED_AUTORESET_MS above.
8157
+ const NEEDS_REVIEW_RESOLVE_CAP = 2;
8158
+ const NEEDS_REVIEW_RESOLVE_MS = process.env.SM_NEEDS_REVIEW_RESOLVE_MINUTES
8159
+ ? Number(process.env.SM_NEEDS_REVIEW_RESOLVE_MINUTES) * 60_000
8160
+ : 30 * 60_000;
8161
+
8162
+ /**
8163
+ * Kill-switch gate for the needs_review auto-resolve pass below
8164
+ * (SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1), same shape as
8165
+ * failedAutoResetDisabled/stuckFailedEscalationDisabled above.
8166
+ */
8167
+ function needsReviewAutoResolveDisabled() {
8168
+ return process.env.SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE === '1';
8169
+ }
8170
+
8171
+ /**
8172
+ * selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) →
8173
+ * [{ slug, cwd, ageMs, attempts }]
8174
+ *
8175
+ * Pure selector — no IO. Selects `needs_review` rows whose auto-fix path is
8176
+ * genuinely spent (isExhaustedAutoFix), whose newest statusHistory entry
8177
+ * with `to === 'needs_review'` is older than `thresholdMs`, and whose
8178
+ * exhaustedResolveAttempts counter has not yet spent its cap.
8179
+ *
8180
+ * The inclusion bound is inclusive of the cap itself (`<= CAP`, not `<
8181
+ * CAP`): NEEDS_REVIEW_RESOLVE_CAP counts REQUEUE attempts already spent, and
8182
+ * the pass that observes attempts === CAP is exactly the one that must fire
8183
+ * the terminal skip (see the interval body's branch below) — excluding that
8184
+ * row here would mean the cap-exhausted row is never selected again and the
8185
+ * dependsOn chain behind it never drains, defeating this PRD's own purpose.
8186
+ */
8187
+ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
8188
+ const targets = [];
8189
+ for (const j of jobs ?? []) {
8190
+ if (j.status !== 'needs_review') continue;
8191
+ if (!isExhaustedAutoFix(j)) continue;
8192
+ if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) continue;
8193
+ const history = j.statusHistory || [];
8194
+ let entry = null;
8195
+ for (let i = history.length - 1; i >= 0; i--) {
8196
+ if (history[i].to === 'needs_review') { entry = history[i]; break; }
8197
+ }
8198
+ if (!entry) continue;
8199
+ const since = Date.parse(entry.at);
8200
+ if (Number.isNaN(since)) continue;
8201
+ const ageMs = now - since;
8202
+ if (ageMs < thresholdMs) continue;
8203
+ targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, attempts: j.exhaustedResolveAttempts ?? 0 });
8204
+ }
8205
+ return targets;
8206
+ }
8207
+
8208
+ /**
8209
+ * Applies the needs_review auto-resolve decision to a single job (mutates in
8210
+ * place; calls transitionJob + appendAuditEvent). Extracted from the
8211
+ * interval body so the three branches are unit-testable without going
8212
+ * through mutate()/queue.json IO. Order of decision:
8213
+ * 1. job.looksDone (the annotation reverifyNeedsReview writes when a
8214
+ * later commit touches the PRD's declared paths) -> 'completed'.
8215
+ * 2. otherwise, one more bounded requeue -> 'pending', incrementing
8216
+ * exhaustedResolveAttempts.
8217
+ * 3. once NEEDS_REVIEW_RESOLVE_CAP requeue attempts are spent -> 'skipped',
8218
+ * with job.error naming the exhausted path so the Queue UI still shows
8219
+ * why, and a marker (needsReviewAutoResolvedSkip) that findBlockingDep
8220
+ * reads to stop treating this SPECIFIC skip as a permanent dependsOn
8221
+ * block — unlike a generic "PRD source vanished" skip, this row was
8222
+ * given every bounded chance to resolve itself.
8223
+ * Re-validates status/exhaustion/cap itself (same race-guard shape as the
8224
+ * failed-autoreset loop above) so a stale target computed before this
8225
+ * mutate() pass can never double-apply. Returns the outcome, or null if the
8226
+ * race guard rejected it.
8227
+ */
8228
+ function applyNeedsReviewAutoResolve(j) {
8229
+ if (!j || j.status !== 'needs_review' || !isExhaustedAutoFix(j)) return null;
8230
+ if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) return null;
8231
+
8232
+ if (j.looksDone) {
8233
+ const attempt = j.exhaustedResolveAttempts ?? 0;
8234
+ transitionJob(j, 'completed', {
8235
+ reason: `needs_review auto-resolve: verifier annotation shows work landed (${j.looksDone.commits.length} commit(s) since this run touch the PRD's declared paths)`,
8236
+ source: 'needsReviewAutoResolve',
8237
+ });
8238
+ appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'completed', attempt });
8239
+ return 'completed';
8240
+ }
8241
+
8242
+ const attemptsSoFar = j.exhaustedResolveAttempts ?? 0;
8243
+ if (attemptsSoFar < NEEDS_REVIEW_RESOLVE_CAP) {
8244
+ const attempt = attemptsSoFar + 1;
8245
+ j.exhaustedResolveAttempts = attempt;
8246
+ transitionJob(j, 'pending', {
8247
+ reason: `needs_review auto-resolve: exhausted auto-fix, no completion evidence — requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`,
8248
+ source: 'needsReviewAutoResolve',
8249
+ });
8250
+ appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'requeued', attempt });
8251
+ return 'requeued';
8252
+ }
8253
+
8254
+ j.needsReviewAutoResolvedSkip = true;
8255
+ j.error = `needs_review auto-resolve: exhausted auto-fix path (autoFixOutcome=${j.autoFixOutcome ?? 'none'}, `
8256
+ + `autoFixRetries=${j.autoFixRetries ?? 0}) with no completion evidence after ${NEEDS_REVIEW_RESOLVE_CAP} `
8257
+ + `requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`;
8258
+ transitionJob(j, 'skipped', {
8259
+ reason: `needs_review auto-resolve: cap exhausted (${NEEDS_REVIEW_RESOLVE_CAP}/${NEEDS_REVIEW_RESOLVE_CAP} requeue attempts) — auto-skipped`,
8260
+ source: 'needsReviewAutoResolve',
8261
+ });
8262
+ appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'skipped', attempt: attemptsSoFar });
8263
+ return 'skipped';
8264
+ }
8265
+
7825
8266
  /**
7826
8267
  * Self-healing pass over needs_review jobs. The verifier runs in-process, so a
7827
8268
  * fix to runVerify.cjs only takes effect for jobs verified AFTER an app
@@ -8879,7 +9320,8 @@ async function init() {
8879
9320
  // else distinguishes "no pending work" from "pending work, never
8880
9321
  // started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
8881
9322
  // Escalation only, same shape as the quarantine/overrun warnings above.
8882
- for (const sp of findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS)) {
9323
+ const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
9324
+ for (const sp of starvedProjects) {
8883
9325
  console.warn(
8884
9326
  `[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
8885
9327
  + `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
@@ -8887,30 +9329,104 @@ async function init() {
8887
9329
  );
8888
9330
  appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
8889
9331
  }
8890
-
8891
- // Stuck-failed escalation (2026-09-10, social-signals-trader): see
8892
- // findStuckFailedJobs' header for why `failed` has no automated way
8893
- // back to pending. Escalation only, same shape as the three warnings
8894
- // above — never an automatic failed → pending requeue (that could
8895
- // discard uncommitted work left by the failed run; see
8896
- // spawnJob:fail-dirty). Kill-switch: SM_STUCK_FAILED_ESCALATE_DISABLE=1.
8897
- if (!stuckFailedEscalationDisabled()) {
8898
- const stuckFailed = findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
8899
- if (stuckFailed.length > 0) {
8900
- mutate((ms) => {
8901
- for (const stuck of stuckFailed) {
8902
- const j = ms.jobs.find((x) => x.slug === stuck.slug);
8903
- if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
8904
- j.stuckFailedNotified = true;
9332
+ // Bounded, automated consequence for a starve that outlives the WARN
9333
+ // above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
9334
+ // project_starved rows and zero consequence). STARVE_ESCALATION_MS is
9335
+ // strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
9336
+ // a subset of the rows already reported above — same verdict, no
9337
+ // re-derivation.
9338
+ runStarveEscalationSweep(starvedProjects);
9339
+
9340
+ // Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
9341
+ // escalation now narrowed to only the rows that auto-reset gave up on.
9342
+ // See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
9343
+ // Computed together, acted on in the SAME mutate(...) pass, so the
9344
+ // stuckFailedNotified race guard below and the auto-reset race guard
9345
+ // above it can never observe two different snapshots of the same row.
9346
+ // Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
9347
+ const autoResetTargets = failedAutoResetDisabled()
9348
+ ? []
9349
+ : selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
9350
+ const stuckFailed = stuckFailedEscalationDisabled()
9351
+ ? []
9352
+ : findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
9353
+ // Bounded automatic terminal decision for exhausted needs_review rows
9354
+ // (this PRD): computed alongside the failed-row passes above and acted
9355
+ // on in the SAME mutate(...) pass below, for the same race-guard reason
9356
+ // — a row's exhaustedResolveAttempts counter must never be read from one
9357
+ // snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
9358
+ const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
9359
+ ? []
9360
+ : selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
9361
+ // Bounded automatic exit for quarantined rows (this PRD): computed
9362
+ // alongside the passes above and acted on in the SAME mutate(...) pass
9363
+ // below, for the same race-guard reason — quarantineResolveAttempts must
9364
+ // never be read from one snapshot and written from another, and the
9365
+ // createdVia re-check inside autoResolveQuarantine must happen in the
9366
+ // same turn as the transition it gates. Kill-switch:
9367
+ // SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
9368
+ const quarantineTargets = quarantineAutoResolveDisabled()
9369
+ ? []
9370
+ : selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
9371
+ if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
9372
+ mutate(async (ms) => {
9373
+ for (const target of autoResetTargets) {
9374
+ const j = ms.jobs.find((x) => x.slug === target.slug);
9375
+ if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
9376
+ const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
9377
+ j.failedAutoResetAttempts = attempt;
9378
+ const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
9379
+ // resetJobFields is the same field-clearing list the admin
9380
+ // scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
9381
+ // it rather than inventing a second list. It also sets job.error to
9382
+ // the reason text passed in; we clear that back to null right
9383
+ // after since this is a clean auto-reset, not a recorded error.
9384
+ if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
9385
+ j.error = null;
9386
+ delete j.stuckFailedNotified;
9387
+ console.warn(
9388
+ `[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
9389
+ + `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
9390
+ );
9391
+ appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
9392
+ }
9393
+ for (const stuck of stuckFailed) {
9394
+ const j = ms.jobs.find((x) => x.slug === stuck.slug);
9395
+ if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
9396
+ // Still has auto-reset attempts left — it will be (or already was,
9397
+ // earlier this same pass) picked up by the loop above instead.
9398
+ // Never log "reset it by hand" for a row that isn't actually stuck.
9399
+ if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
9400
+ j.stuckFailedNotified = true;
9401
+ console.warn(
9402
+ `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
9403
+ + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
9404
+ + `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
9405
+ );
9406
+ appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
9407
+ }
9408
+ for (const target of exhaustedNeedsReviewTargets) {
9409
+ const j = ms.jobs.find((x) => x.slug === target.slug);
9410
+ const outcome = applyNeedsReviewAutoResolve(j);
9411
+ if (outcome) {
8905
9412
  console.warn(
8906
- `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
8907
- + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
8908
- + `no automated recovery reaches a failed row; reset it by hand via scheduler_reset_job`,
9413
+ `[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
9414
+ + `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
8909
9415
  );
8910
- appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
8911
9416
  }
8912
- }).catch(() => {});
8913
- }
9417
+ }
9418
+ for (const target of quarantineTargets) {
9419
+ const j = ms.jobs.find((x) => x.slug === target.slug);
9420
+ if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
9421
+ const outcome = await autoResolveQuarantine(j, target.ageMs);
9422
+ if (outcome) {
9423
+ console.warn(
9424
+ `[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
9425
+ + `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
9426
+ );
9427
+ }
9428
+ }
9429
+ }).catch(() => {});
8914
9430
  }
8915
9431
  }, 10 * 60_000);
8916
9432
 
@@ -9575,8 +10091,12 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
9575
10091
 
9576
10092
  module.exports = {
9577
10093
  classifyQueueStarvation,
10094
+ classifyQueueStarvationByProject,
9578
10095
  runQueueStarvationWatchdog,
9579
10096
  QUEUE_STARVATION_MS,
10097
+ selectStarveEscalations,
10098
+ runStarveEscalationSweep,
10099
+ STARVE_ESCALATION_MS,
9580
10100
  computeBlockedChains,
9581
10101
  stripAppOwnedChurn,
9582
10102
  findOverrunningJobs,
@@ -9616,6 +10136,15 @@ module.exports = {
9616
10136
  findStuckFailedJobs,
9617
10137
  STUCK_FAILED_ESCALATE_MS,
9618
10138
  stuckFailedEscalationDisabled,
10139
+ selectFailedAutoResetTargets,
10140
+ FAILED_AUTORESET_CAP,
10141
+ FAILED_AUTORESET_MS,
10142
+ failedAutoResetDisabled,
10143
+ selectExhaustedNeedsReviewTargets,
10144
+ applyNeedsReviewAutoResolve,
10145
+ NEEDS_REVIEW_RESOLVE_CAP,
10146
+ NEEDS_REVIEW_RESOLVE_MS,
10147
+ needsReviewAutoResolveDisabled,
9619
10148
  isRescanCandidate,
9620
10149
  isFailedUnverifiedShaped,
9621
10150
  computeLooksDone,
@@ -9646,6 +10175,7 @@ module.exports = {
9646
10175
  MAX_INVESTIGATION_DEPTH,
9647
10176
  forceTickOutcome,
9648
10177
  applyPauseCleared,
10178
+ formatLoadGateDetail,
9649
10179
  detectNetworkErrorInLog,
9650
10180
  detectRateLimitInLog,
9651
10181
  classifyFailureOutcome,
@@ -9695,6 +10225,10 @@ module.exports = {
9695
10225
  computeStallSummary,
9696
10226
  findStaleQuarantinedJobs,
9697
10227
  QUARANTINE_ESCALATE_MS,
10228
+ selectQuarantineAutoResolveTargets,
10229
+ autoResolveQuarantine,
10230
+ QUARANTINE_RESOLVE_CAP,
10231
+ quarantineAutoResolveDisabled,
9698
10232
  applyClearQueueVictims,
9699
10233
  PIDLESS_SPAWN_GRACE_MS,
9700
10234
  findStrandedInvestigations,