claude-code-session-manager 0.85.0 → 0.86.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-Bkv-HcP1.js → AgentLibrary-DTFL7y8G.js} +1 -1
- package/dist/assets/{DataModel-DRH-Ty20.js → DataModel-Q4jhl24R.js} +1 -1
- package/dist/assets/{History-CfRhT1Im.js → History-Cj2FejEo.js} +1 -1
- package/dist/assets/{Hooks-wEmh_U6c.js → Hooks-CaelQI6t.js} +1 -1
- package/dist/assets/{HostBilko-D_t7Rbi7.js → HostBilko--v7cMR8I.js} +1 -1
- package/dist/assets/{Library-CpArQ-OJ.js → Library-DgI9oCCZ.js} +1 -1
- package/dist/assets/{ListDetail-pjaKYs84.js → ListDetail-DYUZN-x-.js} +1 -1
- package/dist/assets/{MarkdownEditor-Xc141kjj.js → MarkdownEditor-DCIubYWf.js} +1 -1
- package/dist/assets/{McpServers-ftqaV3kn.js → McpServers-ypCYURh3.js} +1 -1
- package/dist/assets/{Memory-ChMWkNd0.js → Memory-C2qYp-3M.js} +1 -1
- package/dist/assets/{Panel-D9Kr40Ai.js → Panel-Cj2kw-Zv.js} +1 -1
- package/dist/assets/{Permissions-DKoNVgzj.js → Permissions-BiYZNGYW.js} +1 -1
- package/dist/assets/{Plugins-BtChISho.js → Plugins-C1Vj8_dU.js} +2 -2
- package/dist/assets/{ProvenanceBadge-DBA5EcYy.js → ProvenanceBadge-DczPNM5U.js} +1 -1
- package/dist/assets/{SaveBar-I0_dWNTX.js → SaveBar-Cd_7U6Gb.js} +1 -1
- package/dist/assets/{Scheduler-CbES7MC8.js → Scheduler-DcLBiJBq.js} +1 -1
- package/dist/assets/{ScopeSwitcher-5GTEveb2.js → ScopeSwitcher-DVSyI44-.js} +1 -1
- package/dist/assets/{Settings-BX3FElXk.js → Settings-Cv-pRyms.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-DNBFGrYE.js → SkillReferenceGraph-CUv1_Q2c.js} +1 -1
- package/dist/assets/{Skills-DJB6-bBM.js → Skills-C_YHkAy-.js} +1 -1
- package/dist/assets/{SystemPrompt-BiDDrJUA.js → SystemPrompt-B8R7T9xn.js} +1 -1
- package/dist/assets/{TagLibrary-_Wrevtop.js → TagLibrary-dj9YHWyy.js} +1 -1
- package/dist/assets/{TiptapBody-OWWXdLRy.js → TiptapBody-DnSBUjHE.js} +1 -1
- package/dist/assets/{Toggle-B122N0HL.js → Toggle-CjV_BJn6.js} +1 -1
- package/dist/assets/{index-DhvuQL4C.js → index-CXFQIPhO.js} +3 -3
- package/dist/assets/{settingsSchema-sGoCTd7J.js → settingsSchema-BJVciriw.js} +1 -1
- package/dist/index.html +1 -1
- package/package.json +3 -2
- package/src/main/__tests__/health-starve-escalation.test.cjs +94 -0
- package/src/main/__tests__/loadGateDetailTick.test.cjs +31 -0
- package/src/main/__tests__/machineProfile.test.cjs +19 -1
- package/src/main/__tests__/pty-session-open-telemetry.test.cjs +96 -0
- package/src/main/__tests__/queue-starvation-per-project.test.cjs +135 -0
- package/src/main/__tests__/scheduler-failed-autoreset.test.cjs +121 -0
- package/src/main/__tests__/scheduler-needs-review-autoresolve.test.cjs +189 -0
- package/src/main/__tests__/scheduler-no-dead-end-status.test.cjs +152 -0
- package/src/main/__tests__/scheduler-quarantine-autoresolve.test.cjs +165 -0
- package/src/main/__tests__/scheduler-starve-escalation.test.cjs +144 -0
- package/src/main/__tests__/telemetryClient.test.cjs +75 -5
- package/src/main/__tests__/telemetrySettings.test.cjs +32 -0
- package/src/main/health.cjs +75 -2
- package/src/main/lib/__tests__/loadGate.test.cjs +103 -2
- package/src/main/lib/__tests__/telemetryBoot.test.cjs +11 -0
- package/src/main/lib/loadGate.cjs +23 -1
- package/src/main/lib/machineProfile.cjs +15 -0
- package/src/main/lib/schedulerBatch.cjs +12 -1
- package/src/main/lib/schedulerConfig.cjs +13 -0
- package/src/main/lib/telemetryBoot.cjs +11 -8
- package/src/main/lib/telemetryClient.cjs +44 -2
- package/src/main/lib/telemetrySettings.cjs +20 -3
- package/src/main/pty.cjs +9 -0
- package/src/main/scheduler.cjs +600 -66
package/src/main/scheduler.cjs
CHANGED
|
@@ -95,6 +95,7 @@ const {
|
|
|
95
95
|
PIDLESS_SPAWN_GRACE_MS,
|
|
96
96
|
INVESTIGATION_MAX_MS,
|
|
97
97
|
STARVATION_ESCALATE_MS,
|
|
98
|
+
STARVE_ESCALATION_MS,
|
|
98
99
|
} = require('./lib/schedulerConfig.cjs');
|
|
99
100
|
const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
|
|
100
101
|
? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
|
|
@@ -1484,14 +1485,15 @@ function computeBlockedChains(jobs) {
|
|
|
1484
1485
|
* findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
|
|
1485
1486
|
*
|
|
1486
1487
|
* Pure, no IO. A 'quarantined' row (no createdVia provenance) can otherwise
|
|
1487
|
-
* sit forever with nothing looking at it — quarantine
|
|
1488
|
-
* human adopting or archiving it
|
|
1489
|
-
*
|
|
1490
|
-
*
|
|
1491
|
-
*
|
|
1492
|
-
*
|
|
1493
|
-
*
|
|
1494
|
-
*
|
|
1488
|
+
* sit forever with nothing looking at it — quarantine used to clear only via
|
|
1489
|
+
* a human adopting or archiving it; autoResolveQuarantine below now gives it
|
|
1490
|
+
* a bounded automatic exit too. This function stays the escalation/warn half
|
|
1491
|
+
* of that gate: any quarantined row whose recorded quarantine timestamp
|
|
1492
|
+
* (statusHistory's `to === 'quarantined'` entry — stamped at creation, or
|
|
1493
|
+
* backfilled from the PRD file's mtime by reconcile() for rows quarantined
|
|
1494
|
+
* before that stamp existed) is older than `thresholdMs` is reported so the
|
|
1495
|
+
* caller can warn-log and surface it distinctly. A row with no recoverable
|
|
1496
|
+
* timestamp is skipped rather than guessed at.
|
|
1495
1497
|
*/
|
|
1496
1498
|
function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
|
|
1497
1499
|
const stale = [];
|
|
@@ -1507,6 +1509,106 @@ function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
|
|
|
1507
1509
|
return stale;
|
|
1508
1510
|
}
|
|
1509
1511
|
|
|
1512
|
+
// Bounded automatic exit for a quarantined row (this PRD): up to
|
|
1513
|
+
// QUARANTINE_RESOLVE_CAP auto-resolve attempts, each gated on having sat
|
|
1514
|
+
// `quarantined` for QUARANTINE_ESCALATE_MS, before autoResolveQuarantine
|
|
1515
|
+
// below settles the row to 'skipped' rather than leaving it as a dead end
|
|
1516
|
+
// only a human `scheduler_reset_job`/adopt action could ever clear. A single
|
|
1517
|
+
// attempt is enough in practice — the outcome is terminal — but the counter
|
|
1518
|
+
// still guards against two overlapping ticks both trying to resolve the
|
|
1519
|
+
// same row.
|
|
1520
|
+
const QUARANTINE_RESOLVE_CAP = 1;
|
|
1521
|
+
|
|
1522
|
+
/**
|
|
1523
|
+
* Kill-switch gate for the quarantine auto-resolve pass below
|
|
1524
|
+
* (SM_QUARANTINE_AUTORESOLVE_DISABLE=1), same shape as
|
|
1525
|
+
* failedAutoResetDisabled/needsReviewAutoResolveDisabled.
|
|
1526
|
+
*/
|
|
1527
|
+
function quarantineAutoResolveDisabled() {
|
|
1528
|
+
return process.env.SM_QUARANTINE_AUTORESOLVE_DISABLE === '1';
|
|
1529
|
+
}
|
|
1530
|
+
|
|
1531
|
+
/**
|
|
1532
|
+
* selectQuarantineAutoResolveTargets(jobs, now, thresholdMs) →
|
|
1533
|
+
* [{ slug, cwd, ageMs }]
|
|
1534
|
+
*
|
|
1535
|
+
* Pure selector — no IO. Same age computation as findStaleQuarantinedJobs
|
|
1536
|
+
* above, bounded additionally by quarantineResolveAttempts so a row already
|
|
1537
|
+
* auto-resolved (or mid-resolve on a race) is never re-selected. Deliberately
|
|
1538
|
+
* does NOT check createdVia here — that requires a disk read of the PRD
|
|
1539
|
+
* file, and doing it at selection time would let this pass act on a
|
|
1540
|
+
* snapshot that's gone stale by the time the mutate() pass actually runs.
|
|
1541
|
+
* autoResolveQuarantine below re-reads createdVia fresh, immediately before
|
|
1542
|
+
* transitioning, inside the same mutate() callback that applies this
|
|
1543
|
+
* selector's targets — see that function's own header for why.
|
|
1544
|
+
*/
|
|
1545
|
+
function selectQuarantineAutoResolveTargets(jobs, now, thresholdMs) {
|
|
1546
|
+
const targets = [];
|
|
1547
|
+
for (const j of jobs ?? []) {
|
|
1548
|
+
if (j.status !== 'quarantined') continue;
|
|
1549
|
+
if ((j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue;
|
|
1550
|
+
const entry = (j.statusHistory || []).find((h) => h.to === 'quarantined');
|
|
1551
|
+
if (!entry) continue;
|
|
1552
|
+
const since = Date.parse(entry.at);
|
|
1553
|
+
if (Number.isNaN(since)) continue;
|
|
1554
|
+
const ageMs = now - since;
|
|
1555
|
+
if (ageMs < thresholdMs) continue;
|
|
1556
|
+
targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs });
|
|
1557
|
+
}
|
|
1558
|
+
return targets;
|
|
1559
|
+
}
|
|
1560
|
+
|
|
1561
|
+
/**
|
|
1562
|
+
* autoResolveQuarantine(job, ageMs) → Promise<'skipped'|null>
|
|
1563
|
+
*
|
|
1564
|
+
* Applies the bounded automatic exit to a single quarantined row (mutates in
|
|
1565
|
+
* place; calls transitionJob + appendAuditEvent) — extracted so it's
|
|
1566
|
+
* unit-testable without going through mutate()/queue.json IO, same shape as
|
|
1567
|
+
* applyNeedsReviewAutoResolve above.
|
|
1568
|
+
*
|
|
1569
|
+
* Re-validates status + the attempts cap itself (race guard, mirrors the
|
|
1570
|
+
* other auto-resolve loops in the 10-minute interval body), THEN re-reads the
|
|
1571
|
+
* PRD file's createdVia frontmatter fresh from disk before doing anything
|
|
1572
|
+
* else. That ordering is load-bearing: reconcile()'s adopt path (the only
|
|
1573
|
+
* OTHER route off 'quarantined') promotes a row to 'pending' the instant it
|
|
1574
|
+
* observes a createdVia stamp, on its own independent pass — if this
|
|
1575
|
+
* function trusted a snapshot taken before its own turn to run, it could
|
|
1576
|
+
* transition a row to 'skipped' the same tick reconcile() already adopted it
|
|
1577
|
+
* to 'pending', silently discarding a PRD a human just fixed. Checking here,
|
|
1578
|
+
* immediately before the transition, inside the caller's mutate() callback,
|
|
1579
|
+
* closes that window.
|
|
1580
|
+
*
|
|
1581
|
+
* A PRD file that cannot be found or parsed at all is treated as still
|
|
1582
|
+
* lacking provenance — there is no proof it has one, and stalling forever on
|
|
1583
|
+
* an unreadable file would defeat the point of a bounded exit (same
|
|
1584
|
+
* can't-prove-it/don't-guess-but-don't-stall posture as the rest of this
|
|
1585
|
+
* file's stale-row detectors).
|
|
1586
|
+
*/
|
|
1587
|
+
async function autoResolveQuarantine(job, ageMs) {
|
|
1588
|
+
if (!job || job.status !== 'quarantined') return null;
|
|
1589
|
+
if ((job.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) return null;
|
|
1590
|
+
|
|
1591
|
+
let createdVia = null;
|
|
1592
|
+
try {
|
|
1593
|
+
const resolvedDir = await findPrdDir(job.slug);
|
|
1594
|
+
const prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
|
|
1595
|
+
const parsed = await parsePrd(prdPath);
|
|
1596
|
+
createdVia = parsed.createdVia ?? null;
|
|
1597
|
+
} catch { /* unreadable/gone — no provenance found, so it stays "lacking" */ }
|
|
1598
|
+
if (createdVia) return null; // reconcile()'s own adopt path owns this row now
|
|
1599
|
+
|
|
1600
|
+
const attempt = (job.quarantineResolveAttempts ?? 0) + 1;
|
|
1601
|
+
job.quarantineResolveAttempts = attempt;
|
|
1602
|
+
job.error = `quarantined without createdVia provenance past the ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h `
|
|
1603
|
+
+ 'escalation window — auto-resolved to skipped';
|
|
1604
|
+
transitionJob(job, 'skipped', {
|
|
1605
|
+
reason: 'quarantined without createdVia provenance past escalation window',
|
|
1606
|
+
source: 'autoResolveQuarantine',
|
|
1607
|
+
});
|
|
1608
|
+
appendAuditEvent('quarantine_auto_resolved', { slug: job.slug, cwd: job.cwd ?? null, ageMs: ageMs ?? null, attempt });
|
|
1609
|
+
return 'skipped';
|
|
1610
|
+
}
|
|
1611
|
+
|
|
1510
1612
|
/**
|
|
1511
1613
|
* findOverrunningJobs(jobs, now, { factor, floorMs }) → [{ slug, cwd, estimateMinutes, ranMs, ratio }]
|
|
1512
1614
|
*
|
|
@@ -2559,6 +2661,19 @@ function applyPauseCleared(wasPaused, token) {
|
|
|
2559
2661
|
return token;
|
|
2560
2662
|
}
|
|
2561
2663
|
|
|
2664
|
+
/**
|
|
2665
|
+
* Human-readable explanation for a `reason: 'load-deferred'` tick, surfaced
|
|
2666
|
+
* to the renderer via lastTick.detail. Names the gate, the measured ratio,
|
|
2667
|
+
* the threshold and how long the stretch has been held — the box could sit
|
|
2668
|
+
* gated for 80+ minutes with nothing in the UI naming why (PRD: load gate
|
|
2669
|
+
* hysteresis). Pure so it's unit-testable without driving tickQueue's full
|
|
2670
|
+
* fs/worktree machinery.
|
|
2671
|
+
*/
|
|
2672
|
+
function formatLoadGateDetail(load) {
|
|
2673
|
+
const heldMinutes = Math.round(load.gatedSinceMs / 60_000);
|
|
2674
|
+
return `CPU load gate: loadavg1 ${load.loadavg1} / ${load.cores} cores = ${load.ratio} > threshold ${load.threshold}, held for ${heldMinutes}m`;
|
|
2675
|
+
}
|
|
2676
|
+
|
|
2562
2677
|
function attachWindow(w) { mainWindow = w; }
|
|
2563
2678
|
|
|
2564
2679
|
/**
|
|
@@ -6714,7 +6829,7 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
6714
6829
|
}
|
|
6715
6830
|
return recordTick(
|
|
6716
6831
|
{ fired: false, reason: 'load-deferred', deferredCount: gatedBatch.length, ratio: load.ratio, threshold: load.threshold },
|
|
6717
|
-
{ detail:
|
|
6832
|
+
{ detail: formatLoadGateDetail(load), holds },
|
|
6718
6833
|
);
|
|
6719
6834
|
}
|
|
6720
6835
|
|
|
@@ -6860,43 +6975,107 @@ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now,
|
|
|
6860
6975
|
}
|
|
6861
6976
|
|
|
6862
6977
|
/**
|
|
6863
|
-
*
|
|
6864
|
-
*
|
|
6865
|
-
*
|
|
6866
|
-
*
|
|
6867
|
-
*
|
|
6978
|
+
* classifyQueueStarvationByProject({ jobs, paused, runningSet, lastRunAtMs, now, thresholdMs })
|
|
6979
|
+
* → [{ cwd, kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }]
|
|
6980
|
+
*
|
|
6981
|
+
* Per-project driver around classifyQueueStarvation's pure single-project
|
|
6982
|
+
* core. `runningCount > 0` inside that core used to be fed the MACHINE-WIDE
|
|
6983
|
+
* `runningSet.size`, which meant one long-lived job in ANY project disarmed
|
|
6984
|
+
* the watchdog for EVERY other project on the box — observed live
|
|
6985
|
+
* 2026-09-12: a job in starry-night-ships ran 80+ minutes while two other
|
|
6986
|
+
* projects sat starved/blocked for hours, and the watchdog never fired once
|
|
6987
|
+
* because "work is flowing" was true somewhere else. Partitioning by cwd
|
|
6988
|
+
* (the same grouping computeBlockedChains already does) fixes DETECTION only
|
|
6989
|
+
* — the idle clock (`lastRunAtMs`) stays machine-wide, since
|
|
6990
|
+
* `lastDispatchAttemptAt` is machine-level state, and only one tick is ever
|
|
6991
|
+
* forced per watchdog pass regardless of how many cwds are starved.
|
|
6992
|
+
*
|
|
6993
|
+
* Pure, no IO. Returns [] when paused (a DECISION, not a stall) or when no
|
|
6994
|
+
* project has a verdict.
|
|
6995
|
+
*/
|
|
6996
|
+
function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
6997
|
+
if (paused) return [];
|
|
6998
|
+
const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
|
|
6999
|
+
const byCwd = new Map();
|
|
7000
|
+
for (const j of rows) {
|
|
7001
|
+
const key = j.cwd || '(unknown)';
|
|
7002
|
+
if (!byCwd.has(key)) byCwd.set(key, []);
|
|
7003
|
+
byCwd.get(key).push(j);
|
|
7004
|
+
}
|
|
7005
|
+
|
|
7006
|
+
const verdicts = [];
|
|
7007
|
+
for (const [cwd, projectJobs] of byCwd) {
|
|
7008
|
+
// Same source of truth tickQueue itself uses for "is anything running":
|
|
7009
|
+
// the in-process runningSet OR a row already stamped status:'running'.
|
|
7010
|
+
const projRunningCount = projectJobs.filter(
|
|
7011
|
+
(j) => j.status === 'running' || runningSlugs?.has?.(j.slug),
|
|
7012
|
+
).length;
|
|
7013
|
+
const verdict = classifyQueueStarvation({
|
|
7014
|
+
jobs: projectJobs,
|
|
7015
|
+
paused: false,
|
|
7016
|
+
runningCount: projRunningCount,
|
|
7017
|
+
lastRunAtMs,
|
|
7018
|
+
now,
|
|
7019
|
+
thresholdMs,
|
|
7020
|
+
});
|
|
7021
|
+
if (verdict) verdicts.push({ cwd, ...verdict });
|
|
7022
|
+
}
|
|
7023
|
+
return verdicts;
|
|
7024
|
+
}
|
|
7025
|
+
|
|
7026
|
+
/**
|
|
7027
|
+
* The watchdog half: acts on classifyQueueStarvationByProject. Called from
|
|
7028
|
+
* the heartbeat, which already runs on its own timer independent of the
|
|
7029
|
+
* billing poll loop — so a wedged or never-succeeding poll (the
|
|
7030
|
+
* /api/oauth/usage endpoint was itself 429ing all of 2026-09-05) can no
|
|
7031
|
+
* longer leave a queue with ready work idle indefinitely.
|
|
7032
|
+
*
|
|
7033
|
+
* Logs and audits one event PER starved/blocked cwd (each carrying that
|
|
7034
|
+
* cwd), but still forces at most one machine-wide tickQueue() per pass —
|
|
7035
|
+
* the tick itself is machine-wide (it drives whatever the picker finds
|
|
7036
|
+
* across every project), only the DETECTION is per-project.
|
|
6868
7037
|
*/
|
|
6869
7038
|
async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
6870
7039
|
// lastDispatchAttemptAt, not lastRunAt: the latter only advances when a
|
|
6871
7040
|
// batch actually launches, so a poll that keeps succeeding while dispatch
|
|
6872
7041
|
// itself never gets invoked would otherwise mask a stall behind a fresh-
|
|
6873
7042
|
// looking timestamp that was never actually tracking dispatch liveness.
|
|
6874
|
-
const
|
|
7043
|
+
const verdicts = classifyQueueStarvationByProject({
|
|
6875
7044
|
jobs: state?.jobs,
|
|
6876
7045
|
paused: state?.paused,
|
|
6877
|
-
|
|
7046
|
+
runningSet,
|
|
6878
7047
|
lastRunAtMs: Date.parse(state?.lastDispatchAttemptAt ?? ''),
|
|
6879
7048
|
now,
|
|
6880
7049
|
thresholdMs,
|
|
6881
7050
|
});
|
|
6882
|
-
if (
|
|
7051
|
+
if (verdicts.length === 0) return null;
|
|
7052
|
+
|
|
7053
|
+
let anyStarved = false;
|
|
7054
|
+
let primary = null;
|
|
7055
|
+
for (const verdict of verdicts) {
|
|
7056
|
+
const mins = Math.round(verdict.idleMs / 60_000);
|
|
7057
|
+
if (verdict.kind === 'blocked') {
|
|
7058
|
+
console.warn(
|
|
7059
|
+
`[scheduler] QUEUE BLOCKED (${verdict.cwd}): ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
|
|
7060
|
+
+ `terminal or parked dependency, so ticking cannot help. Blockers: `
|
|
7061
|
+
+ verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
|
|
7062
|
+
);
|
|
7063
|
+
appendAuditEvent('queue_blocked_stall', { cwd: verdict.cwd, pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
|
|
7064
|
+
if (!primary) primary = verdict;
|
|
7065
|
+
continue;
|
|
7066
|
+
}
|
|
6883
7067
|
|
|
6884
|
-
const mins = Math.round(verdict.idleMs / 60_000);
|
|
6885
|
-
if (verdict.kind === 'blocked') {
|
|
6886
7068
|
console.warn(
|
|
6887
|
-
`[scheduler] QUEUE
|
|
6888
|
-
+ `
|
|
6889
|
-
+ verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
|
|
7069
|
+
`[scheduler] QUEUE STARVED (${verdict.cwd}): ${verdict.dispatchable} dispatchable job(s) of ${verdict.pending} pending, 0 running, `
|
|
7070
|
+
+ `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
|
|
6890
7071
|
);
|
|
6891
|
-
appendAuditEvent('
|
|
6892
|
-
|
|
7072
|
+
appendAuditEvent('queue_starvation_forced_tick', { cwd: verdict.cwd, pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
|
|
7073
|
+
anyStarved = true;
|
|
7074
|
+
primary = verdict;
|
|
6893
7075
|
}
|
|
6894
7076
|
|
|
6895
|
-
|
|
6896
|
-
|
|
6897
|
-
+ `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
|
|
6898
|
-
);
|
|
6899
|
-
appendAuditEvent('queue_starvation_forced_tick', { pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
|
|
7077
|
+
if (!anyStarved) return primary;
|
|
7078
|
+
|
|
6900
7079
|
// A never-populated utilization reading is itself one of the ways the
|
|
6901
7080
|
// when-available path silently never fires (maybeLaunchWhenAvailable
|
|
6902
7081
|
// returns early on null). Treat unknown as safe here, exactly as the
|
|
@@ -6911,8 +7090,76 @@ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs
|
|
|
6911
7090
|
// actually ticked. The watchdog is the last line of defence against a
|
|
6912
7091
|
// wedged dispatcher, so it must be able to un-wedge this too.
|
|
6913
7092
|
cancelToken.cancelled = false;
|
|
7093
|
+
// A forced tick is machine-wide by construction (the picker considers
|
|
7094
|
+
// every project's rows) — one call here services every starved cwd found
|
|
7095
|
+
// this pass, not one call per cwd.
|
|
6914
7096
|
await tickQueue({ bypassLoadGate: false }).catch((e) => console.error('[scheduler] starvation tick error', e));
|
|
6915
|
-
return
|
|
7097
|
+
return primary;
|
|
7098
|
+
}
|
|
7099
|
+
|
|
7100
|
+
// One-shot latch, keyed per cwd, for the starve-escalation consequence below —
|
|
7101
|
+
// never escalate the same starve stretch twice. Cleared the moment that cwd
|
|
7102
|
+
// stops appearing in findStarvedProjects at all (dispatched, or the machine
|
|
7103
|
+
// went idle/paused), mirroring the heartbeat's stallSince/stallToasted pair.
|
|
7104
|
+
const starveEscalated = new Set();
|
|
7105
|
+
|
|
7106
|
+
/**
|
|
7107
|
+
* selectStarveEscalations(starvedProjects, escalatedCwds, thresholdMs)
|
|
7108
|
+
* → { toEscalate: [...sp], toClear: [cwd, ...] }
|
|
7109
|
+
*
|
|
7110
|
+
* Pure. `starvedProjects` is this sweep's findStarvedProjects() output (the
|
|
7111
|
+
* per-cwd STARVED verdict — cwd, pendingCount, oldestPendingSlug, ageMs);
|
|
7112
|
+
* `escalatedCwds` is the Set already latched from a prior sweep.
|
|
7113
|
+
*
|
|
7114
|
+
* toEscalate: rows crossing thresholdMs for the FIRST time this stretch —
|
|
7115
|
+
* i.e. old enough AND not already latched.
|
|
7116
|
+
* toClear: previously-latched cwds no longer reported as starved at all this
|
|
7117
|
+
* sweep, so a LATER starve on that project escalates again instead of being
|
|
7118
|
+
* silently suppressed forever by a stale latch.
|
|
7119
|
+
*/
|
|
7120
|
+
function selectStarveEscalations(starvedProjects, escalatedCwds, thresholdMs = STARVE_ESCALATION_MS) {
|
|
7121
|
+
const stillStarved = new Set(starvedProjects.map((sp) => sp.cwd));
|
|
7122
|
+
const toClear = [...escalatedCwds].filter((cwd) => !stillStarved.has(cwd));
|
|
7123
|
+
const toEscalate = starvedProjects.filter((sp) => sp.ageMs >= thresholdMs && !escalatedCwds.has(sp.cwd));
|
|
7124
|
+
return { toEscalate, toClear };
|
|
7125
|
+
}
|
|
7126
|
+
|
|
7127
|
+
/**
|
|
7128
|
+
* runStarveEscalationSweep(starvedProjects) — acts on selectStarveEscalations'
|
|
7129
|
+
* verdict: audits a DISTINCT 'project_starve_escalated' event (once per starve
|
|
7130
|
+
* stretch, per cwd) and pushes the same toast-channel error the heartbeat's
|
|
7131
|
+
* stall detector already uses ('schedule:stall' → renderer toast.error), so a
|
|
7132
|
+
* starve that has gone on long enough to matter is visible without grepping
|
|
7133
|
+
* the audit log. The hold reason is read from `lastTick` (recordTick's own
|
|
7134
|
+
* last-computed outcome) — never re-evaluated here, so this can never
|
|
7135
|
+
* disagree with what actually happened on the last tick.
|
|
7136
|
+
*
|
|
7137
|
+
* Escalation only: never mutates a job, never dispatches, never bypasses a
|
|
7138
|
+
* gate. Exported for direct unit testing (attach a fake window via
|
|
7139
|
+
* attachWindow() first to assert the toast send).
|
|
7140
|
+
*/
|
|
7141
|
+
function runStarveEscalationSweep(starvedProjects) {
|
|
7142
|
+
const { toEscalate, toClear } = selectStarveEscalations(starvedProjects, starveEscalated, STARVE_ESCALATION_MS);
|
|
7143
|
+
for (const cwd of toClear) starveEscalated.delete(cwd);
|
|
7144
|
+
for (const sp of toEscalate) {
|
|
7145
|
+
starveEscalated.add(sp.cwd);
|
|
7146
|
+
const holdReason = lastTick?.reason ?? 'unknown';
|
|
7147
|
+
const mins = Math.round(sp.ageMs / 60_000);
|
|
7148
|
+
console.error(
|
|
7149
|
+
`[scheduler] PROJECT STARVE ESCALATED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
|
|
7150
|
+
+ `waiting=${mins}m (>= ${Math.round(STARVE_ESCALATION_MS / 60_000)}m escalation threshold), hold reason=${holdReason} — `
|
|
7151
|
+
+ 'a bounded escalation only; nothing was auto-reset, cancelled, or dispatched',
|
|
7152
|
+
);
|
|
7153
|
+
appendAuditEvent('project_starve_escalated', {
|
|
7154
|
+
cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs, holdReason,
|
|
7155
|
+
});
|
|
7156
|
+
sendIfAlive(mainWindow, 'schedule:stall', {
|
|
7157
|
+
message: `Project starved: ${sp.cwd} has ${sp.pendingCount} pending PRD(s), oldest waiting ~${mins}m `
|
|
7158
|
+
+ `(hold reason: ${holdReason}) — check the Scheduler tab.`,
|
|
7159
|
+
total: sp.pendingCount,
|
|
7160
|
+
byProject: { [sp.cwd]: { starved: sp.pendingCount } },
|
|
7161
|
+
});
|
|
7162
|
+
}
|
|
6916
7163
|
}
|
|
6917
7164
|
|
|
6918
7165
|
// ---------- dead-process reaper ----------
|
|
@@ -7781,24 +8028,90 @@ function stuckFailedEscalationDisabled() {
|
|
|
7781
8028
|
return process.env.SM_STUCK_FAILED_ESCALATE_DISABLE === '1';
|
|
7782
8029
|
}
|
|
7783
8030
|
|
|
8031
|
+
// Bounded automatic failed -> pending recovery (PRD 1151): a failed row gets
|
|
8032
|
+
// up to FAILED_AUTORESET_CAP auto-reset attempts, each gated on having sat
|
|
8033
|
+
// `failed` for FAILED_AUTORESET_MS, before the stuck-failed escalation below
|
|
8034
|
+
// is allowed to page a human. Same env-override shape as
|
|
8035
|
+
// STUCK_FAILED_ESCALATE_MS/QUARANTINE_ESCALATE_MS above.
|
|
8036
|
+
const FAILED_AUTORESET_CAP = 3;
|
|
8037
|
+
const FAILED_AUTORESET_MS = process.env.SM_FAILED_AUTORESET_MINUTES
|
|
8038
|
+
? Number(process.env.SM_FAILED_AUTORESET_MINUTES) * 60_000
|
|
8039
|
+
: 10 * 60_000;
|
|
8040
|
+
|
|
8041
|
+
/**
|
|
8042
|
+
* Kill-switch gate for the failed-autoreset pass below
|
|
8043
|
+
* (SM_FAILED_AUTORESET_DISABLE=1), same shape as stuckFailedEscalationDisabled
|
|
8044
|
+
* above.
|
|
8045
|
+
*/
|
|
8046
|
+
function failedAutoResetDisabled() {
|
|
8047
|
+
return process.env.SM_FAILED_AUTORESET_DISABLE === '1';
|
|
8048
|
+
}
|
|
8049
|
+
|
|
8050
|
+
/**
|
|
8051
|
+
* selectFailedAutoResetTargets(jobs, now, thresholdMs) →
|
|
8052
|
+
* [{ slug, cwd, ageMs, attempts }]
|
|
8053
|
+
*
|
|
8054
|
+
* Pure selector — no IO, no `require` inside the function. Selects `failed`
|
|
8055
|
+
* rows whose newest statusHistory entry with `to === 'failed'` is older than
|
|
8056
|
+
* `thresholdMs` and whose failedAutoResetAttempts counter hasn't yet spent
|
|
8057
|
+
* FAILED_AUTORESET_CAP attempts. "Newest" (not first) matters because a row
|
|
8058
|
+
* can have failed more than once across its lifetime (an earlier auto-reset
|
|
8059
|
+
* attempt that itself failed again) — only the most recent failed-since
|
|
8060
|
+
* timestamp should gate the next attempt.
|
|
8061
|
+
*
|
|
8062
|
+
* Excludes a row whose newest failed-entry came from spawnJob:fail-dirty —
|
|
8063
|
+
* that source means a transient failure left genuinely uncommitted work in
|
|
8064
|
+
* the job's worktree and the system already decided once, deliberately, not
|
|
8065
|
+
* to auto-requeue it (see that call site's own comment: "could discard
|
|
8066
|
+
* uncommitted work left by the failed run"). This bounded auto-reset is a
|
|
8067
|
+
* different, slower mechanism and must not quietly override that decision
|
|
8068
|
+
* 10 minutes later — a human should look at a dirty worktree before it gets
|
|
8069
|
+
* re-driven.
|
|
8070
|
+
*/
|
|
8071
|
+
function selectFailedAutoResetTargets(jobs, now, thresholdMs) {
|
|
8072
|
+
const targets = [];
|
|
8073
|
+
for (const j of jobs ?? []) {
|
|
8074
|
+
if (j.status !== 'failed') continue;
|
|
8075
|
+
const attempts = j.failedAutoResetAttempts ?? 0;
|
|
8076
|
+
if (attempts >= FAILED_AUTORESET_CAP) continue;
|
|
8077
|
+
const history = j.statusHistory || [];
|
|
8078
|
+
let entry = null;
|
|
8079
|
+
for (let i = history.length - 1; i >= 0; i--) {
|
|
8080
|
+
if (history[i].to === 'failed') { entry = history[i]; break; }
|
|
8081
|
+
}
|
|
8082
|
+
if (!entry) continue;
|
|
8083
|
+
if (entry.source === 'spawnJob:fail-dirty') continue;
|
|
8084
|
+
const since = Date.parse(entry.at);
|
|
8085
|
+
if (Number.isNaN(since)) continue;
|
|
8086
|
+
const ageMs = now - since;
|
|
8087
|
+
if (ageMs < thresholdMs) continue;
|
|
8088
|
+
targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, attempts });
|
|
8089
|
+
}
|
|
8090
|
+
return targets;
|
|
8091
|
+
}
|
|
8092
|
+
|
|
7784
8093
|
/**
|
|
7785
8094
|
* findStuckFailedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
|
|
7786
8095
|
*
|
|
7787
8096
|
* Pure predicate (isRescanCandidate's own resolveRunId/classifyRunOutcome log
|
|
7788
8097
|
* read is the only IO, gated per-job exactly like shouldRunPeriodicReverify
|
|
7789
|
-
* above). `failed`
|
|
7790
|
-
* path — selectResumeRecoveryTarget/selectAutoFixTargets both
|
|
7791
|
-
* needs_review, reapDeadRunningJobs only ever writes running →
|
|
7792
|
-
* reconcile-repair's to-pending is for structurally invalid rows.
|
|
7793
|
-
*
|
|
7794
|
-
*
|
|
7795
|
-
*
|
|
7796
|
-
*
|
|
7797
|
-
*
|
|
7798
|
-
*
|
|
7799
|
-
*
|
|
7800
|
-
*
|
|
7801
|
-
*
|
|
8098
|
+
* above). `failed` used to be a fully terminal state for every automated
|
|
8099
|
+
* recovery path — selectResumeRecoveryTarget/selectAutoFixTargets both
|
|
8100
|
+
* require needs_review, reapDeadRunningJobs only ever writes running →
|
|
8101
|
+
* failed, and reconcile-repair's to-pending is for structurally invalid rows.
|
|
8102
|
+
* That is no longer true: selectFailedAutoResetTargets above now drives a
|
|
8103
|
+
* bounded failed → pending auto-reset (LEGAL_TRANSITIONS already allowed the
|
|
8104
|
+
* edge). This escalation now only fires once that auto-reset budget is
|
|
8105
|
+
* genuinely spent (see the interval body's filter on failedAutoResetAttempts)
|
|
8106
|
+
* — a rescan candidate (isRescanCandidate) that has sat failed longer than
|
|
8107
|
+
* `thresholdMs` AND exhausted its auto-reset attempts can therefore go
|
|
8108
|
+
* silently stuck forever — job 4056-outcome-stats sat `failed` for five days
|
|
8109
|
+
* with no operator signal (reported 2026-09-10, social-signals-trader) even
|
|
8110
|
+
* though the periodic reverify pass (once shouldRunPeriodicReverify's guard
|
|
8111
|
+
* was fixed) WAS firing on it — reverifyNeedsReview's failed branch can
|
|
8112
|
+
* annotate looksDone but can never resolve a failed row itself (see its own
|
|
8113
|
+
* header). This is the visibility half that guard fix was missing: escalate
|
|
8114
|
+
* once per exhausted row, never requeue from here.
|
|
7802
8115
|
*
|
|
7803
8116
|
* `stuckFailedNotified` gates this to exactly once per row — once the caller
|
|
7804
8117
|
* stamps it, this always excludes that row so a human is never re-paged on
|
|
@@ -7812,7 +8125,18 @@ function findStuckFailedJobs(jobs, now, thresholdMs) {
|
|
|
7812
8125
|
if (j.status !== 'failed') continue;
|
|
7813
8126
|
if (j.stuckFailedNotified === true) continue;
|
|
7814
8127
|
if (!isRescanCandidate(j)) continue;
|
|
7815
|
-
|
|
8128
|
+
// Newest (not first) to === 'failed' entry — same rationale as
|
|
8129
|
+
// selectFailedAutoResetTargets above: a row can have failed more than
|
|
8130
|
+
// once across its lifetime (an earlier auto-reset attempt that itself
|
|
8131
|
+
// failed again), and only the CURRENT failure episode's age should gate
|
|
8132
|
+
// escalation. Using the first/oldest entry would report a stale age
|
|
8133
|
+
// (and become instantly escalation-eligible) for a row that failed
|
|
8134
|
+
// months ago, recovered, and has only just failed again.
|
|
8135
|
+
const history = j.statusHistory || [];
|
|
8136
|
+
let entry = null;
|
|
8137
|
+
for (let i = history.length - 1; i >= 0; i--) {
|
|
8138
|
+
if (history[i].to === 'failed') { entry = history[i]; break; }
|
|
8139
|
+
}
|
|
7816
8140
|
if (!entry) continue;
|
|
7817
8141
|
const since = Date.parse(entry.at);
|
|
7818
8142
|
if (Number.isNaN(since)) continue;
|
|
@@ -7822,6 +8146,123 @@ function findStuckFailedJobs(jobs, now, thresholdMs) {
|
|
|
7822
8146
|
return stuck;
|
|
7823
8147
|
}
|
|
7824
8148
|
|
|
8149
|
+
// Bounded automatic terminal decision for an EXHAUSTED needs_review row
|
|
8150
|
+
// (isExhaustedAutoFix === true — auto-fix attempted, no plan produced,
|
|
8151
|
+
// retries spent): up to NEEDS_REVIEW_RESOLVE_CAP requeue attempts (a
|
|
8152
|
+
// needs_review -> pending -> ... -> needs_review round trip counts as one
|
|
8153
|
+
// spent attempt), each gated on having sat exhausted-needs_review for
|
|
8154
|
+
// NEEDS_REVIEW_RESOLVE_MS, before the row is auto-skipped so a `dependsOn`
|
|
8155
|
+
// chain behind it always drains without an operator. Same env-override
|
|
8156
|
+
// shape as FAILED_AUTORESET_MS above.
|
|
8157
|
+
const NEEDS_REVIEW_RESOLVE_CAP = 2;
|
|
8158
|
+
const NEEDS_REVIEW_RESOLVE_MS = process.env.SM_NEEDS_REVIEW_RESOLVE_MINUTES
|
|
8159
|
+
? Number(process.env.SM_NEEDS_REVIEW_RESOLVE_MINUTES) * 60_000
|
|
8160
|
+
: 30 * 60_000;
|
|
8161
|
+
|
|
8162
|
+
/**
|
|
8163
|
+
* Kill-switch gate for the needs_review auto-resolve pass below
|
|
8164
|
+
* (SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1), same shape as
|
|
8165
|
+
* failedAutoResetDisabled/stuckFailedEscalationDisabled above.
|
|
8166
|
+
*/
|
|
8167
|
+
function needsReviewAutoResolveDisabled() {
|
|
8168
|
+
return process.env.SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE === '1';
|
|
8169
|
+
}
|
|
8170
|
+
|
|
8171
|
+
/**
|
|
8172
|
+
* selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) →
|
|
8173
|
+
* [{ slug, cwd, ageMs, attempts }]
|
|
8174
|
+
*
|
|
8175
|
+
* Pure selector — no IO. Selects `needs_review` rows whose auto-fix path is
|
|
8176
|
+
* genuinely spent (isExhaustedAutoFix), whose newest statusHistory entry
|
|
8177
|
+
* with `to === 'needs_review'` is older than `thresholdMs`, and whose
|
|
8178
|
+
* exhaustedResolveAttempts counter has not yet spent its cap.
|
|
8179
|
+
*
|
|
8180
|
+
* The inclusion bound is inclusive of the cap itself (`<= CAP`, not `<
|
|
8181
|
+
* CAP`): NEEDS_REVIEW_RESOLVE_CAP counts REQUEUE attempts already spent, and
|
|
8182
|
+
* the pass that observes attempts === CAP is exactly the one that must fire
|
|
8183
|
+
* the terminal skip (see the interval body's branch below) — excluding that
|
|
8184
|
+
* row here would mean the cap-exhausted row is never selected again and the
|
|
8185
|
+
* dependsOn chain behind it never drains, defeating this PRD's own purpose.
|
|
8186
|
+
*/
|
|
8187
|
+
function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
|
|
8188
|
+
const targets = [];
|
|
8189
|
+
for (const j of jobs ?? []) {
|
|
8190
|
+
if (j.status !== 'needs_review') continue;
|
|
8191
|
+
if (!isExhaustedAutoFix(j)) continue;
|
|
8192
|
+
if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) continue;
|
|
8193
|
+
const history = j.statusHistory || [];
|
|
8194
|
+
let entry = null;
|
|
8195
|
+
for (let i = history.length - 1; i >= 0; i--) {
|
|
8196
|
+
if (history[i].to === 'needs_review') { entry = history[i]; break; }
|
|
8197
|
+
}
|
|
8198
|
+
if (!entry) continue;
|
|
8199
|
+
const since = Date.parse(entry.at);
|
|
8200
|
+
if (Number.isNaN(since)) continue;
|
|
8201
|
+
const ageMs = now - since;
|
|
8202
|
+
if (ageMs < thresholdMs) continue;
|
|
8203
|
+
targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, attempts: j.exhaustedResolveAttempts ?? 0 });
|
|
8204
|
+
}
|
|
8205
|
+
return targets;
|
|
8206
|
+
}
|
|
8207
|
+
|
|
8208
|
+
/**
|
|
8209
|
+
* Applies the needs_review auto-resolve decision to a single job (mutates in
|
|
8210
|
+
* place; calls transitionJob + appendAuditEvent). Extracted from the
|
|
8211
|
+
* interval body so the three branches are unit-testable without going
|
|
8212
|
+
* through mutate()/queue.json IO. Order of decision:
|
|
8213
|
+
* 1. job.looksDone (the annotation reverifyNeedsReview writes when a
|
|
8214
|
+
* later commit touches the PRD's declared paths) -> 'completed'.
|
|
8215
|
+
* 2. otherwise, one more bounded requeue -> 'pending', incrementing
|
|
8216
|
+
* exhaustedResolveAttempts.
|
|
8217
|
+
* 3. once NEEDS_REVIEW_RESOLVE_CAP requeue attempts are spent -> 'skipped',
|
|
8218
|
+
* with job.error naming the exhausted path so the Queue UI still shows
|
|
8219
|
+
* why, and a marker (needsReviewAutoResolvedSkip) that findBlockingDep
|
|
8220
|
+
* reads to stop treating this SPECIFIC skip as a permanent dependsOn
|
|
8221
|
+
* block — unlike a generic "PRD source vanished" skip, this row was
|
|
8222
|
+
* given every bounded chance to resolve itself.
|
|
8223
|
+
* Re-validates status/exhaustion/cap itself (same race-guard shape as the
|
|
8224
|
+
* failed-autoreset loop above) so a stale target computed before this
|
|
8225
|
+
* mutate() pass can never double-apply. Returns the outcome, or null if the
|
|
8226
|
+
* race guard rejected it.
|
|
8227
|
+
*/
|
|
8228
|
+
function applyNeedsReviewAutoResolve(j) {
|
|
8229
|
+
if (!j || j.status !== 'needs_review' || !isExhaustedAutoFix(j)) return null;
|
|
8230
|
+
if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) return null;
|
|
8231
|
+
|
|
8232
|
+
if (j.looksDone) {
|
|
8233
|
+
const attempt = j.exhaustedResolveAttempts ?? 0;
|
|
8234
|
+
transitionJob(j, 'completed', {
|
|
8235
|
+
reason: `needs_review auto-resolve: verifier annotation shows work landed (${j.looksDone.commits.length} commit(s) since this run touch the PRD's declared paths)`,
|
|
8236
|
+
source: 'needsReviewAutoResolve',
|
|
8237
|
+
});
|
|
8238
|
+
appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'completed', attempt });
|
|
8239
|
+
return 'completed';
|
|
8240
|
+
}
|
|
8241
|
+
|
|
8242
|
+
const attemptsSoFar = j.exhaustedResolveAttempts ?? 0;
|
|
8243
|
+
if (attemptsSoFar < NEEDS_REVIEW_RESOLVE_CAP) {
|
|
8244
|
+
const attempt = attemptsSoFar + 1;
|
|
8245
|
+
j.exhaustedResolveAttempts = attempt;
|
|
8246
|
+
transitionJob(j, 'pending', {
|
|
8247
|
+
reason: `needs_review auto-resolve: exhausted auto-fix, no completion evidence — requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`,
|
|
8248
|
+
source: 'needsReviewAutoResolve',
|
|
8249
|
+
});
|
|
8250
|
+
appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'requeued', attempt });
|
|
8251
|
+
return 'requeued';
|
|
8252
|
+
}
|
|
8253
|
+
|
|
8254
|
+
j.needsReviewAutoResolvedSkip = true;
|
|
8255
|
+
j.error = `needs_review auto-resolve: exhausted auto-fix path (autoFixOutcome=${j.autoFixOutcome ?? 'none'}, `
|
|
8256
|
+
+ `autoFixRetries=${j.autoFixRetries ?? 0}) with no completion evidence after ${NEEDS_REVIEW_RESOLVE_CAP} `
|
|
8257
|
+
+ `requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`;
|
|
8258
|
+
transitionJob(j, 'skipped', {
|
|
8259
|
+
reason: `needs_review auto-resolve: cap exhausted (${NEEDS_REVIEW_RESOLVE_CAP}/${NEEDS_REVIEW_RESOLVE_CAP} requeue attempts) — auto-skipped`,
|
|
8260
|
+
source: 'needsReviewAutoResolve',
|
|
8261
|
+
});
|
|
8262
|
+
appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'skipped', attempt: attemptsSoFar });
|
|
8263
|
+
return 'skipped';
|
|
8264
|
+
}
|
|
8265
|
+
|
|
7825
8266
|
/**
|
|
7826
8267
|
* Self-healing pass over needs_review jobs. The verifier runs in-process, so a
|
|
7827
8268
|
* fix to runVerify.cjs only takes effect for jobs verified AFTER an app
|
|
@@ -8879,7 +9320,8 @@ async function init() {
|
|
|
8879
9320
|
// else distinguishes "no pending work" from "pending work, never
|
|
8880
9321
|
// started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
|
|
8881
9322
|
// Escalation only, same shape as the quarantine/overrun warnings above.
|
|
8882
|
-
|
|
9323
|
+
const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
|
|
9324
|
+
for (const sp of starvedProjects) {
|
|
8883
9325
|
console.warn(
|
|
8884
9326
|
`[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
|
|
8885
9327
|
+ `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
|
|
@@ -8887,30 +9329,104 @@ async function init() {
|
|
|
8887
9329
|
);
|
|
8888
9330
|
appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
|
|
8889
9331
|
}
|
|
8890
|
-
|
|
8891
|
-
//
|
|
8892
|
-
//
|
|
8893
|
-
//
|
|
8894
|
-
//
|
|
8895
|
-
//
|
|
8896
|
-
|
|
8897
|
-
|
|
8898
|
-
|
|
8899
|
-
|
|
8900
|
-
|
|
8901
|
-
|
|
8902
|
-
|
|
8903
|
-
|
|
8904
|
-
|
|
9332
|
+
// Bounded, automated consequence for a starve that outlives the WARN
|
|
9333
|
+
// above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
|
|
9334
|
+
// project_starved rows and zero consequence). STARVE_ESCALATION_MS is
|
|
9335
|
+
// strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
|
|
9336
|
+
// a subset of the rows already reported above — same verdict, no
|
|
9337
|
+
// re-derivation.
|
|
9338
|
+
runStarveEscalationSweep(starvedProjects);
|
|
9339
|
+
|
|
9340
|
+
// Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
|
|
9341
|
+
// escalation now narrowed to only the rows that auto-reset gave up on.
|
|
9342
|
+
// See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
|
|
9343
|
+
// Computed together, acted on in the SAME mutate(...) pass, so the
|
|
9344
|
+
// stuckFailedNotified race guard below and the auto-reset race guard
|
|
9345
|
+
// above it can never observe two different snapshots of the same row.
|
|
9346
|
+
// Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
|
|
9347
|
+
const autoResetTargets = failedAutoResetDisabled()
|
|
9348
|
+
? []
|
|
9349
|
+
: selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
|
|
9350
|
+
const stuckFailed = stuckFailedEscalationDisabled()
|
|
9351
|
+
? []
|
|
9352
|
+
: findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
|
|
9353
|
+
// Bounded automatic terminal decision for exhausted needs_review rows
|
|
9354
|
+
// (this PRD): computed alongside the failed-row passes above and acted
|
|
9355
|
+
// on in the SAME mutate(...) pass below, for the same race-guard reason
|
|
9356
|
+
// — a row's exhaustedResolveAttempts counter must never be read from one
|
|
9357
|
+
// snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
|
|
9358
|
+
const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
|
|
9359
|
+
? []
|
|
9360
|
+
: selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
|
|
9361
|
+
// Bounded automatic exit for quarantined rows (this PRD): computed
|
|
9362
|
+
// alongside the passes above and acted on in the SAME mutate(...) pass
|
|
9363
|
+
// below, for the same race-guard reason — quarantineResolveAttempts must
|
|
9364
|
+
// never be read from one snapshot and written from another, and the
|
|
9365
|
+
// createdVia re-check inside autoResolveQuarantine must happen in the
|
|
9366
|
+
// same turn as the transition it gates. Kill-switch:
|
|
9367
|
+
// SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
|
|
9368
|
+
const quarantineTargets = quarantineAutoResolveDisabled()
|
|
9369
|
+
? []
|
|
9370
|
+
: selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
|
|
9371
|
+
if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
|
|
9372
|
+
mutate(async (ms) => {
|
|
9373
|
+
for (const target of autoResetTargets) {
|
|
9374
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
9375
|
+
if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
|
|
9376
|
+
const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
|
|
9377
|
+
j.failedAutoResetAttempts = attempt;
|
|
9378
|
+
const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
|
|
9379
|
+
// resetJobFields is the same field-clearing list the admin
|
|
9380
|
+
// scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
|
|
9381
|
+
// it rather than inventing a second list. It also sets job.error to
|
|
9382
|
+
// the reason text passed in; we clear that back to null right
|
|
9383
|
+
// after since this is a clean auto-reset, not a recorded error.
|
|
9384
|
+
if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
|
|
9385
|
+
j.error = null;
|
|
9386
|
+
delete j.stuckFailedNotified;
|
|
9387
|
+
console.warn(
|
|
9388
|
+
`[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
9389
|
+
+ `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
|
|
9390
|
+
);
|
|
9391
|
+
appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
|
|
9392
|
+
}
|
|
9393
|
+
for (const stuck of stuckFailed) {
|
|
9394
|
+
const j = ms.jobs.find((x) => x.slug === stuck.slug);
|
|
9395
|
+
if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
|
|
9396
|
+
// Still has auto-reset attempts left — it will be (or already was,
|
|
9397
|
+
// earlier this same pass) picked up by the loop above instead.
|
|
9398
|
+
// Never log "reset it by hand" for a row that isn't actually stuck.
|
|
9399
|
+
if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
|
|
9400
|
+
j.stuckFailedNotified = true;
|
|
9401
|
+
console.warn(
|
|
9402
|
+
`[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
|
|
9403
|
+
+ `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
9404
|
+
+ `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
|
|
9405
|
+
);
|
|
9406
|
+
appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
|
|
9407
|
+
}
|
|
9408
|
+
for (const target of exhaustedNeedsReviewTargets) {
|
|
9409
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
9410
|
+
const outcome = applyNeedsReviewAutoResolve(j);
|
|
9411
|
+
if (outcome) {
|
|
8905
9412
|
console.warn(
|
|
8906
|
-
`[scheduler]
|
|
8907
|
-
+ `
|
|
8908
|
-
+ `no automated recovery reaches a failed row; reset it by hand via scheduler_reset_job`,
|
|
9413
|
+
`[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
9414
|
+
+ `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
|
|
8909
9415
|
);
|
|
8910
|
-
appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
|
|
8911
9416
|
}
|
|
8912
|
-
}
|
|
8913
|
-
|
|
9417
|
+
}
|
|
9418
|
+
for (const target of quarantineTargets) {
|
|
9419
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
9420
|
+
if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
|
|
9421
|
+
const outcome = await autoResolveQuarantine(j, target.ageMs);
|
|
9422
|
+
if (outcome) {
|
|
9423
|
+
console.warn(
|
|
9424
|
+
`[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
9425
|
+
+ `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
|
|
9426
|
+
);
|
|
9427
|
+
}
|
|
9428
|
+
}
|
|
9429
|
+
}).catch(() => {});
|
|
8914
9430
|
}
|
|
8915
9431
|
}, 10 * 60_000);
|
|
8916
9432
|
|
|
@@ -9575,8 +10091,12 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
9575
10091
|
|
|
9576
10092
|
module.exports = {
|
|
9577
10093
|
classifyQueueStarvation,
|
|
10094
|
+
classifyQueueStarvationByProject,
|
|
9578
10095
|
runQueueStarvationWatchdog,
|
|
9579
10096
|
QUEUE_STARVATION_MS,
|
|
10097
|
+
selectStarveEscalations,
|
|
10098
|
+
runStarveEscalationSweep,
|
|
10099
|
+
STARVE_ESCALATION_MS,
|
|
9580
10100
|
computeBlockedChains,
|
|
9581
10101
|
stripAppOwnedChurn,
|
|
9582
10102
|
findOverrunningJobs,
|
|
@@ -9616,6 +10136,15 @@ module.exports = {
|
|
|
9616
10136
|
findStuckFailedJobs,
|
|
9617
10137
|
STUCK_FAILED_ESCALATE_MS,
|
|
9618
10138
|
stuckFailedEscalationDisabled,
|
|
10139
|
+
selectFailedAutoResetTargets,
|
|
10140
|
+
FAILED_AUTORESET_CAP,
|
|
10141
|
+
FAILED_AUTORESET_MS,
|
|
10142
|
+
failedAutoResetDisabled,
|
|
10143
|
+
selectExhaustedNeedsReviewTargets,
|
|
10144
|
+
applyNeedsReviewAutoResolve,
|
|
10145
|
+
NEEDS_REVIEW_RESOLVE_CAP,
|
|
10146
|
+
NEEDS_REVIEW_RESOLVE_MS,
|
|
10147
|
+
needsReviewAutoResolveDisabled,
|
|
9619
10148
|
isRescanCandidate,
|
|
9620
10149
|
isFailedUnverifiedShaped,
|
|
9621
10150
|
computeLooksDone,
|
|
@@ -9646,6 +10175,7 @@ module.exports = {
|
|
|
9646
10175
|
MAX_INVESTIGATION_DEPTH,
|
|
9647
10176
|
forceTickOutcome,
|
|
9648
10177
|
applyPauseCleared,
|
|
10178
|
+
formatLoadGateDetail,
|
|
9649
10179
|
detectNetworkErrorInLog,
|
|
9650
10180
|
detectRateLimitInLog,
|
|
9651
10181
|
classifyFailureOutcome,
|
|
@@ -9695,6 +10225,10 @@ module.exports = {
|
|
|
9695
10225
|
computeStallSummary,
|
|
9696
10226
|
findStaleQuarantinedJobs,
|
|
9697
10227
|
QUARANTINE_ESCALATE_MS,
|
|
10228
|
+
selectQuarantineAutoResolveTargets,
|
|
10229
|
+
autoResolveQuarantine,
|
|
10230
|
+
QUARANTINE_RESOLVE_CAP,
|
|
10231
|
+
quarantineAutoResolveDisabled,
|
|
9698
10232
|
applyClearQueueVictims,
|
|
9699
10233
|
PIDLESS_SPAWN_GRACE_MS,
|
|
9700
10234
|
findStrandedInvestigations,
|