claude-code-session-manager 0.83.0 → 0.84.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-B-OM-4gk.js → AgentLibrary-CCVIpoSz.js} +1 -1
- package/dist/assets/{DataModel-BC6ppcBr.js → DataModel-BltREYde.js} +1 -1
- package/dist/assets/{History-D3Qw2BWn.js → History-BZxFkOp6.js} +2 -2
- package/dist/assets/{Hooks-OxBQ0nIc.js → Hooks-Bznlfaaa.js} +1 -1
- package/dist/assets/{HostBilko-BNFC6bNZ.js → HostBilko-DUG5YHA_.js} +1 -1
- package/dist/assets/{Library-DAMyrUtY.js → Library-Bi2Fn3w9.js} +1 -1
- package/dist/assets/{ListDetail-BwywrpvE.js → ListDetail-4VBvXKrz.js} +1 -1
- package/dist/assets/{MarkdownEditor-HEwjN1bp.js → MarkdownEditor-D2ft_v5j.js} +1 -1
- package/dist/assets/{McpServers-D1VZF7Ix.js → McpServers-FDekKkyE.js} +1 -1
- package/dist/assets/{Memory-CgYJ_QqD.js → Memory-Dkl80uj_.js} +6 -6
- package/dist/assets/{Panel-dTNWC59Q.js → Panel-LurG5VfD.js} +1 -1
- package/dist/assets/{Permissions-qI_NKi1b.js → Permissions-D2wHBCFA.js} +1 -1
- package/dist/assets/{Plugins-DZ0_CHiF.js → Plugins-CTw_wwbI.js} +2 -2
- package/dist/assets/{ProvenanceBadge-B2i_uxCu.js → ProvenanceBadge-CW6HNUv6.js} +1 -1
- package/dist/assets/{SaveBar-4tNMCjUd.js → SaveBar-alDHG6cP.js} +1 -1
- package/dist/assets/{Scheduler-CUen18Gc.js → Scheduler-DxiPcaiW.js} +7 -7
- package/dist/assets/{ScopeSwitcher-Co3tkLTT.js → ScopeSwitcher-DzFXUKLZ.js} +1 -1
- package/dist/assets/Settings-B6H3v2am.js +3 -0
- package/dist/assets/{SkillReferenceGraph-BKcHtMPc.js → SkillReferenceGraph-B8EZalpN.js} +1 -1
- package/dist/assets/{Skills-DHb4INtF.js → Skills-DwuHuA08.js} +2 -2
- package/dist/assets/SystemPrompt-CMMqpGYn.js +1 -0
- package/dist/assets/{TagLibrary-xLP5dJvh.js → TagLibrary-C2N5C1n9.js} +1 -1
- package/dist/assets/{TiptapBody-T-ULTjK4.js → TiptapBody-btlID-dQ.js} +1 -1
- package/dist/assets/{Toggle-B-5M2hGe.js → Toggle-BGOn3DZj.js} +1 -1
- package/dist/assets/{index-CzAxC432.js → index-QLRf0epp.js} +316 -314
- package/dist/assets/{index-CYhdtisq.css → index-mnjNDpb1.css} +1 -1
- package/dist/assets/{settingsSchema-C_nsemcF.js → settingsSchema-DA3N2Up3.js} +1 -1
- package/dist/index.html +2 -2
- package/package.json +4 -1
- package/scripts/hooks/guard-destructive-git.cjs +514 -0
- package/scripts/hooks/guard-inline-implementation.cjs +219 -0
- package/scripts/hooks/guard-prd-writes.cjs +200 -0
- package/src/main/__tests__/epicMintTelemetryTap.test.cjs +64 -0
- package/src/main/__tests__/health-delegation-chain.test.cjs +2 -1
- package/src/main/__tests__/health-queue-dispatch.test.cjs +84 -0
- package/src/main/__tests__/health-usage-poller.test.cjs +97 -0
- package/src/main/__tests__/health-worktree-cap-blocked.test.cjs +65 -0
- package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +143 -0
- package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +120 -0
- package/src/main/__tests__/promptSessionTranscript.test.cjs +0 -0
- package/src/main/__tests__/queue-starvation-dispatch-driver.test.cjs +143 -0
- package/src/main/__tests__/rateLimitPollerStreak.test.cjs +79 -0
- package/src/main/__tests__/scheduleJobTransitionsTelemetryTap.test.cjs +72 -0
- package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +74 -0
- package/src/main/__tests__/scheduler-job-overrun.test.cjs +58 -0
- package/src/main/__tests__/scheduler-notify-originating-tab-transcript.test.cjs +1 -0
- package/src/main/__tests__/scheduler-periodic-reverify-guard.test.cjs +134 -2
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +33 -0
- package/src/main/__tests__/scheduler-stuck-failed-escalation.test.cjs +136 -0
- package/src/main/__tests__/telemetryClient.test.cjs +66 -0
- package/src/main/__tests__/telemetryContract.test.cjs +883 -0
- package/src/main/crashDiagnostics.cjs +29 -1
- package/src/main/health.cjs +197 -3
- package/src/main/index.cjs +65 -6
- package/src/main/ipcSchemas.cjs +19 -2
- package/src/main/lib/__tests__/crashTelemetry.test.cjs +97 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +302 -4
- package/src/main/lib/__tests__/fixtures/scheduler-machine.json.corrupt-1789147548 +34 -0
- package/src/main/lib/__tests__/gitWorktree.test.cjs +413 -1
- package/src/main/lib/__tests__/jobWorktreeBootLive.test.cjs +71 -0
- package/src/main/lib/__tests__/queueStoreAtomicWrite.test.cjs +88 -0
- package/src/main/lib/__tests__/queueStoreMachineStateRecovery.test.cjs +123 -0
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +58 -0
- package/src/main/lib/__tests__/telemetryBacklog.test.cjs +620 -0
- package/src/main/lib/__tests__/telemetryBoot.test.cjs +125 -0
- package/src/main/lib/__tests__/telemetryConsent.test.cjs +130 -0
- package/src/main/lib/__tests__/telemetryCounters.test.cjs +57 -0
- package/src/main/lib/__tests__/telemetryCountersMetadataColumn.test.cjs +89 -0
- package/src/main/lib/crashTelemetry.cjs +37 -0
- package/src/main/lib/delegationReadiness.cjs +119 -1
- package/src/main/lib/epicMint.cjs +2 -0
- package/src/main/lib/gitWorktree.cjs +427 -17
- package/src/main/lib/jobWorktree.cjs +2 -1
- package/src/main/lib/jobWorktreeBootLive.cjs +51 -0
- package/src/main/lib/jobWorktreeTerminalOrphanLive.cjs +68 -0
- package/src/main/lib/opsErrorLog.cjs +78 -25
- package/src/main/lib/queueStore.cjs +233 -20
- package/src/main/lib/reaperHelpers.cjs +23 -1
- package/src/main/lib/scheduleJobSchema.cjs +7 -0
- package/src/main/lib/scheduleJobTransitions.cjs +12 -0
- package/src/main/lib/telemetryBacklog.cjs +601 -0
- package/src/main/lib/telemetryBoot.cjs +71 -0
- package/src/main/lib/telemetryClient.cjs +31 -0
- package/src/main/lib/telemetryConsent.cjs +34 -0
- package/src/main/lib/telemetryCounters.cjs +45 -0
- package/src/main/promptSessionTranscript.cjs +0 -0
- package/src/main/pty.cjs +2 -0
- package/src/main/scheduler.cjs +481 -33
- package/src/preload/api.d.ts +84 -4
- package/src/preload/index.cjs +9 -0
- package/dist/assets/Settings-DXVgKLUx.js +0 -3
- package/dist/assets/SystemPrompt-BjqFOHSk.js +0 -1
package/src/main/scheduler.cjs
CHANGED
|
@@ -143,6 +143,9 @@ function resolveOriginSessionId(cwd, epicId) {
|
|
|
143
143
|
const sessionSlots = require('./lib/sessionSlots.cjs');
|
|
144
144
|
const quietMachineLease = require('./lib/quietMachineLease.cjs');
|
|
145
145
|
const jobWorktree = require('./lib/jobWorktree.cjs');
|
|
146
|
+
const gitWorktree = require('./lib/gitWorktree.cjs');
|
|
147
|
+
const { buildJobWorktreeIsLive } = require('./lib/jobWorktreeBootLive.cjs');
|
|
148
|
+
const { buildTerminalOrphanIsLive } = require('./lib/jobWorktreeTerminalOrphanLive.cjs');
|
|
146
149
|
const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
|
|
147
150
|
const queueStore = require('./lib/queueStore.cjs');
|
|
148
151
|
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
@@ -1294,6 +1297,7 @@ function loadSchedulerState() {
|
|
|
1294
1297
|
if (typeof s.backoffMs === 'number') backoffMs = s.backoffMs;
|
|
1295
1298
|
if (typeof s.pauseClearedManuallyAt === 'number') pauseClearedManuallyAt = s.pauseClearedManuallyAt;
|
|
1296
1299
|
if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
|
|
1300
|
+
if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
|
|
1297
1301
|
} catch { /* first boot or corrupt — start fresh */ }
|
|
1298
1302
|
}
|
|
1299
1303
|
|
|
@@ -1313,6 +1317,7 @@ function persistSchedulerState() {
|
|
|
1313
1317
|
pausedReason: null,
|
|
1314
1318
|
pausedSince: null,
|
|
1315
1319
|
pauseClearedManuallyAt,
|
|
1320
|
+
failureStreakWarned,
|
|
1316
1321
|
});
|
|
1317
1322
|
} catch (e) {
|
|
1318
1323
|
console.warn('[scheduler] failed to persist scheduler state', e?.message);
|
|
@@ -1629,14 +1634,16 @@ function findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
|
|
|
1629
1634
|
// (The single-file EMPTY_QUEUE/shapeQueue readers were retired with the
|
|
1630
1635
|
// global queue.json — queueStore.cjs's merged readers own the shape now.)
|
|
1631
1636
|
|
|
1632
|
-
// Quarantine a corrupt shard alongside itself (once per
|
|
1633
|
-
//
|
|
1634
|
-
//
|
|
1635
|
-
|
|
1637
|
+
// Quarantine a corrupt shard alongside itself (once per PATH, not once per
|
|
1638
|
+
// process — a module-level boolean let one file's tear consume the latch and
|
|
1639
|
+
// silently swallow every other file's `.corrupt-*` copy for the rest of the
|
|
1640
|
+
// process's life, observed live 2026-09-11) so a human can diff it against
|
|
1641
|
+
// the .bak-* snapshots.
|
|
1642
|
+
const quarantinedPaths = new Set();
|
|
1636
1643
|
function flagUnreadable(state) {
|
|
1637
1644
|
if (!state.unreadable) return state;
|
|
1638
|
-
if (
|
|
1639
|
-
|
|
1645
|
+
if (state.unreadablePath && !quarantinedPaths.has(state.unreadablePath)) {
|
|
1646
|
+
quarantinedPaths.add(state.unreadablePath);
|
|
1640
1647
|
try {
|
|
1641
1648
|
fs.copyFileSync(state.unreadablePath, `${state.unreadablePath}.corrupt-${Date.now()}`);
|
|
1642
1649
|
} catch { /* best-effort: the read already failed, the copy may too */ }
|
|
@@ -2411,6 +2418,48 @@ let firstFailureAt = null;
|
|
|
2411
2418
|
let firstNon429FailureAt = null; // tracks only transient/config failures; 429s don't count toward network-pause threshold
|
|
2412
2419
|
let lastFailureKind = null; // 'transient' | 'meter_rate_limited' | 'auth' | null
|
|
2413
2420
|
let pauseClearedManuallyAt = null;
|
|
2421
|
+
// PRD: the usage-poller silent-failure-streak WARN is emitted once per streak,
|
|
2422
|
+
// not once per failure (57 failures must produce ONE opsErrorLog line, not 57).
|
|
2423
|
+
// Reset alongside consecutiveFailures everywhere that resets to 0.
|
|
2424
|
+
let failureStreakWarned = false;
|
|
2425
|
+
// Ceiling on pollLoop's exponential poll backoff (both the 'transient'/'config'
|
|
2426
|
+
// branch and the 'meter_rate_limited' branch below share this cap — a single
|
|
2427
|
+
// constant so the two never drift to different ceilings).
|
|
2428
|
+
const BACKOFF_MAX_MS = 480_000; // 8 minutes
|
|
2429
|
+
// Threshold past which a growing consecutiveFailures streak stops being normal
|
|
2430
|
+
// jitter and becomes worth a human's attention. health.cjs imports this so the
|
|
2431
|
+
// WARN and the `npm run health` non-GREEN trip at the exact same count.
|
|
2432
|
+
const FAILURE_STREAK_WARN_THRESHOLD = 5;
|
|
2433
|
+
|
|
2434
|
+
/** Pure: exponential backoff with a cap, shared by every pollLoop failure branch. Exported for unit testing. */
|
|
2435
|
+
function nextBackoffMs(prevBackoffMs) {
|
|
2436
|
+
return prevBackoffMs ? Math.min(prevBackoffMs * 2, BACKOFF_MAX_MS) : 30_000;
|
|
2437
|
+
}
|
|
2438
|
+
|
|
2439
|
+
/**
|
|
2440
|
+
* Pure: should this failure count trigger the one-time streak WARN? Exported
|
|
2441
|
+
* for unit testing. `alreadyWarned` is the current streak's warned flag —
|
|
2442
|
+
* true for every failure after the threshold-crossing one, so this returns
|
|
2443
|
+
* true exactly once per streak.
|
|
2444
|
+
*/
|
|
2445
|
+
function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold = FAILURE_STREAK_WARN_THRESHOLD) {
|
|
2446
|
+
return consecutiveFailures >= threshold && !alreadyWarned;
|
|
2447
|
+
}
|
|
2448
|
+
|
|
2449
|
+
/** Emits the one-time opsErrorLog WARN for a failure streak crossing the threshold, if not already warned this streak. */
|
|
2450
|
+
function warnFailureStreakIfNeeded() {
|
|
2451
|
+
if (!shouldWarnFailureStreak(consecutiveFailures, failureStreakWarned)) return;
|
|
2452
|
+
failureStreakWarned = true;
|
|
2453
|
+
try {
|
|
2454
|
+
appendError({
|
|
2455
|
+
cwd: DEFAULT_PROJECT_CWD,
|
|
2456
|
+
scope: 'scheduler',
|
|
2457
|
+
level: 'warn',
|
|
2458
|
+
message: `usage/rate-limit poller has failed ${consecutiveFailures} consecutive times (kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${SCHEDULER_STATE_PATH}`,
|
|
2459
|
+
meta: { consecutiveFailures, backoffMs, lastFailureKind },
|
|
2460
|
+
});
|
|
2461
|
+
} catch { /* durable logging must never break the poll loop */ }
|
|
2462
|
+
}
|
|
2414
2463
|
// PRD 1119: consecutive-rapid-rate-limit hard-pause tracking, keyed per slug.
|
|
2415
2464
|
// See isCooldownSuppressed/nextRapidRateLimitCount below for the pure rules.
|
|
2416
2465
|
const consecutiveRapidRateLimitsBySlug = new Map();
|
|
@@ -2783,6 +2832,7 @@ async function clearPause(source) {
|
|
|
2783
2832
|
firstFailureAt = null;
|
|
2784
2833
|
firstNon429FailureAt = null;
|
|
2785
2834
|
lastFailureKind = null;
|
|
2835
|
+
failureStreakWarned = false;
|
|
2786
2836
|
persistSchedulerState();
|
|
2787
2837
|
}
|
|
2788
2838
|
if (wasPaused) await broadcast({ flush: true });
|
|
@@ -2836,6 +2886,9 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
2836
2886
|
// human-driven reset must genuinely start the auto-fix budget over,
|
|
2837
2887
|
// including the one-time dead-fix-plan-child reopen (PRD 1129).
|
|
2838
2888
|
delete job.autoFixReopened;
|
|
2889
|
+
// Same category — an overrun badge is this run's outcome (see
|
|
2890
|
+
// findOverrunningJobs's header), never durable across a reset.
|
|
2891
|
+
delete job.overrun;
|
|
2839
2892
|
// Like exitCode: this run's outcome, not durable across a reset — a stale
|
|
2840
2893
|
// leak badge from a prior attempt must not linger once the job re-fires.
|
|
2841
2894
|
delete job.leakedDescendants;
|
|
@@ -3140,6 +3193,7 @@ async function notifyOriginatingTab(job, {
|
|
|
3140
3193
|
await appendTranscriptTurn(job.cwd, epicIdForTranscript, {
|
|
3141
3194
|
role: 'assistant',
|
|
3142
3195
|
text: resultText || message,
|
|
3196
|
+
eventId: `prd-result:${job.slug}:${job.runId || ''}`,
|
|
3143
3197
|
});
|
|
3144
3198
|
} catch (e) {
|
|
3145
3199
|
console.error('[scheduler] notifyOriginatingTab transcript append error', job?.slug, e);
|
|
@@ -3997,13 +4051,20 @@ function pickRunDir() {
|
|
|
3997
4051
|
/**
|
|
3998
4052
|
* Execute a single PRD job. Writes stdout/stderr to a log file and a meta
|
|
3999
4053
|
* JSON sidecar. Accepts an optional onPid(pid) callback called synchronously
|
|
4000
|
-
* after spawn so callers can persist the pid before the job finishes
|
|
4054
|
+
* after spawn so callers can persist the pid before the job finishes, and an
|
|
4055
|
+
* optional onPhase(phase) callback invoked at the top of this function (see
|
|
4056
|
+
* spawnJob's dispatchPhase breadcrumb) — this function has no access to
|
|
4057
|
+
* `mutate`, so the caller injects the stamping side effect instead.
|
|
4001
4058
|
*
|
|
4002
4059
|
* Uses withChildAndLog for the child lifecycle (fd open/close, watchdog timers).
|
|
4003
4060
|
* Watchdogs are declared as an array; the result-tailer's exit-code mapping
|
|
4004
4061
|
* (success+killedBySignal → 0) is scheduler-specific and lives in onExit.
|
|
4005
4062
|
*/
|
|
4006
|
-
async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget = null, foreignWip = null, launchEnv = null) {
|
|
4063
|
+
async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget = null, foreignWip = null, launchEnv = null, onPhase = null) {
|
|
4064
|
+
// First statement — stamps the 'exec-entered' dispatch-phase breadcrumb
|
|
4065
|
+
// before openLog below, so a hang inside openLog/spawn itself still shows
|
|
4066
|
+
// execution reached this function (see spawnJob's dispatchPhase comment).
|
|
4067
|
+
if (onPhase) await onPhase('exec-entered');
|
|
4007
4068
|
const logPath = path.join(runDir, `${job.slug}.log`);
|
|
4008
4069
|
const metaPath = path.join(runDir, `${job.slug}.meta.json`);
|
|
4009
4070
|
// `cwd` stays the MAIN tree throughout — PRD lookup (findPrdDir/prdPathForJob)
|
|
@@ -4049,6 +4110,12 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4049
4110
|
|
|
4050
4111
|
let prompt;
|
|
4051
4112
|
let prdPath = null;
|
|
4113
|
+
// Captured on a successful parsePrd() so the raw, untruncated PRD body can
|
|
4114
|
+
// be recorded as a 'user' turn in the durable per-Epic transcript below —
|
|
4115
|
+
// the durable store must see the actual prompt the executor ran, not just
|
|
4116
|
+
// the scheduler's own short status chip (that's notifyOriginatingTab's job
|
|
4117
|
+
// for the assistant side).
|
|
4118
|
+
let parsedPrdMeta = null;
|
|
4052
4119
|
if (resumeTarget) {
|
|
4053
4120
|
// Resume mode (PRD 1111): a short deterministic preamble naming the
|
|
4054
4121
|
// recorded dirty paths, NEVER the original PRD body — the resumed
|
|
@@ -4071,6 +4138,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4071
4138
|
// resolved — never concatenated here, so it can't end up ahead of the
|
|
4072
4139
|
// digest in the composed prompt (PRD 992).
|
|
4073
4140
|
prompt = parsed.body;
|
|
4141
|
+
parsedPrdMeta = parsed;
|
|
4074
4142
|
} catch (e) {
|
|
4075
4143
|
// The project-scoped dir isn't the only place a PRD source can live — a
|
|
4076
4144
|
// writer that hasn't migrated to prdLocations.cjs yet (or a not-yet-run
|
|
@@ -4084,6 +4152,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4084
4152
|
const parsed = await parsePrd(fallbackPath);
|
|
4085
4153
|
prompt = parsed.body;
|
|
4086
4154
|
prdPath = fallbackPath;
|
|
4155
|
+
parsedPrdMeta = parsed;
|
|
4087
4156
|
} catch (e2) {
|
|
4088
4157
|
// Found the dir a moment ago but the read still failed. Case A: the
|
|
4089
4158
|
// source has since been archived — stale-skip as usual. Case B: it
|
|
@@ -4121,6 +4190,26 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4121
4190
|
}
|
|
4122
4191
|
} // end resumeTarget ? preamble : normal-PRD-read
|
|
4123
4192
|
|
|
4193
|
+
// Record the raw, untruncated PRD body as a 'user' turn before any digest/
|
|
4194
|
+
// finish-protocol boilerplate is concatenated onto `prompt` below — mirrors
|
|
4195
|
+
// notifyOriginatingTab's epicId resolution order (sourcePromptId, then
|
|
4196
|
+
// sourceTabId, then job.epicId) so the same Epic's user/assistant turns
|
|
4197
|
+
// land in the same transcript file. Best-effort: never blocks a spawn.
|
|
4198
|
+
if (!resumeTarget && parsedPrdMeta) {
|
|
4199
|
+
const epicIdForPromptTurn = parsedPrdMeta.sourcePromptId || parsedPrdMeta.sourceTabId || job.epicId || null;
|
|
4200
|
+
if (epicIdForPromptTurn && cwd) {
|
|
4201
|
+
try {
|
|
4202
|
+
await promptSessionTranscript.appendTurn(cwd, epicIdForPromptTurn, {
|
|
4203
|
+
role: 'user',
|
|
4204
|
+
text: prompt,
|
|
4205
|
+
eventId: `prd:${job.slug}`,
|
|
4206
|
+
});
|
|
4207
|
+
} catch (e) {
|
|
4208
|
+
safeLog(`[scheduler] transcript append (PRD prompt) failed: ${e?.message ?? e}\n`);
|
|
4209
|
+
}
|
|
4210
|
+
}
|
|
4211
|
+
}
|
|
4212
|
+
|
|
4124
4213
|
let contextDigestApplied = false;
|
|
4125
4214
|
let originSessionId = null;
|
|
4126
4215
|
if (!resumeTarget) {
|
|
@@ -5165,10 +5254,30 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5165
5254
|
: await jobWorktree.createJobWorktree({ cwd: job.cwd || defaultCwd, slug: job.slug });
|
|
5166
5255
|
if (!preflightWorktree.ok && /^worktree cap reached\b/.test(preflightWorktree.reason || '')) {
|
|
5167
5256
|
console.log(`[scheduler] ${job.slug}: deferring — ${preflightWorktree.reason}`);
|
|
5168
|
-
|
|
5257
|
+
// A leaked worktree cap count used to strand a job silently — nothing
|
|
5258
|
+
// but this console.log + a heldReason field the human had to go
|
|
5259
|
+
// looking for (see gitWorktree.cjs's reserveWorktreeSlot: the cap
|
|
5260
|
+
// itself now self-heals, but a GENUINE block, while it lasts, must
|
|
5261
|
+
// still reach a durable, visible channel). Fired only the first time
|
|
5262
|
+
// THIS slug gets held on this exact reason — a job stuck for hours
|
|
5263
|
+
// must not flood opsErrorLog with one line per dispatch attempt.
|
|
5264
|
+
const wasAlreadyFlaggedForThisReason = await mutate((s) => {
|
|
5169
5265
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
5266
|
+
const already = idx >= 0 && s.jobs[idx].heldReason === preflightWorktree.reason;
|
|
5170
5267
|
if (idx >= 0) s.jobs[idx].heldReason = preflightWorktree.reason;
|
|
5268
|
+
return already;
|
|
5171
5269
|
});
|
|
5270
|
+
if (!wasAlreadyFlaggedForThisReason) {
|
|
5271
|
+
try {
|
|
5272
|
+
appendError({
|
|
5273
|
+
cwd: job.cwd || defaultCwd,
|
|
5274
|
+
scope: 'scheduler',
|
|
5275
|
+
level: 'error',
|
|
5276
|
+
message: `${job.slug}: dispatch blocked — ${preflightWorktree.reason}`,
|
|
5277
|
+
meta: { slug: job.slug, reason: preflightWorktree.reason },
|
|
5278
|
+
});
|
|
5279
|
+
} catch { /* durable logging must never break the queue */ }
|
|
5280
|
+
}
|
|
5172
5281
|
await broadcast({ flush: true });
|
|
5173
5282
|
return;
|
|
5174
5283
|
}
|
|
@@ -5290,6 +5399,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5290
5399
|
delete s.jobs[idx].heldReason;
|
|
5291
5400
|
s.jobs[idx].runId = runId;
|
|
5292
5401
|
s.jobs[idx].startedAt = new Date().toISOString();
|
|
5402
|
+
// Dispatch-phase breadcrumb (PRD: dispatch-region diagnostic
|
|
5403
|
+
// breadcrumb) — a transient marker of how far THIS dispatch got,
|
|
5404
|
+
// exactly like heldReason: stamped at each step of the
|
|
5405
|
+
// running-transition → executeJob → onPid region and deleted at
|
|
5406
|
+
// finalize (see the two finalize mutates below) so a
|
|
5407
|
+
// completed/failed row never carries a stale one.
|
|
5408
|
+
s.jobs[idx].dispatchPhase = 'running-stamped';
|
|
5409
|
+
s.jobs[idx].dispatchPhaseAt = s.jobs[idx].startedAt;
|
|
5293
5410
|
dispatchStartedAtMs = Date.parse(s.jobs[idx].startedAt);
|
|
5294
5411
|
if (job.quietMachine === true) {
|
|
5295
5412
|
s.jobs[idx].quietMachine = true;
|
|
@@ -5308,6 +5425,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5308
5425
|
await broadcast({ flush: true });
|
|
5309
5426
|
if (dispatchSkippedAlreadyCompleted) return;
|
|
5310
5427
|
|
|
5428
|
+
await mutate((s) => {
|
|
5429
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
5430
|
+
if (idx >= 0) {
|
|
5431
|
+
s.jobs[idx].dispatchPhase = 'pre-run-git-snapshot';
|
|
5432
|
+
s.jobs[idx].dispatchPhaseAt = new Date().toISOString();
|
|
5433
|
+
}
|
|
5434
|
+
});
|
|
5435
|
+
|
|
5311
5436
|
// Commit-guard baseline: snapshot the working tree BEFORE the run so the
|
|
5312
5437
|
// post-run check flags only paths THIS job left dirty, not pre-existing WIP.
|
|
5313
5438
|
const guardCwd = job.cwd || defaultCwd;
|
|
@@ -5340,6 +5465,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5340
5465
|
s.jobs[idx].guardHeadBefore = guardHeadBefore || null;
|
|
5341
5466
|
if (preRunDirtyPaths.length) s.jobs[idx].preRunDirtyPaths = preRunDirtyPaths;
|
|
5342
5467
|
else delete s.jobs[idx].preRunDirtyPaths;
|
|
5468
|
+
s.jobs[idx].dispatchPhase = 'baseline-persisted';
|
|
5469
|
+
s.jobs[idx].dispatchPhaseAt = new Date().toISOString();
|
|
5343
5470
|
}
|
|
5344
5471
|
});
|
|
5345
5472
|
|
|
@@ -5361,17 +5488,21 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5361
5488
|
console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
|
|
5362
5489
|
} else {
|
|
5363
5490
|
console.log(`[scheduler] ${job.slug}: running in main tree (worktree not used: ${worktree.reason})`);
|
|
5364
|
-
// Surface any degraded-isolation fallback on the job row itself so it's
|
|
5365
|
-
// queryable from the queue instead of console-only — except the
|
|
5366
|
-
// deliberate env-disable flag, which is an intentional opt-out, not a
|
|
5367
|
-
// degradation worth flagging.
|
|
5368
|
-
if (!jobWorktree.isWorktreeDisabled()) {
|
|
5369
|
-
await mutate((s) => {
|
|
5370
|
-
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
5371
|
-
if (idx >= 0) s.jobs[idx].worktreeFallbackReason = worktree.reason;
|
|
5372
|
-
});
|
|
5373
|
-
}
|
|
5374
5491
|
}
|
|
5492
|
+
// dispatchPhase stamp folded into a single unconditional mutate covering
|
|
5493
|
+
// both branches above — the degraded-isolation fallback flag (skipped
|
|
5494
|
+
// for the deliberate env-disable opt-out) rides along in the SAME
|
|
5495
|
+
// mutate rather than a second one.
|
|
5496
|
+
await mutate((s) => {
|
|
5497
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
5498
|
+
if (idx >= 0) {
|
|
5499
|
+
s.jobs[idx].dispatchPhase = 'worktree-resolved';
|
|
5500
|
+
s.jobs[idx].dispatchPhaseAt = new Date().toISOString();
|
|
5501
|
+
if (!worktree.ok && !jobWorktree.isWorktreeDisabled()) {
|
|
5502
|
+
s.jobs[idx].worktreeFallbackReason = worktree.reason;
|
|
5503
|
+
}
|
|
5504
|
+
}
|
|
5505
|
+
});
|
|
5375
5506
|
// Base-tree WIP carried into the worktree (createWorktree, PRD 1094) —
|
|
5376
5507
|
// recorded on the job row so integration can exclude these paths from
|
|
5377
5508
|
// the branch diff below, and so it's queryable from the queue.
|
|
@@ -5422,10 +5553,20 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5422
5553
|
if (idx >= 0) {
|
|
5423
5554
|
s.jobs[idx].sessionId = sessionId;
|
|
5424
5555
|
s.jobs[idx].runtime = { pid, runId, startedAt: s.jobs[idx].startedAt, sessionId, cwd };
|
|
5556
|
+
s.jobs[idx].dispatchPhase = 'spawned';
|
|
5557
|
+
s.jobs[idx].dispatchPhaseAt = new Date().toISOString();
|
|
5425
5558
|
}
|
|
5426
5559
|
});
|
|
5427
5560
|
await broadcast({ flush: true });
|
|
5428
|
-
}, worktree.ok ? worktree.dir : undefined, resumeTarget, foreignWip, launchEnv)
|
|
5561
|
+
}, worktree.ok ? worktree.dir : undefined, resumeTarget, foreignWip, launchEnv, async (phase) => {
|
|
5562
|
+
await mutate((s) => {
|
|
5563
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
5564
|
+
if (idx >= 0) {
|
|
5565
|
+
s.jobs[idx].dispatchPhase = phase;
|
|
5566
|
+
s.jobs[idx].dispatchPhaseAt = new Date().toISOString();
|
|
5567
|
+
}
|
|
5568
|
+
});
|
|
5569
|
+
});
|
|
5429
5570
|
} finally {
|
|
5430
5571
|
if (worktree.ok) {
|
|
5431
5572
|
worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
|
|
@@ -6033,6 +6174,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6033
6174
|
delete s.jobs[i2].sharedTreeGuard;
|
|
6034
6175
|
}
|
|
6035
6176
|
delete s.jobs[i2].runtime;
|
|
6177
|
+
// Dispatch-phase breadcrumb was only ever meant to cover THIS
|
|
6178
|
+
// dispatch's pre-spawn/spawn region — a terminal row must not
|
|
6179
|
+
// carry a stale one into its next dispatch (unlike guardBaseline's
|
|
6180
|
+
// preRunDirtyPaths sibling, which deliberately survives to
|
|
6181
|
+
// history.jsonl — this has no such carry-forward use).
|
|
6182
|
+
delete s.jobs[i2].dispatchPhase;
|
|
6183
|
+
delete s.jobs[i2].dispatchPhaseAt;
|
|
6184
|
+
// A completed/failed/needs_review row is no longer running — its
|
|
6185
|
+
// overrun badge (if any) was this run's outcome and must not
|
|
6186
|
+
// linger onto whatever the next dispatch of this slug does.
|
|
6187
|
+
delete s.jobs[i2].overrun;
|
|
6036
6188
|
// Pre-run baseline no longer needed once this run has finalized —
|
|
6037
6189
|
// its whole purpose (letting THIS finalize compute a truthful
|
|
6038
6190
|
// delta) is done; a fresh one is captured at the next dispatch.
|
|
@@ -6326,6 +6478,78 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6326
6478
|
}
|
|
6327
6479
|
}
|
|
6328
6480
|
|
|
6481
|
+
// Throttle for reclaimTerminalJobOrphansThrottled below — a `git worktree
|
|
6482
|
+
// list` + per-candidate `status`/`merge-base` call per known project cwd is
|
|
6483
|
+
// cheap once, but tickQueue can fire every few seconds; there is no value in
|
|
6484
|
+
// re-running this on every single tick.
|
|
6485
|
+
const TERMINAL_ORPHAN_RECLAIM_INTERVAL_MS = 10 * 60_000; // 10 minutes
|
|
6486
|
+
let lastTerminalOrphanReclaimAt = 0;
|
|
6487
|
+
|
|
6488
|
+
/**
|
|
6489
|
+
* Fire-and-forget, throttled sweep for job-kind worktrees whose owning queue
|
|
6490
|
+
* row has already resolved (completed/failed/skipped) but whose checkout is
|
|
6491
|
+
* still on disk — see gitWorktree.cjs's reclaimTerminalJobOrphans header for
|
|
6492
|
+
* the two independent safety proofs it requires before deleting anything,
|
|
6493
|
+
* and jobWorktreeTerminalOrphanLive.cjs for the liveness gate checked BEFORE
|
|
6494
|
+
* those proofs (PRD 1163: neither proof is itself a liveness check, so a
|
|
6495
|
+
* still-running executor behind a prematurely-terminal row could otherwise
|
|
6496
|
+
* satisfy both).
|
|
6497
|
+
*
|
|
6498
|
+
* Runs once per TERMINAL_ORPHAN_RECLAIM_INTERVAL_MS across every project cwd
|
|
6499
|
+
* this tick's queue state knows about. The `/proc` holder scan
|
|
6500
|
+
* (`listCwdHolders`) is computed ONCE per sweep here and reused for every
|
|
6501
|
+
* candidate across every cwd, rather than re-scanning `/proc` per candidate.
|
|
6502
|
+
* Never throws (each per-cwd call is itself never-throws; this wrapper is
|
|
6503
|
+
* just the throttle + fan-out + queue-row stamping).
|
|
6504
|
+
*/
|
|
6505
|
+
async function reclaimTerminalJobOrphansThrottled(state) {
|
|
6506
|
+
const now = Date.now();
|
|
6507
|
+
if (now - lastTerminalOrphanReclaimAt < TERMINAL_ORPHAN_RECLAIM_INTERVAL_MS) return;
|
|
6508
|
+
lastTerminalOrphanReclaimAt = now;
|
|
6509
|
+
|
|
6510
|
+
const terminalJobsByCwd = new Map();
|
|
6511
|
+
for (const j of state.jobs || []) {
|
|
6512
|
+
if (j.status !== 'completed' && j.status !== 'failed' && j.status !== 'skipped') continue;
|
|
6513
|
+
const cwd = j.cwd || DEFAULT_PROJECT_CWD;
|
|
6514
|
+
if (!terminalJobsByCwd.has(cwd)) terminalJobsByCwd.set(cwd, []);
|
|
6515
|
+
terminalJobsByCwd.get(cwd).push(j);
|
|
6516
|
+
}
|
|
6517
|
+
|
|
6518
|
+
const cwdHolders = gitWorktree.listCwdHolders();
|
|
6519
|
+
// { slug -> reason }, collected across every cwd this sweep touches, so a
|
|
6520
|
+
// single mutate() at the end can stamp every blocked-live row at once.
|
|
6521
|
+
const blockedLive = new Map();
|
|
6522
|
+
|
|
6523
|
+
for (const [cwd, terminalJobs] of terminalJobsByCwd) {
|
|
6524
|
+
const terminalSlugs = new Set(terminalJobs.map((j) => j.slug));
|
|
6525
|
+
const isLive = buildTerminalOrphanIsLive({
|
|
6526
|
+
terminalJobs,
|
|
6527
|
+
claudePidAlive,
|
|
6528
|
+
hasLiveHolder: gitWorktree.hasLiveHolder,
|
|
6529
|
+
cwdHolders,
|
|
6530
|
+
onLive: (slug, reason) => blockedLive.set(slug, reason),
|
|
6531
|
+
});
|
|
6532
|
+
const reclaimed = await jobWorktree.reclaimTerminalJobOrphans({ cwd, terminalSlugs, isLive });
|
|
6533
|
+
if (reclaimed.length) {
|
|
6534
|
+
console.log(`[scheduler] reclaimed ${reclaimed.length} terminal job-worktree orphan(s) in ${cwd}: ${reclaimed.join(', ')}`);
|
|
6535
|
+
}
|
|
6536
|
+
}
|
|
6537
|
+
|
|
6538
|
+
if (blockedLive.size > 0) {
|
|
6539
|
+
const stampedAt = new Date().toISOString();
|
|
6540
|
+
await mutate((s) => {
|
|
6541
|
+
for (const j of s.jobs || []) {
|
|
6542
|
+
if (!blockedLive.has(j.slug)) continue;
|
|
6543
|
+
j.worktreeReclaimBlockedLive = true;
|
|
6544
|
+
j.worktreeReclaimBlockedLiveAt = stampedAt;
|
|
6545
|
+
j.worktreeReclaimBlockedLiveReason = blockedLive.get(j.slug);
|
|
6546
|
+
}
|
|
6547
|
+
}).catch((e) => {
|
|
6548
|
+
console.warn('[scheduler] failed to stamp worktreeReclaimBlockedLive', e?.message);
|
|
6549
|
+
});
|
|
6550
|
+
}
|
|
6551
|
+
}
|
|
6552
|
+
|
|
6329
6553
|
/**
|
|
6330
6554
|
* Dispatch a resume-recovery attempt (PRD 1111) for a job already found
|
|
6331
6555
|
* eligible by selectResumeRecoveryTarget. Thin wrapper around spawnJob —
|
|
@@ -6365,10 +6589,29 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
6365
6589
|
}
|
|
6366
6590
|
if (cancelToken.cancelled) return { fired: false, reason: 'cancelled' };
|
|
6367
6591
|
|
|
6592
|
+
// Stamped here — the moment tickQueue actually reaches the picker,
|
|
6593
|
+
// regardless of whether this pass ends in a launch — so
|
|
6594
|
+
// classifyQueueStarvation can tell "the engine keeps evaluating the
|
|
6595
|
+
// queue" apart from "nothing has invoked tickQueue in a long time".
|
|
6596
|
+
// Distinct from `lastRunAt` below, which stays true to its existing
|
|
6597
|
+
// meaning (a batch actually launched) since other readers depend on that.
|
|
6598
|
+
await mutate((s) => { s.lastDispatchAttemptAt = new Date().toISOString(); });
|
|
6599
|
+
|
|
6368
6600
|
// The retired-flat-dir sweep now lives inside reconcile() itself (see its
|
|
6369
6601
|
// own comment) so every caller of reconcile — not just this tick — gets
|
|
6370
6602
|
// the guarantee.
|
|
6371
6603
|
await reconcile(state);
|
|
6604
|
+
// Reclaim any job-kind worktree whose owning row already resolved
|
|
6605
|
+
// (completed/failed/skipped) without the run ever reaching
|
|
6606
|
+
// cleanupWorktree — a leaked checkout that would otherwise sit until the
|
|
6607
|
+
// stale-age sweep (default 24h) or the next app restart
|
|
6608
|
+
// (reconcileWorktreesOnBoot only runs once, at boot). Throttled to once
|
|
6609
|
+
// per interval (not every tick) since it's a `git worktree list` +
|
|
6610
|
+
// per-candidate git call per known project cwd. Best-effort and
|
|
6611
|
+
// fire-and-forget: must never hold up dispatch.
|
|
6612
|
+
reclaimTerminalJobOrphansThrottled(state).catch((e) => {
|
|
6613
|
+
console.warn('[scheduler] terminal job-worktree orphan reclaim failed', e?.message);
|
|
6614
|
+
});
|
|
6372
6615
|
// Session-Manager's machine-wide slot pool is the ONLY concurrency limit
|
|
6373
6616
|
// the picker answers to (plus the memory gate below). The scheduler used
|
|
6374
6617
|
// to also carry a private `concurrencyCap` of 3 — the exact per-consumer
|
|
@@ -6624,11 +6867,15 @@ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now,
|
|
|
6624
6867
|
* with ready work idle indefinitely.
|
|
6625
6868
|
*/
|
|
6626
6869
|
async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
6870
|
+
// lastDispatchAttemptAt, not lastRunAt: the latter only advances when a
|
|
6871
|
+
// batch actually launches, so a poll that keeps succeeding while dispatch
|
|
6872
|
+
// itself never gets invoked would otherwise mask a stall behind a fresh-
|
|
6873
|
+
// looking timestamp that was never actually tracking dispatch liveness.
|
|
6627
6874
|
const verdict = classifyQueueStarvation({
|
|
6628
6875
|
jobs: state?.jobs,
|
|
6629
6876
|
paused: state?.paused,
|
|
6630
6877
|
runningCount: runningSet.size,
|
|
6631
|
-
lastRunAtMs: Date.parse(state?.
|
|
6878
|
+
lastRunAtMs: Date.parse(state?.lastDispatchAttemptAt ?? ''),
|
|
6632
6879
|
now,
|
|
6633
6880
|
thresholdMs,
|
|
6634
6881
|
});
|
|
@@ -6655,6 +6902,15 @@ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs
|
|
|
6655
6902
|
// returns early on null). Treat unknown as safe here, exactly as the
|
|
6656
6903
|
// billing meter's own 429 fallback already does.
|
|
6657
6904
|
if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
|
|
6905
|
+
// The in-process cancelToken is only ever reset by runDueJobs() (force-tick
|
|
6906
|
+
// / run-now / resume-timer) — every other path that clears a pause
|
|
6907
|
+
// (clearPause(), the poll loop's own auto-recovery) leaves it untouched
|
|
6908
|
+
// (see applyPauseCleared's header for the 2026-07-14 incident this caused).
|
|
6909
|
+
// A watchdog that fires but doesn't clear a stuck cancelToken would forever
|
|
6910
|
+
// hit tickQueue's very first guard and report a forced tick that never
|
|
6911
|
+
// actually ticked. The watchdog is the last line of defence against a
|
|
6912
|
+
// wedged dispatcher, so it must be able to un-wedge this too.
|
|
6913
|
+
cancelToken.cancelled = false;
|
|
6658
6914
|
await tickQueue({ bypassLoadGate: false }).catch((e) => console.error('[scheduler] starvation tick error', e));
|
|
6659
6915
|
return verdict;
|
|
6660
6916
|
}
|
|
@@ -6794,8 +7050,17 @@ async function reapDeadRunningJobs() {
|
|
|
6794
7050
|
// a true claim when the run dir produced no log output at all; a log
|
|
6795
7051
|
// with real content proves the job DID run (see
|
|
6796
7052
|
// resolvePidlessGateOutcome's header).
|
|
6797
|
-
|
|
6798
|
-
|
|
7053
|
+
// A pidless row was stamped with the runId of whatever batch dir
|
|
7054
|
+
// tickQueue handed its dispatch (pickRunDir's header: "tickQueue hands
|
|
7055
|
+
// ONE shared batch dir to every spawnJob in the batch") BEFORE the
|
|
7056
|
+
// spawn that never completed — so that dir may hold nothing of this
|
|
7057
|
+
// slug's own, or only a sibling's `<other-slug>.log` from the same
|
|
7058
|
+
// batch. logHasOutput's own existsSync+size check is exactly "does
|
|
7059
|
+
// THIS slug have any artifact in there" — reused rather than
|
|
7060
|
+
// re-derived so a phantom link never survives the reap.
|
|
7061
|
+
const hasOwnArtifact = pidless ? logHasOutput(logPath) : true;
|
|
7062
|
+
const gateOutcome = pidless ? resolvePidlessGateOutcome(outcome, hasOwnArtifact) : mapOutcomeToGateOutcome(outcome);
|
|
7063
|
+
dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath, noOwnArtifact: pidless && !hasOwnArtifact });
|
|
6799
7064
|
}
|
|
6800
7065
|
|
|
6801
7066
|
queueHealthSweepCycle += 1;
|
|
@@ -6894,7 +7159,7 @@ async function reapDeadRunningJobs() {
|
|
|
6894
7159
|
}
|
|
6895
7160
|
|
|
6896
7161
|
await mutate(async (s) => {
|
|
6897
|
-
for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
|
|
7162
|
+
for (const { slug, pid, outcome, gateOutcome, pidless, reason, noOwnArtifact } of dead) {
|
|
6898
7163
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
6899
7164
|
if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
|
|
6900
7165
|
const rateLimited = outcome === 'rate_limited';
|
|
@@ -6984,7 +7249,22 @@ async function reapDeadRunningJobs() {
|
|
|
6984
7249
|
}
|
|
6985
7250
|
if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
|
|
6986
7251
|
}
|
|
7252
|
+
// A pidless spawn that never wrote its own '<slug>.log' into the
|
|
7253
|
+
// batch runId dir it was stamped with must not keep that runId —
|
|
7254
|
+
// it points at a directory with no artifacts for THIS slug (at best
|
|
7255
|
+
// empty, at worst only a sibling's log from the same batch — see the
|
|
7256
|
+
// 'dead' push above). resolveRunId's own backfill scan
|
|
7257
|
+
// (existsSync-per-dir on '<slug>.log') would find nothing here
|
|
7258
|
+
// either, so clearing this makes job.runId and resolveRunId(job)
|
|
7259
|
+
// agree: both null, rather than one falsely pointing at a run this
|
|
7260
|
+
// slug never produced.
|
|
7261
|
+
if (!rateLimited && noOwnArtifact) {
|
|
7262
|
+
s.jobs[idx].runId = null;
|
|
7263
|
+
}
|
|
6987
7264
|
delete s.jobs[idx].runtime;
|
|
7265
|
+
delete s.jobs[idx].dispatchPhase;
|
|
7266
|
+
delete s.jobs[idx].dispatchPhaseAt;
|
|
7267
|
+
delete s.jobs[idx].overrun;
|
|
6988
7268
|
delete s.jobs[idx].guardBaseline;
|
|
6989
7269
|
delete s.jobs[idx].guardHeadBefore;
|
|
6990
7270
|
applyLeftoverFields(s.jobs[idx], deltaPaths);
|
|
@@ -7046,6 +7326,7 @@ async function pollLoop() {
|
|
|
7046
7326
|
firstFailureAt = null;
|
|
7047
7327
|
firstNon429FailureAt = null;
|
|
7048
7328
|
lastFailureKind = null;
|
|
7329
|
+
failureStreakWarned = false;
|
|
7049
7330
|
lastPollAt = Date.now();
|
|
7050
7331
|
lastPollOk = true;
|
|
7051
7332
|
persistSchedulerState();
|
|
@@ -7072,6 +7353,7 @@ async function pollLoop() {
|
|
|
7072
7353
|
firstFailureAt = null;
|
|
7073
7354
|
firstNon429FailureAt = null;
|
|
7074
7355
|
lastFailureKind = null;
|
|
7356
|
+
failureStreakWarned = false;
|
|
7075
7357
|
lastPollAt = Date.now();
|
|
7076
7358
|
lastPollOk = true;
|
|
7077
7359
|
persistSchedulerState();
|
|
@@ -7089,13 +7371,23 @@ async function pollLoop() {
|
|
|
7089
7371
|
} else if (r.kind === 'meter_rate_limited') {
|
|
7090
7372
|
// Billing meter is itself being rate-limited. Treat as "utilization unknown but safe":
|
|
7091
7373
|
// fire available jobs anyway at utilization=0 rather than pausing the queue.
|
|
7374
|
+
// Still back off the POLL cadence itself (same curve/cap as the transient
|
|
7375
|
+
// branch) and persist state every cycle — without this, a sustained 429
|
|
7376
|
+
// streak hammered the already-rate-limited endpoint every POLL_INTERVAL_MS
|
|
7377
|
+
// forever AND never wrote lastPollAt/consecutiveFailures back to
|
|
7378
|
+
// scheduler-state.json, so the sidecar froze stale while the loop kept
|
|
7379
|
+
// failing silently underneath it (the 57-consecutive-failure incident).
|
|
7092
7380
|
lastPollAt = Date.now();
|
|
7093
7381
|
lastPollOk = false;
|
|
7094
7382
|
consecutiveFailures++;
|
|
7095
7383
|
lastFailureKind = 'meter_rate_limited';
|
|
7096
7384
|
// Don't update firstNon429FailureAt — 429s don't count toward the 30-min network-pause threshold.
|
|
7385
|
+
backoffMs = nextBackoffMs(backoffMs);
|
|
7386
|
+
backoffNextAt = Date.now() + backoffMs;
|
|
7097
7387
|
cachedUtilization = 0; // assume safe; fire any pending work
|
|
7098
|
-
console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on heuristic (failure #${consecutiveFailures})`);
|
|
7388
|
+
console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on heuristic (failure #${consecutiveFailures}); retry in ${backoffMs / 1000}s`);
|
|
7389
|
+
warnFailureStreakIfNeeded();
|
|
7390
|
+
persistSchedulerState();
|
|
7099
7391
|
const cur = await readQueue();
|
|
7100
7392
|
await maybeLaunchWhenAvailable(cur);
|
|
7101
7393
|
await broadcast();
|
|
@@ -7113,7 +7405,7 @@ async function pollLoop() {
|
|
|
7113
7405
|
// transient or config — apply exponential backoff and count toward 30-min threshold.
|
|
7114
7406
|
lastFailureKind = 'transient';
|
|
7115
7407
|
if (!firstNon429FailureAt) firstNon429FailureAt = Date.now();
|
|
7116
|
-
backoffMs =
|
|
7408
|
+
backoffMs = nextBackoffMs(backoffMs);
|
|
7117
7409
|
const totalNon429FailureMs = Date.now() - firstNon429FailureAt;
|
|
7118
7410
|
console.log(`[scheduler] transient failure #${consecutiveFailures}: ${r.kind} ${r.message ?? ''}; retry in ${backoffMs / 1000}s`);
|
|
7119
7411
|
|
|
@@ -7127,7 +7419,21 @@ async function pollLoop() {
|
|
|
7127
7419
|
}
|
|
7128
7420
|
|
|
7129
7421
|
backoffNextAt = Date.now() + backoffMs;
|
|
7422
|
+
warnFailureStreakIfNeeded();
|
|
7130
7423
|
persistSchedulerState();
|
|
7424
|
+
// A failed billing poll must not silently stop dispatch — only the
|
|
7425
|
+
// 'ok' and 'meter_rate_limited' branches used to reach
|
|
7426
|
+
// maybeLaunchWhenAvailable, so auth/transient failures left ready
|
|
7427
|
+
// pending work untouched until either the queue-starvation watchdog's
|
|
7428
|
+
// 10-minute safety net fired or the poll itself recovered. Utilization
|
|
7429
|
+
// is unknown during a failed poll, not unsafe — treated the same way
|
|
7430
|
+
// the meter_rate_limited branch above already treats a 429 as safe to
|
|
7431
|
+
// fire through. maybeLaunchWhenAvailable itself still honors an
|
|
7432
|
+
// 'auth'/'network' pause (state.paused), so this is a no-op whenever
|
|
7433
|
+
// setPaused() above actually engaged one.
|
|
7434
|
+
if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
|
|
7435
|
+
await maybeLaunchWhenAvailable(await readQueue());
|
|
7436
|
+
await broadcast();
|
|
7131
7437
|
}
|
|
7132
7438
|
} catch (e) {
|
|
7133
7439
|
// Unexpected error (e.g., IPC transport failure)
|
|
@@ -7137,9 +7443,17 @@ async function pollLoop() {
|
|
|
7137
7443
|
lastFailureKind = 'transient';
|
|
7138
7444
|
if (!firstFailureAt) firstFailureAt = Date.now();
|
|
7139
7445
|
if (!firstNon429FailureAt) firstNon429FailureAt = Date.now();
|
|
7140
|
-
backoffMs =
|
|
7446
|
+
backoffMs = nextBackoffMs(backoffMs);
|
|
7141
7447
|
backoffNextAt = Date.now() + backoffMs;
|
|
7448
|
+
warnFailureStreakIfNeeded();
|
|
7142
7449
|
persistSchedulerState();
|
|
7450
|
+
// Same rationale as the auth/transient branch above: the outer catch
|
|
7451
|
+
// must not be a silent dispatch dead-end either.
|
|
7452
|
+
try {
|
|
7453
|
+
if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
|
|
7454
|
+
await maybeLaunchWhenAvailable(await readQueue());
|
|
7455
|
+
await broadcast();
|
|
7456
|
+
} catch { /* best-effort — the poll loop must still re-arm below */ }
|
|
7143
7457
|
} finally {
|
|
7144
7458
|
const delay = backoffMs || POLL_INTERVAL_MS;
|
|
7145
7459
|
pollLoopTimer = setTimeout(() => { pollLoop().catch(() => {}); }, delay);
|
|
@@ -7422,13 +7736,90 @@ function isRescanCandidate(job) {
|
|
|
7422
7736
|
* the unguarded boot call, could reach it). Reported 2026-09-10 from
|
|
7423
7737
|
* social-signals-trader.
|
|
7424
7738
|
*
|
|
7425
|
-
*
|
|
7426
|
-
*
|
|
7427
|
-
*
|
|
7428
|
-
*
|
|
7739
|
+
* Reopened through a different door 2026-09-12 (starry-night-ships
|
|
7740
|
+
* 231-saturn-record-and-docs / 243-neptune-kurama-mode): the guard above only
|
|
7741
|
+
* covered isRescanCandidate (re-verification), but reverifyNeedsReview's auto-
|
|
7742
|
+
* fix loop ALSO lives behind this same guard, and a `needs_review` row whose
|
|
7743
|
+
* mechanical recovery already ran and failed (mechanicalRecoveryAttempted:
|
|
7744
|
+
* true, verdict still 'worktree_integration_failed', not itself a
|
|
7745
|
+
* RESCANNABLE_VERDICTS member) is invisible to isRescanCandidate — so the
|
|
7746
|
+
* next rung (a fix-plan investigation via selectAutoFixTargets) never got a
|
|
7747
|
+
* chance to fire either. Widened to OR in every live target of the recovery
|
|
7748
|
+
* ladder the periodic pass actually drives (selectMechanicalRecoveryTarget /
|
|
7749
|
+
* selectResumeRecoveryTarget / selectAutoFixTargets) so the guard can never
|
|
7750
|
+
* again be narrower than the work reverifyNeedsReview performs.
|
|
7751
|
+
*
|
|
7752
|
+
* Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget are pure
|
|
7753
|
+
* (no I/O). selectAutoFixTargets is called with an injected fixSlugExists
|
|
7754
|
+
* that always returns false — cheap and deliberately over-inclusive (a false
|
|
7755
|
+
* positive here just means one extra periodic pass, never a missed one) so
|
|
7756
|
+
* this guard never pays selectAutoFixTargets's production fs.existsSync scan
|
|
7757
|
+
* per tick. resolveRunId's IO only fires for rows missing job.runId, same as
|
|
7758
|
+
* isRescanCandidate already incurs above.
|
|
7429
7759
|
*/
|
|
7430
7760
|
function shouldRunPeriodicReverify(jobs) {
|
|
7431
|
-
|
|
7761
|
+
if (!Array.isArray(jobs)) return false;
|
|
7762
|
+
if (jobs.some((j) => isRescanCandidate(j))) return true;
|
|
7763
|
+
if (jobs.some((j) => selectMechanicalRecoveryTarget(j) || selectResumeRecoveryTarget(j))) return true;
|
|
7764
|
+
return selectAutoFixTargets(jobs, { fixSlugExists: () => false }).length > 0;
|
|
7765
|
+
}
|
|
7766
|
+
|
|
7767
|
+
// Default 24h, overridable via SM_STUCK_FAILED_ESCALATE_HOURS — same
|
|
7768
|
+
// env-override shape as QUARANTINE_ESCALATE_MS above.
|
|
7769
|
+
const STUCK_FAILED_ESCALATE_MS = process.env.SM_STUCK_FAILED_ESCALATE_HOURS
|
|
7770
|
+
? Number(process.env.SM_STUCK_FAILED_ESCALATE_HOURS) * 60 * 60_000
|
|
7771
|
+
: 24 * 60 * 60_000;
|
|
7772
|
+
|
|
7773
|
+
/**
|
|
7774
|
+
* Kill-switch gate for the stuck-failed escalation below
|
|
7775
|
+
* (SM_STUCK_FAILED_ESCALATE_DISABLE=1), mirroring the
|
|
7776
|
+
* SM_REVERIFY_PERIODIC_DISABLE / SM_RCA_DISABLE convention. A tiny wrapper
|
|
7777
|
+
* so the disable path is unit-testable without invoking the 10-minute
|
|
7778
|
+
* setInterval body directly.
|
|
7779
|
+
*/
|
|
7780
|
+
function stuckFailedEscalationDisabled() {
|
|
7781
|
+
return process.env.SM_STUCK_FAILED_ESCALATE_DISABLE === '1';
|
|
7782
|
+
}
|
|
7783
|
+
|
|
7784
|
+
/**
|
|
7785
|
+
* findStuckFailedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
|
|
7786
|
+
*
|
|
7787
|
+
* Pure predicate (isRescanCandidate's own resolveRunId/classifyRunOutcome log
|
|
7788
|
+
* read is the only IO, gated per-job exactly like shouldRunPeriodicReverify
|
|
7789
|
+
* above). `failed` is a fully terminal state for every automated recovery
|
|
7790
|
+
* path — selectResumeRecoveryTarget/selectAutoFixTargets both require
|
|
7791
|
+
* needs_review, reapDeadRunningJobs only ever writes running → failed, and
|
|
7792
|
+
* reconcile-repair's to-pending is for structurally invalid rows. Only a
|
|
7793
|
+
* human's scheduler_reset_job ever takes failed → pending (LEGAL_TRANSITIONS).
|
|
7794
|
+
* A rescan candidate (isRescanCandidate) that has sat failed longer than
|
|
7795
|
+
* `thresholdMs` can therefore go silently stuck forever — job
|
|
7796
|
+
* 4056-outcome-stats sat `failed` for five days with no operator signal
|
|
7797
|
+
* (reported 2026-09-10, social-signals-trader) even though the periodic
|
|
7798
|
+
* reverify pass (once shouldRunPeriodicReverify's guard was fixed) WAS firing
|
|
7799
|
+
* on it — reverifyNeedsReview's failed branch can annotate looksDone but can
|
|
7800
|
+
* never resolve a failed row itself (see its own header). This is the
|
|
7801
|
+
* visibility half that guard fix was missing: escalate once, never requeue.
|
|
7802
|
+
*
|
|
7803
|
+
* `stuckFailedNotified` gates this to exactly once per row — once the caller
|
|
7804
|
+
* stamps it, this always excludes that row so a human is never re-paged on
|
|
7805
|
+
* every 10-minute tick for the same stuck job. A row with no recoverable
|
|
7806
|
+
* 'failed' timestamp is skipped rather than guessed at (mirrors
|
|
7807
|
+
* findStaleQuarantinedJobs above).
|
|
7808
|
+
*/
|
|
7809
|
+
function findStuckFailedJobs(jobs, now, thresholdMs) {
|
|
7810
|
+
const stuck = [];
|
|
7811
|
+
for (const j of jobs ?? []) {
|
|
7812
|
+
if (j.status !== 'failed') continue;
|
|
7813
|
+
if (j.stuckFailedNotified === true) continue;
|
|
7814
|
+
if (!isRescanCandidate(j)) continue;
|
|
7815
|
+
const entry = (j.statusHistory || []).find((h) => h.to === 'failed');
|
|
7816
|
+
if (!entry) continue;
|
|
7817
|
+
const since = Date.parse(entry.at);
|
|
7818
|
+
if (Number.isNaN(since)) continue;
|
|
7819
|
+
const ageMs = now - since;
|
|
7820
|
+
if (ageMs >= thresholdMs) stuck.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs });
|
|
7821
|
+
}
|
|
7822
|
+
return stuck;
|
|
7432
7823
|
}
|
|
7433
7824
|
|
|
7434
7825
|
/**
|
|
@@ -8264,7 +8655,18 @@ async function init() {
|
|
|
8264
8655
|
try {
|
|
8265
8656
|
const worktreeCwds = new Set(bootSnap.jobs.map((j) => j.cwd).filter(Boolean));
|
|
8266
8657
|
worktreeCwds.add(DEFAULT_PROJECT_CWD);
|
|
8267
|
-
|
|
8658
|
+
// A job is spawned `detached: true`, so its `claude -p` executor can
|
|
8659
|
+
// survive this very app restart — a worktree found at boot is NOT, by
|
|
8660
|
+
// itself, proof its run already died. isLive checks the already-read
|
|
8661
|
+
// bootSnap (no extra queue read) for a live running-row pid, OR a live
|
|
8662
|
+
// /proc cwd holder under the checkout itself. See jobWorktreeBootLive.cjs.
|
|
8663
|
+
const isLive = buildJobWorktreeIsLive({
|
|
8664
|
+
bootJobs: bootSnap.jobs,
|
|
8665
|
+
claudePidAlive,
|
|
8666
|
+
hasLiveHolder: gitWorktree.hasLiveHolder,
|
|
8667
|
+
cwdHolders: gitWorktree.listCwdHolders(),
|
|
8668
|
+
});
|
|
8669
|
+
await jobWorktree.reconcileWorktreesOnBoot([...worktreeCwds], { isLive });
|
|
8268
8670
|
} catch (e) {
|
|
8269
8671
|
console.error('[scheduler] boot worktree reconciliation failed', e?.message);
|
|
8270
8672
|
}
|
|
@@ -8430,6 +8832,17 @@ async function init() {
|
|
|
8430
8832
|
appendAuditEvent('job_overrunning_estimate', {
|
|
8431
8833
|
slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
|
|
8432
8834
|
});
|
|
8835
|
+
// Durable stamp so schedule:state (and therefore the renderer) can see
|
|
8836
|
+
// this without re-deriving it — the console.warn/audit event above are
|
|
8837
|
+
// visible only in the log, never on the row itself. Display-only
|
|
8838
|
+
// advisory field; re-stamped in place every sweep, never appended.
|
|
8839
|
+
mutate((state) => {
|
|
8840
|
+
const j = state.jobs.find((x) => x.slug === over.slug);
|
|
8841
|
+
if (!j) return;
|
|
8842
|
+
j.overrun = {
|
|
8843
|
+
ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
|
|
8844
|
+
};
|
|
8845
|
+
}).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
|
|
8433
8846
|
}
|
|
8434
8847
|
|
|
8435
8848
|
// Stranded-investigation restore. Unlike the two escalations above, this
|
|
@@ -8474,6 +8887,31 @@ async function init() {
|
|
|
8474
8887
|
);
|
|
8475
8888
|
appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
|
|
8476
8889
|
}
|
|
8890
|
+
|
|
8891
|
+
// Stuck-failed escalation (2026-09-10, social-signals-trader): see
|
|
8892
|
+
// findStuckFailedJobs' header for why `failed` has no automated way
|
|
8893
|
+
// back to pending. Escalation only, same shape as the three warnings
|
|
8894
|
+
// above — never an automatic failed → pending requeue (that could
|
|
8895
|
+
// discard uncommitted work left by the failed run; see
|
|
8896
|
+
// spawnJob:fail-dirty). Kill-switch: SM_STUCK_FAILED_ESCALATE_DISABLE=1.
|
|
8897
|
+
if (!stuckFailedEscalationDisabled()) {
|
|
8898
|
+
const stuckFailed = findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
|
|
8899
|
+
if (stuckFailed.length > 0) {
|
|
8900
|
+
mutate((ms) => {
|
|
8901
|
+
for (const stuck of stuckFailed) {
|
|
8902
|
+
const j = ms.jobs.find((x) => x.slug === stuck.slug);
|
|
8903
|
+
if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
|
|
8904
|
+
j.stuckFailedNotified = true;
|
|
8905
|
+
console.warn(
|
|
8906
|
+
`[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
|
|
8907
|
+
+ `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
8908
|
+
+ `no automated recovery reaches a failed row; reset it by hand via scheduler_reset_job`,
|
|
8909
|
+
);
|
|
8910
|
+
appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
|
|
8911
|
+
}
|
|
8912
|
+
}).catch(() => {});
|
|
8913
|
+
}
|
|
8914
|
+
}
|
|
8477
8915
|
}, 10 * 60_000);
|
|
8478
8916
|
|
|
8479
8917
|
// Self-rescheduling poll loop with exponential backoff. Replaces the
|
|
@@ -9149,6 +9587,11 @@ module.exports = {
|
|
|
9149
9587
|
init,
|
|
9150
9588
|
ROOT,
|
|
9151
9589
|
PRDS_DIR,
|
|
9590
|
+
SCHEDULER_STATE_PATH,
|
|
9591
|
+
BACKOFF_MAX_MS,
|
|
9592
|
+
FAILURE_STREAK_WARN_THRESHOLD,
|
|
9593
|
+
nextBackoffMs,
|
|
9594
|
+
shouldWarnFailureStreak,
|
|
9152
9595
|
healRefusalReason,
|
|
9153
9596
|
writeQueue,
|
|
9154
9597
|
reconcile,
|
|
@@ -9170,6 +9613,9 @@ module.exports = {
|
|
|
9170
9613
|
availableForJobs,
|
|
9171
9614
|
reverifyNeedsReview,
|
|
9172
9615
|
shouldRunPeriodicReverify,
|
|
9616
|
+
findStuckFailedJobs,
|
|
9617
|
+
STUCK_FAILED_ESCALATE_MS,
|
|
9618
|
+
stuckFailedEscalationDisabled,
|
|
9173
9619
|
isRescanCandidate,
|
|
9174
9620
|
isFailedUnverifiedShaped,
|
|
9175
9621
|
computeLooksDone,
|
|
@@ -9281,6 +9727,8 @@ module.exports = {
|
|
|
9281
9727
|
clearPause,
|
|
9282
9728
|
tickQueue,
|
|
9283
9729
|
runDueJobs,
|
|
9730
|
+
pollLoop,
|
|
9731
|
+
maybeLaunchWhenAvailable,
|
|
9284
9732
|
isCooldownSuppressed,
|
|
9285
9733
|
nextRapidRateLimitCount,
|
|
9286
9734
|
CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
|