claude-code-session-manager 0.75.3 → 0.76.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-CzQqcObq.js → AgentLibrary-CBx9l4zN.js} +1 -1
- package/dist/assets/{DataModel-Bj_WlLz8.js → DataModel-Bf0EIE_t.js} +1 -1
- package/dist/assets/{History-DnSi_OHm.js → History-CpdtWhC8.js} +1 -1
- package/dist/assets/{Hooks-0BB0dp3S.js → Hooks-DyUbMDmg.js} +1 -1
- package/dist/assets/{HostBilko-DHpwwsLQ.js → HostBilko-By-wIpry.js} +1 -1
- package/dist/assets/{Library-CaJVqVvi.js → Library-CQmo4QVC.js} +1 -1
- package/dist/assets/{ListDetail-C1W2HmC2.js → ListDetail-BQMd6NOm.js} +1 -1
- package/dist/assets/{MarkdownEditor-5Ob9FW3z.js → MarkdownEditor-DEp43FXX.js} +1 -1
- package/dist/assets/{McpServers-JxCSfm1S.js → McpServers-CLarzwqA.js} +1 -1
- package/dist/assets/{Memory-BDeqlqwH.js → Memory-B0sCdIy1.js} +1 -1
- package/dist/assets/{Panel-Dh9ZHuEj.js → Panel-BhWPVOCD.js} +1 -1
- package/dist/assets/{Permissions-DXy-CbEY.js → Permissions-Ddlq8T_O.js} +1 -1
- package/dist/assets/{Plugins-_n1Iuc8T.js → Plugins-D2oA_2Jl.js} +2 -2
- package/dist/assets/{ProvenanceBadge-BP_evfxE.js → ProvenanceBadge-DgAgavUM.js} +1 -1
- package/dist/assets/{SaveBar-D-gCUx4n.js → SaveBar-Qvc4Ek-H.js} +1 -1
- package/dist/assets/{Scheduler-Bpd4OGju.js → Scheduler-BmYJvNzK.js} +1 -1
- package/dist/assets/{ScopeSwitcher-CAWzM6RI.js → ScopeSwitcher-C_zWEtIl.js} +1 -1
- package/dist/assets/{Settings-DRRozLyT.js → Settings-2Vx3X5SI.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-DGHDWlz4.js → SkillReferenceGraph-BDEUjlTQ.js} +1 -1
- package/dist/assets/{Skills-D8L66eiX.js → Skills-Cmrz_LeN.js} +1 -1
- package/dist/assets/{SystemPrompt-CYtUsonD.js → SystemPrompt-DVA1eYDP.js} +1 -1
- package/dist/assets/{TagLibrary-E5CLeuVk.js → TagLibrary-DYJGAKZu.js} +1 -1
- package/dist/assets/{TiptapBody-B2hRgbPE.js → TiptapBody-DmPc3amD.js} +1 -1
- package/dist/assets/{Toggle-BTwsbxam.js → Toggle-zfd5LJkK.js} +1 -1
- package/dist/assets/{index-DijufvkJ.js → index-B_4PNh9T.js} +676 -676
- package/dist/assets/{index-CMLnzdZC.css → index-DIjnPkRN.css} +1 -1
- package/dist/assets/{settingsSchema-D6wzxAi6.js → settingsSchema-B9es6fdA.js} +1 -1
- package/dist/index.html +2 -2
- package/package.json +1 -1
- package/scripts/lib/activeSessions.cjs +116 -6
- package/scripts/scheduler-mcp-server.cjs +154 -95
- package/src/main/__tests__/epicStatusMirror.test.cjs +110 -0
- package/src/main/__tests__/health-delegation-chain.test.cjs +105 -0
- package/src/main/__tests__/prdAdminRoutes.test.cjs +295 -0
- package/src/main/__tests__/prdCreate.test.cjs +109 -0
- package/src/main/__tests__/scheduler-autofix-select.test.cjs +15 -3
- package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +41 -0
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +60 -1
- package/src/main/__tests__/scheduler-stranded-investigation.test.cjs +185 -0
- package/src/main/__tests__/seedSchedulerMcp.test.cjs +66 -0
- package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -5
- package/src/main/bilkoHost.cjs +4 -3
- package/src/main/chatRunner.cjs +6 -1
- package/src/main/config.cjs +22 -33
- package/src/main/health.cjs +153 -2
- package/src/main/index.cjs +56 -4
- package/src/main/ipcSchemas.cjs +18 -1
- package/src/main/lib/__tests__/activeIndexRebuild.test.cjs +179 -0
- package/src/main/lib/__tests__/childWithLog.test.cjs +63 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +241 -42
- package/src/main/lib/__tests__/ephemeralCwd.test.cjs +91 -0
- package/src/main/lib/__tests__/epicWorktreeMint.test.cjs +1 -1
- package/src/main/lib/__tests__/gitWorktree.test.cjs +14 -2
- package/src/main/lib/__tests__/gitWorktreeSalvage.test.cjs +107 -0
- package/src/main/lib/__tests__/jobWorktree.test.cjs +2 -2
- package/src/main/lib/__tests__/loadGate.test.cjs +159 -0
- package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +101 -0
- package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +151 -0
- package/src/main/lib/__tests__/opsRootResolve.test.cjs +149 -0
- package/src/main/lib/__tests__/projectRootResolve.test.cjs +148 -0
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +112 -0
- package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +19 -9
- package/src/main/lib/__tests__/schedulerBatchFairness.test.cjs +213 -0
- package/src/main/lib/__tests__/schedulerBatchProjectCap.test.cjs +127 -0
- package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +217 -0
- package/src/main/lib/activeIndexMerge.cjs +15 -0
- package/src/main/lib/activeIndexRebuild.cjs +133 -0
- package/src/main/lib/buildTarget.cjs +3 -2
- package/src/main/lib/childWithLog.cjs +33 -1
- package/src/main/lib/crossProjectFeedback.cjs +8 -1
- package/src/main/lib/delegationReadiness.cjs +408 -26
- package/src/main/lib/ephemeralCwd.cjs +78 -0
- package/src/main/lib/epicDelegationStats.cjs +2 -1
- package/src/main/lib/epicMint.cjs +17 -1
- package/src/main/lib/epicStatusMirror.cjs +95 -0
- package/src/main/lib/epicValidationHook.cjs +2 -1
- package/src/main/lib/gitWorktree.cjs +56 -2
- package/src/main/lib/jobWorktree.cjs +1 -0
- package/src/main/lib/loadGate.cjs +134 -0
- package/src/main/lib/mcpToolCatalog.cjs +285 -0
- package/src/main/lib/opsErrorLog.cjs +12 -1
- package/src/main/lib/opsOwnership.cjs +94 -0
- package/src/main/lib/prdAdminRoutes.cjs +43 -3
- package/src/main/lib/prdCreate.cjs +46 -14
- package/src/main/lib/prdLocations.cjs +13 -6
- package/src/main/lib/projectRootResolve.cjs +134 -0
- package/src/main/lib/promptSessionSchema.cjs +7 -0
- package/src/main/lib/queueStore.cjs +31 -5
- package/src/main/lib/rcaReport.cjs +1 -1
- package/src/main/lib/reaperHelpers.cjs +47 -1
- package/src/main/lib/schedulerBatch.cjs +171 -29
- package/src/main/lib/schedulerConfig.cjs +80 -0
- package/src/main/projectBrief.cjs +3 -2
- package/src/main/projectPages.cjs +2 -1
- package/src/main/promptSessionTranscript.cjs +0 -0
- package/src/main/pty.cjs +5 -0
- package/src/main/queueOps.cjs +15 -8
- package/src/main/scheduler.cjs +346 -49
- package/src/main/seedSchedulerMcp.cjs +58 -4
- package/src/preload/api.d.ts +69 -1
- package/src/preload/index.cjs +2 -0
package/src/main/scheduler.cjs
CHANGED
|
@@ -55,7 +55,8 @@ const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
|
|
|
55
55
|
const supervisor = require('./supervisor.cjs');
|
|
56
56
|
const { resolveClaudeBin } = require('./lib/claudeBin.cjs');
|
|
57
57
|
const { readTail } = require('./lib/fileTail.cjs');
|
|
58
|
-
const { claudePidAlive, classifyRunOutcome, ORPHAN_REQUEUE_CAP } = require('./lib/reaperHelpers.cjs');
|
|
58
|
+
const { claudePidAlive, classifyRunOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
|
|
59
|
+
const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
|
|
59
60
|
const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
|
|
60
61
|
const { sendIfAlive } = require('./lib/sendToRenderer.cjs');
|
|
61
62
|
const { createBroadcastCoalescer } = require('./lib/broadcastCoalescer.cjs');
|
|
@@ -68,7 +69,7 @@ const promptSessionTranscript = require('./promptSessionTranscript.cjs');
|
|
|
68
69
|
const { verifyRun } = require('./runVerify.cjs');
|
|
69
70
|
const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
|
|
70
71
|
const logs = require('./logs.cjs');
|
|
71
|
-
const { schemas, validated } = require('./ipcSchemas.cjs');
|
|
72
|
+
const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
|
|
72
73
|
const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
|
|
73
74
|
const {
|
|
74
75
|
POLL_INTERVAL_MS,
|
|
@@ -78,6 +79,9 @@ const {
|
|
|
78
79
|
QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
|
|
79
80
|
JOB_OVERRUN_FACTOR: JOB_OVERRUN_FACTOR_DEFAULT,
|
|
80
81
|
JOB_OVERRUN_FLOOR_MS: JOB_OVERRUN_FLOOR_MS_DEFAULT,
|
|
82
|
+
PIDLESS_SPAWN_GRACE_MS,
|
|
83
|
+
INVESTIGATION_MAX_MS,
|
|
84
|
+
STARVATION_ESCALATE_MS,
|
|
81
85
|
} = require('./lib/schedulerConfig.cjs');
|
|
82
86
|
const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
|
|
83
87
|
? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
|
|
@@ -88,7 +92,7 @@ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
|
|
|
88
92
|
const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
|
|
89
93
|
? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
|
|
90
94
|
: JOB_OVERRUN_FLOOR_MS_DEFAULT;
|
|
91
|
-
const { pickForProject, pickNextBatch, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
|
|
95
|
+
const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
|
|
92
96
|
const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
|
|
93
97
|
const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
|
|
94
98
|
const queueHistory = require('./lib/queueHistory.cjs');
|
|
@@ -100,7 +104,7 @@ const queueOps = require('./queueOps.cjs');
|
|
|
100
104
|
// home-dir layout.
|
|
101
105
|
const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
|
|
102
106
|
const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
|
|
103
|
-
const { transitionJob, STATUS_HISTORY_CAP } = require('./lib/scheduleJobTransitions.cjs');
|
|
107
|
+
const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
|
|
104
108
|
const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
|
|
105
109
|
const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
|
|
106
110
|
const { appendAuditEvent } = require('./lib/auditLog.cjs');
|
|
@@ -205,6 +209,18 @@ const FINISH_PROTOCOL = `
|
|
|
205
209
|
Once every acceptance-criteria line above is satisfied, finish in this EXACT
|
|
206
210
|
sequence. Do not stop before the commit lands; committing is part of the job.
|
|
207
211
|
|
|
212
|
+
RUN VERIFICATION IN THE FOREGROUND — this applies to the whole run, not just
|
|
213
|
+
step 3 below: every test/typecheck/lint/build command you run, whether while
|
|
214
|
+
implementing the AC or during VERIFY, must run SYNCHRONOUSLY and you must wait
|
|
215
|
+
for it to return. Never start a verification command as a background task
|
|
216
|
+
(no background Bash) and then call Monitor, TaskOutput, or ScheduleWakeup to
|
|
217
|
+
pick up its result later — a headless \`claude -p\` run has no later turn, so
|
|
218
|
+
nothing ever delivers that notification and the run dies mid-verification with
|
|
219
|
+
no commit and no verdict. For a long-running command, bound it yourself with
|
|
220
|
+
the shell (e.g. \`timeout 300 npm test\`) and a matching foreground tool
|
|
221
|
+
timeout; if it still cannot finish inside budget, stop and emit
|
|
222
|
+
SCHEDULER_VERDICT: FAIL with the reason instead of deferring it.
|
|
223
|
+
|
|
208
224
|
1. CODE REVIEW — run \`/code-review --fix\` on your changes and apply the fixes it
|
|
209
225
|
surfaces (correctness first). For any finding you judge a false positive, say
|
|
210
226
|
why in your result; do not silently skip it. If \`/code-review\` is not
|
|
@@ -690,6 +706,43 @@ async function safeSlugPath(slug) {
|
|
|
690
706
|
return safeSlugPathIn(dir, slug);
|
|
691
707
|
}
|
|
692
708
|
|
|
709
|
+
/**
|
|
710
|
+
* The two distinct failure modes safeSlugPath collapses into one nullable
|
|
711
|
+
* return (the defect this fixes — see the PRD that added this helper's
|
|
712
|
+
* Goal): a slug that fails SCHEDULE_SLUG_RE is a caller mistake ("invalid
|
|
713
|
+
* slug"), while a well-formed slug that exists in no candidate PRD dir is a
|
|
714
|
+
* lookup miss ("unknown slug") — an agent retrying the first as if it were
|
|
715
|
+
* the second (or vice versa) burns a turn on the wrong fix. Returns
|
|
716
|
+
* `{ ok: true, path }` or `{ ok: false, reason: 'invalid-slug' | 'not-found' }`.
|
|
717
|
+
* `cwd`, if given, narrows the search to that one project's own PRD dirs
|
|
718
|
+
* (prdDirForCwd + its Epic-scoped dirs — same pattern as getPrdParsed);
|
|
719
|
+
* omitted, it searches every candidate dir machine-wide via findPrdDir.
|
|
720
|
+
*/
|
|
721
|
+
async function resolveSlugOrReason(slug, cwd) {
|
|
722
|
+
if (!SCHEDULE_SLUG_RE.test(slug)) return { ok: false, reason: 'invalid-slug' };
|
|
723
|
+
if (cwd) {
|
|
724
|
+
for (const dir of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
|
|
725
|
+
const p = safeSlugPathIn(dir, slug);
|
|
726
|
+
if (!p) continue;
|
|
727
|
+
try {
|
|
728
|
+
await fsp.access(p);
|
|
729
|
+
return { ok: true, path: p };
|
|
730
|
+
} catch { /* not in this dir — try the next candidate */ }
|
|
731
|
+
}
|
|
732
|
+
return { ok: false, reason: 'not-found' };
|
|
733
|
+
}
|
|
734
|
+
const dir = await findPrdDir(slug);
|
|
735
|
+
if (!dir) return { ok: false, reason: 'not-found' };
|
|
736
|
+
const p = safeSlugPathIn(dir, slug);
|
|
737
|
+
if (!p) return { ok: false, reason: 'not-found' };
|
|
738
|
+
return { ok: true, path: p };
|
|
739
|
+
}
|
|
740
|
+
|
|
741
|
+
/** Actionable message for `resolveSlugOrReason`'s 'not-found' reason. */
|
|
742
|
+
function unknownSlugMessage(slug) {
|
|
743
|
+
return `unknown slug "${slug}": no PRD file with that name in any known project — call scheduler_list_prds (optionally with cwd) to see what exists`;
|
|
744
|
+
}
|
|
745
|
+
|
|
693
746
|
/**
|
|
694
747
|
* Move a completed job's `<slug>.md` out of its PRD dir into that dir's
|
|
695
748
|
* sibling `prds-archived/`, so a finished slug can't be re-fired by the
|
|
@@ -1098,6 +1151,73 @@ function findOverrunningJobs(jobs, now, { factor, floorMs } = {}) {
|
|
|
1098
1151
|
return out;
|
|
1099
1152
|
}
|
|
1100
1153
|
|
|
1154
|
+
/**
|
|
1155
|
+
* findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
|
|
1156
|
+
* → [{ slug, cwd, ageMs, restoreStatus }]
|
|
1157
|
+
*
|
|
1158
|
+
* Pure (besides the warn-log side effect on the two unprovable-age cases
|
|
1159
|
+
* below), no other IO. spawnInvestigation's own restore of a job's
|
|
1160
|
+
* pre-investigation status runs entirely inside the process that spawned the
|
|
1161
|
+
* probe (its withChildAndLog onExit handler, or the synchronous-throw catch
|
|
1162
|
+
* path) — so a job left 'investigating' when the app itself dies or restarts
|
|
1163
|
+
* has NOTHING left to restore it. The comment at spawnInvestigation's onExit
|
|
1164
|
+
* asserts "'investigating' must never be the job's resting state"; this is
|
|
1165
|
+
* the sweep that makes that true across a restart, not just within one.
|
|
1166
|
+
*
|
|
1167
|
+
* A row qualifies only when ALL of:
|
|
1168
|
+
* - status is 'investigating'
|
|
1169
|
+
* - its most recent transition INTO 'investigating' (statusHistory's last
|
|
1170
|
+
* `to === 'investigating'` entry — a job can be investigated more than
|
|
1171
|
+
* once across its life, e.g. a retried auto-fix) is older than `maxMs`
|
|
1172
|
+
* - it has no live probe process behind it (checked via runtime.pid, set by
|
|
1173
|
+
* spawnInvestigation once its child spawns and cleared on every restore
|
|
1174
|
+
* path, the same shape reapDeadRunningJobs already uses for 'running' rows)
|
|
1175
|
+
*
|
|
1176
|
+
* `restoreStatus` is that transition entry's `from` — the exact value
|
|
1177
|
+
* spawnInvestigation itself would have restored to (`failedJob.status ||
|
|
1178
|
+
* 'failed'`), which for a row that already finished and recorded
|
|
1179
|
+
* finishedAt+exitCode (the burrow-834 shape) is whatever terminal status was
|
|
1180
|
+
* computed for that outcome BEFORE the probe was spawned — this sweep never
|
|
1181
|
+
* re-derives it from exitCode, only replays the already-recorded decision.
|
|
1182
|
+
*
|
|
1183
|
+
* A row with no recoverable transition timestamp cannot have its age proven,
|
|
1184
|
+
* so it is warn-logged and left alone rather than guessed at — same posture
|
|
1185
|
+
* as findStaleQuarantinedJobs/findOverrunningJobs.
|
|
1186
|
+
*
|
|
1187
|
+
* `restoreStatus` is validated against LEGAL_TRANSITIONS['investigating']
|
|
1188
|
+
* before being returned — `statusHistory`'s `from` should only ever be
|
|
1189
|
+
* 'failed' or 'needs_review' (the only two states LEGAL_TRANSITIONS allows
|
|
1190
|
+
* into 'investigating'), but a corrupted/unexpected value must not be handed
|
|
1191
|
+
* straight to transitionJob: an illegal target is refused outright (row stays
|
|
1192
|
+
* stuck at 'investigating', re-detected as stranded every sweep with no path
|
|
1193
|
+
* out), so an out-of-set `from` falls back to 'failed' here instead.
|
|
1194
|
+
*/
|
|
1195
|
+
const INVESTIGATING_RESTORE_TARGETS = new Set(LEGAL_TRANSITIONS.investigating);
|
|
1196
|
+
function findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive) {
|
|
1197
|
+
const out = [];
|
|
1198
|
+
for (const j of jobs ?? []) {
|
|
1199
|
+
if (j.status !== 'investigating') continue;
|
|
1200
|
+
const entries = (j.statusHistory || []).filter((h) => h.to === 'investigating');
|
|
1201
|
+
const entry = entries[entries.length - 1];
|
|
1202
|
+
if (!entry) {
|
|
1203
|
+
console.warn(`[scheduler] findStrandedInvestigations: ${j.slug} is 'investigating' with no statusHistory entry recording the transition — cannot prove age, leaving alone`);
|
|
1204
|
+
continue;
|
|
1205
|
+
}
|
|
1206
|
+
const since = Date.parse(entry.at ?? '');
|
|
1207
|
+
if (Number.isNaN(since)) {
|
|
1208
|
+
console.warn(`[scheduler] findStrandedInvestigations: ${j.slug} has an unparseable investigating-transition timestamp (${entry.at}) — cannot prove age, leaving alone`);
|
|
1209
|
+
continue;
|
|
1210
|
+
}
|
|
1211
|
+
const ageMs = now - since;
|
|
1212
|
+
if (ageMs < maxMs) continue; // a live probe must not be yanked out from under itself
|
|
1213
|
+
const pid = j.runtime?.pid;
|
|
1214
|
+
if (pid && isAlive(pid)) continue; // probe genuinely still running — not stranded
|
|
1215
|
+
const restoreStatus = INVESTIGATING_RESTORE_TARGETS.has(entry.from) ? entry.from : 'failed';
|
|
1216
|
+
out.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, restoreStatus });
|
|
1217
|
+
}
|
|
1218
|
+
return out;
|
|
1219
|
+
}
|
|
1220
|
+
|
|
1101
1221
|
// An empty queue and an unreadable queue are NOT the same thing, and
|
|
1102
1222
|
// conflating them is destructive: reconcile() treats every PRD .md with no
|
|
1103
1223
|
// matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
|
|
@@ -1677,6 +1797,10 @@ async function reconcile(state) {
|
|
|
1677
1797
|
originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
1678
1798
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1679
1799
|
status: 'pending',
|
|
1800
|
+
// Enqueue time (PRD 1086/1087): the cross-project fairness tiebreak and
|
|
1801
|
+
// the starvation escalation both need a provable age for a pending row;
|
|
1802
|
+
// before this stamp a freshly minted row carried no timestamp at all.
|
|
1803
|
+
queuedAt: new Date().toISOString(),
|
|
1680
1804
|
runId: null,
|
|
1681
1805
|
startedAt: null,
|
|
1682
1806
|
finishedAt: null,
|
|
@@ -1864,6 +1988,9 @@ function drainDeferredInvestigation() {
|
|
|
1864
1988
|
let cancelToken = { cancelled: false };
|
|
1865
1989
|
// Last memory-gate observation; included in snapshot for renderer visibility.
|
|
1866
1990
|
let lastMemGate = null;
|
|
1991
|
+
// CPU-load launch gate (PRD 1085, lib/loadGate.cjs) — innermost launch
|
|
1992
|
+
// predicate after pool → project cap → memory. Withholds launches only.
|
|
1993
|
+
const loadGate = createLoadGate();
|
|
1867
1994
|
|
|
1868
1995
|
// Last tickQueue outcome, kept for the UI. tickQueue already computes a precise
|
|
1869
1996
|
// reason for every way a batch can come back empty (dependency holds, slot
|
|
@@ -1932,6 +2059,8 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
|
|
|
1932
2059
|
lastFailureKind,
|
|
1933
2060
|
},
|
|
1934
2061
|
memGate: lastMemGate,
|
|
2062
|
+
// Why nothing is launching when the box is CPU-saturated (PRD 1085).
|
|
2063
|
+
loadGate: loadGate.snapshot(),
|
|
1935
2064
|
lastTick,
|
|
1936
2065
|
// The machine-wide slot pool IS the concurrency limit — there is no
|
|
1937
2066
|
// separate scheduler cap any more. `source` distinguishes the
|
|
@@ -2609,9 +2738,15 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
|
|
|
2609
2738
|
* tree was dirty — closed by widening that call site's condition, not by
|
|
2610
2739
|
* changing this function's four defenses below, which still apply to both
|
|
2611
2740
|
* shapes identically:
|
|
2612
|
-
* - siblingRunning: a concurrent job in the same
|
|
2613
|
-
* evidence unreliable in both directions (extra
|
|
2614
|
-
* that isn't this job's doing).
|
|
2741
|
+
* - siblingRunning: on a SHARED tree only — a concurrent job in the same
|
|
2742
|
+
* cwd makes working-tree evidence unreliable in both directions (extra
|
|
2743
|
+
* dirt OR a clean tree that isn't this job's doing). Suppressed by
|
|
2744
|
+
* ranInWorktree: when this job ran in its own git worktree, the
|
|
2745
|
+
* newly-dirty set and the integrated HEAD are attributable to this job
|
|
2746
|
+
* alone regardless of what siblings were doing concurrently in their own
|
|
2747
|
+
* worktrees, so the excuse does not apply (PRD 109 shipped 'completed'
|
|
2748
|
+
* with nothing committed specifically because this carve-out fired
|
|
2749
|
+
* unconditionally during a high-concurrency run).
|
|
2615
2750
|
* - jobSelfCommitted: HEAD moved during the run, so the job's deliverable
|
|
2616
2751
|
* landed even if dirt (from a concurrent actor) remains.
|
|
2617
2752
|
* - legitimateNoOp (COMPLETED_EQUIVALENT_VERDICTS): runVerify.cjs's own
|
|
@@ -2631,8 +2766,8 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
|
|
|
2631
2766
|
* still a genuine finish-protocol violation (incident:
|
|
2632
2767
|
* 523-fix-bounded-fix-plan-retry, 2026-07-12).
|
|
2633
2768
|
*/
|
|
2634
|
-
function commitGuardVerdict({ newlyDirty, siblingRunning, jobSelfCommitted, legitimateNoOp, isFixPlanJob, verifyResult }) {
|
|
2635
|
-
if (siblingRunning || jobSelfCommitted || legitimateNoOp) return null;
|
|
2769
|
+
function commitGuardVerdict({ newlyDirty, siblingRunning, ranInWorktree, jobSelfCommitted, legitimateNoOp, isFixPlanJob, verifyResult, salvagePatch }) {
|
|
2770
|
+
if ((siblingRunning && !ranInWorktree) || jobSelfCommitted || legitimateNoOp) return null;
|
|
2636
2771
|
const dirty = newlyDirty || [];
|
|
2637
2772
|
if (dirty.length === 0 && isFixPlanJob) return null;
|
|
2638
2773
|
|
|
@@ -2651,9 +2786,10 @@ function commitGuardVerdict({ newlyDirty, siblingRunning, jobSelfCommitted, legi
|
|
|
2651
2786
|
}
|
|
2652
2787
|
|
|
2653
2788
|
const sample = dirty.slice(0, 3).join(', ');
|
|
2789
|
+
const salvageNote = salvagePatch ? ` — recoverable from salvage patch ${salvagePatch}` : '';
|
|
2654
2790
|
return {
|
|
2655
2791
|
verdict: 'uncommitted_changes',
|
|
2656
|
-
reason: `finish protocol incomplete: ${dirty.length} uncommitted file(s) left in working tree (e.g. ${sample})`,
|
|
2792
|
+
reason: `finish protocol incomplete: ${dirty.length} uncommitted file(s) left in working tree (e.g. ${sample})${salvageNote}`,
|
|
2657
2793
|
downgradeTo: 'needs_review',
|
|
2658
2794
|
annotations: carried.length ? carried : undefined,
|
|
2659
2795
|
};
|
|
@@ -2813,7 +2949,12 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2813
2949
|
// overrides `--model sonnet`, so scheduled jobs burn Opus credits silently.
|
|
2814
2950
|
// PATH must include Homebrew/user bins or the job's node/git children ENOENT
|
|
2815
2951
|
// when Electron was launched from Finder/Dock on macOS (stripped PATH).
|
|
2816
|
-
|
|
2952
|
+
// SM_PROJECT_ROOT is the main-tree cwd (never spawnCwd, which may be a
|
|
2953
|
+
// job/epic worktree) — forwarded by scheduler-mcp-server.cjs as
|
|
2954
|
+
// originProjectRoot so a job running inside its own worktree can still
|
|
2955
|
+
// resolve the real project for create-prd/open-session/readiness. See
|
|
2956
|
+
// projectRootResolve.cjs.
|
|
2957
|
+
const childEnv = cleanChildEnv({ PATH: pathWithUserBins(), SM_PROJECT_ROOT: cwd });
|
|
2817
2958
|
|
|
2818
2959
|
// Track whether the agent has emitted a `result` event in its JSONL stream.
|
|
2819
2960
|
// null until seen; then one of "success" | "error_max_turns" | … per the
|
|
@@ -3317,7 +3458,10 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3317
3458
|
// 'investigating' must never be the job's resting state.
|
|
3318
3459
|
mutate((s) => {
|
|
3319
3460
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
3320
|
-
if (j && j.status === 'investigating')
|
|
3461
|
+
if (j && j.status === 'investigating') {
|
|
3462
|
+
transitionJob(j, failedJob.status || 'failed', { reason: 'investigation probe exited — restoring prior status', source: 'spawnInvestigation:onExit' });
|
|
3463
|
+
delete j.runtime;
|
|
3464
|
+
}
|
|
3321
3465
|
})
|
|
3322
3466
|
.then(() => broadcast({ flush: true }))
|
|
3323
3467
|
.catch(() => {});
|
|
@@ -3374,6 +3518,15 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3374
3518
|
|
|
3375
3519
|
if (child) {
|
|
3376
3520
|
safeLog(`[scheduler] investigation pid=${child.pid}\n\n`);
|
|
3521
|
+
// Recorded so findStrandedInvestigations (a post-restart maintenance
|
|
3522
|
+
// sweep — the live process has no other way to know a probe is still
|
|
3523
|
+
// running) can tell a live probe apart from one whose owning process is
|
|
3524
|
+
// long gone, the same way reapDeadRunningJobs checks a running job's
|
|
3525
|
+
// runtime.pid.
|
|
3526
|
+
mutate((s) => {
|
|
3527
|
+
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
3528
|
+
if (j && j.status === 'investigating') j.runtime = { pid: child.pid };
|
|
3529
|
+
}).catch(() => {});
|
|
3377
3530
|
}
|
|
3378
3531
|
return { deferred: false };
|
|
3379
3532
|
} catch (e) {
|
|
@@ -3383,7 +3536,10 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3383
3536
|
releaseSlot();
|
|
3384
3537
|
mutate((s) => {
|
|
3385
3538
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
3386
|
-
if (j && j.status === 'investigating')
|
|
3539
|
+
if (j && j.status === 'investigating') {
|
|
3540
|
+
transitionJob(j, failedJob.status || 'failed', { reason: 'investigation spawn threw before exiting — restoring prior status', source: 'spawnInvestigation:catch' });
|
|
3541
|
+
delete j.runtime;
|
|
3542
|
+
}
|
|
3387
3543
|
})
|
|
3388
3544
|
.then(() => broadcast({ flush: true }))
|
|
3389
3545
|
.catch(() => {});
|
|
@@ -3430,6 +3586,16 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3430
3586
|
console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
|
|
3431
3587
|
} else {
|
|
3432
3588
|
console.log(`[scheduler] ${job.slug}: running in main tree (worktree not used: ${worktree.reason})`);
|
|
3589
|
+
// Surface any degraded-isolation fallback on the job row itself so it's
|
|
3590
|
+
// queryable from the queue instead of console-only — except the
|
|
3591
|
+
// deliberate env-disable flag, which is an intentional opt-out, not a
|
|
3592
|
+
// degradation worth flagging.
|
|
3593
|
+
if (!jobWorktree.isWorktreeDisabled()) {
|
|
3594
|
+
await mutate((s) => {
|
|
3595
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3596
|
+
if (idx >= 0) s.jobs[idx].worktreeFallbackReason = worktree.reason;
|
|
3597
|
+
});
|
|
3598
|
+
}
|
|
3433
3599
|
}
|
|
3434
3600
|
|
|
3435
3601
|
// Integrate the job's branch back into guardCwd's own HEAD, THEN tear the
|
|
@@ -3448,6 +3614,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3448
3614
|
let res;
|
|
3449
3615
|
let worktreeLeftoverDirty = [];
|
|
3450
3616
|
let worktreeIntegrationFailure = null;
|
|
3617
|
+
let worktreeSalvagePatch = null;
|
|
3451
3618
|
try {
|
|
3452
3619
|
res = await executeJob(job, runDir, defaultCwd, async (pid, sessionId, cwd) => {
|
|
3453
3620
|
await mutate((s) => {
|
|
@@ -3462,6 +3629,19 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3462
3629
|
} finally {
|
|
3463
3630
|
if (worktree.ok) {
|
|
3464
3631
|
worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
|
|
3632
|
+
// Salvage the worktree's full diff (tracked + untracked) to the run
|
|
3633
|
+
// dir BEFORE the checkout is removed below — otherwise a job killed
|
|
3634
|
+
// before its finish-protocol commit loses that work outright, with
|
|
3635
|
+
// no branch, no stash, no patch anywhere. Best-effort: never blocks
|
|
3636
|
+
// integration/cleanup and never changes the job's verdict.
|
|
3637
|
+
if (worktreeLeftoverDirty.length) {
|
|
3638
|
+
const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
|
|
3639
|
+
const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
|
|
3640
|
+
if (salvage && salvage.ok) {
|
|
3641
|
+
worktreeSalvagePatch = salvagePath;
|
|
3642
|
+
console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
|
|
3643
|
+
}
|
|
3644
|
+
}
|
|
3465
3645
|
const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug });
|
|
3466
3646
|
if (!integration.ok) {
|
|
3467
3647
|
worktreeIntegrationFailure = integration.reason;
|
|
@@ -3614,10 +3794,12 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3614
3794
|
const guardVerdict = commitGuardVerdict({
|
|
3615
3795
|
newlyDirty,
|
|
3616
3796
|
siblingRunning,
|
|
3797
|
+
ranInWorktree: worktree.ok,
|
|
3617
3798
|
jobSelfCommitted,
|
|
3618
3799
|
legitimateNoOp: guardIsLegitimateNoOp,
|
|
3619
3800
|
isFixPlanJob: isFixPlanSlug(job.slug),
|
|
3620
3801
|
verifyResult,
|
|
3802
|
+
salvagePatch: worktreeSalvagePatch,
|
|
3621
3803
|
});
|
|
3622
3804
|
if (guardVerdict) {
|
|
3623
3805
|
verifyResult = guardVerdict;
|
|
@@ -3709,6 +3891,11 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3709
3891
|
transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
|
|
3710
3892
|
s.jobs[i2].finishedAt = new Date().toISOString();
|
|
3711
3893
|
s.jobs[i2].exitCode = res.exitCode;
|
|
3894
|
+
if (worktreeSalvagePatch) {
|
|
3895
|
+
s.jobs[i2].worktreeSalvagePatch = worktreeSalvagePatch;
|
|
3896
|
+
} else {
|
|
3897
|
+
delete s.jobs[i2].worktreeSalvagePatch;
|
|
3898
|
+
}
|
|
3712
3899
|
s.jobs[i2].error = effectiveStatus === 'needs_review'
|
|
3713
3900
|
? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
|
|
3714
3901
|
// A failed job (non-zero exit) never consults verifyResult above,
|
|
@@ -3887,12 +4074,13 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3887
4074
|
});
|
|
3888
4075
|
await broadcast({ flush: true });
|
|
3889
4076
|
} else if (decision.action === 'fail-dirty') {
|
|
3890
|
-
|
|
4077
|
+
const salvageNote = worktreeSalvagePatch ? ` — recoverable from salvage patch ${worktreeSalvagePatch}` : '';
|
|
4078
|
+
console.log(`[scheduler] transient failure (${decision.transientKind}) for ${job.slug} left ${newlyDirtyCount} uncommitted file(s) (e.g. ${dirtySample})${salvageNote} — not auto-requeuing`);
|
|
3891
4079
|
await mutate((s) => {
|
|
3892
4080
|
const i = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3893
4081
|
if (i >= 0) {
|
|
3894
4082
|
transitionJob(s.jobs[i], 'failed', { reason: `transient failure (${decision.transientKind}) left uncommitted work — not auto-requeued`, source: 'spawnJob:fail-dirty' });
|
|
3895
|
-
s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample}) — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
|
|
4083
|
+
s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample})${salvageNote} — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
|
|
3896
4084
|
}
|
|
3897
4085
|
});
|
|
3898
4086
|
await broadcast({ flush: true });
|
|
@@ -3937,7 +4125,10 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3937
4125
|
// is synchronous and spawnJob is fire-and-forget.
|
|
3938
4126
|
let tickTail = Promise.resolve();
|
|
3939
4127
|
|
|
3940
|
-
|
|
4128
|
+
// `bypassLoadGate` is set only by the explicit human run-now / force-tick
|
|
4129
|
+
// paths (via runDueJobs): the human is asking, so the CPU-load gate yields
|
|
4130
|
+
// and logs that it did. Every automatic caller leaves it false.
|
|
4131
|
+
function tickQueue({ bypassLoadGate = false } = {}) {
|
|
3941
4132
|
const next = tickTail.then(async () => {
|
|
3942
4133
|
const state = await readQueue();
|
|
3943
4134
|
// Never reconcile against an unreadable queue: reconcile() would see zero
|
|
@@ -4025,6 +4216,35 @@ function tickQueue() {
|
|
|
4025
4216
|
lastMemGate = null;
|
|
4026
4217
|
}
|
|
4027
4218
|
|
|
4219
|
+
// Load gate (PRD 1085) — the INNERMOST launch predicate, evaluated only
|
|
4220
|
+
// once every outer gate (sessionSlots pool → per-project cap inside
|
|
4221
|
+
// pickNextBatch → memory above) has already admitted `gatedBatch`. It
|
|
4222
|
+
// never touches running jobs and never becomes a second pool: it only
|
|
4223
|
+
// withholds this tick's launches while the 1-minute loadavg per core is
|
|
4224
|
+
// over LOAD_GATE_PER_CORE. An explicit human Run now bypasses it.
|
|
4225
|
+
const load = loadGate.evaluate({ bypass: bypassLoadGate });
|
|
4226
|
+
if (load.bypassed) {
|
|
4227
|
+
console.log(`[scheduler] load gate: BYPASSED by run-now (loadavg1=${load.loadavg1} cores=${load.cores} ratio=${load.ratio} > ${load.threshold})`);
|
|
4228
|
+
} else if (load.gated) {
|
|
4229
|
+
const line = `[scheduler] load gate: loadavg1=${load.loadavg1} cores=${load.cores} ratio=${load.ratio} > ${load.threshold} — holding ${gatedBatch.length} eligible job(s)`;
|
|
4230
|
+
if (load.escalate) {
|
|
4231
|
+
const top = topCpuConsumers(3);
|
|
4232
|
+
console.warn(`${line} for ${Math.round(load.gatedSinceMs / 60_000)}m; top CPU: ${top.length ? top.join(' | ') : 'n/a'}`);
|
|
4233
|
+
} else {
|
|
4234
|
+
console.log(line);
|
|
4235
|
+
}
|
|
4236
|
+
if (load.shouldAudit) {
|
|
4237
|
+
appendAuditEvent('launch_load_gated', {
|
|
4238
|
+
loadavg1: load.loadavg1, cores: load.cores, ratio: load.ratio, threshold: load.threshold,
|
|
4239
|
+
held: gatedBatch.map((j) => j.slug), gatedSinceMs: load.gatedSinceMs,
|
|
4240
|
+
});
|
|
4241
|
+
}
|
|
4242
|
+
return recordTick(
|
|
4243
|
+
{ fired: false, reason: 'load-deferred', deferredCount: gatedBatch.length, ratio: load.ratio, threshold: load.threshold },
|
|
4244
|
+
{ detail: `load gate: ${load.loadavg1} / ${load.cores} cores = ${load.ratio} > ${load.threshold}`, holds },
|
|
4245
|
+
);
|
|
4246
|
+
}
|
|
4247
|
+
|
|
4028
4248
|
await mutate((s) => { s.lastRunAt = new Date().toISOString(); });
|
|
4029
4249
|
await broadcast();
|
|
4030
4250
|
|
|
@@ -4072,7 +4292,7 @@ function forceTickOutcome(result) {
|
|
|
4072
4292
|
}
|
|
4073
4293
|
}
|
|
4074
4294
|
|
|
4075
|
-
async function runDueJobs() {
|
|
4295
|
+
async function runDueJobs({ bypassLoadGate = false } = {}) {
|
|
4076
4296
|
const state = await readQueue();
|
|
4077
4297
|
if (state.unreadable) {
|
|
4078
4298
|
console.error('[scheduler] runDueJobs skipped: queue.json unreadable');
|
|
@@ -4083,7 +4303,7 @@ async function runDueJobs() {
|
|
|
4083
4303
|
return { fired: false, reason: 'paused' };
|
|
4084
4304
|
}
|
|
4085
4305
|
cancelToken = { cancelled: false };
|
|
4086
|
-
const result = await tickQueue();
|
|
4306
|
+
const result = await tickQueue({ bypassLoadGate });
|
|
4087
4307
|
// Clear the one-shot scheduledFor without waiting for jobs to settle.
|
|
4088
4308
|
await mutate((s) => { s.scheduledFor = null; });
|
|
4089
4309
|
await broadcast();
|
|
@@ -4109,11 +4329,13 @@ async function maybeLaunchWhenAvailable(state) {
|
|
|
4109
4329
|
// ---------- dead-process reaper ----------
|
|
4110
4330
|
|
|
4111
4331
|
/**
|
|
4112
|
-
* Scan running jobs, identify those whose claude process is provably dead
|
|
4113
|
-
*
|
|
4114
|
-
*
|
|
4115
|
-
*
|
|
4116
|
-
*
|
|
4332
|
+
* Scan running jobs, identify those whose claude process is provably dead OR
|
|
4333
|
+
* whose spawn never got far enough to record a runtime.pid in the first
|
|
4334
|
+
* place, and finalize them to completed/failed by reading the run log.
|
|
4335
|
+
* Called once per poll cycle. A job whose pid is alive (claudePidAlive) is
|
|
4336
|
+
* always skipped. A pidless job younger than PIDLESS_SPAWN_GRACE_MS is
|
|
4337
|
+
* skipped too (spawn may still be mid-flight) — see selectReapableJobs for
|
|
4338
|
+
* the full predicate. Exported so unit tests can invoke it directly.
|
|
4117
4339
|
*/
|
|
4118
4340
|
async function reapDeadRunningJobs() {
|
|
4119
4341
|
try {
|
|
@@ -4123,32 +4345,45 @@ async function reapDeadRunningJobs() {
|
|
|
4123
4345
|
// status:"running" with no slug left in runningSet to trigger reconciliation.
|
|
4124
4346
|
// queue.json is the source of truth for which jobs are actually running.
|
|
4125
4347
|
const state = await readQueue();
|
|
4348
|
+
const { reapable, warnings } = selectReapableJobs(state.jobs, Date.now(), {
|
|
4349
|
+
pidAlive: claudePidAlive,
|
|
4350
|
+
grace: PIDLESS_SPAWN_GRACE_MS,
|
|
4351
|
+
});
|
|
4352
|
+
for (const w of warnings) {
|
|
4353
|
+
console.warn(`[scheduler] reapDeadRunningJobs: ${w.reason} slug=${w.slug} — leaving row alone`);
|
|
4354
|
+
}
|
|
4355
|
+
|
|
4126
4356
|
const dead = [];
|
|
4127
|
-
for (const
|
|
4128
|
-
|
|
4129
|
-
const
|
|
4130
|
-
if (!pid) continue; // spawn may be mid-flight; give it a cycle
|
|
4131
|
-
if (claudePidAlive(pid)) continue;
|
|
4132
|
-
const logPath = j.runId
|
|
4357
|
+
for (const { slug, pid, pidless, reason } of reapable) {
|
|
4358
|
+
const j = state.jobs.find((x) => x.slug === slug);
|
|
4359
|
+
const logPath = j?.runId
|
|
4133
4360
|
? path.join(RUNS_DIR, j.runId, `${j.slug}.log`)
|
|
4134
4361
|
: null;
|
|
4362
|
+
// Absent/empty run dir → classifyRunOutcome finds no result event →
|
|
4363
|
+
// 'no_result' → non-success below → filed as failed, never completed.
|
|
4135
4364
|
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
4136
|
-
dead.push({ slug
|
|
4365
|
+
dead.push({ slug, pid, outcome, pidless, reason });
|
|
4137
4366
|
}
|
|
4138
4367
|
if (dead.length === 0) return;
|
|
4139
4368
|
|
|
4140
4369
|
await mutate((s) => {
|
|
4141
|
-
for (const { slug, pid, outcome } of dead) {
|
|
4370
|
+
for (const { slug, pid, outcome, pidless, reason } of dead) {
|
|
4142
4371
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
4143
4372
|
if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
|
|
4144
4373
|
const success = outcome === 'success';
|
|
4145
|
-
|
|
4374
|
+
const transitionReason = pidless ? reason : `reaped: process gone (outcome=${outcome})`;
|
|
4375
|
+
transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
|
|
4146
4376
|
s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
|
|
4147
4377
|
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
4148
|
-
s.jobs[idx].error = success ? null :
|
|
4378
|
+
s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
|
|
4149
4379
|
delete s.jobs[idx].runtime;
|
|
4150
4380
|
runningSet.delete(slug);
|
|
4151
|
-
|
|
4381
|
+
if (pidless) {
|
|
4382
|
+
console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
|
|
4383
|
+
appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
|
|
4384
|
+
} else {
|
|
4385
|
+
console.log(`[scheduler] reaped dead job slug=${slug} pid=${pid} outcome=${outcome}`);
|
|
4386
|
+
}
|
|
4152
4387
|
}
|
|
4153
4388
|
});
|
|
4154
4389
|
|
|
@@ -4337,12 +4572,17 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
|
|
|
4337
4572
|
// their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
|
|
4338
4573
|
const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'pass_no_commit', 'pass_no_commit_already_shipped']);
|
|
4339
4574
|
|
|
4340
|
-
// Bounds fix-plan recursion:
|
|
4341
|
-
// (
|
|
4342
|
-
//
|
|
4343
|
-
//
|
|
4344
|
-
//
|
|
4345
|
-
|
|
4575
|
+
// Bounds fix-plan recursion: cap N permits at most N+1 fix jobs per original
|
|
4576
|
+
// slug (depth 1 = the original job, depth 2 = its `-fix`, depth 3+ is
|
|
4577
|
+
// excluded). With N=1 that's `<slug>-fix` and `<slug>-fix-fix`, never a third
|
|
4578
|
+
// `-fix-fix-fix`. Lowered from 2 to 1 on 2026-08-31 (starry-night-ships):
|
|
4579
|
+
// three concurrent chains (115-fix-fix, 113-fix-fix, 111-fix-fix-fix) were
|
|
4580
|
+
// riding the old cap, and 115-fix-fix's own root-cause section read "The
|
|
4581
|
+
// code was already CORRECT. Only verification and commit failed." — a third
|
|
4582
|
+
// auto-retry re-runs an entire PRD and test battery to redo a `git commit`,
|
|
4583
|
+
// at near-zero marginal success probability. Shared by selectAutoFixTargets
|
|
4584
|
+
// and spawnInvestigation so both call sites agree on one threshold.
|
|
4585
|
+
const MAX_INVESTIGATION_DEPTH = 1;
|
|
4346
4586
|
|
|
4347
4587
|
/**
|
|
4348
4588
|
* True when a fix-plan job's investigationDepth is at or past the recursion
|
|
@@ -4841,7 +5081,7 @@ function registerScheduleHandlers() {
|
|
|
4841
5081
|
// Clears any existing pause first (same semantics as run-now).
|
|
4842
5082
|
await clearPause('run-now');
|
|
4843
5083
|
try {
|
|
4844
|
-
const result = await runDueJobs();
|
|
5084
|
+
const result = await runDueJobs({ bypassLoadGate: true });
|
|
4845
5085
|
return forceTickOutcome(result);
|
|
4846
5086
|
} catch (e) {
|
|
4847
5087
|
logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (force-tick)', meta: { error: e?.message } });
|
|
@@ -4917,7 +5157,7 @@ function registerScheduleHandlers() {
|
|
|
4917
5157
|
ipcMain.handle('schedule:run-now', async () => {
|
|
4918
5158
|
// Manual run-now overrides any auto-pause. Clear it first.
|
|
4919
5159
|
await clearPause('run-now');
|
|
4920
|
-
runDueJobs().catch((e) => logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (run-now)', meta: { error: e?.message } }));
|
|
5160
|
+
runDueJobs({ bypassLoadGate: true }).catch((e) => logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (run-now)', meta: { error: e?.message } }));
|
|
4921
5161
|
return { ok: true };
|
|
4922
5162
|
});
|
|
4923
5163
|
|
|
@@ -5286,6 +5526,49 @@ async function init() {
|
|
|
5286
5526
|
slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
|
|
5287
5527
|
});
|
|
5288
5528
|
}
|
|
5529
|
+
|
|
5530
|
+
// Stranded-investigation restore. Unlike the two escalations above, this
|
|
5531
|
+
// one ACTS: 'investigating' is a transient status whose restore
|
|
5532
|
+
// (spawnInvestigation's onExit/catch) only runs inside the process that
|
|
5533
|
+
// spawned the probe, so an app restart mid-probe leaves the row frozen
|
|
5534
|
+
// there forever (see findStrandedInvestigations' header, and the
|
|
5535
|
+
// "'investigating' must never be the job's resting state" comment at
|
|
5536
|
+
// spawnInvestigation's onExit). This restores each stranded row to the
|
|
5537
|
+
// exact terminal status it already carried before the probe was
|
|
5538
|
+
// spawned — it never re-runs or re-investigates anything.
|
|
5539
|
+
const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
|
|
5540
|
+
if (stranded.length > 0) {
|
|
5541
|
+
mutate((ms) => {
|
|
5542
|
+
for (const st of stranded) {
|
|
5543
|
+
const j = ms.jobs.find((x) => x.slug === st.slug);
|
|
5544
|
+
if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
|
|
5545
|
+
transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
|
|
5546
|
+
delete j.runtime;
|
|
5547
|
+
console.warn(
|
|
5548
|
+
`[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
|
|
5549
|
+
+ `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
|
|
5550
|
+
+ `restored to '${st.restoreStatus}'`,
|
|
5551
|
+
);
|
|
5552
|
+
appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
|
|
5553
|
+
}
|
|
5554
|
+
})
|
|
5555
|
+
.then(() => broadcast({ flush: true }))
|
|
5556
|
+
.catch(() => {});
|
|
5557
|
+
}
|
|
5558
|
+
|
|
5559
|
+
// Per-project starvation (PRD 1087): a project with pending work that has
|
|
5560
|
+
// been passed over on every tick while OTHER projects dispatch. Nothing
|
|
5561
|
+
// else distinguishes "no pending work" from "pending work, never
|
|
5562
|
+
// started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
|
|
5563
|
+
// Escalation only, same shape as the quarantine/overrun warnings above.
|
|
5564
|
+
for (const sp of findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS)) {
|
|
5565
|
+
console.warn(
|
|
5566
|
+
`[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
|
|
5567
|
+
+ `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
|
|
5568
|
+
+ `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
|
|
5569
|
+
);
|
|
5570
|
+
appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
|
|
5571
|
+
}
|
|
5289
5572
|
}, 10 * 60_000);
|
|
5290
5573
|
|
|
5291
5574
|
// Self-rescheduling poll loop with exponential backoff. Replaces the
|
|
@@ -5638,9 +5921,16 @@ const remote = {
|
|
|
5638
5921
|
},
|
|
5639
5922
|
|
|
5640
5923
|
async resetJob(slug, opts = {}) {
|
|
5641
|
-
|
|
5924
|
+
const resolved = await resolveSlugOrReason(slug, opts.cwd);
|
|
5925
|
+
if (!resolved.ok) {
|
|
5926
|
+
return { ok: false, error: resolved.reason === 'invalid-slug' ? 'invalid slug' : unknownSlugMessage(slug) };
|
|
5927
|
+
}
|
|
5642
5928
|
const outcome = await mutate((state) => {
|
|
5643
|
-
|
|
5929
|
+
// Same cwd filter as resolveSlugOrReason's file lookup above — slugs are
|
|
5930
|
+
// derived from title text with no cwd salt, so two different projects
|
|
5931
|
+
// can independently produce the identical slug; an opts.cwd caller must
|
|
5932
|
+
// reset THAT project's job, not just any queue row matching the string.
|
|
5933
|
+
const idx = state.jobs.findIndex((j) => j.slug === slug && (!opts.cwd || j.cwd === opts.cwd));
|
|
5644
5934
|
if (idx < 0) return { kind: 'not-found' };
|
|
5645
5935
|
// Terminal-status guard lives in resetJobFields itself; force:true
|
|
5646
5936
|
// threads through to override it.
|
|
@@ -5828,10 +6118,16 @@ const remote = {
|
|
|
5828
6118
|
// cancelled job lands in 'failed' with an error naming the cause,
|
|
5829
6119
|
// consistent with every other non-success terminal outcome. Refuses a
|
|
5830
6120
|
// slug that's already terminal — nothing left to cancel.
|
|
5831
|
-
async cancelJob(slug) {
|
|
6121
|
+
async cancelJob(slug, opts = {}) {
|
|
6122
|
+
if (!SCHEDULE_SLUG_RE.test(slug)) return { ok: false, error: 'invalid slug' };
|
|
5832
6123
|
const state = await readQueue();
|
|
5833
|
-
const job = state.jobs.find((j) => j.slug === slug);
|
|
5834
|
-
if (!job)
|
|
6124
|
+
const job = state.jobs.find((j) => j.slug === slug && (!opts.cwd || j.cwd === opts.cwd));
|
|
6125
|
+
if (!job) {
|
|
6126
|
+
return {
|
|
6127
|
+
ok: false,
|
|
6128
|
+
error: `unknown slug "${slug}": no queued job with that name${opts.cwd ? ` in cwd ${opts.cwd}` : ''} — call scheduler_list_jobs to see what exists`,
|
|
6129
|
+
};
|
|
6130
|
+
}
|
|
5835
6131
|
if (job.status === 'completed' || job.status === 'failed' || job.status === 'needs_review' || job.status === 'skipped') {
|
|
5836
6132
|
return { ok: false, error: `job already terminal (status: "${job.status}") — nothing to cancel` };
|
|
5837
6133
|
}
|
|
@@ -5889,9 +6185,10 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
5889
6185
|
return;
|
|
5890
6186
|
}
|
|
5891
6187
|
const force = parsed.force === true;
|
|
5892
|
-
const
|
|
6188
|
+
const cwd = typeof parsed.cwd === 'string' ? parsed.cwd : undefined;
|
|
6189
|
+
const result = await remoteObj.resetJob(slug, { force, cwd });
|
|
5893
6190
|
sendJson(res, 200, result);
|
|
5894
6191
|
});
|
|
5895
6192
|
}
|
|
5896
6193
|
|
|
5897
|
-
module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims };
|
|
6194
|
+
module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS };
|