claude-code-session-manager 0.75.3 → 0.77.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-CzQqcObq.js → AgentLibrary-B2ie8bbw.js} +2 -2
- package/dist/assets/{DataModel-Bj_WlLz8.js → DataModel-BIJPYw32.js} +1 -1
- package/dist/assets/{History-DnSi_OHm.js → History-CeY6dk9S.js} +2 -2
- package/dist/assets/{Hooks-0BB0dp3S.js → Hooks-BFH2ocKg.js} +2 -2
- package/dist/assets/{HostBilko-DHpwwsLQ.js → HostBilko-36gj9wLz.js} +1 -1
- package/dist/assets/{Library-CaJVqVvi.js → Library-C-hBct39.js} +1 -1
- package/dist/assets/{ListDetail-C1W2HmC2.js → ListDetail-CNq64VWV.js} +1 -1
- package/dist/assets/{MarkdownEditor-5Ob9FW3z.js → MarkdownEditor-Bh3qt5-1.js} +1 -1
- package/dist/assets/{McpServers-JxCSfm1S.js → McpServers-DpGN0oyz.js} +1 -1
- package/dist/assets/{Memory-BDeqlqwH.js → Memory-D59hUjC4.js} +6 -6
- package/dist/assets/{Panel-Dh9ZHuEj.js → Panel-DCgbaoci.js} +1 -1
- package/dist/assets/{Permissions-DXy-CbEY.js → Permissions-DAmQ0DYV.js} +2 -2
- package/dist/assets/{Plugins-_n1Iuc8T.js → Plugins-Dyfgn6Is.js} +2 -2
- package/dist/assets/{ProvenanceBadge-BP_evfxE.js → ProvenanceBadge-BiYhPO1U.js} +1 -1
- package/dist/assets/SaveBar-RV7B6sOh.js +1 -0
- package/dist/assets/Scheduler-BPaNqx1b.js +14 -0
- package/dist/assets/{ScopeSwitcher-CAWzM6RI.js → ScopeSwitcher-P4mdLGNU.js} +1 -1
- package/dist/assets/{Settings-DRRozLyT.js → Settings-BL4vf5aX.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-DGHDWlz4.js → SkillReferenceGraph-BRBDyi1_.js} +1 -1
- package/dist/assets/{Skills-D8L66eiX.js → Skills-BV08gDUH.js} +2 -2
- package/dist/assets/{SystemPrompt-CYtUsonD.js → SystemPrompt-CLftSsDw.js} +1 -1
- package/dist/assets/TagLibrary-Bp8jGsd5.js +1 -0
- package/dist/assets/{TiptapBody-B2hRgbPE.js → TiptapBody-jCpuB6E5.js} +1 -1
- package/dist/assets/{Toggle-BTwsbxam.js → Toggle-D2paA1xf.js} +1 -1
- package/dist/assets/{index-DijufvkJ.js → index-BDRSqBl3.js} +704 -704
- package/dist/assets/{index-CMLnzdZC.css → index-CYhdtisq.css} +1 -1
- package/dist/assets/{settingsSchema-D6wzxAi6.js → settingsSchema-6IOLjZZN.js} +1 -1
- package/dist/index.html +2 -2
- package/package.json +8 -2
- package/plugins/session-manager-dev/skills/develop/standards.md +1 -1
- package/scripts/lib/activeSessions.cjs +116 -6
- package/scripts/project-pages-logic/dist/logic.cjs +4709 -0
- package/scripts/render-project-pages/dist/renderer.cjs +18900 -0
- package/scripts/render-project-pages.cjs +70 -0
- package/scripts/scheduler-mcp-server.cjs +269 -96
- package/scripts/validate-project-pages-summary.cjs +62 -0
- package/src/main/__tests__/agentModelResolve.test.cjs +66 -0
- package/src/main/__tests__/epicStatusMirror.test.cjs +110 -0
- package/src/main/__tests__/health-delegation-chain.test.cjs +106 -0
- package/src/main/__tests__/prdAdminRoutes.test.cjs +295 -0
- package/src/main/__tests__/prdAgentType.test.cjs +103 -0
- package/src/main/__tests__/prdCreate.test.cjs +247 -0
- package/src/main/__tests__/prdFrontmatterAgentType.test.cjs +117 -0
- package/src/main/__tests__/prdFrontmatterQuietMachine.test.cjs +108 -0
- package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +485 -0
- package/src/main/__tests__/projectPages.test.cjs +73 -1
- package/src/main/__tests__/rcaReport.test.cjs +54 -0
- package/src/main/__tests__/runVerify.test.cjs +94 -0
- package/src/main/__tests__/scheduler-autofix-select.test.cjs +58 -3
- package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +103 -0
- package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +41 -0
- package/src/main/__tests__/scheduler-effective-concurrency.test.cjs +10 -0
- package/src/main/__tests__/scheduler-foreign-wip-manifest.test.cjs +78 -0
- package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +242 -0
- package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +31 -0
- package/src/main/__tests__/scheduler-launch-failure.test.cjs +201 -0
- package/src/main/__tests__/scheduler-leftover-fields.test.cjs +52 -0
- package/src/main/__tests__/scheduler-looks-done.test.cjs +241 -0
- package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +135 -0
- package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +222 -0
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +207 -1
- package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +212 -0
- package/src/main/__tests__/scheduler-stranded-investigation.test.cjs +185 -0
- package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +194 -0
- package/src/main/__tests__/seedAgentPersonas.test.cjs +75 -14
- package/src/main/__tests__/seedSchedulerMcp.test.cjs +66 -0
- package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -5
- package/src/main/bilkoHost.cjs +4 -3
- package/src/main/chatRunner.cjs +6 -1
- package/src/main/config.cjs +25 -33
- package/src/main/health.cjs +153 -2
- package/src/main/index.cjs +64 -5
- package/src/main/ipcSchemas.cjs +69 -1
- package/src/main/lib/__tests__/activeIndexRebuild.test.cjs +179 -0
- package/src/main/lib/__tests__/childWithLog.test.cjs +141 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +391 -42
- package/src/main/lib/__tests__/ephemeralCwd.test.cjs +91 -0
- package/src/main/lib/__tests__/epicWorktreeMint.test.cjs +5 -3
- package/src/main/lib/__tests__/fixChainDepth.test.cjs +40 -0
- package/src/main/lib/__tests__/gitWorktree.test.cjs +290 -5
- package/src/main/lib/__tests__/gitWorktreeSalvage.test.cjs +107 -0
- package/src/main/lib/__tests__/gitWorktreeSalvageDelta.test.cjs +153 -0
- package/src/main/lib/__tests__/jobWorktree.test.cjs +6 -4
- package/src/main/lib/__tests__/landedSinceRun.test.cjs +73 -0
- package/src/main/lib/__tests__/launchFailure.test.cjs +220 -0
- package/src/main/lib/__tests__/loadGate.test.cjs +159 -0
- package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +102 -0
- package/src/main/lib/__tests__/opsOwnership.test.cjs +7 -0
- package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +151 -0
- package/src/main/lib/__tests__/opsRootResolve.test.cjs +149 -0
- package/src/main/lib/__tests__/prdDeclaredPaths.test.cjs +82 -0
- package/src/main/lib/__tests__/projectRootResolve.test.cjs +148 -0
- package/src/main/lib/__tests__/queueHealth.test.cjs +58 -0
- package/src/main/lib/__tests__/quietMachineLease.test.cjs +39 -0
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +133 -0
- package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +19 -9
- package/src/main/lib/__tests__/schedulerBatchFairness.test.cjs +213 -0
- package/src/main/lib/__tests__/schedulerBatchLaunchHold.test.cjs +125 -0
- package/src/main/lib/__tests__/schedulerBatchProjectCap.test.cjs +127 -0
- package/src/main/lib/__tests__/schedulerBatchQuietMachine.test.cjs +109 -0
- package/src/main/lib/__tests__/schedulerMcpServerHeadlessRefusal.test.cjs +71 -0
- package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +217 -0
- package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +350 -0
- package/src/main/lib/activeIndexMerge.cjs +15 -0
- package/src/main/lib/activeIndexRebuild.cjs +133 -0
- package/src/main/lib/agentModelResolve.cjs +58 -0
- package/src/main/lib/buildTarget.cjs +3 -2
- package/src/main/lib/childWithLog.cjs +69 -2
- package/src/main/lib/claudeBin.cjs +54 -1
- package/src/main/lib/crossProjectFeedback.cjs +8 -1
- package/src/main/lib/definitionOfDone.cjs +3 -2
- package/src/main/lib/delegationReadiness.cjs +514 -26
- package/src/main/lib/ephemeralCwd.cjs +78 -0
- package/src/main/lib/epicDelegationStats.cjs +2 -1
- package/src/main/lib/epicMint.cjs +17 -1
- package/src/main/lib/epicStatusMirror.cjs +95 -0
- package/src/main/lib/epicValidationHook.cjs +2 -1
- package/src/main/lib/epicWorktreeMint.cjs +5 -2
- package/src/main/lib/fixChainDepth.cjs +45 -0
- package/src/main/lib/gitWorktree.cjs +520 -21
- package/src/main/lib/jobWorktree.cjs +2 -0
- package/src/main/lib/landedSinceRun.cjs +55 -0
- package/src/main/lib/launchFailure.cjs +357 -0
- package/src/main/lib/loadGate.cjs +134 -0
- package/src/main/lib/mcpToolCatalog.cjs +370 -0
- package/src/main/lib/opsErrorLog.cjs +12 -1
- package/src/main/lib/opsOwnership.cjs +106 -0
- package/src/main/lib/prdAdminRoutes.cjs +43 -3
- package/src/main/lib/prdAgentType.cjs +84 -0
- package/src/main/lib/prdCreate.cjs +103 -15
- package/src/main/lib/prdDeclaredPaths.cjs +70 -0
- package/src/main/lib/prdFrontmatter.cjs +17 -3
- package/src/main/lib/prdLocations.cjs +13 -6
- package/src/main/lib/projectHomeAdminRoutes.cjs +402 -0
- package/src/main/lib/projectPageSummarySchema.cjs +181 -0
- package/src/main/lib/projectRootResolve.cjs +134 -0
- package/src/main/lib/promptSessionSchema.cjs +7 -0
- package/src/main/lib/queueHealth.cjs +38 -0
- package/src/main/lib/queueStore.cjs +40 -7
- package/src/main/lib/quietMachineLease.cjs +48 -0
- package/src/main/lib/rcaReport.cjs +54 -4
- package/src/main/lib/reaperHelpers.cjs +64 -1
- package/src/main/lib/scheduleJobSchema.cjs +31 -0
- package/src/main/lib/scheduleJobTransitions.cjs +6 -2
- package/src/main/lib/schedulerBatch.cjs +301 -55
- package/src/main/lib/schedulerConfig.cjs +99 -0
- package/src/main/projectBrief.cjs +3 -2
- package/src/main/projectPages.cjs +162 -3
- package/src/main/promptSessionTranscript.cjs +0 -0
- package/src/main/pty.cjs +5 -0
- package/src/main/queueOps.cjs +15 -8
- package/src/main/runVerify.cjs +50 -9
- package/src/main/scheduler/prdParser.cjs +18 -1
- package/src/main/scheduler.cjs +1701 -130
- package/src/main/seedAgentPersonas.cjs +62 -21
- package/src/main/seedSchedulerMcp.cjs +58 -4
- package/src/main/templates/project-pages-catalog.json +741 -0
- package/src/main/templates/project-pages-pipeline.md +417 -0
- package/src/preload/api.d.ts +187 -3
- package/src/preload/index.cjs +9 -0
- package/src/seed/agents/project-home-builder.md +59 -0
- package/dist/assets/SaveBar-D-gCUx4n.js +0 -1
- package/dist/assets/Scheduler-Bpd4OGju.js +0 -14
- package/dist/assets/TagLibrary-E5CLeuVk.js +0 -1
package/src/main/scheduler.cjs
CHANGED
|
@@ -53,9 +53,13 @@ const { ipcMain } = require('electron');
|
|
|
53
53
|
const billing = require('./usage.cjs');
|
|
54
54
|
const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
|
|
55
55
|
const supervisor = require('./supervisor.cjs');
|
|
56
|
-
const { resolveClaudeBin } = require('./lib/claudeBin.cjs');
|
|
56
|
+
const { resolveClaudeBin, probeClaudeVersion } = require('./lib/claudeBin.cjs');
|
|
57
|
+
const launchFailure = require('./lib/launchFailure.cjs');
|
|
58
|
+
const { appendError } = require('./lib/opsErrorLog.cjs');
|
|
57
59
|
const { readTail } = require('./lib/fileTail.cjs');
|
|
58
|
-
const { claudePidAlive, classifyRunOutcome, ORPHAN_REQUEUE_CAP } = require('./lib/reaperHelpers.cjs');
|
|
60
|
+
const { claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
|
|
61
|
+
const { computeQueueHealth } = require('./lib/queueHealth.cjs');
|
|
62
|
+
const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
|
|
59
63
|
const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
|
|
60
64
|
const { sendIfAlive } = require('./lib/sendToRenderer.cjs');
|
|
61
65
|
const { createBroadcastCoalescer } = require('./lib/broadcastCoalescer.cjs');
|
|
@@ -67,8 +71,10 @@ const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs')
|
|
|
67
71
|
const promptSessionTranscript = require('./promptSessionTranscript.cjs');
|
|
68
72
|
const { verifyRun } = require('./runVerify.cjs');
|
|
69
73
|
const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
|
|
74
|
+
const { landedSinceRun } = require('./lib/landedSinceRun.cjs');
|
|
75
|
+
const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
|
|
70
76
|
const logs = require('./logs.cjs');
|
|
71
|
-
const { schemas, validated } = require('./ipcSchemas.cjs');
|
|
77
|
+
const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
|
|
72
78
|
const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
|
|
73
79
|
const {
|
|
74
80
|
POLL_INTERVAL_MS,
|
|
@@ -78,6 +84,9 @@ const {
|
|
|
78
84
|
QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
|
|
79
85
|
JOB_OVERRUN_FACTOR: JOB_OVERRUN_FACTOR_DEFAULT,
|
|
80
86
|
JOB_OVERRUN_FLOOR_MS: JOB_OVERRUN_FLOOR_MS_DEFAULT,
|
|
87
|
+
PIDLESS_SPAWN_GRACE_MS,
|
|
88
|
+
INVESTIGATION_MAX_MS,
|
|
89
|
+
STARVATION_ESCALATE_MS,
|
|
81
90
|
} = require('./lib/schedulerConfig.cjs');
|
|
82
91
|
const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
|
|
83
92
|
? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
|
|
@@ -88,7 +97,7 @@ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
|
|
|
88
97
|
const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
|
|
89
98
|
? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
|
|
90
99
|
: JOB_OVERRUN_FLOOR_MS_DEFAULT;
|
|
91
|
-
const { pickForProject, pickNextBatch, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
|
|
100
|
+
const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
|
|
92
101
|
const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
|
|
93
102
|
const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
|
|
94
103
|
const queueHistory = require('./lib/queueHistory.cjs');
|
|
@@ -100,7 +109,8 @@ const queueOps = require('./queueOps.cjs');
|
|
|
100
109
|
// home-dir layout.
|
|
101
110
|
const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
|
|
102
111
|
const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
|
|
103
|
-
const
|
|
112
|
+
const agentModelResolve = require('./lib/agentModelResolve.cjs');
|
|
113
|
+
const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
|
|
104
114
|
const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
|
|
105
115
|
const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
|
|
106
116
|
const { appendAuditEvent } = require('./lib/auditLog.cjs');
|
|
@@ -123,6 +133,7 @@ function resolveOriginSessionId(cwd, epicId) {
|
|
|
123
133
|
return session && typeof session.claudeSessionId === 'string' ? session.claudeSessionId : null;
|
|
124
134
|
}
|
|
125
135
|
const sessionSlots = require('./lib/sessionSlots.cjs');
|
|
136
|
+
const quietMachineLease = require('./lib/quietMachineLease.cjs');
|
|
126
137
|
const jobWorktree = require('./lib/jobWorktree.cjs');
|
|
127
138
|
const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
|
|
128
139
|
const queueStore = require('./lib/queueStore.cjs');
|
|
@@ -180,6 +191,21 @@ const RESULT_TEXT_TAIL_BYTES = 64 * 1024;
|
|
|
180
191
|
const IDLE_OUTPUT_KILL_MS = 20 * 60_000;
|
|
181
192
|
const IDLE_CHECK_INTERVAL_MS = 60_000;
|
|
182
193
|
|
|
194
|
+
// Foreground Bash budget for every spawned `claude -p` job (executor +
|
|
195
|
+
// investigation). The Claude Code harness auto-backgrounds any foreground
|
|
196
|
+
// Bash command past its own default (120s) or max (600s) timeout and returns
|
|
197
|
+
// a tool result promising a later notification — but a headless single-shot
|
|
198
|
+
// run has no later turn, so that notification can never arrive and the run
|
|
199
|
+
// dead-ends mid-verification with no commit and no verdict. Raising these
|
|
200
|
+
// via the child's env moves that trap out of reach of normal gate commands
|
|
201
|
+
// (test suites, builds). BASH_MAX_TIMEOUT_MS MUST stay strictly below
|
|
202
|
+
// IDLE_OUTPUT_KILL_MS with real margin: a long foreground Bash emits no
|
|
203
|
+
// stream-json events while it runs, so the log mtime stalls and the
|
|
204
|
+
// idle-tail watchdog above would SIGTERM the job mid-gate if the two ever
|
|
205
|
+
// crossed — trading one silent failure for another.
|
|
206
|
+
const BASH_DEFAULT_TIMEOUT_MS = 600_000; // 10 min
|
|
207
|
+
const BASH_MAX_TIMEOUT_MS = 900_000; // 15 min — must stay below IDLE_OUTPUT_KILL_MS
|
|
208
|
+
|
|
183
209
|
// Boot reconciliation: a job left 'running' by an app restart/crash whose log
|
|
184
210
|
// shows neither success nor a real failure result was merely interrupted — the
|
|
185
211
|
// host died, the PRD didn't. Re-queue it up to this many times before giving up
|
|
@@ -205,6 +231,20 @@ const FINISH_PROTOCOL = `
|
|
|
205
231
|
Once every acceptance-criteria line above is satisfied, finish in this EXACT
|
|
206
232
|
sequence. Do not stop before the commit lands; committing is part of the job.
|
|
207
233
|
|
|
234
|
+
RUN VERIFICATION IN THE FOREGROUND — this applies to the whole run, not just
|
|
235
|
+
step 3 below: every test/typecheck/lint/build command you run, whether while
|
|
236
|
+
implementing the AC or during VERIFY, must run SYNCHRONOUSLY and you must wait
|
|
237
|
+
for it to return. Never start a verification command as a background task
|
|
238
|
+
(no background Bash) and then call Monitor, TaskOutput, or ScheduleWakeup to
|
|
239
|
+
pick up its result later — a headless \`claude -p\` run has no later turn, so
|
|
240
|
+
nothing ever delivers that notification and the run dies mid-verification with
|
|
241
|
+
no commit and no verdict. Your foreground Bash budget for this run is
|
|
242
|
+
${BASH_DEFAULT_TIMEOUT_MS / 1000}s by default, up to ${BASH_MAX_TIMEOUT_MS / 1000}s max
|
|
243
|
+
— size your own \`timeout <n>\` wrapper (e.g. \`timeout ${Math.floor(BASH_MAX_TIMEOUT_MS / 1000)} npm test\`)
|
|
244
|
+
to fit inside that ceiling; if a gate command still cannot finish inside
|
|
245
|
+
budget, stop and emit SCHEDULER_VERDICT: FAIL with the reason instead of
|
|
246
|
+
deferring it.
|
|
247
|
+
|
|
208
248
|
1. CODE REVIEW — run \`/code-review --fix\` on your changes and apply the fixes it
|
|
209
249
|
surfaces (correctness first). For any finding you judge a false positive, say
|
|
210
250
|
why in your result; do not silently skip it. If \`/code-review\` is not
|
|
@@ -287,6 +327,156 @@ function gitHead(cwd) {
|
|
|
287
327
|
});
|
|
288
328
|
}
|
|
289
329
|
|
|
330
|
+
// Return the current `git stash list` entries in cwd as raw lines
|
|
331
|
+
// "<hash> <ref> <subject>" (hash is stable even as ref indices shift when a
|
|
332
|
+
// new entry is pushed on top), or null when the guard does not apply (cwd is
|
|
333
|
+
// not a git work tree, git is missing, or the call errors). Never throws.
|
|
334
|
+
function stashList(cwd) {
|
|
335
|
+
return new Promise((resolve) => {
|
|
336
|
+
if (!cwd) { resolve(null); return; }
|
|
337
|
+
execFile(
|
|
338
|
+
'git',
|
|
339
|
+
['-C', cwd, 'stash', 'list', '--format=%H %gd %gs'],
|
|
340
|
+
{ timeout: 10_000, windowsHide: true },
|
|
341
|
+
(err, stdout) => {
|
|
342
|
+
if (err) { resolve(null); return; }
|
|
343
|
+
resolve(String(stdout || '').split('\n').filter(Boolean));
|
|
344
|
+
},
|
|
345
|
+
);
|
|
346
|
+
});
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
// Parse one `stashList()` line into { hash, ref, subject }. Pure, exported
|
|
350
|
+
// for unit testing. Returns null for a malformed line.
|
|
351
|
+
function parseStashLine(line) {
|
|
352
|
+
const m = /^(\S+)\s+(\S+)\s+(.*)$/.exec(String(line || ''));
|
|
353
|
+
return m ? { hash: m[1], ref: m[2], subject: m[3] } : null;
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
// Paths touched by any commit landed in cwd strictly between headBefore and
|
|
357
|
+
// headAfter. Returns [] when no commit landed (headBefore === headAfter, or
|
|
358
|
+
// either is missing) — used by the shared-tree guard below to tell a path
|
|
359
|
+
// the job legitimately committed apart from a path that just silently went
|
|
360
|
+
// quiet with nothing to explain it. Never throws.
|
|
361
|
+
function pathsChangedSince(cwd, headBefore, headAfter) {
|
|
362
|
+
return new Promise((resolve) => {
|
|
363
|
+
if (!cwd || !headBefore || !headAfter || headBefore === headAfter) { resolve([]); return; }
|
|
364
|
+
execFile(
|
|
365
|
+
'git',
|
|
366
|
+
['-C', cwd, 'diff', '--name-only', `${headBefore}..${headAfter}`],
|
|
367
|
+
{ timeout: 10_000, windowsHide: true },
|
|
368
|
+
(err, stdout) => { resolve(err ? [] : String(stdout || '').split('\n').filter(Boolean)); },
|
|
369
|
+
);
|
|
370
|
+
});
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
// Restore ONE specific stash ref (never a blanket pop of "whatever is on
|
|
374
|
+
// top") into cwd: apply, then drop only on a clean apply. On conflict the
|
|
375
|
+
// entry is left in place — never dropped, never forced — so the operator's
|
|
376
|
+
// own `git stash pop`/`apply` still works afterward. Never throws.
|
|
377
|
+
function restoreSpecificStash(cwd, ref) {
|
|
378
|
+
return new Promise((resolve) => {
|
|
379
|
+
execFile('git', ['-C', cwd, 'stash', 'apply', ref], { timeout: 10_000, windowsHide: true }, (applyErr, _stdout, applyStderr) => {
|
|
380
|
+
if (applyErr) {
|
|
381
|
+
resolve({ ok: false, error: String(applyStderr || applyErr.message || applyErr).trim().split('\n')[0] });
|
|
382
|
+
return;
|
|
383
|
+
}
|
|
384
|
+
execFile('git', ['-C', cwd, 'stash', 'drop', ref], { timeout: 10_000, windowsHide: true }, () => {
|
|
385
|
+
resolve({ ok: true });
|
|
386
|
+
});
|
|
387
|
+
});
|
|
388
|
+
});
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
// Diff a before/after `stashList()` pair plus a before/after dirty-path pair
|
|
392
|
+
// to find what an in-place job silently discarded from a tree it shares with
|
|
393
|
+
// something else (Incident: social-signals-trader 2026-09-01, a blanket
|
|
394
|
+
// `git stash` reverted a live operator config edit with no error anywhere).
|
|
395
|
+
// Two independent signals, either of which means the job discarded state it
|
|
396
|
+
// did not create:
|
|
397
|
+
// - newStashes: a stash entry now present that wasn't in the baseline —
|
|
398
|
+
// the job ran `git stash` itself.
|
|
399
|
+
// - reverted: a path that was dirty in the baseline, is clean now, and was
|
|
400
|
+
// not touched by any commit landed during the run — the job reset/
|
|
401
|
+
// checked-out over pre-existing uncommitted work without stashing it.
|
|
402
|
+
// Pure/no I/O — the guard's git calls happen at the call site
|
|
403
|
+
// (checkSharedTreeGuard). Exported for unit testing.
|
|
404
|
+
function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAfter, pathsCommittedDuringRun }) {
|
|
405
|
+
const beforeHashes = new Set((stashBefore || []).map((l) => parseStashLine(l)?.hash).filter(Boolean));
|
|
406
|
+
const newStashes = (stashAfter || [])
|
|
407
|
+
.map(parseStashLine)
|
|
408
|
+
.filter((e) => e && !beforeHashes.has(e.hash));
|
|
409
|
+
const dirtyAfterSet = new Set(dirtyAfter || []);
|
|
410
|
+
const committedSet = new Set(pathsCommittedDuringRun || []);
|
|
411
|
+
const reverted = (dirtyBefore || []).filter((p) => !dirtyAfterSet.has(p) && !committedSet.has(p));
|
|
412
|
+
return { newStashes, reverted };
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// Post-run shared-tree guard for an IN-PLACE job (worktree.ok === false —
|
|
416
|
+
// callers must gate on that; a worktree-isolated run's git state can never
|
|
417
|
+
// leak into guardCwd, so there is nothing here to check). Best-effort: never
|
|
418
|
+
// throws, never changes the job's exit code. Restores exactly one
|
|
419
|
+
// executor-created stash (never guesses when there are 2+); reports anything
|
|
420
|
+
// it can't safely resolve on the returned object so the caller can surface it
|
|
421
|
+
// on the job row instead of finishing silently green.
|
|
422
|
+
async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug }) {
|
|
423
|
+
try {
|
|
424
|
+
const [stashAfter, headAfter] = await Promise.all([
|
|
425
|
+
module.exports.stashList(cwd),
|
|
426
|
+
module.exports.gitHead(cwd),
|
|
427
|
+
]);
|
|
428
|
+
const pathsCommittedDuringRun = await module.exports.pathsChangedSince(cwd, headBefore, headAfter);
|
|
429
|
+
// First pass: which stashes are new. Decided before charging anything
|
|
430
|
+
// against dirtyBaseline — a path this run's own stash covers must not be
|
|
431
|
+
// judged "reverted" using dirty state captured before the restore below
|
|
432
|
+
// has had a chance to bring it back.
|
|
433
|
+
const { newStashes } = module.exports.evaluateSharedTreeGuard({
|
|
434
|
+
stashBefore: stashBaseline,
|
|
435
|
+
stashAfter,
|
|
436
|
+
dirtyBefore: [],
|
|
437
|
+
dirtyAfter: [],
|
|
438
|
+
pathsCommittedDuringRun,
|
|
439
|
+
});
|
|
440
|
+
|
|
441
|
+
const result = {};
|
|
442
|
+
if (newStashes.length === 1) {
|
|
443
|
+
const [entry] = newStashes;
|
|
444
|
+
const restore = await module.exports.restoreSpecificStash(cwd, entry.ref);
|
|
445
|
+
if (restore.ok) {
|
|
446
|
+
result.restoredStash = entry.ref;
|
|
447
|
+
console.log(`[scheduler] ${slug}: restored a stash the job created in the shared tree (${entry.ref})`);
|
|
448
|
+
} else {
|
|
449
|
+
result.restoreFailed = `${entry.ref}: ${restore.error || 'apply failed'}`;
|
|
450
|
+
console.error(`[scheduler] ${slug}: shared-tree guard could not restore ${entry.ref}: ${restore.error}`);
|
|
451
|
+
}
|
|
452
|
+
} else if (newStashes.length > 1) {
|
|
453
|
+
result.ambiguousStashes = newStashes.map((e) => e.ref);
|
|
454
|
+
console.error(`[scheduler] ${slug}: shared-tree guard found ${newStashes.length} stashes the job created — ambiguous, not auto-restoring (${result.ambiguousStashes.join(', ')})`);
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
// Second pass: recompute "reverted" against the tree's dirty state AFTER
|
|
458
|
+
// any restore attempt above, so a path that came back via a successfully
|
|
459
|
+
// restored stash is not ALSO reported as an unexplained revert (it was
|
|
460
|
+
// explained — by the stash this guard just restored).
|
|
461
|
+
const dirtyAfter = await module.exports.uncommittedChanges(cwd);
|
|
462
|
+
const { reverted } = module.exports.evaluateSharedTreeGuard({
|
|
463
|
+
stashBefore: stashBaseline,
|
|
464
|
+
stashAfter,
|
|
465
|
+
dirtyBefore: dirtyBaseline,
|
|
466
|
+
dirtyAfter,
|
|
467
|
+
pathsCommittedDuringRun,
|
|
468
|
+
});
|
|
469
|
+
if (reverted.length) {
|
|
470
|
+
result.reverted = reverted;
|
|
471
|
+
console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
|
|
472
|
+
}
|
|
473
|
+
return (result.restoredStash || result.restoreFailed || result.ambiguousStashes || result.reverted) ? result : null;
|
|
474
|
+
} catch (e) {
|
|
475
|
+
console.error(`[scheduler] ${slug}: shared-tree guard error`, e);
|
|
476
|
+
return null;
|
|
477
|
+
}
|
|
478
|
+
}
|
|
479
|
+
|
|
290
480
|
// True when cwd is inside a git repository. Used to keep a non-git cwd (e.g.
|
|
291
481
|
// a scratch dir like /tmp) from ever being handed to an investigation's
|
|
292
482
|
// fix-plan as its cwd — the commit guard, worktree isolation, and
|
|
@@ -690,6 +880,43 @@ async function safeSlugPath(slug) {
|
|
|
690
880
|
return safeSlugPathIn(dir, slug);
|
|
691
881
|
}
|
|
692
882
|
|
|
883
|
+
/**
|
|
884
|
+
* The two distinct failure modes safeSlugPath collapses into one nullable
|
|
885
|
+
* return (the defect this fixes — see the PRD that added this helper's
|
|
886
|
+
* Goal): a slug that fails SCHEDULE_SLUG_RE is a caller mistake ("invalid
|
|
887
|
+
* slug"), while a well-formed slug that exists in no candidate PRD dir is a
|
|
888
|
+
* lookup miss ("unknown slug") — an agent retrying the first as if it were
|
|
889
|
+
* the second (or vice versa) burns a turn on the wrong fix. Returns
|
|
890
|
+
* `{ ok: true, path }` or `{ ok: false, reason: 'invalid-slug' | 'not-found' }`.
|
|
891
|
+
* `cwd`, if given, narrows the search to that one project's own PRD dirs
|
|
892
|
+
* (prdDirForCwd + its Epic-scoped dirs — same pattern as getPrdParsed);
|
|
893
|
+
* omitted, it searches every candidate dir machine-wide via findPrdDir.
|
|
894
|
+
*/
|
|
895
|
+
async function resolveSlugOrReason(slug, cwd) {
|
|
896
|
+
if (!SCHEDULE_SLUG_RE.test(slug)) return { ok: false, reason: 'invalid-slug' };
|
|
897
|
+
if (cwd) {
|
|
898
|
+
for (const dir of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
|
|
899
|
+
const p = safeSlugPathIn(dir, slug);
|
|
900
|
+
if (!p) continue;
|
|
901
|
+
try {
|
|
902
|
+
await fsp.access(p);
|
|
903
|
+
return { ok: true, path: p };
|
|
904
|
+
} catch { /* not in this dir — try the next candidate */ }
|
|
905
|
+
}
|
|
906
|
+
return { ok: false, reason: 'not-found' };
|
|
907
|
+
}
|
|
908
|
+
const dir = await findPrdDir(slug);
|
|
909
|
+
if (!dir) return { ok: false, reason: 'not-found' };
|
|
910
|
+
const p = safeSlugPathIn(dir, slug);
|
|
911
|
+
if (!p) return { ok: false, reason: 'not-found' };
|
|
912
|
+
return { ok: true, path: p };
|
|
913
|
+
}
|
|
914
|
+
|
|
915
|
+
/** Actionable message for `resolveSlugOrReason`'s 'not-found' reason. */
|
|
916
|
+
function unknownSlugMessage(slug) {
|
|
917
|
+
return `unknown slug "${slug}": no PRD file with that name in any known project — call scheduler_list_prds (optionally with cwd) to see what exists`;
|
|
918
|
+
}
|
|
919
|
+
|
|
693
920
|
/**
|
|
694
921
|
* Move a completed job's `<slug>.md` out of its PRD dir into that dir's
|
|
695
922
|
* sibling `prds-archived/`, so a finished slug can't be re-fired by the
|
|
@@ -1098,6 +1325,73 @@ function findOverrunningJobs(jobs, now, { factor, floorMs } = {}) {
|
|
|
1098
1325
|
return out;
|
|
1099
1326
|
}
|
|
1100
1327
|
|
|
1328
|
+
/**
|
|
1329
|
+
* findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
|
|
1330
|
+
* → [{ slug, cwd, ageMs, restoreStatus }]
|
|
1331
|
+
*
|
|
1332
|
+
* Pure (besides the warn-log side effect on the two unprovable-age cases
|
|
1333
|
+
* below), no other IO. spawnInvestigation's own restore of a job's
|
|
1334
|
+
* pre-investigation status runs entirely inside the process that spawned the
|
|
1335
|
+
* probe (its withChildAndLog onExit handler, or the synchronous-throw catch
|
|
1336
|
+
* path) — so a job left 'investigating' when the app itself dies or restarts
|
|
1337
|
+
* has NOTHING left to restore it. The comment at spawnInvestigation's onExit
|
|
1338
|
+
* asserts "'investigating' must never be the job's resting state"; this is
|
|
1339
|
+
* the sweep that makes that true across a restart, not just within one.
|
|
1340
|
+
*
|
|
1341
|
+
* A row qualifies only when ALL of:
|
|
1342
|
+
* - status is 'investigating'
|
|
1343
|
+
* - its most recent transition INTO 'investigating' (statusHistory's last
|
|
1344
|
+
* `to === 'investigating'` entry — a job can be investigated more than
|
|
1345
|
+
* once across its life, e.g. a retried auto-fix) is older than `maxMs`
|
|
1346
|
+
* - it has no live probe process behind it (checked via runtime.pid, set by
|
|
1347
|
+
* spawnInvestigation once its child spawns and cleared on every restore
|
|
1348
|
+
* path, the same shape reapDeadRunningJobs already uses for 'running' rows)
|
|
1349
|
+
*
|
|
1350
|
+
* `restoreStatus` is that transition entry's `from` — the exact value
|
|
1351
|
+
* spawnInvestigation itself would have restored to (`failedJob.status ||
|
|
1352
|
+
* 'failed'`), which for a row that already finished and recorded
|
|
1353
|
+
* finishedAt+exitCode (the burrow-834 shape) is whatever terminal status was
|
|
1354
|
+
* computed for that outcome BEFORE the probe was spawned — this sweep never
|
|
1355
|
+
* re-derives it from exitCode, only replays the already-recorded decision.
|
|
1356
|
+
*
|
|
1357
|
+
* A row with no recoverable transition timestamp cannot have its age proven,
|
|
1358
|
+
* so it is warn-logged and left alone rather than guessed at — same posture
|
|
1359
|
+
* as findStaleQuarantinedJobs/findOverrunningJobs.
|
|
1360
|
+
*
|
|
1361
|
+
* `restoreStatus` is validated against LEGAL_TRANSITIONS['investigating']
|
|
1362
|
+
* before being returned — `statusHistory`'s `from` should only ever be
|
|
1363
|
+
* 'failed' or 'needs_review' (the only two states LEGAL_TRANSITIONS allows
|
|
1364
|
+
* into 'investigating'), but a corrupted/unexpected value must not be handed
|
|
1365
|
+
* straight to transitionJob: an illegal target is refused outright (row stays
|
|
1366
|
+
* stuck at 'investigating', re-detected as stranded every sweep with no path
|
|
1367
|
+
* out), so an out-of-set `from` falls back to 'failed' here instead.
|
|
1368
|
+
*/
|
|
1369
|
+
const INVESTIGATING_RESTORE_TARGETS = new Set(LEGAL_TRANSITIONS.investigating);
|
|
1370
|
+
function findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive) {
|
|
1371
|
+
const out = [];
|
|
1372
|
+
for (const j of jobs ?? []) {
|
|
1373
|
+
if (j.status !== 'investigating') continue;
|
|
1374
|
+
const entries = (j.statusHistory || []).filter((h) => h.to === 'investigating');
|
|
1375
|
+
const entry = entries[entries.length - 1];
|
|
1376
|
+
if (!entry) {
|
|
1377
|
+
console.warn(`[scheduler] findStrandedInvestigations: ${j.slug} is 'investigating' with no statusHistory entry recording the transition — cannot prove age, leaving alone`);
|
|
1378
|
+
continue;
|
|
1379
|
+
}
|
|
1380
|
+
const since = Date.parse(entry.at ?? '');
|
|
1381
|
+
if (Number.isNaN(since)) {
|
|
1382
|
+
console.warn(`[scheduler] findStrandedInvestigations: ${j.slug} has an unparseable investigating-transition timestamp (${entry.at}) — cannot prove age, leaving alone`);
|
|
1383
|
+
continue;
|
|
1384
|
+
}
|
|
1385
|
+
const ageMs = now - since;
|
|
1386
|
+
if (ageMs < maxMs) continue; // a live probe must not be yanked out from under itself
|
|
1387
|
+
const pid = j.runtime?.pid;
|
|
1388
|
+
if (pid && isAlive(pid)) continue; // probe genuinely still running — not stranded
|
|
1389
|
+
const restoreStatus = INVESTIGATING_RESTORE_TARGETS.has(entry.from) ? entry.from : 'failed';
|
|
1390
|
+
out.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, restoreStatus });
|
|
1391
|
+
}
|
|
1392
|
+
return out;
|
|
1393
|
+
}
|
|
1394
|
+
|
|
1101
1395
|
// An empty queue and an unreadable queue are NOT the same thing, and
|
|
1102
1396
|
// conflating them is destructive: reconcile() treats every PRD .md with no
|
|
1103
1397
|
// matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
|
|
@@ -1443,9 +1737,11 @@ async function reconcile(state) {
|
|
|
1443
1737
|
// membership, so moving the file between Epic dirs must re-point the row.
|
|
1444
1738
|
epicId: p.epicId ?? job.epicId ?? null,
|
|
1445
1739
|
dependsOn: p.dependsOn,
|
|
1740
|
+
quietMachine: p.quietMachine === true,
|
|
1446
1741
|
originSessionId: job.originSessionId
|
|
1447
1742
|
?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
|
|
1448
1743
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1744
|
+
agentType: p.agentType ?? job.agentType ?? null,
|
|
1449
1745
|
};
|
|
1450
1746
|
// Adopt path: a row parked 'quarantined' (no createdVia provenance when
|
|
1451
1747
|
// discovered) whose PRD file now carries a stamp — written via the
|
|
@@ -1554,8 +1850,10 @@ async function reconcile(state) {
|
|
|
1554
1850
|
sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
|
|
1555
1851
|
epicId: p.epicId ?? inv.row?.epicId ?? null,
|
|
1556
1852
|
dependsOn: p.dependsOn,
|
|
1853
|
+
quietMachine: p.quietMachine === true,
|
|
1557
1854
|
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
1558
1855
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1856
|
+
agentType: p.agentType ?? inv.row?.agentType ?? null,
|
|
1559
1857
|
};
|
|
1560
1858
|
const reason = `reconcile: repaired invalid status ${JSON.stringify(oldStatus)}`;
|
|
1561
1859
|
// A repair is not a lifecycle transition — the corrupted `status` was
|
|
@@ -1674,9 +1972,15 @@ async function reconcile(state) {
|
|
|
1674
1972
|
sourceTabId: p.sourceTabId,
|
|
1675
1973
|
epicId: p.epicId ?? null,
|
|
1676
1974
|
dependsOn: p.dependsOn,
|
|
1975
|
+
quietMachine: p.quietMachine === true,
|
|
1677
1976
|
originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
1678
1977
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1978
|
+
agentType: p.agentType ?? null,
|
|
1679
1979
|
status: 'pending',
|
|
1980
|
+
// Enqueue time (PRD 1086/1087): the cross-project fairness tiebreak and
|
|
1981
|
+
// the starvation escalation both need a provable age for a pending row;
|
|
1982
|
+
// before this stamp a freshly minted row carried no timestamp at all.
|
|
1983
|
+
queuedAt: new Date().toISOString(),
|
|
1680
1984
|
runId: null,
|
|
1681
1985
|
startedAt: null,
|
|
1682
1986
|
finishedAt: null,
|
|
@@ -1864,6 +2168,9 @@ function drainDeferredInvestigation() {
|
|
|
1864
2168
|
let cancelToken = { cancelled: false };
|
|
1865
2169
|
// Last memory-gate observation; included in snapshot for renderer visibility.
|
|
1866
2170
|
let lastMemGate = null;
|
|
2171
|
+
// CPU-load launch gate (PRD 1085, lib/loadGate.cjs) — innermost launch
|
|
2172
|
+
// predicate after pool → project cap → memory. Withholds launches only.
|
|
2173
|
+
const loadGate = createLoadGate();
|
|
1867
2174
|
|
|
1868
2175
|
// Last tickQueue outcome, kept for the UI. tickQueue already computes a precise
|
|
1869
2176
|
// reason for every way a batch can come back empty (dependency holds, slot
|
|
@@ -1924,6 +2231,10 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
|
|
|
1924
2231
|
lastRunAt: state.lastRunAt,
|
|
1925
2232
|
nextReset: getNextResetCached(),
|
|
1926
2233
|
paused: state.paused,
|
|
2234
|
+
// Launch circuit breaker (issue #11): which personas cannot launch right
|
|
2235
|
+
// now and why, plus any degraded-mode env in force. Empty objects when healthy.
|
|
2236
|
+
launchBlocks: state.launchBlocks ?? {},
|
|
2237
|
+
launchMitigations: state.launchMitigations ?? {},
|
|
1927
2238
|
utilization: cachedUtilization,
|
|
1928
2239
|
pollHealth: {
|
|
1929
2240
|
lastPollAt,
|
|
@@ -1932,6 +2243,8 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
|
|
|
1932
2243
|
lastFailureKind,
|
|
1933
2244
|
},
|
|
1934
2245
|
memGate: lastMemGate,
|
|
2246
|
+
// Why nothing is launching when the box is CPU-saturated (PRD 1085).
|
|
2247
|
+
loadGate: loadGate.snapshot(),
|
|
1935
2248
|
lastTick,
|
|
1936
2249
|
// The machine-wide slot pool IS the concurrency limit — there is no
|
|
1937
2250
|
// separate scheduler cap any more. `source` distinguishes the
|
|
@@ -2065,7 +2378,17 @@ async function setPaused(reason, resumeAtIso) {
|
|
|
2065
2378
|
|
|
2066
2379
|
async function clearPause(source) {
|
|
2067
2380
|
if (resumeTimer) { clearTimeout(resumeTimer); resumeTimer = null; }
|
|
2381
|
+
const humanOverride = source === 'manual' || source === 'run-now';
|
|
2068
2382
|
const wasPaused = await mutate((s) => {
|
|
2383
|
+
// A human Resume / Run now also re-closes every launch circuit breaker:
|
|
2384
|
+
// the operator is asserting the environment is fixed (CLI updated,
|
|
2385
|
+
// re-logged-in). The next dispatch of each persona is its probe; if the
|
|
2386
|
+
// environment is still broken the breaker simply re-arms.
|
|
2387
|
+
if (humanOverride && s.launchBlocks && Object.keys(s.launchBlocks).length) {
|
|
2388
|
+
console.log(`[scheduler] clearPause (${source}): clearing launch blocks [${Object.keys(s.launchBlocks).join(', ')}]`);
|
|
2389
|
+
for (const j of s.jobs) if (j.status === 'pending' && j.heldReason && /^launch blocked/.test(j.heldReason)) delete j.heldReason;
|
|
2390
|
+
s.launchBlocks = {};
|
|
2391
|
+
}
|
|
2069
2392
|
if (!s.paused) return false;
|
|
2070
2393
|
console.log(`[scheduler] clearPause (${source || 'manual'})`);
|
|
2071
2394
|
s.paused = null;
|
|
@@ -2113,6 +2436,29 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
2113
2436
|
job.error = errorMsg ?? null;
|
|
2114
2437
|
delete job.runtime;
|
|
2115
2438
|
delete job.verifierVerdict;
|
|
2439
|
+
delete job.uncommittedPaths;
|
|
2440
|
+
delete job.resumeRecoveryAttempted;
|
|
2441
|
+
// Same "this run's outcome, not durable across a reset" category as the
|
|
2442
|
+
// fields above — a stale 'archive' recoveryAction from a prior life of this
|
|
2443
|
+
// slug must never survive a reset and silently exclude a genuinely-new
|
|
2444
|
+
// needs_review episode from selectAutoFixTargets (applyRcaClassification
|
|
2445
|
+
// only overwrites these on a successful RCA write, so without this they
|
|
2446
|
+
// can otherwise linger forever when RCA is disabled or errors).
|
|
2447
|
+
delete job.rcaFailureClass;
|
|
2448
|
+
delete job.rcaRecoveryAction;
|
|
2449
|
+
// Like exitCode: this run's outcome, not durable across a reset — a stale
|
|
2450
|
+
// leak badge from a prior attempt must not linger once the job re-fires.
|
|
2451
|
+
delete job.leakedDescendants;
|
|
2452
|
+
// A pending row is about to re-run fresh — a stale leftover badge or a
|
|
2453
|
+
// stale pre-run baseline from the attempt that just ended must not linger
|
|
2454
|
+
// and be mistaken for THIS (not-yet-run) attempt's own output. spawnJob
|
|
2455
|
+
// persists a brand-new guardBaseline at the next dispatch.
|
|
2456
|
+
delete job.guardBaseline;
|
|
2457
|
+
delete job.guardHeadBefore;
|
|
2458
|
+
delete job.leftoverPaths;
|
|
2459
|
+
delete job.leftoverCount;
|
|
2460
|
+
delete job.leftoverPathsTruncated;
|
|
2461
|
+
delete job.preRunDirtyPaths;
|
|
2116
2462
|
// Deliberately NOT deleting job.landedCommit: it must outlive a reset so a
|
|
2117
2463
|
// re-fired run of this same slug can pass it to verifyRun as
|
|
2118
2464
|
// priorLandedCommit (pass_no_commit_prior_run_verified exemption).
|
|
@@ -2609,9 +2955,15 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
|
|
|
2609
2955
|
* tree was dirty — closed by widening that call site's condition, not by
|
|
2610
2956
|
* changing this function's four defenses below, which still apply to both
|
|
2611
2957
|
* shapes identically:
|
|
2612
|
-
* - siblingRunning: a concurrent job in the same
|
|
2613
|
-
* evidence unreliable in both directions (extra
|
|
2614
|
-
* that isn't this job's doing).
|
|
2958
|
+
* - siblingRunning: on a SHARED tree only — a concurrent job in the same
|
|
2959
|
+
* cwd makes working-tree evidence unreliable in both directions (extra
|
|
2960
|
+
* dirt OR a clean tree that isn't this job's doing). Suppressed by
|
|
2961
|
+
* ranInWorktree: when this job ran in its own git worktree, the
|
|
2962
|
+
* newly-dirty set and the integrated HEAD are attributable to this job
|
|
2963
|
+
* alone regardless of what siblings were doing concurrently in their own
|
|
2964
|
+
* worktrees, so the excuse does not apply (PRD 109 shipped 'completed'
|
|
2965
|
+
* with nothing committed specifically because this carve-out fired
|
|
2966
|
+
* unconditionally during a high-concurrency run).
|
|
2615
2967
|
* - jobSelfCommitted: HEAD moved during the run, so the job's deliverable
|
|
2616
2968
|
* landed even if dirt (from a concurrent actor) remains.
|
|
2617
2969
|
* - legitimateNoOp (COMPLETED_EQUIVALENT_VERDICTS): runVerify.cjs's own
|
|
@@ -2631,8 +2983,8 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
|
|
|
2631
2983
|
* still a genuine finish-protocol violation (incident:
|
|
2632
2984
|
* 523-fix-bounded-fix-plan-retry, 2026-07-12).
|
|
2633
2985
|
*/
|
|
2634
|
-
function commitGuardVerdict({ newlyDirty, siblingRunning, jobSelfCommitted, legitimateNoOp, isFixPlanJob, verifyResult }) {
|
|
2635
|
-
if (siblingRunning || jobSelfCommitted || legitimateNoOp) return null;
|
|
2986
|
+
function commitGuardVerdict({ newlyDirty, siblingRunning, ranInWorktree, jobSelfCommitted, legitimateNoOp, isFixPlanJob, verifyResult, salvagePatch }) {
|
|
2987
|
+
if ((siblingRunning && !ranInWorktree) || jobSelfCommitted || legitimateNoOp) return null;
|
|
2636
2988
|
const dirty = newlyDirty || [];
|
|
2637
2989
|
if (dirty.length === 0 && isFixPlanJob) return null;
|
|
2638
2990
|
|
|
@@ -2651,12 +3003,196 @@ function commitGuardVerdict({ newlyDirty, siblingRunning, jobSelfCommitted, legi
|
|
|
2651
3003
|
}
|
|
2652
3004
|
|
|
2653
3005
|
const sample = dirty.slice(0, 3).join(', ');
|
|
3006
|
+
const salvageNote = salvagePatch ? ` — recoverable from salvage patch ${salvagePatch}` : '';
|
|
2654
3007
|
return {
|
|
2655
3008
|
verdict: 'uncommitted_changes',
|
|
2656
|
-
reason: `finish protocol incomplete: ${dirty.length} uncommitted file(s) left in working tree (e.g. ${sample})`,
|
|
3009
|
+
reason: `finish protocol incomplete: ${dirty.length} uncommitted file(s) left in working tree (e.g. ${sample})${salvageNote}`,
|
|
2657
3010
|
downgradeTo: 'needs_review',
|
|
2658
3011
|
annotations: carried.length ? carried : undefined,
|
|
3012
|
+
// The exact dirty-path list, persisted on the job row (see the
|
|
3013
|
+
// commit-guard call site) so a later resume-recovery attempt
|
|
3014
|
+
// (selectResumeRecoveryTarget) can name these paths without re-running
|
|
3015
|
+
// `git status` against a tree that may have moved on since.
|
|
3016
|
+
dirtyPaths: dirty,
|
|
3017
|
+
};
|
|
3018
|
+
}
|
|
3019
|
+
|
|
3020
|
+
// Every path list this job leaves attributed on the row is capped here so a
|
|
3021
|
+
// pathological run (thousands of newly-dirty files) never bloats queue.json
|
|
3022
|
+
// or history.jsonl — the count is still recorded in full via leftoverCount,
|
|
3023
|
+
// only the displayed sample is capped.
|
|
3024
|
+
const LEFTOVER_PATHS_CAP = 50;
|
|
3025
|
+
|
|
3026
|
+
/**
|
|
3027
|
+
* Pure: turn a newly-dirty path list (or null, meaning "couldn't tell" —
|
|
3028
|
+
* never "left nothing") into the `leftoverPaths`/`leftoverCount`/
|
|
3029
|
+
* `leftoverPathsTruncated` triple stamped on a terminal job row, or null when
|
|
3030
|
+
* there is nothing to attribute (empty list, or the list itself is
|
|
3031
|
+
* unavailable). One shape for both the worktree-leftover path and the
|
|
3032
|
+
* in-place baseline-delta path — see this function's callers in spawnJob and
|
|
3033
|
+
* reapDeadRunningJobs, both of which diff against a persisted pre-run
|
|
3034
|
+
* baseline so a human's or a sibling's pre-existing WIP is never
|
|
3035
|
+
* misattributed to this job.
|
|
3036
|
+
*/
|
|
3037
|
+
function leftoverFieldsFrom(paths) {
|
|
3038
|
+
if (!Array.isArray(paths) || paths.length === 0) return null;
|
|
3039
|
+
const fields = {
|
|
3040
|
+
leftoverPaths: paths.slice(0, LEFTOVER_PATHS_CAP),
|
|
3041
|
+
leftoverCount: paths.length,
|
|
2659
3042
|
};
|
|
3043
|
+
if (paths.length > LEFTOVER_PATHS_CAP) fields.leftoverPathsTruncated = true;
|
|
3044
|
+
return fields;
|
|
3045
|
+
}
|
|
3046
|
+
|
|
3047
|
+
/** Stamps (or clears) the leftover-attribution fields on a job row in place. */
|
|
3048
|
+
function applyLeftoverFields(row, paths) {
|
|
3049
|
+
delete row.leftoverPaths;
|
|
3050
|
+
delete row.leftoverCount;
|
|
3051
|
+
delete row.leftoverPathsTruncated;
|
|
3052
|
+
const fields = leftoverFieldsFrom(paths);
|
|
3053
|
+
if (fields) Object.assign(row, fields);
|
|
3054
|
+
}
|
|
3055
|
+
|
|
3056
|
+
// Same bloat concern as LEFTOVER_PATHS_CAP, applied to the PRE-run dirty
|
|
3057
|
+
// snapshot (foreign WIP the job did not create) instead of the post-run
|
|
3058
|
+
// leftover delta.
|
|
3059
|
+
const PRE_RUN_DIRTY_PATHS_CAP = 200;
|
|
3060
|
+
|
|
3061
|
+
/**
|
|
3062
|
+
* Pure: cap a dirty-path list at PRE_RUN_DIRTY_PATHS_CAP, appending a
|
|
3063
|
+
* `+N more` marker entry when truncated, so queue.json/history.jsonl never
|
|
3064
|
+
* take on an unbounded row for a pathologically dirty shared tree. Returns
|
|
3065
|
+
* [] for null/empty input (never null) — callers gate storage/prompt
|
|
3066
|
+
* injection on `.length` the same way carriedPaths already does.
|
|
3067
|
+
*/
|
|
3068
|
+
function capDirtyPaths(paths, cap = PRE_RUN_DIRTY_PATHS_CAP) {
|
|
3069
|
+
if (!Array.isArray(paths) || paths.length === 0) return [];
|
|
3070
|
+
if (paths.length <= cap) return paths.slice();
|
|
3071
|
+
return [...paths.slice(0, cap), `+${paths.length - cap} more`];
|
|
3072
|
+
}
|
|
3073
|
+
|
|
3074
|
+
// Stable, machine-greppable delimiter — a downstream PRD (verifier scoring
|
|
3075
|
+
// foreign-WIP test failures separately) greps the executor log for this
|
|
3076
|
+
// exact marker, so its text must never be reworded casually.
|
|
3077
|
+
const FOREIGN_WIP_DELIMITER = '--- FOREIGN WORKING-TREE STATE (not your work) ---';
|
|
3078
|
+
const FOREIGN_WIP_END_DELIMITER = '--- END FOREIGN WORKING-TREE STATE ---';
|
|
3079
|
+
|
|
3080
|
+
/**
|
|
3081
|
+
* Pure: build the executor-prompt section warning about pre-existing dirty
|
|
3082
|
+
* paths this job does not own — either base WIP carried into an isolated
|
|
3083
|
+
* worktree (PRD 1094's carriedPaths, checked first since it's the more
|
|
3084
|
+
* specific/authoritative case) or the raw pre-run dirty snapshot of a shared
|
|
3085
|
+
* (non-isolated) tree. Returns '' when both lists are empty so a clean spawn
|
|
3086
|
+
* produces a byte-identical prompt to before this section existed.
|
|
3087
|
+
*/
|
|
3088
|
+
function buildForeignWipSection({ preRunDirtyPaths, carriedPaths } = {}) {
|
|
3089
|
+
const carried = Array.isArray(carriedPaths) ? carriedPaths.filter(Boolean) : [];
|
|
3090
|
+
if (carried.length) {
|
|
3091
|
+
return [
|
|
3092
|
+
FOREIGN_WIP_DELIMITER,
|
|
3093
|
+
'This job is running in an isolated git worktree, but the following paths carry uncommitted base-tree work-in-progress that was carried into this checkout so the tree is self-consistent. The authoritative copy of these files lives in the MAIN tree, not this worktree.',
|
|
3094
|
+
'These files were already modified before this job started. They are NOT this job\'s work:',
|
|
3095
|
+
...carried.map((p) => ` ${p}`),
|
|
3096
|
+
'Do not stage, commit, revert, or stash these paths. A test failure confined to these paths is not this job\'s regression.',
|
|
3097
|
+
FOREIGN_WIP_END_DELIMITER,
|
|
3098
|
+
].join('\n');
|
|
3099
|
+
}
|
|
3100
|
+
const dirty = Array.isArray(preRunDirtyPaths) ? preRunDirtyPaths.filter(Boolean) : [];
|
|
3101
|
+
if (dirty.length) {
|
|
3102
|
+
return [
|
|
3103
|
+
FOREIGN_WIP_DELIMITER,
|
|
3104
|
+
'This job is running in a SHARED working tree (not isolated in its own worktree). The following paths were already modified when this job started:',
|
|
3105
|
+
...dirty.map((p) => ` ${p}`),
|
|
3106
|
+
'These files are NOT this job\'s work. Do not stage, commit, revert, or stash them. A test failure confined to these paths is not this job\'s regression.',
|
|
3107
|
+
FOREIGN_WIP_END_DELIMITER,
|
|
3108
|
+
].join('\n');
|
|
3109
|
+
}
|
|
3110
|
+
return '';
|
|
3111
|
+
}
|
|
3112
|
+
|
|
3113
|
+
/**
|
|
3114
|
+
* Resume-first recovery (PRD 1111). A job parked in needs_review with verdict
|
|
3115
|
+
* 'uncommitted_changes' has a live claude session (job.sessionId, minted by
|
|
3116
|
+
* spawnJob's `--session-id`) that already has full context of the work it
|
|
3117
|
+
* left uncommitted — resuming it via `claude -p --resume <sessionId>` lets it
|
|
3118
|
+
* finish its own finish-protocol COMMIT step, instead of spawnInvestigation
|
|
3119
|
+
* cold-reading the log to author a fix-plan PRD that a FRESH session then has
|
|
3120
|
+
* to re-derive that same context for. Pure/no I/O so the eligibility rule can
|
|
3121
|
+
* be unit-tested directly, matching classifyFailureOutcome/commitGuardVerdict.
|
|
3122
|
+
*
|
|
3123
|
+
* Bounded to exactly one attempt via job.resumeRecoveryAttempted, stamped
|
|
3124
|
+
* atomically with the 'running' transition inside spawnJob's own dispatch
|
|
3125
|
+
* mutate (see spawnJob) — never here — so a crash between this function
|
|
3126
|
+
* returning a target and the resume child actually spawning cannot leave the
|
|
3127
|
+
* job re-eligible.
|
|
3128
|
+
*
|
|
3129
|
+
* Kill-switch: SM_RESUME_RECOVERY_DISABLE=1 restores today's behaviour
|
|
3130
|
+
* exactly (always returns null), mirroring SM_RCA_DISABLE/SM_DOD_DISABLE.
|
|
3131
|
+
*/
|
|
3132
|
+
function selectResumeRecoveryTarget(job) {
|
|
3133
|
+
if (process.env.SM_RESUME_RECOVERY_DISABLE === '1') return null;
|
|
3134
|
+
if (!job || job.status !== 'needs_review') return null;
|
|
3135
|
+
if (job.verifierVerdict !== 'uncommitted_changes') return null;
|
|
3136
|
+
if (typeof job.sessionId !== 'string' || job.sessionId.length === 0) return null;
|
|
3137
|
+
if (job.resumeRecoveryAttempted === true) return null;
|
|
3138
|
+
const dirtyPaths = Array.isArray(job.uncommittedPaths)
|
|
3139
|
+
? job.uncommittedPaths.filter((p) => typeof p === 'string' && p.length > 0)
|
|
3140
|
+
: [];
|
|
3141
|
+
if (!dirtyPaths.length) return null;
|
|
3142
|
+
return { slug: job.slug, sessionId: job.sessionId, dirtyPaths, salvagePatch: job.salvagePatch || null };
|
|
3143
|
+
}
|
|
3144
|
+
|
|
3145
|
+
/**
|
|
3146
|
+
* Short deterministic preamble for a resume-recovery dispatch — NEVER the
|
|
3147
|
+
* original PRD body (the resumed session already has that in its own
|
|
3148
|
+
* conversation history; re-embedding it would just waste context and risk
|
|
3149
|
+
* contradicting whatever state the session actually left behind). Names the
|
|
3150
|
+
* exact paths recorded on the parked job row so the resumed run can verify
|
|
3151
|
+
* them on disk before trusting them, rather than re-deriving them itself.
|
|
3152
|
+
*/
|
|
3153
|
+
function buildResumeRecoveryPreamble({ dirtyPaths, salvagePatch }) {
|
|
3154
|
+
const pathList = dirtyPaths.map((p) => `- ${p}`).join('\n');
|
|
3155
|
+
const salvageLine = salvagePatch
|
|
3156
|
+
? `\nA salvage patch of this work was also captured at: ${salvagePatch} — apply it if any of the paths above are missing from the working tree.\n`
|
|
3157
|
+
: '';
|
|
3158
|
+
return `RESUME RECOVERY: your previous run in this same session left uncommitted work on disk and exited before the finish protocol's COMMIT step ran. This is a continuation of that same session, not a new task — do not restart from scratch.
|
|
3159
|
+
|
|
3160
|
+
The following path(s) were recorded as uncommitted when this job was parked for review:
|
|
3161
|
+
${pathList}
|
|
3162
|
+
${salvageLine}
|
|
3163
|
+
Do the following now:
|
|
3164
|
+
1. Run \`git status\` and verify each path above is present on disk and reflects your intended work. If a path is missing, investigate before recreating it — don't blindly redo work that may already be committed or salvaged elsewhere.
|
|
3165
|
+
2. Run the project's verification gate (typecheck/lint/tests) in the FOREGROUND — wait for it to finish and read its real exit code before proceeding. Do not background it.
|
|
3166
|
+
3. If the gate is green, stage exactly the paths you created or modified for this work and commit them: \`git add <path> [<path>...] && git commit -m "<type>(<scope>): <summary>"\`.
|
|
3167
|
+
4. If the gate is red, fix it, then commit.
|
|
3168
|
+
|
|
3169
|
+
As the LAST LINE of your final result text, emit exactly one of:
|
|
3170
|
+
SCHEDULER_VERDICT: PASS
|
|
3171
|
+
SCHEDULER_VERDICT: FAIL <one-line reason>
|
|
3172
|
+
Print PASS only once the commit above has actually landed.`;
|
|
3173
|
+
}
|
|
3174
|
+
|
|
3175
|
+
/**
|
|
3176
|
+
* Pure argv builder for a `claude -p` child spawn, shared so the
|
|
3177
|
+
* resume-vs-fresh-session choice is made in exactly one place. `resume`
|
|
3178
|
+
* selects `--resume <sessionId>` (reconnect) INSTEAD of `--session-id
|
|
3179
|
+
* <sessionId>` (mint) — the two flags are mutually exclusive, never both.
|
|
3180
|
+
* `--model` is always explicit (never left to the CLI's drifting default —
|
|
3181
|
+
* see conventions.md). `systemPrompt`, when given (the PRD's `agentType`
|
|
3182
|
+
* persona body, resolved by agentModelResolve.cjs's resolvePrdPersonaForSpawn),
|
|
3183
|
+
* is passed as `--append-system-prompt` so the executor IS that persona at
|
|
3184
|
+
* launch rather than being asked in prose to adopt one.
|
|
3185
|
+
*/
|
|
3186
|
+
function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }) {
|
|
3187
|
+
return [
|
|
3188
|
+
'-p', prompt,
|
|
3189
|
+
'--model', model,
|
|
3190
|
+
...(systemPrompt ? ['--append-system-prompt', systemPrompt] : []),
|
|
3191
|
+
'--dangerously-skip-permissions',
|
|
3192
|
+
'--output-format', 'stream-json',
|
|
3193
|
+
'--verbose',
|
|
3194
|
+
...(resume ? ['--resume', sessionId] : ['--session-id', sessionId]),
|
|
3195
|
+
];
|
|
2660
3196
|
}
|
|
2661
3197
|
|
|
2662
3198
|
// ---------- execution ----------
|
|
@@ -2677,7 +3213,7 @@ function pickRunDir() {
|
|
|
2677
3213
|
* Watchdogs are declared as an array; the result-tailer's exit-code mapping
|
|
2678
3214
|
* (success+killedBySignal → 0) is scheduler-specific and lives in onExit.
|
|
2679
3215
|
*/
|
|
2680
|
-
async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
3216
|
+
async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget = null, foreignWip = null, launchEnv = null) {
|
|
2681
3217
|
const logPath = path.join(runDir, `${job.slug}.log`);
|
|
2682
3218
|
const metaPath = path.join(runDir, `${job.slug}.meta.json`);
|
|
2683
3219
|
// `cwd` stays the MAIN tree throughout — PRD lookup (findPrdDir/prdPathForJob)
|
|
@@ -2687,7 +3223,10 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2687
3223
|
const cwd = job.cwd || defaultCwd;
|
|
2688
3224
|
const spawnCwd = execCwd || cwd;
|
|
2689
3225
|
const startedAt = Date.now();
|
|
2690
|
-
|
|
3226
|
+
// Resume mode (PRD 1111) reconnects to the SAME session that left the
|
|
3227
|
+
// uncommitted work — reusing its id via `--resume` instead of minting a
|
|
3228
|
+
// fresh one via `--session-id` is the entire point of the recovery.
|
|
3229
|
+
const sessionId = resumeTarget ? resumeTarget.sessionId : randomUUID();
|
|
2691
3230
|
|
|
2692
3231
|
// Phase 1: open log fd so we can emit pre-spawn diagnostics (early-exit
|
|
2693
3232
|
// error paths) before the child is created. withChildAndLog takes ownership
|
|
@@ -2712,14 +3251,23 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2712
3251
|
return { exitCode: -1, durationMs: 0, error: errMsg, sessionId };
|
|
2713
3252
|
}
|
|
2714
3253
|
|
|
3254
|
+
let prompt;
|
|
3255
|
+
let prdPath = null;
|
|
3256
|
+
if (resumeTarget) {
|
|
3257
|
+
// Resume mode (PRD 1111): a short deterministic preamble naming the
|
|
3258
|
+
// recorded dirty paths, NEVER the original PRD body — the resumed
|
|
3259
|
+
// session already has that in its own conversation history via
|
|
3260
|
+
// --resume, and re-embedding it here would just contradict whatever
|
|
3261
|
+
// state the session actually left on disk.
|
|
3262
|
+
prompt = buildResumeRecoveryPreamble({ dirtyPaths: resumeTarget.dirtyPaths, salvagePatch: resumeTarget.salvagePatch });
|
|
3263
|
+
} else {
|
|
2715
3264
|
// Read full PRD body fresh from disk (queue stored only the preview).
|
|
2716
3265
|
// Resolve through findPrdDir's full candidate search (legacy flat dir +
|
|
2717
3266
|
// every project's Epic-scoped dirs) first, so the common case — a live
|
|
2718
3267
|
// Epic-scoped PRD — is a first-try hit instead of probing the retired flat
|
|
2719
3268
|
// dir and only then falling back.
|
|
2720
|
-
let prompt;
|
|
2721
3269
|
const resolvedDir = await findPrdDir(job.slug);
|
|
2722
|
-
|
|
3270
|
+
prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
|
|
2723
3271
|
try {
|
|
2724
3272
|
const parsed = await parsePrd(prdPath);
|
|
2725
3273
|
// The review → security-review → verify → commit finish sequence is
|
|
@@ -2775,14 +3323,17 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2775
3323
|
return { exitCode: -1, durationMs: 0, error: e?.message };
|
|
2776
3324
|
}
|
|
2777
3325
|
}
|
|
3326
|
+
} // end resumeTarget ? preamble : normal-PRD-read
|
|
2778
3327
|
|
|
3328
|
+
let contextDigestApplied = false;
|
|
3329
|
+
let originSessionId = null;
|
|
3330
|
+
if (!resumeTarget) {
|
|
2779
3331
|
// Prepend the Epic's own session digest (PRD 950/958) when this job traces
|
|
2780
3332
|
// back to a known Epic — additive only, never mutates the PRD body itself.
|
|
2781
3333
|
// A missing/unresolved epicId or a digest build failure is a silent no-op:
|
|
2782
3334
|
// the PRD's own body must remain sufficient to complete the job on its own.
|
|
2783
3335
|
const digestEpicId = job.epicId ?? job.sourcePromptId ?? null;
|
|
2784
|
-
|
|
2785
|
-
let contextDigestApplied = false;
|
|
3336
|
+
originSessionId = resolveOriginSessionId(cwd, digestEpicId);
|
|
2786
3337
|
let digestText = '';
|
|
2787
3338
|
if (originSessionId) {
|
|
2788
3339
|
try {
|
|
@@ -2793,12 +3344,39 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2793
3344
|
digestText = '';
|
|
2794
3345
|
}
|
|
2795
3346
|
}
|
|
3347
|
+
// Quiet-machine degraded dispatch (PRD 1107): this job opted into
|
|
3348
|
+
// `quietMachine: true` but waited past quietMachineWaitMs() without the
|
|
3349
|
+
// machine ever going quiet, so pickNextBatch dispatched it anyway rather
|
|
3350
|
+
// than wedge the queue forever. Told to the executor as a plain prompt
|
|
3351
|
+
// line — its own wall-clock/timing acceptance criteria were measured (or
|
|
3352
|
+
// will be measured) under CPU contention from sibling jobs, not on a
|
|
3353
|
+
// quiet machine, so it should not report a timing result as trustworthy
|
|
3354
|
+
// without saying so.
|
|
3355
|
+
if (job.quietLeaseDegraded === true) {
|
|
3356
|
+
prompt = `NOTE: this job requested \`quietMachine: true\` but the machine never went idle within the `
|
|
3357
|
+
+ `configured wait window, so it was dispatched anyway (degraded). Any timing/frame-rate/performance `
|
|
3358
|
+
+ `measurement in this run may be affected by CPU contention from other concurrent jobs — say so explicitly `
|
|
3359
|
+
+ `in your result rather than reporting it as a clean measurement.\n\n${prompt}`;
|
|
3360
|
+
}
|
|
2796
3361
|
// Always route through composeExecutorPrompt (even with an empty digest)
|
|
2797
3362
|
// so the finish protocol is appended in the prompt's tail exactly once,
|
|
2798
3363
|
// after any digest fence rather than concatenated ahead of it.
|
|
2799
3364
|
prompt = composeExecutorPrompt({ prdBody: prompt, digestText, finishProtocol: FINISH_PROTOCOL });
|
|
2800
3365
|
|
|
2801
|
-
|
|
3366
|
+
// Foreign-WIP manifest (starry-night-ships PRD 148 postmortem): the
|
|
3367
|
+
// scheduler already knows, at spawn time, which dirty paths this job did
|
|
3368
|
+
// not create — either a shared tree's pre-existing dirty set or worktree
|
|
3369
|
+
// WIP carried in from the base tree (PRD 1094). Telling the executor
|
|
3370
|
+
// explicitly here means it never has to bisect by content to prove a test
|
|
3371
|
+
// failure isn't its own regression. '' (clean spawn) leaves prompt
|
|
3372
|
+
// byte-identical to before this section existed.
|
|
3373
|
+
const foreignWipSection = buildForeignWipSection(foreignWip || {});
|
|
3374
|
+
if (foreignWipSection) {
|
|
3375
|
+
prompt = `${prompt}\n\n${foreignWipSection}`;
|
|
3376
|
+
}
|
|
3377
|
+
} // end !resumeTarget digest/finish-protocol composition
|
|
3378
|
+
|
|
3379
|
+
const promptCheck = validatePromptForSpawn(prompt, resumeTarget ? `<resume recovery preamble for ${job.slug}>` : prdPath);
|
|
2802
3380
|
if (!promptCheck.ok) {
|
|
2803
3381
|
safeLog(`[scheduler] ${promptCheck.error}\n`);
|
|
2804
3382
|
closeFd();
|
|
@@ -2806,6 +3384,15 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2806
3384
|
return { exitCode: -1, durationMs: 0, error: promptCheck.error, sessionId };
|
|
2807
3385
|
}
|
|
2808
3386
|
|
|
3387
|
+
// PRD agentType → persona + model (PRD 1115): resolved for both fresh and
|
|
3388
|
+
// resume dispatches, keyed off job.agentType (persisted on the queue row
|
|
3389
|
+
// by reconcile()) rather than re-reading the PRD file — a resumed session
|
|
3390
|
+
// must keep launching as the SAME persona it started as. Never throws;
|
|
3391
|
+
// a dangling/absent agentType falls back to no persona + FALLBACK_MODEL
|
|
3392
|
+
// and is logged once by resolvePrdPersonaForSpawn itself.
|
|
3393
|
+
const personaResolution = await agentModelResolve.resolvePrdPersonaForSpawn({ cwd, agentType: job.agentType });
|
|
3394
|
+
safeLog(`[scheduler] agentType=${job.agentType || '(none)'} persona=${personaResolution.personaPath || '(fallback — no persona applied)'} model=${personaResolution.model}\n`);
|
|
3395
|
+
|
|
2809
3396
|
return await new Promise((resolve) => {
|
|
2810
3397
|
const claudeBin = resolveClaudeBin();
|
|
2811
3398
|
// Strip Claude Code env and secrets that leak in when session-manager is
|
|
@@ -2813,7 +3400,31 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2813
3400
|
// overrides `--model sonnet`, so scheduled jobs burn Opus credits silently.
|
|
2814
3401
|
// PATH must include Homebrew/user bins or the job's node/git children ENOENT
|
|
2815
3402
|
// when Electron was launched from Finder/Dock on macOS (stripped PATH).
|
|
2816
|
-
|
|
3403
|
+
// SM_PROJECT_ROOT is the main-tree cwd (never spawnCwd, which may be a
|
|
3404
|
+
// job/epic worktree) — forwarded by scheduler-mcp-server.cjs as
|
|
3405
|
+
// originProjectRoot so a job running inside its own worktree can still
|
|
3406
|
+
// resolve the real project for create-prd/open-session/readiness. See
|
|
3407
|
+
// projectRootResolve.cjs.
|
|
3408
|
+
// SM_SCHEDULER_JOB_SLUG marks the child (and the MCP servers it
|
|
3409
|
+
// inherits its env to) as a headless scheduled executor, so
|
|
3410
|
+
// scheduler-mcp-server.cjs can refuse scheduler_create_prd from inside a
|
|
3411
|
+
// run (issue #11 list C1 — the PRD 460 self-queue incident). Only a
|
|
3412
|
+
// persona whose whole job is decomposition may still queue.
|
|
3413
|
+
// `launchEnv` is the launch circuit breaker's degraded-mode env (e.g.
|
|
3414
|
+
// MAX_THINKING_TOKENS=0 while an outdated CLI's thinking parameter is
|
|
3415
|
+
// being rejected — lib/launchFailure.cjs); applied last so it wins.
|
|
3416
|
+
const childEnv = cleanChildEnv({
|
|
3417
|
+
PATH: pathWithUserBins(),
|
|
3418
|
+
SM_PROJECT_ROOT: cwd,
|
|
3419
|
+
SM_SCHEDULER_JOB_SLUG: job.slug,
|
|
3420
|
+
SM_SCHEDULER_JOB_MAY_QUEUE: job.agentType === 'architect' ? '1' : '0',
|
|
3421
|
+
BASH_DEFAULT_TIMEOUT_MS: String(BASH_DEFAULT_TIMEOUT_MS),
|
|
3422
|
+
BASH_MAX_TIMEOUT_MS: String(BASH_MAX_TIMEOUT_MS),
|
|
3423
|
+
...(launchEnv && typeof launchEnv === 'object' ? launchEnv : {}),
|
|
3424
|
+
});
|
|
3425
|
+
if (launchEnv && Object.keys(launchEnv).length) {
|
|
3426
|
+
safeLog(`[scheduler] launch mitigation env applied: ${Object.entries(launchEnv).map(([k, v]) => `${k}=${v}`).join(' ')}\n`);
|
|
3427
|
+
}
|
|
2817
3428
|
|
|
2818
3429
|
// Track whether the agent has emitted a `result` event in its JSONL stream.
|
|
2819
3430
|
// null until seen; then one of "success" | "error_max_turns" | … per the
|
|
@@ -2920,14 +3531,16 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2920
3531
|
closeFd,
|
|
2921
3532
|
spawn: {
|
|
2922
3533
|
command: claudeBin,
|
|
2923
|
-
|
|
2924
|
-
|
|
2925
|
-
|
|
2926
|
-
|
|
2927
|
-
|
|
2928
|
-
|
|
2929
|
-
|
|
2930
|
-
|
|
3534
|
+
// Resume mode passes `--resume <sessionId>` (reconnect to the SAME
|
|
3535
|
+
// session) INSTEAD of `--session-id <sessionId>` (mint a new one) —
|
|
3536
|
+
// never both, see buildClaudeSpawnArgs.
|
|
3537
|
+
args: buildClaudeSpawnArgs({
|
|
3538
|
+
prompt,
|
|
3539
|
+
model: personaResolution.model,
|
|
3540
|
+
sessionId,
|
|
3541
|
+
resume: !!resumeTarget,
|
|
3542
|
+
systemPrompt: personaResolution.systemPrompt,
|
|
3543
|
+
}),
|
|
2931
3544
|
options: {
|
|
2932
3545
|
cwd: spawnCwd,
|
|
2933
3546
|
env: childEnv,
|
|
@@ -2940,8 +3553,13 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2940
3553
|
},
|
|
2941
3554
|
},
|
|
2942
3555
|
watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog],
|
|
2943
|
-
onExit({ exitCode, signal, killedByWatchdog: _kbw, error, spawnFailed, safeLog: sl }) {
|
|
3556
|
+
onExit({ exitCode, signal, killedByWatchdog: _kbw, error, spawnFailed, leakedDescendants, safeLog: sl }) {
|
|
2944
3557
|
const durationMs = Date.now() - startedAt;
|
|
3558
|
+
const leaked = leakedDescendants ?? [];
|
|
3559
|
+
if (leaked.length > 0) {
|
|
3560
|
+
sl(`\n[scheduler] leaked ${leaked.length} descendant(s) swept from job process group: ` +
|
|
3561
|
+
`${leaked.map((p) => `pid=${p.pid} comm=${p.comm} pcpu=${p.pcpu} etimes=${p.etimes}s`).join(', ')}\n`);
|
|
3562
|
+
}
|
|
2945
3563
|
|
|
2946
3564
|
if (error) {
|
|
2947
3565
|
// Covers both synchronous spawn failure and child 'error' events.
|
|
@@ -2951,8 +3569,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2951
3569
|
sl(`\n[scheduler] ${errMsg}\n`);
|
|
2952
3570
|
// Sync write: inside a Promise executor callback; must flush meta
|
|
2953
3571
|
// before resolve() so the spawnJob mutate() that follows sees it.
|
|
2954
|
-
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA, originSessionId, contextDigestApplied });
|
|
2955
|
-
resolve({ exitCode: -1, durationMs, error: errMsg, sessionId });
|
|
3572
|
+
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA, originSessionId, contextDigestApplied });
|
|
3573
|
+
resolve({ exitCode: -1, durationMs, error: errMsg, leakedDescendants: leaked, sessionId });
|
|
2956
3574
|
return;
|
|
2957
3575
|
}
|
|
2958
3576
|
|
|
@@ -2975,16 +3593,33 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2975
3593
|
`duration=${Math.round(durationMs / 1000)}s\n`);
|
|
2976
3594
|
const rateLimited = effectiveCode !== 0 && detectRateLimitInLog(logPath);
|
|
2977
3595
|
const networkError = effectiveCode !== 0 && !rateLimited && detectNetworkErrorInLog(logPath);
|
|
3596
|
+
// Non-run detection (issue #11 lists A1–A3): the harness's `result`
|
|
3597
|
+
// event tells us whether the model ever got a turn. A first-request
|
|
3598
|
+
// API rejection (num_turns ≤ 1, output_tokens 0, `API Error:` text)
|
|
3599
|
+
// is a broken ENVIRONMENT, not a failed PRD — spawnJob routes it to
|
|
3600
|
+
// the launch circuit breaker instead of failed/investigation.
|
|
3601
|
+
const resultStats = launchFailure.readResultEvent(logPath);
|
|
3602
|
+
const launchFailed = (effectiveCode !== 0 && !rateLimited && !networkError)
|
|
3603
|
+
? launchFailure.classifyLaunchFailure(resultStats)
|
|
3604
|
+
: null;
|
|
3605
|
+
if (launchFailed) {
|
|
3606
|
+
sl(`\n[scheduler] LAUNCH FAILURE (${launchFailed.kind}${launchFailed.httpStatus ? ` HTTP ${launchFailed.httpStatus}` : ''}): ` +
|
|
3607
|
+
`${launchFailed.message} — no turn was taken; this is not a PRD failure\n`);
|
|
3608
|
+
}
|
|
2978
3609
|
// Sync write: child 'exit' handler must flush meta before resolve()
|
|
2979
3610
|
// so the spawnJob mutate() that follows sees the persisted exit code.
|
|
2980
3611
|
config.writeJsonSync(metaPath, {
|
|
2981
3612
|
slug: job.slug, cwd, sessionId, exitCode: effectiveCode, rateLimited, networkError,
|
|
2982
|
-
|
|
3613
|
+
launchFailure: launchFailed,
|
|
3614
|
+
numTurns: resultStats?.numTurns ?? null, outputTokens: resultStats?.outputTokens ?? null,
|
|
3615
|
+
totalCostUsd: resultStats?.totalCostUsd ?? null, terminalReasonFromHarness: resultStats?.terminalReason ?? null,
|
|
3616
|
+
launchEnvApplied: launchEnv && Object.keys(launchEnv).length ? Object.keys(launchEnv) : [],
|
|
3617
|
+
startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
|
|
2983
3618
|
agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
|
|
2984
3619
|
schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
|
|
2985
3620
|
originSessionId, contextDigestApplied,
|
|
2986
3621
|
});
|
|
2987
|
-
resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, sessionId });
|
|
3622
|
+
resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, launchFailure: launchFailed, resultStats, leakedDescendants: leaked, sessionId });
|
|
2988
3623
|
},
|
|
2989
3624
|
});
|
|
2990
3625
|
|
|
@@ -3049,7 +3684,27 @@ function healTargetForFix(fixSlug, jobs) {
|
|
|
3049
3684
|
* spawnInvestigation computes.
|
|
3050
3685
|
*/
|
|
3051
3686
|
function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group }) {
|
|
3052
|
-
|
|
3687
|
+
const abandonedBackgroundTaskNote = failedJob.verifierVerdict === 'abandoned_background_task' ? `
|
|
3688
|
+
|
|
3689
|
+
# Known failure class: abandoned background task
|
|
3690
|
+
This job's verifier verdict is \`abandoned_background_task\`: the transcript shows a Bash command
|
|
3691
|
+
auto-backgrounded past its foreground timeout, and the run ended waiting for a "you will be
|
|
3692
|
+
notified when it completes" callback a headless run structurally cannot receive. This is NOT
|
|
3693
|
+
evidence the work failed — it is evidence the run stopped short of its finish protocol. The work is
|
|
3694
|
+
usually already written and correct; only the commit is missing.
|
|
3695
|
+
|
|
3696
|
+
By the time this investigation runs, the failed job's isolated worktree (if it ran in one) has
|
|
3697
|
+
already been cleaned up — \`${cwd}\` is the BASE repo, not that worktree, so a plain \`git status\`/
|
|
3698
|
+
\`git diff\` there will usually show nothing even though real work was produced. The scheduler
|
|
3699
|
+
salvages any uncommitted diff from a killed job's worktree BEFORE deleting it${
|
|
3700
|
+
failedJob.salvagePatch ? `, and this job's salvage patch was captured at:\n\n ${failedJob.salvagePatch}` : ', to a `.uncommitted.patch` file next to the run log — check the run dir for one'
|
|
3701
|
+
}.
|
|
3702
|
+
|
|
3703
|
+
The fix-plan PRD you write for this MUST instruct its executor to, in order:
|
|
3704
|
+
1. Check for a salvage patch (named \`<slug>.uncommitted.patch\` in the run directory${failedJob.salvagePatch ? `, e.g. \`${failedJob.salvagePatch}\`` : ''}) and, if found, apply it to the working tree BEFORE inspecting \`git status\`/\`git diff\` in ${cwd} for uncommitted changes matching the original PRD's acceptance criteria.
|
|
3705
|
+
2. If the work is present (via the applied patch or already in the tree) and satisfies the acceptance criteria, run the project's verify commands and COMMIT it — do not re-implement or re-plan the PRD from scratch.
|
|
3706
|
+
3. Only fall back to re-implementing whatever acceptance criteria are genuinely missing after applying any salvage patch, not the whole PRD.` : '';
|
|
3707
|
+
return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${abandonedBackgroundTaskNote}
|
|
3053
3708
|
|
|
3054
3709
|
# Failed job
|
|
3055
3710
|
- Slug: ${failedJob.slug}
|
|
@@ -3167,7 +3822,37 @@ function readRunOutcomeSidecars(runDir, slug) {
|
|
|
3167
3822
|
* or if the run being investigated actually verified clean (nothing to fix —
|
|
3168
3823
|
* see shouldSkipInvestigationForCleanRun).
|
|
3169
3824
|
*/
|
|
3825
|
+
const INVESTIGATION_LAUNCH_KEY = 'investigation';
|
|
3826
|
+
|
|
3170
3827
|
async function spawnInvestigation(failedJob, runDir) {
|
|
3828
|
+
// The probe launches with the same CLI as the job it diagnoses. While
|
|
3829
|
+
// that CLI cannot launch at all (launch circuit breaker, issue #11 list
|
|
3830
|
+
// B1: probes e4f82da2/d374e6bf died on the same HTTP 400 as the runs
|
|
3831
|
+
// they were investigating) there is nothing to diagnose — skip, loudly.
|
|
3832
|
+
{
|
|
3833
|
+
const state = await readQueue().catch(() => null);
|
|
3834
|
+
const block = state?.launchBlocks?.[INVESTIGATION_LAUNCH_KEY];
|
|
3835
|
+
const jobBlock = state?.launchBlocks?.[launchFailure.launchBlockKeyFor(failedJob)];
|
|
3836
|
+
const gate = launchFailure.evaluateLaunchGate(block || jobBlock, { now: Date.now(), claudeVersion: await probeClaudeVersion() });
|
|
3837
|
+
if (gate.state === 'blocked') {
|
|
3838
|
+
console.log(`[scheduler] skip investigation: ${failedJob.slug} — ${gate.reason}`);
|
|
3839
|
+
await mutate((s) => {
|
|
3840
|
+
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
3841
|
+
if (j) { j.autoFixOutcome = 'launch-blocked'; j.autoFixNote = gate.reason; }
|
|
3842
|
+
}).catch(() => {});
|
|
3843
|
+
return { deferred: false };
|
|
3844
|
+
}
|
|
3845
|
+
}
|
|
3846
|
+
// Resume-first recovery (PRD 1111) always gets first refusal — a job
|
|
3847
|
+
// eligible for a bounded `--resume` dispatch must never also get a
|
|
3848
|
+
// cold-read fix-plan PRD authored in the same pass. selectResumeRecoveryTarget
|
|
3849
|
+
// returns null for every job shape spawnInvestigation is normally called
|
|
3850
|
+
// with (e.g. plain 'failed' jobs never carry verifierVerdict
|
|
3851
|
+
// 'uncommitted_changes'), so this is a no-op for the common case.
|
|
3852
|
+
if (selectResumeRecoveryTarget(failedJob)) {
|
|
3853
|
+
console.log(`[scheduler] skip investigation: ${failedJob.slug} is resume-recovery eligible`);
|
|
3854
|
+
return { deferred: false };
|
|
3855
|
+
}
|
|
3171
3856
|
if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth)) {
|
|
3172
3857
|
console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
|
|
3173
3858
|
return { deferred: false };
|
|
@@ -3277,7 +3962,11 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3277
3962
|
await broadcast({ flush: true });
|
|
3278
3963
|
|
|
3279
3964
|
const claudeBin = resolveClaudeBin();
|
|
3280
|
-
const childEnv = cleanChildEnv({
|
|
3965
|
+
const childEnv = cleanChildEnv({
|
|
3966
|
+
PATH: pathWithUserBins(), // Homebrew/user bins for macOS
|
|
3967
|
+
BASH_DEFAULT_TIMEOUT_MS: String(BASH_DEFAULT_TIMEOUT_MS),
|
|
3968
|
+
BASH_MAX_TIMEOUT_MS: String(BASH_MAX_TIMEOUT_MS),
|
|
3969
|
+
});
|
|
3281
3970
|
|
|
3282
3971
|
// Investigation needs only a deadman watchdog — no idle-tail or result-tail
|
|
3283
3972
|
// since investigations are short-running Opus probes with a hard ceiling.
|
|
@@ -3317,7 +4006,10 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3317
4006
|
// 'investigating' must never be the job's resting state.
|
|
3318
4007
|
mutate((s) => {
|
|
3319
4008
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
3320
|
-
if (j && j.status === 'investigating')
|
|
4009
|
+
if (j && j.status === 'investigating') {
|
|
4010
|
+
transitionJob(j, failedJob.status || 'failed', { reason: 'investigation probe exited — restoring prior status', source: 'spawnInvestigation:onExit' });
|
|
4011
|
+
delete j.runtime;
|
|
4012
|
+
}
|
|
3321
4013
|
})
|
|
3322
4014
|
.then(() => broadcast({ flush: true }))
|
|
3323
4015
|
.catch(() => {});
|
|
@@ -3336,6 +4028,23 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3336
4028
|
return;
|
|
3337
4029
|
}
|
|
3338
4030
|
sl(`\n[scheduler] investigation exit code=${exitCode}\n`);
|
|
4031
|
+
if (exitCode !== 0) {
|
|
4032
|
+
const probeResult = launchFailure.readResultEvent(investigationLogPath);
|
|
4033
|
+
const probeLaunchFailure = launchFailure.classifyLaunchFailure(probeResult);
|
|
4034
|
+
if (probeLaunchFailure) {
|
|
4035
|
+
sl(`\n[scheduler] investigation LAUNCH FAILURE (${probeLaunchFailure.kind}): ${probeLaunchFailure.message} — arming '${INVESTIGATION_LAUNCH_KEY}' launch block\n`);
|
|
4036
|
+
probeClaudeVersion().then((claudeVersion) => mutate((s) => {
|
|
4037
|
+
s.launchBlocks = s.launchBlocks || {};
|
|
4038
|
+
s.launchBlocks[INVESTIGATION_LAUNCH_KEY] = launchFailure.armLaunchBlock(s.launchBlocks[INVESTIGATION_LAUNCH_KEY] || null, {
|
|
4039
|
+
kind: probeLaunchFailure.kind, httpStatus: probeLaunchFailure.httpStatus, message: probeLaunchFailure.message,
|
|
4040
|
+
now: Date.now(), claudeVersion, slug: failedJob.slug, runId: failedJob.runId ?? null,
|
|
4041
|
+
});
|
|
4042
|
+
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
4043
|
+
if (j) { j.autoFixOutcome = 'launch-blocked'; j.autoFixNote = `investigation probe never ran: ${probeLaunchFailure.message}`; }
|
|
4044
|
+
})).catch(() => {});
|
|
4045
|
+
return;
|
|
4046
|
+
}
|
|
4047
|
+
}
|
|
3339
4048
|
// Fold the investigation's <RCA> summary into the root-cause report already
|
|
3340
4049
|
// written for this job (needs_review jobs only — writeRcaReport no-ops when
|
|
3341
4050
|
// failedJob has no verifierVerdict, e.g. plain 'failed' jobs never got one).
|
|
@@ -3374,6 +4083,15 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3374
4083
|
|
|
3375
4084
|
if (child) {
|
|
3376
4085
|
safeLog(`[scheduler] investigation pid=${child.pid}\n\n`);
|
|
4086
|
+
// Recorded so findStrandedInvestigations (a post-restart maintenance
|
|
4087
|
+
// sweep — the live process has no other way to know a probe is still
|
|
4088
|
+
// running) can tell a live probe apart from one whose owning process is
|
|
4089
|
+
// long gone, the same way reapDeadRunningJobs checks a running job's
|
|
4090
|
+
// runtime.pid.
|
|
4091
|
+
mutate((s) => {
|
|
4092
|
+
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
4093
|
+
if (j && j.status === 'investigating') j.runtime = { pid: child.pid };
|
|
4094
|
+
}).catch(() => {});
|
|
3377
4095
|
}
|
|
3378
4096
|
return { deferred: false };
|
|
3379
4097
|
} catch (e) {
|
|
@@ -3383,7 +4101,10 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3383
4101
|
releaseSlot();
|
|
3384
4102
|
mutate((s) => {
|
|
3385
4103
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
3386
|
-
if (j && j.status === 'investigating')
|
|
4104
|
+
if (j && j.status === 'investigating') {
|
|
4105
|
+
transitionJob(j, failedJob.status || 'failed', { reason: 'investigation spawn threw before exiting — restoring prior status', source: 'spawnInvestigation:catch' });
|
|
4106
|
+
delete j.runtime;
|
|
4107
|
+
}
|
|
3387
4108
|
})
|
|
3388
4109
|
.then(() => broadcast({ flush: true }))
|
|
3389
4110
|
.catch(() => {});
|
|
@@ -3391,7 +4112,122 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3391
4112
|
}
|
|
3392
4113
|
}
|
|
3393
4114
|
|
|
3394
|
-
|
|
4115
|
+
/**
|
|
4116
|
+
* computeLaunchHolds(state) → Map<slug, reason>
|
|
4117
|
+
*
|
|
4118
|
+
* The launch circuit breaker's per-tick view (lib/launchFailure.cjs, issue
|
|
4119
|
+
* #11): every pending row whose persona is blocked is held with its reason;
|
|
4120
|
+
* when a persona's backoff has elapsed exactly ONE of its pending rows is
|
|
4121
|
+
* left pickable (the half-open probe) and the rest are held behind it. A
|
|
4122
|
+
* CLI version change drops the block outright — that is the incident's real
|
|
4123
|
+
* fix (`claude update`) and the queue must resume on the next tick.
|
|
4124
|
+
* Mutates nothing; spawnJob makes the durable decision at dispatch.
|
|
4125
|
+
*/
|
|
4126
|
+
async function computeLaunchHolds(state, { now = Date.now(), claudeVersion } = {}) {
|
|
4127
|
+
const held = new Map();
|
|
4128
|
+
const blocks = state?.launchBlocks;
|
|
4129
|
+
if (!blocks || typeof blocks !== 'object' || !Object.keys(blocks).length) return held;
|
|
4130
|
+
const version = claudeVersion === undefined ? await probeClaudeVersion() : claudeVersion;
|
|
4131
|
+
const probeAllowed = new Set();
|
|
4132
|
+
for (const j of state.jobs || []) {
|
|
4133
|
+
if (j.status !== 'pending') continue;
|
|
4134
|
+
const key = launchFailure.launchBlockKeyFor(j);
|
|
4135
|
+
const block = blocks[key];
|
|
4136
|
+
if (!block) continue;
|
|
4137
|
+
const gate = launchFailure.evaluateLaunchGate(block, { now, claudeVersion: version });
|
|
4138
|
+
if (gate.state === 'open') continue;
|
|
4139
|
+
if (gate.state === 'probe' && !probeAllowed.has(key)) {
|
|
4140
|
+
probeAllowed.add(key);
|
|
4141
|
+
continue;
|
|
4142
|
+
}
|
|
4143
|
+
held.set(j.slug, gate.state === 'probe'
|
|
4144
|
+
? `launch blocked (${block.kind}) — waiting for this tick's probe of '${key}'`
|
|
4145
|
+
: gate.reason);
|
|
4146
|
+
}
|
|
4147
|
+
return held;
|
|
4148
|
+
}
|
|
4149
|
+
|
|
4150
|
+
/**
|
|
4151
|
+
* A run that never got a turn (res.launchFailure — see executeJob's onExit)
|
|
4152
|
+
* is routed here instead of the failed/investigation path (issue #11 lists
|
|
4153
|
+
* A1–A3, B1): the row goes back to `pending` carrying the API's own message
|
|
4154
|
+
* as its error, no retry budget is consumed, no auto-fix probe is spawned
|
|
4155
|
+
* (it would die the same way), and the persona's launch circuit breaker is
|
|
4156
|
+
* armed so the queue stops re-dispatching identical doomed launches while
|
|
4157
|
+
* still self-healing on backoff / CLI update / human Retry.
|
|
4158
|
+
*/
|
|
4159
|
+
/**
|
|
4160
|
+
* Pure state mutation behind handleLaunchFailure (exported for tests): arms
|
|
4161
|
+
* the persona's breaker and returns the job's `running` row to `pending`
|
|
4162
|
+
* carrying the API message. Returns the armed block.
|
|
4163
|
+
*/
|
|
4164
|
+
function applyLaunchFailure(s, { job, lf, runId, launchKey, mitigationApplied, claudeVersion, now = Date.now() }) {
|
|
4165
|
+
s.launchBlocks = s.launchBlocks || {};
|
|
4166
|
+
s.launchMitigations = s.launchMitigations || {};
|
|
4167
|
+
const prev = s.launchBlocks[launchKey] || null;
|
|
4168
|
+
const armed = launchFailure.armLaunchBlock(prev, {
|
|
4169
|
+
kind: lf.kind, httpStatus: lf.httpStatus, message: lf.message, now, claudeVersion,
|
|
4170
|
+
slug: job.slug, runId, mitigationApplied,
|
|
4171
|
+
});
|
|
4172
|
+
s.launchBlocks[launchKey] = armed;
|
|
4173
|
+
// A mitigation that was in force and still failed is no longer proven —
|
|
4174
|
+
// drop it so the hint and the next probe are honest.
|
|
4175
|
+
if (mitigationApplied && s.launchMitigations[launchKey]) delete s.launchMitigations[launchKey];
|
|
4176
|
+
const i = (s.jobs || []).findIndex((x) => x.slug === job.slug);
|
|
4177
|
+
if (i >= 0 && s.jobs[i].status === 'running') {
|
|
4178
|
+
const prevCount = s.jobs[i].launchFailure?.count ?? 0;
|
|
4179
|
+
const msg = `launch failure (${lf.kind}${lf.httpStatus ? ` HTTP ${lf.httpStatus}` : ''}): ${lf.message}`;
|
|
4180
|
+
resetJobFields(s.jobs[i], msg, { source: 'spawnJob:launch-failure' });
|
|
4181
|
+
s.jobs[i].launchFailure = {
|
|
4182
|
+
kind: lf.kind, httpStatus: lf.httpStatus ?? null, message: lf.message,
|
|
4183
|
+
at: new Date(now).toISOString(), runId, count: prevCount + 1, mitigationApplied,
|
|
4184
|
+
};
|
|
4185
|
+
s.jobs[i].terminalReason = `launch_failure:${lf.kind}`;
|
|
4186
|
+
s.jobs[i].heldReason = armed.exhausted
|
|
4187
|
+
? `launch blocked (${lf.kind}) after ${armed.attempts} failed probe(s) — ${armed.hint}`
|
|
4188
|
+
: `launch blocked (${lf.kind}) — re-probe at ${armed.until}. ${armed.hint}`;
|
|
4189
|
+
}
|
|
4190
|
+
return armed;
|
|
4191
|
+
}
|
|
4192
|
+
|
|
4193
|
+
async function handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion }) {
|
|
4194
|
+
const lf = res.launchFailure;
|
|
4195
|
+
const now = Date.now();
|
|
4196
|
+
const mitigationApplied = !!(launchEnv && Object.keys(launchEnv).length);
|
|
4197
|
+
let armed = null;
|
|
4198
|
+
await mutate((s) => {
|
|
4199
|
+
armed = applyLaunchFailure(s, { job, lf, runId, launchKey, mitigationApplied, claudeVersion, now });
|
|
4200
|
+
});
|
|
4201
|
+
launchFailure.writeOutcomeSidecar(runDir, job.slug, {
|
|
4202
|
+
runId,
|
|
4203
|
+
exitCode: res.exitCode,
|
|
4204
|
+
durationMs: res.durationMs ?? null,
|
|
4205
|
+
numTurns: res.resultStats?.numTurns ?? null,
|
|
4206
|
+
outputTokens: res.resultStats?.outputTokens ?? null,
|
|
4207
|
+
totalCostUsd: res.resultStats?.totalCostUsd ?? null,
|
|
4208
|
+
verdict: null,
|
|
4209
|
+
status: 'pending',
|
|
4210
|
+
terminalReason: `launch_failure:${lf.kind}`,
|
|
4211
|
+
launchFailure: { kind: lf.kind, httpStatus: lf.httpStatus ?? null, message: lf.message },
|
|
4212
|
+
launchEnvApplied: launchEnv ? Object.keys(launchEnv) : [],
|
|
4213
|
+
filesChanged: 0,
|
|
4214
|
+
landedCommit: null,
|
|
4215
|
+
});
|
|
4216
|
+
try {
|
|
4217
|
+
appendError({
|
|
4218
|
+
cwd: job.cwd || DEFAULT_PROJECT_CWD,
|
|
4219
|
+
scope: 'scheduler',
|
|
4220
|
+
level: 'error',
|
|
4221
|
+
message: `launch failure (${lf.kind}) for ${job.slug}: ${lf.message} — persona '${launchKey}' blocked, attempt ${armed?.attempts}${armed?.exhausted ? ' (exhausted; needs CLI update or Retry)' : ''}`,
|
|
4222
|
+
meta: { slug: job.slug, runId, kind: lf.kind, httpStatus: lf.httpStatus ?? null, claudeVersion: claudeVersion ?? null, mitigationApplied, hint: armed?.hint },
|
|
4223
|
+
});
|
|
4224
|
+
} catch { /* durable logging must never break the queue */ }
|
|
4225
|
+
console.error(`[scheduler] ${job.slug}: LAUNCH FAILURE (${lf.kind}${lf.httpStatus ? ` HTTP ${lf.httpStatus}` : ''}) — ${lf.message}. ` +
|
|
4226
|
+
`Persona '${launchKey}' blocked (attempt ${armed?.attempts}${armed?.until ? `, re-probe at ${armed.until}` : ', exhausted'}). ${armed?.hint}`);
|
|
4227
|
+
await broadcast({ flush: true });
|
|
4228
|
+
}
|
|
4229
|
+
|
|
4230
|
+
async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
3395
4231
|
// Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
|
|
3396
4232
|
// — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
|
|
3397
4233
|
// leaves the job pending; the next tick retries when a slot frees up.
|
|
@@ -3401,13 +4237,108 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3401
4237
|
return;
|
|
3402
4238
|
}
|
|
3403
4239
|
runningSet.add(job.slug);
|
|
4240
|
+
// Exclusive quiet-machine lease (PRD 1107) — acquired here, in the same
|
|
4241
|
+
// slot-acquire/dispatch step as sessionSlots, and released in this
|
|
4242
|
+
// function's own finally below alongside sessionSlots.release, so every
|
|
4243
|
+
// exit path (normal exit, timeout, SIGTERM, crash) that already frees the
|
|
4244
|
+
// session slot also frees the lease. pickNextBatch only ever hands this
|
|
4245
|
+
// function a quietMachine job when the lease was free at pick time, so
|
|
4246
|
+
// acquire() here should never fail in practice — but check anyway rather
|
|
4247
|
+
// than assume, since a lease held by a stale slug would otherwise wedge
|
|
4248
|
+
// silently.
|
|
4249
|
+
const quietLeaseAcquired = job.quietMachine === true && quietMachineLease.acquire(job.slug);
|
|
3404
4250
|
try {
|
|
4251
|
+
// Worktree isolation cap check (PRD 1112) — probed BEFORE the job is
|
|
4252
|
+
// marked 'running', so a job that can't get isolation right now is a
|
|
4253
|
+
// DEFERRAL, not a fallback: it stays 'pending' and is retried on the
|
|
4254
|
+
// next dispatch pass, exactly like the sessionSlots miss above, instead
|
|
4255
|
+
// of degrading into an in-place run in a tree a sibling job may be
|
|
4256
|
+
// actively writing to (the shared-tree collision this cap exists to
|
|
4257
|
+
// prevent). Every OTHER worktree.ok===false reason (not a git repo,
|
|
4258
|
+
// disabled, carry-over failure) keeps the existing in-place fallback —
|
|
4259
|
+
// only the cap-reached reason is a deferral, checked here via
|
|
4260
|
+
// createJobWorktree's own reason string so the two paths never
|
|
4261
|
+
// silently drift out of sync with gitWorktree.cjs's actual wording.
|
|
4262
|
+
const preflightWorktree = resumeTarget
|
|
4263
|
+
? { ok: false, reason: 'resume-recovery: running in place to reuse the session\'s prior working tree' }
|
|
4264
|
+
: await jobWorktree.createJobWorktree({ cwd: job.cwd || defaultCwd, slug: job.slug });
|
|
4265
|
+
if (!preflightWorktree.ok && /^worktree cap reached\b/.test(preflightWorktree.reason || '')) {
|
|
4266
|
+
console.log(`[scheduler] ${job.slug}: deferring — ${preflightWorktree.reason}`);
|
|
4267
|
+
await mutate((s) => {
|
|
4268
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4269
|
+
if (idx >= 0) s.jobs[idx].heldReason = preflightWorktree.reason;
|
|
4270
|
+
});
|
|
4271
|
+
await broadcast({ flush: true });
|
|
4272
|
+
return;
|
|
4273
|
+
}
|
|
4274
|
+
// Launch circuit breaker (lib/launchFailure.cjs, issue #11). Re-evaluated
|
|
4275
|
+
// here, not just in tickQueue, because the block can change between the
|
|
4276
|
+
// pick and this dispatch (another job's probe just failed). 'blocked' →
|
|
4277
|
+
// hold the row; 'probe' → this job is the single half-open probe and is
|
|
4278
|
+
// stamped as such so no sibling probes the same broken persona at once.
|
|
4279
|
+
const launchKey = launchFailure.launchBlockKeyFor(job);
|
|
4280
|
+
const claudeVersionNow = await probeClaudeVersion();
|
|
4281
|
+
let launchEnv = null;
|
|
4282
|
+
let launchProbe = false;
|
|
4283
|
+
const launchGate = await mutate((s) => {
|
|
4284
|
+
s.launchBlocks = s.launchBlocks || {};
|
|
4285
|
+
s.launchMitigations = s.launchMitigations || {};
|
|
4286
|
+
const mitigation = s.launchMitigations[launchKey];
|
|
4287
|
+
if (mitigation && claudeVersionNow && mitigation.claudeVersion && mitigation.claudeVersion !== claudeVersionNow) {
|
|
4288
|
+
console.log(`[scheduler] launch gate: CLI version changed (${mitigation.claudeVersion} → ${claudeVersionNow}) — dropping ${launchKey} mitigation to retry a clean launch`);
|
|
4289
|
+
delete s.launchMitigations[launchKey];
|
|
4290
|
+
}
|
|
4291
|
+
const block = s.launchBlocks[launchKey];
|
|
4292
|
+
const gate = launchFailure.evaluateLaunchGate(block, { now: Date.now(), claudeVersion: claudeVersionNow });
|
|
4293
|
+
if (gate.state === 'open' && block) {
|
|
4294
|
+
console.log(`[scheduler] launch gate: clearing ${launchKey} block (${gate.reason})`);
|
|
4295
|
+
delete s.launchBlocks[launchKey];
|
|
4296
|
+
}
|
|
4297
|
+
if (gate.state === 'blocked') {
|
|
4298
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4299
|
+
if (idx >= 0) s.jobs[idx].heldReason = gate.reason;
|
|
4300
|
+
return gate;
|
|
4301
|
+
}
|
|
4302
|
+
if (gate.state === 'probe') {
|
|
4303
|
+
block.probing = { slug: job.slug, at: new Date().toISOString() };
|
|
4304
|
+
launchProbe = true;
|
|
4305
|
+
launchEnv = block.mitigationEnv || null;
|
|
4306
|
+
} else if (s.launchMitigations[launchKey]?.env) {
|
|
4307
|
+
launchEnv = { ...s.launchMitigations[launchKey].env };
|
|
4308
|
+
}
|
|
4309
|
+
return gate;
|
|
4310
|
+
});
|
|
4311
|
+
if (launchGate.state === 'blocked') {
|
|
4312
|
+
console.log(`[scheduler] ${job.slug}: deferring — ${launchGate.reason}`);
|
|
4313
|
+
await broadcast({ flush: true });
|
|
4314
|
+
return;
|
|
4315
|
+
}
|
|
4316
|
+
if (launchProbe) {
|
|
4317
|
+
console.log(`[scheduler] ${job.slug}: dispatching as launch probe for '${launchKey}'${launchEnv ? ` with mitigation ${JSON.stringify(launchEnv)}` : ''}`);
|
|
4318
|
+
}
|
|
4319
|
+
|
|
3405
4320
|
await mutate((s) => {
|
|
3406
4321
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3407
4322
|
if (idx >= 0) {
|
|
3408
|
-
transitionJob(s.jobs[idx], 'running', {
|
|
4323
|
+
transitionJob(s.jobs[idx], 'running', {
|
|
4324
|
+
reason: resumeTarget ? 'dispatched for resume-recovery' : 'dispatched for execution',
|
|
4325
|
+
source: 'spawnJob:dispatch',
|
|
4326
|
+
});
|
|
4327
|
+
delete s.jobs[idx].heldReason;
|
|
3409
4328
|
s.jobs[idx].runId = runId;
|
|
3410
4329
|
s.jobs[idx].startedAt = new Date().toISOString();
|
|
4330
|
+
if (job.quietMachine === true) {
|
|
4331
|
+
s.jobs[idx].quietMachine = true;
|
|
4332
|
+
s.jobs[idx].quietLeaseDegraded = job.quietLeaseDegraded === true;
|
|
4333
|
+
}
|
|
4334
|
+
// Stamp the bounded one-attempt marker BEFORE the resume spawn, in
|
|
4335
|
+
// the SAME mutate as the 'running' transition, so an app crash
|
|
4336
|
+
// between here and the child actually spawning still leaves this
|
|
4337
|
+
// job un-retriable (selectResumeRecoveryTarget returns null once
|
|
4338
|
+
// this is true) rather than silently re-firing forever.
|
|
4339
|
+
if (resumeTarget) {
|
|
4340
|
+
s.jobs[idx].resumeRecoveryAttempted = true;
|
|
4341
|
+
}
|
|
3411
4342
|
}
|
|
3412
4343
|
});
|
|
3413
4344
|
await broadcast({ flush: true });
|
|
@@ -3417,19 +4348,74 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3417
4348
|
const guardCwd = job.cwd || defaultCwd;
|
|
3418
4349
|
const guardBaseline = await uncommittedChanges(guardCwd);
|
|
3419
4350
|
const guardHeadBefore = await gitHead(guardCwd);
|
|
4351
|
+
// Shared-tree stash guard baseline (incident 2026-09-01): captured
|
|
4352
|
+
// unconditionally, before worktree isolation is even attempted, so an
|
|
4353
|
+
// in-place run always has a true pre-run snapshot to diff against. See
|
|
4354
|
+
// checkSharedTreeGuard below, gated to in-place runs only.
|
|
4355
|
+
const stashBaseline = await stashList(guardCwd);
|
|
4356
|
+
|
|
4357
|
+
// Persist the pre-run baseline onto the row itself (not just the local
|
|
4358
|
+
// variable) so a finalizer that never reaches the rest of THIS function
|
|
4359
|
+
// — namely reapDeadRunningJobs, when the process vanishes mid-run — can
|
|
4360
|
+
// still compute a truthful newly-dirty delta instead of having no
|
|
4361
|
+
// baseline at all. `runtime` (unlike this) is deleted on finalize; this
|
|
4362
|
+
// survives until the finalize mutate below explicitly clears it.
|
|
4363
|
+
//
|
|
4364
|
+
// preRunDirtyPaths is the SAME snapshot, capped and reworked into the
|
|
4365
|
+
// executor-facing manifest (buildForeignWipSection) telling the job which
|
|
4366
|
+
// paths it does not own — unlike guardBaseline/guardHeadBefore, it is
|
|
4367
|
+
// deliberately left on the row through to history.jsonl (not deleted at
|
|
4368
|
+
// finalize) so a post-hoc reader can tell whether a completed job ran
|
|
4369
|
+
// against foreign WIP.
|
|
4370
|
+
const preRunDirtyPaths = capDirtyPaths(guardBaseline);
|
|
4371
|
+
await mutate((s) => {
|
|
4372
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4373
|
+
if (idx >= 0) {
|
|
4374
|
+
s.jobs[idx].guardBaseline = guardBaseline || [];
|
|
4375
|
+
s.jobs[idx].guardHeadBefore = guardHeadBefore || null;
|
|
4376
|
+
if (preRunDirtyPaths.length) s.jobs[idx].preRunDirtyPaths = preRunDirtyPaths;
|
|
4377
|
+
else delete s.jobs[idx].preRunDirtyPaths;
|
|
4378
|
+
}
|
|
4379
|
+
});
|
|
3420
4380
|
|
|
3421
4381
|
// Worktree isolation (PRD 994): give this job its own linked `git worktree`
|
|
3422
4382
|
// checkout so its edits/tests/commit never collide with a sibling job or
|
|
3423
4383
|
// an interactive session in the SAME repo. `worktree.ok` is false (with a
|
|
3424
|
-
// logged reason) for a non-git cwd, a dirty base tree,
|
|
3425
|
-
//
|
|
3426
|
-
// never a hard failure.
|
|
3427
|
-
//
|
|
3428
|
-
|
|
4384
|
+
// logged reason) for a non-git cwd, a dirty base tree, or
|
|
4385
|
+
// SM_JOB_WORKTREE_DISABLE=1 — every case falls back to running in place,
|
|
4386
|
+
// never a hard failure. (The cap-reached reason was already handled above
|
|
4387
|
+
// as a pre-dispatch DEFERRAL — a job never reaches this point with that
|
|
4388
|
+
// reason.) See jobWorktree.cjs's header comment for why job.cwd
|
|
4389
|
+
// (guardCwd) itself is NEVER repointed at the worktree dir. Reuses
|
|
4390
|
+
// `preflightWorktree` computed above the 'running' transition — it
|
|
4391
|
+
// already IS this job's worktree attempt (or already-created checkout),
|
|
4392
|
+
// so calling createJobWorktree a second time here would double-create
|
|
4393
|
+
// (or double-count the cap) for the exact same job.
|
|
4394
|
+
const worktree = preflightWorktree;
|
|
3429
4395
|
if (worktree.ok) {
|
|
3430
4396
|
console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
|
|
3431
4397
|
} else {
|
|
3432
4398
|
console.log(`[scheduler] ${job.slug}: running in main tree (worktree not used: ${worktree.reason})`);
|
|
4399
|
+
// Surface any degraded-isolation fallback on the job row itself so it's
|
|
4400
|
+
// queryable from the queue instead of console-only — except the
|
|
4401
|
+
// deliberate env-disable flag, which is an intentional opt-out, not a
|
|
4402
|
+
// degradation worth flagging.
|
|
4403
|
+
if (!jobWorktree.isWorktreeDisabled()) {
|
|
4404
|
+
await mutate((s) => {
|
|
4405
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4406
|
+
if (idx >= 0) s.jobs[idx].worktreeFallbackReason = worktree.reason;
|
|
4407
|
+
});
|
|
4408
|
+
}
|
|
4409
|
+
}
|
|
4410
|
+
// Base-tree WIP carried into the worktree (createWorktree, PRD 1094) —
|
|
4411
|
+
// recorded on the job row so integration can exclude these paths from
|
|
4412
|
+
// the branch diff below, and so it's queryable from the queue.
|
|
4413
|
+
const carriedPaths = (worktree.ok && Array.isArray(worktree.carriedPaths)) ? worktree.carriedPaths : [];
|
|
4414
|
+
if (carriedPaths.length) {
|
|
4415
|
+
await mutate((s) => {
|
|
4416
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4417
|
+
if (idx >= 0) s.jobs[idx].carriedPaths = carriedPaths;
|
|
4418
|
+
});
|
|
3433
4419
|
}
|
|
3434
4420
|
|
|
3435
4421
|
// Integrate the job's branch back into guardCwd's own HEAD, THEN tear the
|
|
@@ -3448,6 +4434,17 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3448
4434
|
let res;
|
|
3449
4435
|
let worktreeLeftoverDirty = [];
|
|
3450
4436
|
let worktreeIntegrationFailure = null;
|
|
4437
|
+
// A job's uncommitted-work patch, whichever isolation mode produced it —
|
|
4438
|
+
// set by EITHER branch below, never both (worktree.ok picks exactly one
|
|
4439
|
+
// shape for the whole run). Named generically (not "worktree...") because
|
|
4440
|
+
// an in-place run salvages one too (PRD 1098).
|
|
4441
|
+
let salvagePatch = null;
|
|
4442
|
+
// Which foreign-WIP shape applies to THIS run: an isolated worktree only
|
|
4443
|
+
// ever needs to disclose carriedPaths (its checkout starts clean apart
|
|
4444
|
+
// from those carried paths); an in-place/shared-tree run discloses the
|
|
4445
|
+
// raw pre-run dirty snapshot instead. Never both — see
|
|
4446
|
+
// buildForeignWipSection.
|
|
4447
|
+
const foreignWip = worktree.ok ? { carriedPaths } : { preRunDirtyPaths };
|
|
3451
4448
|
try {
|
|
3452
4449
|
res = await executeJob(job, runDir, defaultCwd, async (pid, sessionId, cwd) => {
|
|
3453
4450
|
await mutate((s) => {
|
|
@@ -3458,11 +4455,27 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3458
4455
|
}
|
|
3459
4456
|
});
|
|
3460
4457
|
await broadcast({ flush: true });
|
|
3461
|
-
}, worktree.ok ? worktree.dir : undefined);
|
|
4458
|
+
}, worktree.ok ? worktree.dir : undefined, resumeTarget, foreignWip, launchEnv);
|
|
3462
4459
|
} finally {
|
|
3463
4460
|
if (worktree.ok) {
|
|
3464
4461
|
worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
|
|
3465
|
-
|
|
4462
|
+
// Salvage the worktree's full diff (tracked + untracked) to the run
|
|
4463
|
+
// dir BEFORE the checkout is removed below — otherwise a job killed
|
|
4464
|
+
// before its finish-protocol commit loses that work outright, with
|
|
4465
|
+
// no branch, no stash, no patch anywhere. Best-effort: never blocks
|
|
4466
|
+
// integration/cleanup and never changes the job's verdict.
|
|
4467
|
+
if (worktreeLeftoverDirty.length) {
|
|
4468
|
+
const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
|
|
4469
|
+
const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
|
|
4470
|
+
if (salvage && salvage.ok) {
|
|
4471
|
+
salvagePatch = salvagePath;
|
|
4472
|
+
console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
|
|
4473
|
+
}
|
|
4474
|
+
}
|
|
4475
|
+
const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
|
|
4476
|
+
if (integration.ok && integration.reason === 'carried-wip-only') {
|
|
4477
|
+
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
|
|
4478
|
+
}
|
|
3466
4479
|
if (!integration.ok) {
|
|
3467
4480
|
worktreeIntegrationFailure = integration.reason;
|
|
3468
4481
|
console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
|
|
@@ -3475,9 +4488,86 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3475
4488
|
branch: worktree.branch,
|
|
3476
4489
|
keepBranch: !integration.ok,
|
|
3477
4490
|
});
|
|
4491
|
+
} else {
|
|
4492
|
+
// In-place run (non-git cwd, cap reached, env-disabled, or a carry-over
|
|
4493
|
+
// failure) — there is no throwaway checkout to diff, so salvage only
|
|
4494
|
+
// the DELTA this job itself dirtied: paths in guardBaseline are a
|
|
4495
|
+
// human's or a sibling job's pre-existing WIP and must never appear in
|
|
4496
|
+
// this job's patch. Runs for every exit code (finally always fires
|
|
4497
|
+
// once `res` resolves, success or not) including signal deaths and the
|
|
4498
|
+
// rate-limited/halt path — a killed in-place run is exactly the case
|
|
4499
|
+
// this exists to cover. Never mutates guardCwd's index or stashes:
|
|
4500
|
+
// salvageDirtyDelta is read-only (git status + git diff only).
|
|
4501
|
+
try {
|
|
4502
|
+
const after = await uncommittedChanges(guardCwd);
|
|
4503
|
+
if (after) {
|
|
4504
|
+
const baseSet = new Set(guardBaseline || []);
|
|
4505
|
+
const deltaPaths = after.filter((p) => !baseSet.has(p));
|
|
4506
|
+
if (deltaPaths.length) {
|
|
4507
|
+
const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
|
|
4508
|
+
const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: guardCwd, paths: deltaPaths, outFile: salvagePath });
|
|
4509
|
+
if (salvage && salvage.ok) {
|
|
4510
|
+
salvagePatch = salvagePath;
|
|
4511
|
+
console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted in-place diff (${deltaPaths.length} path(s)) to ${salvagePath}`);
|
|
4512
|
+
}
|
|
4513
|
+
}
|
|
4514
|
+
}
|
|
4515
|
+
} catch (e) {
|
|
4516
|
+
console.error(`[scheduler] ${job.slug}: in-place salvage failed`, e);
|
|
4517
|
+
}
|
|
3478
4518
|
}
|
|
3479
4519
|
}
|
|
3480
4520
|
|
|
4521
|
+
// Newly-dirty leftover computation — hoisted OUT of the exit===0 branch
|
|
4522
|
+
// (below) so it runs for every terminal outcome: exit 0, any non-zero
|
|
4523
|
+
// exit including 137/143, and the rate-limited/halt path alike. This is
|
|
4524
|
+
// the exact same shape the exit=0 commit-guard and the transient-failure
|
|
4525
|
+
// classifier each used to compute independently (guardCwd's own
|
|
4526
|
+
// baseline-delta UNION worktreeLeftoverDirty, which is already
|
|
4527
|
+
// inherently-new since it came from a fresh worktree checkout with no
|
|
4528
|
+
// baseline to diff against) — computed once here and reused by both
|
|
4529
|
+
// below, plus by the terminal-finalize mutate for leftoverPaths/
|
|
4530
|
+
// leftoverCount. null only when git-status itself is unavailable
|
|
4531
|
+
// (non-git cwd / git errored) — NEVER treated as "left nothing", exactly
|
|
4532
|
+
// like every other best-effort git-state check in this function.
|
|
4533
|
+
const afterGuardCwd = await uncommittedChanges(guardCwd);
|
|
4534
|
+
const newlyDirtyAll = afterGuardCwd === null
|
|
4535
|
+
? null
|
|
4536
|
+
: [...new Set([
|
|
4537
|
+
...afterGuardCwd.filter((p) => !new Set(guardBaseline || []).has(p)),
|
|
4538
|
+
...worktreeLeftoverDirty,
|
|
4539
|
+
])];
|
|
4540
|
+
|
|
4541
|
+
if (res.launchFailure) {
|
|
4542
|
+
await handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion: claudeVersionNow });
|
|
4543
|
+
return;
|
|
4544
|
+
}
|
|
4545
|
+
if (launchFailure.resultShowsRealTurn(res.resultStats)) {
|
|
4546
|
+
// The launch worked (whatever happens to the run next) — close the
|
|
4547
|
+
// breaker for this persona. If the probe only got through thanks to a
|
|
4548
|
+
// mitigation env, keep applying that env to every later launch of the
|
|
4549
|
+
// persona until the CLI version changes; otherwise the very next job
|
|
4550
|
+
// would fail the same way and re-arm the block (a flap per job).
|
|
4551
|
+
await mutate((s) => {
|
|
4552
|
+
const block = s.launchBlocks?.[launchKey];
|
|
4553
|
+
if (!block) return;
|
|
4554
|
+
delete s.launchBlocks[launchKey];
|
|
4555
|
+
if (launchEnv && Object.keys(launchEnv).length) {
|
|
4556
|
+
s.launchMitigations = s.launchMitigations || {};
|
|
4557
|
+
s.launchMitigations[launchKey] = {
|
|
4558
|
+
kind: block.kind,
|
|
4559
|
+
env: { ...launchEnv },
|
|
4560
|
+
since: new Date().toISOString(),
|
|
4561
|
+
claudeVersion: claudeVersionNow ?? block.claudeVersion ?? null,
|
|
4562
|
+
hint: launchFailure.launchFailureHint(block.kind, { claudeVersion: claudeVersionNow ?? block.claudeVersion }),
|
|
4563
|
+
};
|
|
4564
|
+
console.log(`[scheduler] launch gate: ${launchKey} recovered via mitigation ${JSON.stringify(launchEnv)} — kept in force until the CLI version changes`);
|
|
4565
|
+
} else {
|
|
4566
|
+
console.log(`[scheduler] launch gate: ${launchKey} recovered — block cleared after ${block.attempts} failed probe(s)`);
|
|
4567
|
+
}
|
|
4568
|
+
});
|
|
4569
|
+
}
|
|
4570
|
+
|
|
3481
4571
|
if (res.rateLimited) {
|
|
3482
4572
|
const resetIso = await refreshNextReset().catch(() => cachedNextReset);
|
|
3483
4573
|
await setPaused('rate_limit', resetIso);
|
|
@@ -3519,6 +4609,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3519
4609
|
// pass it back into verifyRun as priorLandedCommit (see the
|
|
3520
4610
|
// pass_no_commit_prior_run_verified exemption in runVerify.cjs).
|
|
3521
4611
|
let jobLandedCommitThisRun = null;
|
|
4612
|
+
let sharedTreeGuard = null;
|
|
3522
4613
|
if (res.exitCode === 0 && !res.rateLimited) {
|
|
3523
4614
|
// Detect whether the job self-committed by comparing HEAD before/after.
|
|
3524
4615
|
// Used by the sentinel override: SCHEDULER_VERDICT: PASS + a landed
|
|
@@ -3591,20 +4682,12 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3591
4682
|
const guardWillRefire = verifyResult && verifyResult.downgradeTo === 'pending';
|
|
3592
4683
|
const guardIsLegitimateNoOp = verifyResult && COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict);
|
|
3593
4684
|
if (res.exitCode === 0 && !res.rateLimited && !guardWillRefire && !guardIsLegitimateNoOp) {
|
|
3594
|
-
|
|
3595
|
-
//
|
|
3596
|
-
//
|
|
3597
|
-
//
|
|
3598
|
-
if (
|
|
3599
|
-
const
|
|
3600
|
-
// worktreeLeftoverDirty was captured from a FRESH checkout (no baseline
|
|
3601
|
-
// to diff against — every path in it is inherently new) right before
|
|
3602
|
-
// the worktree was torn down, so it must be counted here or a job's
|
|
3603
|
-
// uncommitted leftovers silently vanish with the worktree.
|
|
3604
|
-
const newlyDirty = [...new Set([
|
|
3605
|
-
...after.filter((p) => !baseSet.has(p)),
|
|
3606
|
-
...worktreeLeftoverDirty,
|
|
3607
|
-
])];
|
|
4685
|
+
// afterGuardCwd === null means non-git cwd (or git errored) —
|
|
4686
|
+
// best-effort skip, same as always; only a git-status result (even an
|
|
4687
|
+
// empty one) counts as evidence for the zero-edit path. newlyDirtyAll
|
|
4688
|
+
// was computed once, above, right after the try/finally.
|
|
4689
|
+
if (afterGuardCwd !== null) {
|
|
4690
|
+
const newlyDirty = newlyDirtyAll;
|
|
3608
4691
|
const guardState = await readQueue().catch(() => ({ jobs: [] }));
|
|
3609
4692
|
const siblingRunning = (guardState.jobs || []).some(
|
|
3610
4693
|
(j) => j.slug !== job.slug && j.status === 'running' && (j.cwd || defaultCwd) === guardCwd,
|
|
@@ -3614,10 +4697,12 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3614
4697
|
const guardVerdict = commitGuardVerdict({
|
|
3615
4698
|
newlyDirty,
|
|
3616
4699
|
siblingRunning,
|
|
4700
|
+
ranInWorktree: worktree.ok,
|
|
3617
4701
|
jobSelfCommitted,
|
|
3618
4702
|
legitimateNoOp: guardIsLegitimateNoOp,
|
|
3619
4703
|
isFixPlanJob: isFixPlanSlug(job.slug),
|
|
3620
4704
|
verifyResult,
|
|
4705
|
+
salvagePatch,
|
|
3621
4706
|
});
|
|
3622
4707
|
if (guardVerdict) {
|
|
3623
4708
|
verifyResult = guardVerdict;
|
|
@@ -3641,6 +4726,36 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3641
4726
|
};
|
|
3642
4727
|
}
|
|
3643
4728
|
|
|
4729
|
+
// Shared-tree stash guard (incident 2026-09-01): only meaningful for an
|
|
4730
|
+
// IN-PLACE run — worktree.ok isolates the job's git state into its own
|
|
4731
|
+
// checkout, so nothing there can leak into guardCwd. Best-effort and run
|
|
4732
|
+
// regardless of exit code: a job can discard shared state on its way to
|
|
4733
|
+
// a non-zero exit just as easily as on a clean one.
|
|
4734
|
+
if (!worktree.ok) {
|
|
4735
|
+
sharedTreeGuard = await module.exports.checkSharedTreeGuard({
|
|
4736
|
+
cwd: guardCwd,
|
|
4737
|
+
stashBaseline,
|
|
4738
|
+
dirtyBaseline: guardBaseline,
|
|
4739
|
+
headBefore: guardHeadBefore,
|
|
4740
|
+
slug: job.slug,
|
|
4741
|
+
});
|
|
4742
|
+
// A restored stash alone isn't silence — it's logged loudly above and
|
|
4743
|
+
// surfaced on the job row below — but a path that's still missing
|
|
4744
|
+
// (restore failed, or two-plus stashes we refused to guess between, or
|
|
4745
|
+
// a revert with no stash to restore at all) must not finish green.
|
|
4746
|
+
if (sharedTreeGuard && (sharedTreeGuard.restoreFailed || sharedTreeGuard.ambiguousStashes || sharedTreeGuard.reverted)) {
|
|
4747
|
+
verifyResult = {
|
|
4748
|
+
verdict: 'shared_tree_reverted',
|
|
4749
|
+
reason: sharedTreeGuard.reverted
|
|
4750
|
+
? `job discarded pre-existing state in the shared tree: ${sharedTreeGuard.reverted.length} path(s) reverted with no commit to explain it (${sharedTreeGuard.reverted.slice(0, 3).join(', ')})`
|
|
4751
|
+
: sharedTreeGuard.restoreFailed
|
|
4752
|
+
? `job stashed the shared tree and the stash could not be auto-restored: ${sharedTreeGuard.restoreFailed}`
|
|
4753
|
+
: `job created ${sharedTreeGuard.ambiguousStashes.length} stashes in the shared tree — ambiguous, not auto-restored (${sharedTreeGuard.ambiguousStashes.join(', ')})`,
|
|
4754
|
+
downgradeTo: 'needs_review',
|
|
4755
|
+
};
|
|
4756
|
+
}
|
|
4757
|
+
}
|
|
4758
|
+
|
|
3644
4759
|
// SIGTERM commit check: reuse the same commit-window scan the exit=0
|
|
3645
4760
|
// guard uses above (one commit-detection path, not two) to see whether a
|
|
3646
4761
|
// 143 (SIGTERM) run still landed a deliverable before it died. Scoped
|
|
@@ -3663,6 +4778,8 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3663
4778
|
let needsInvestigationNow = false;
|
|
3664
4779
|
let investigationJobSnapshot = null;
|
|
3665
4780
|
let needsReviewRcaSnapshot = null;
|
|
4781
|
+
let resumeRecoveryJob = null;
|
|
4782
|
+
let resumeRecoveryTarget = null;
|
|
3666
4783
|
let terminalNotifySnapshot = null;
|
|
3667
4784
|
const newlyCompletedPrds = [];
|
|
3668
4785
|
await mutate((s) => {
|
|
@@ -3709,6 +4826,12 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3709
4826
|
transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
|
|
3710
4827
|
s.jobs[i2].finishedAt = new Date().toISOString();
|
|
3711
4828
|
s.jobs[i2].exitCode = res.exitCode;
|
|
4829
|
+
s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
|
|
4830
|
+
if (salvagePatch) {
|
|
4831
|
+
s.jobs[i2].salvagePatch = salvagePatch;
|
|
4832
|
+
} else {
|
|
4833
|
+
delete s.jobs[i2].salvagePatch;
|
|
4834
|
+
}
|
|
3712
4835
|
s.jobs[i2].error = effectiveStatus === 'needs_review'
|
|
3713
4836
|
? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
|
|
3714
4837
|
// A failed job (non-zero exit) never consults verifyResult above,
|
|
@@ -3731,6 +4854,26 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3731
4854
|
} else {
|
|
3732
4855
|
delete s.jobs[i2].verifierVerdict;
|
|
3733
4856
|
}
|
|
4857
|
+
// Closed-set outcome taxonomy (issue #11 list A2) so a queue row
|
|
4858
|
+
// says WHY it ended without anyone opening the transcript.
|
|
4859
|
+
s.jobs[i2].terminalReason = launchFailure.deriveTerminalReason({
|
|
4860
|
+
effectiveStatus, exitCode: res.exitCode, verifyResult, sigtermOverride, worktreeIntegrationFailure,
|
|
4861
|
+
});
|
|
4862
|
+
delete s.jobs[i2].launchFailure;
|
|
4863
|
+
delete s.jobs[i2].heldReason;
|
|
4864
|
+
// Persist the commit-guard's exact dirty-path list (verdict
|
|
4865
|
+
// 'uncommitted_changes' only) so a later resume-recovery attempt
|
|
4866
|
+
// (selectResumeRecoveryTarget) can name these paths without
|
|
4867
|
+
// re-running `git status` against a tree that may have moved on.
|
|
4868
|
+
if (verifyResult?.verdict === 'uncommitted_changes' && Array.isArray(verifyResult.dirtyPaths)) {
|
|
4869
|
+
// Capped the same way preRunDirtyPaths/leftoverPaths are — an
|
|
4870
|
+
// uncapped list here would let a pathologically dirty tree bloat
|
|
4871
|
+
// queue.json/history.jsonl and the resume-recovery prompt built
|
|
4872
|
+
// from it (buildResumeRecoveryPreamble/selectResumeRecoveryTarget).
|
|
4873
|
+
s.jobs[i2].uncommittedPaths = capDirtyPaths(verifyResult.dirtyPaths);
|
|
4874
|
+
} else {
|
|
4875
|
+
delete s.jobs[i2].uncommittedPaths;
|
|
4876
|
+
}
|
|
3734
4877
|
// Non-blocking notes (e.g. a recovered missing-dependency probe, or a
|
|
3735
4878
|
// pattern hit demoted because a materially-checkable verdict outranked
|
|
3736
4879
|
// it) — surfaced even on completed jobs so the signal isn't lost.
|
|
@@ -3741,7 +4884,29 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3741
4884
|
} else {
|
|
3742
4885
|
delete s.jobs[i2].verifierAnnotations;
|
|
3743
4886
|
}
|
|
4887
|
+
// Shared-tree guard outcome (restored stash / unresolved revert /
|
|
4888
|
+
// ambiguous stashes) — visible on the row even when a restored
|
|
4889
|
+
// stash left the run otherwise green, so it's never silent.
|
|
4890
|
+
if (sharedTreeGuard) {
|
|
4891
|
+
s.jobs[i2].sharedTreeGuard = sharedTreeGuard;
|
|
4892
|
+
} else {
|
|
4893
|
+
delete s.jobs[i2].sharedTreeGuard;
|
|
4894
|
+
}
|
|
3744
4895
|
delete s.jobs[i2].runtime;
|
|
4896
|
+
// Pre-run baseline no longer needed once this run has finalized —
|
|
4897
|
+
// its whole purpose (letting THIS finalize compute a truthful
|
|
4898
|
+
// delta) is done; a fresh one is captured at the next dispatch.
|
|
4899
|
+
delete s.jobs[i2].guardBaseline;
|
|
4900
|
+
delete s.jobs[i2].guardHeadBefore;
|
|
4901
|
+
// Leftover-attribution fields (PRD: capture+surface uncommitted
|
|
4902
|
+
// work on every terminal path, not just exit=0) — set for EVERY
|
|
4903
|
+
// terminal outcome above (completed/failed/needs_review alike),
|
|
4904
|
+
// not just the exit=0 commit-guard branch, so a bare `failed` row
|
|
4905
|
+
// is visually distinguishable from one that quietly left work
|
|
4906
|
+
// behind. newlyDirtyAll is null when git-status was unavailable
|
|
4907
|
+
// (non-git cwd) — applyLeftoverFields treats null like "nothing to
|
|
4908
|
+
// attribute" via its Array.isArray guard, same as an empty array.
|
|
4909
|
+
applyLeftoverFields(s.jobs[i2], newlyDirtyAll);
|
|
3745
4910
|
|
|
3746
4911
|
if (isNotifiableTerminalStatus(effectiveStatus)) {
|
|
3747
4912
|
terminalNotifySnapshot = { ...s.jobs[i2] };
|
|
@@ -3759,6 +4924,19 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3759
4924
|
// takes the treatAsPending branch above and never reaches here).
|
|
3760
4925
|
needsReviewRcaSnapshot = { ...s.jobs[i2] };
|
|
3761
4926
|
|
|
4927
|
+
// Resume-first recovery (PRD 1111): evaluated BEFORE the auto-fix
|
|
4928
|
+
// eligibility check below — a job whose verdict is
|
|
4929
|
+
// 'uncommitted_changes' with a live sessionId gets one bounded
|
|
4930
|
+
// `--resume` dispatch instead of a cold-read fix-plan
|
|
4931
|
+
// investigation. Snapshot only (no I/O inside mutate()); the
|
|
4932
|
+
// actual dispatch happens outside mutate(), below. Never sets
|
|
4933
|
+
// needsInvestigationNow — the two are mutually exclusive for the
|
|
4934
|
+
// same tick, mirroring the `else if` used outside mutate().
|
|
4935
|
+
const target = selectResumeRecoveryTarget(s.jobs[i2]);
|
|
4936
|
+
if (target) {
|
|
4937
|
+
resumeRecoveryJob = { ...s.jobs[i2] };
|
|
4938
|
+
resumeRecoveryTarget = target;
|
|
4939
|
+
} else {
|
|
3762
4940
|
// Same-tick auto-fix (feedback 2026-07-12): rather than waiting up to
|
|
3763
4941
|
// 10 min for reverifyNeedsReview()'s periodic pass, check right here
|
|
3764
4942
|
// whether this job qualifies for auto-fix (same eligibility rule
|
|
@@ -3782,6 +4960,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3782
4960
|
needsInvestigationNow = true;
|
|
3783
4961
|
investigationJobSnapshot = { ...s.jobs[i2] };
|
|
3784
4962
|
}
|
|
4963
|
+
}
|
|
3785
4964
|
}
|
|
3786
4965
|
// Auto-promote: when a fix-* PRD completes successfully, the original
|
|
3787
4966
|
// failed PRD's work is logically done. Flip its status to 'completed'
|
|
@@ -3797,6 +4976,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3797
4976
|
orig.exitCode = 0;
|
|
3798
4977
|
orig.error = null;
|
|
3799
4978
|
orig.completedBy = job.slug;
|
|
4979
|
+
delete orig.looksDone;
|
|
3800
4980
|
if (priorStatus === 'needs_review') {
|
|
3801
4981
|
delete orig.verifierVerdict;
|
|
3802
4982
|
}
|
|
@@ -3811,6 +4991,24 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3811
4991
|
}
|
|
3812
4992
|
await broadcast({ flush: true });
|
|
3813
4993
|
|
|
4994
|
+
// Per-run outcome sidecar (issue #11 list B5): turns/tokens/verdict in one
|
|
4995
|
+
// small JSON next to the log so fleet health never needs a transcript parse.
|
|
4996
|
+
launchFailure.writeOutcomeSidecar(runDir, job.slug, {
|
|
4997
|
+
runId,
|
|
4998
|
+
exitCode: res.exitCode,
|
|
4999
|
+
durationMs: res.durationMs ?? null,
|
|
5000
|
+
numTurns: res.resultStats?.numTurns ?? null,
|
|
5001
|
+
outputTokens: res.resultStats?.outputTokens ?? null,
|
|
5002
|
+
totalCostUsd: res.resultStats?.totalCostUsd ?? null,
|
|
5003
|
+
verdict: verifyResult?.verdict ?? (res.exitCode === 0 ? 'clean' : null),
|
|
5004
|
+
status: terminalNotifySnapshot?.status ?? failedJobSnapshot?.status ?? null,
|
|
5005
|
+
terminalReason: terminalNotifySnapshot?.terminalReason ?? failedJobSnapshot?.terminalReason ?? null,
|
|
5006
|
+
launchFailure: null,
|
|
5007
|
+
launchEnvApplied: launchEnv ? Object.keys(launchEnv) : [],
|
|
5008
|
+
filesChanged: Array.isArray(newlyDirtyAll) ? newlyDirtyAll.length : null,
|
|
5009
|
+
landedCommit: jobLandedCommitThisRun ?? null,
|
|
5010
|
+
});
|
|
5011
|
+
|
|
3814
5012
|
if (terminalNotifySnapshot) {
|
|
3815
5013
|
notifyOriginatingTab(terminalNotifySnapshot).catch((e) => {
|
|
3816
5014
|
console.error('[scheduler] notifyOriginatingTab error', job.slug, e);
|
|
@@ -3826,12 +5024,28 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3826
5024
|
verdict: needsReviewRcaSnapshot.verifierVerdict,
|
|
3827
5025
|
annotations: needsReviewRcaSnapshot.verifierAnnotations,
|
|
3828
5026
|
})
|
|
3829
|
-
.then((report) =>
|
|
5027
|
+
.then(async (report) => {
|
|
5028
|
+
// Persist the classification onto the parked job row so the scheduler
|
|
5029
|
+
// can route on it (e.g. selectAutoFixTargets excluding 'archive')
|
|
5030
|
+
// without re-parsing the RCA markdown on every pass.
|
|
5031
|
+
await mutate((s) => {
|
|
5032
|
+
const j = s.jobs.find((x) => x.slug === needsReviewRcaSnapshot.slug);
|
|
5033
|
+
applyRcaClassification(j, report);
|
|
5034
|
+
}).catch(() => {});
|
|
5035
|
+
return notifyNeedsReview(needsReviewRcaSnapshot, report);
|
|
5036
|
+
})
|
|
3830
5037
|
.catch((e) => {
|
|
3831
5038
|
console.error('[scheduler] writeRcaReport error', job.slug, e);
|
|
3832
5039
|
});
|
|
3833
5040
|
}
|
|
3834
5041
|
|
|
5042
|
+
if (resumeRecoveryJob && resumeRecoveryTarget) {
|
|
5043
|
+
console.log(`[scheduler] needs_review ${job.slug} → resume-recovery (session ${resumeRecoveryTarget.sessionId}, ${resumeRecoveryTarget.dirtyPaths.length} dirty path(s))`);
|
|
5044
|
+
spawnResumeRecovery(resumeRecoveryJob, resumeRecoveryTarget).catch((e) => {
|
|
5045
|
+
console.error('[scheduler] spawnResumeRecovery error', job.slug, e);
|
|
5046
|
+
});
|
|
5047
|
+
}
|
|
5048
|
+
|
|
3835
5049
|
if (actuallyFailed && failedJobSnapshot) {
|
|
3836
5050
|
// Transient-failure detector. A 143/137 exit is ALWAYS a signal kill — the
|
|
3837
5051
|
// agent never self-exits with those — so the only question is WHO killed it.
|
|
@@ -3849,25 +5063,11 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3849
5063
|
// the threshold and still fall through to investigation.
|
|
3850
5064
|
const ec = failedJobSnapshot.exitCode;
|
|
3851
5065
|
const retries = failedJobSnapshot.transientRetries ?? 0;
|
|
3852
|
-
// Only pay for the extra git status call when the failure is plausibly
|
|
3853
|
-
// transient — a real code failure never needs the dirty-tree check.
|
|
3854
5066
|
const maybeTransient = (ec === 143 || ec === 137) || res.networkError === true;
|
|
3855
|
-
|
|
3856
|
-
|
|
3857
|
-
|
|
3858
|
-
|
|
3859
|
-
const baseSet = new Set(guardBaseline || []);
|
|
3860
|
-
// See the commit-guard block above: worktreeLeftoverDirty was captured
|
|
3861
|
-
// (and the checkout already torn down) before this point, so it must
|
|
3862
|
-
// be folded in here too or a transiently-killed job's leftover WIP
|
|
3863
|
-
// silently disappears with its worktree.
|
|
3864
|
-
const newlyDirty = [...new Set([
|
|
3865
|
-
...(afterFailure || []).filter((p) => !baseSet.has(p)),
|
|
3866
|
-
...worktreeLeftoverDirty,
|
|
3867
|
-
])];
|
|
3868
|
-
newlyDirtyCount = newlyDirty.length;
|
|
3869
|
-
dirtySample = newlyDirty.slice(0, 3).join(', ');
|
|
3870
|
-
}
|
|
5067
|
+
// newlyDirtyAll was computed once, above, right after the try/finally —
|
|
5068
|
+
// reused here rather than re-querying git status a third time.
|
|
5069
|
+
const newlyDirtyCount = maybeTransient ? (newlyDirtyAll || []).length : 0;
|
|
5070
|
+
const dirtySample = maybeTransient ? (newlyDirtyAll || []).slice(0, 3).join(', ') : '';
|
|
3871
5071
|
const decision = classifyFailureOutcome({
|
|
3872
5072
|
exitCode: ec,
|
|
3873
5073
|
networkError: res.networkError,
|
|
@@ -3887,12 +5087,13 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3887
5087
|
});
|
|
3888
5088
|
await broadcast({ flush: true });
|
|
3889
5089
|
} else if (decision.action === 'fail-dirty') {
|
|
3890
|
-
|
|
5090
|
+
const salvageNote = salvagePatch ? ` — recoverable from salvage patch ${salvagePatch}` : '';
|
|
5091
|
+
console.log(`[scheduler] transient failure (${decision.transientKind}) for ${job.slug} left ${newlyDirtyCount} uncommitted file(s) (e.g. ${dirtySample})${salvageNote} — not auto-requeuing`);
|
|
3891
5092
|
await mutate((s) => {
|
|
3892
5093
|
const i = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3893
5094
|
if (i >= 0) {
|
|
3894
5095
|
transitionJob(s.jobs[i], 'failed', { reason: `transient failure (${decision.transientKind}) left uncommitted work — not auto-requeued`, source: 'spawnJob:fail-dirty' });
|
|
3895
|
-
s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample}) — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
|
|
5096
|
+
s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample})${salvageNote} — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
|
|
3896
5097
|
}
|
|
3897
5098
|
});
|
|
3898
5099
|
await broadcast({ flush: true });
|
|
@@ -3927,17 +5128,41 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3927
5128
|
runningSet.delete(job.slug);
|
|
3928
5129
|
// Slot release notifies subscribed pumps (chat lane) machine-wide.
|
|
3929
5130
|
sessionSlots.release(slotToken);
|
|
5131
|
+
// Release the exclusive quiet-machine lease on EVERY exit path this
|
|
5132
|
+
// finally covers (normal exit, timeout, SIGTERM, crash) — see the
|
|
5133
|
+
// acquire-site comment above. Bounded: a lease this function never
|
|
5134
|
+
// acquired is simply a no-op release.
|
|
5135
|
+
if (quietLeaseAcquired) quietMachineLease.release(job.slug);
|
|
3930
5136
|
// Each job completion is a signal to advance the queue.
|
|
3931
5137
|
tickQueue().catch(() => {});
|
|
3932
5138
|
}
|
|
3933
5139
|
}
|
|
3934
5140
|
|
|
5141
|
+
/**
|
|
5142
|
+
* Dispatch a resume-recovery attempt (PRD 1111) for a job already found
|
|
5143
|
+
* eligible by selectResumeRecoveryTarget. Thin wrapper around spawnJob —
|
|
5144
|
+
* reuses its entire slot-acquire/worktree/verify/commit-guard/finalize
|
|
5145
|
+
* machinery unchanged, so a resume run that itself parks or fails falls
|
|
5146
|
+
* through to the SAME spawnInvestigation fallback any other run would, with
|
|
5147
|
+
* zero special-casing. `job` and `resumeTarget` must be snapshots taken
|
|
5148
|
+
* BEFORE this call (this function does no eligibility re-check — spawnJob's
|
|
5149
|
+
* own dispatch mutate is what stamps resumeRecoveryAttempted, atomically
|
|
5150
|
+
* with the 'running' transition).
|
|
5151
|
+
*/
|
|
5152
|
+
async function spawnResumeRecovery(job, resumeTarget) {
|
|
5153
|
+
const { runId, dir: runDir } = pickRunDir();
|
|
5154
|
+
await spawnJob(job, runId, runDir, job.cwd || DEFAULT_PROJECT_CWD, resumeTarget);
|
|
5155
|
+
}
|
|
5156
|
+
|
|
3935
5157
|
// Serialized ticker: prevents two concurrent tickQueue() calls from racing
|
|
3936
5158
|
// on the same pending jobs. A simple promise tail suffices since pickNextBatch
|
|
3937
5159
|
// is synchronous and spawnJob is fire-and-forget.
|
|
3938
5160
|
let tickTail = Promise.resolve();
|
|
3939
5161
|
|
|
3940
|
-
|
|
5162
|
+
// `bypassLoadGate` is set only by the explicit human run-now / force-tick
|
|
5163
|
+
// paths (via runDueJobs): the human is asking, so the CPU-load gate yields
|
|
5164
|
+
// and logs that it did. Every automatic caller leaves it false.
|
|
5165
|
+
function tickQueue({ bypassLoadGate = false } = {}) {
|
|
3941
5166
|
const next = tickTail.then(async () => {
|
|
3942
5167
|
const state = await readQueue();
|
|
3943
5168
|
// Never reconcile against an unreadable queue: reconcile() would see zero
|
|
@@ -3962,7 +5187,13 @@ function tickQueue() {
|
|
|
3962
5187
|
// cap that sessionSlots.cjs was written to replace — which silently
|
|
3963
5188
|
// ceilinged the queue at 3 while the pool the user configured said 5.
|
|
3964
5189
|
const freeSlots = sessionSlots.available();
|
|
3965
|
-
const
|
|
5190
|
+
const heldSlugs = await computeLaunchHolds(state);
|
|
5191
|
+
const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
|
|
5192
|
+
leaseHeld: quietMachineLease.isHeld(),
|
|
5193
|
+
machineInUse: sessionSlots.inUse(),
|
|
5194
|
+
now: Date.now(),
|
|
5195
|
+
heldSlugs,
|
|
5196
|
+
});
|
|
3966
5197
|
if (batch.length === 0 && freeSlots === 0) {
|
|
3967
5198
|
const snap = sessionSlots.snapshot();
|
|
3968
5199
|
const pendingCount = state.jobs.filter((j) => j.status === 'pending').length;
|
|
@@ -4025,6 +5256,35 @@ function tickQueue() {
|
|
|
4025
5256
|
lastMemGate = null;
|
|
4026
5257
|
}
|
|
4027
5258
|
|
|
5259
|
+
// Load gate (PRD 1085) — the INNERMOST launch predicate, evaluated only
|
|
5260
|
+
// once every outer gate (sessionSlots pool → per-project cap inside
|
|
5261
|
+
// pickNextBatch → memory above) has already admitted `gatedBatch`. It
|
|
5262
|
+
// never touches running jobs and never becomes a second pool: it only
|
|
5263
|
+
// withholds this tick's launches while the 1-minute loadavg per core is
|
|
5264
|
+
// over LOAD_GATE_PER_CORE. An explicit human Run now bypasses it.
|
|
5265
|
+
const load = loadGate.evaluate({ bypass: bypassLoadGate });
|
|
5266
|
+
if (load.bypassed) {
|
|
5267
|
+
console.log(`[scheduler] load gate: BYPASSED by run-now (loadavg1=${load.loadavg1} cores=${load.cores} ratio=${load.ratio} > ${load.threshold})`);
|
|
5268
|
+
} else if (load.gated) {
|
|
5269
|
+
const line = `[scheduler] load gate: loadavg1=${load.loadavg1} cores=${load.cores} ratio=${load.ratio} > ${load.threshold} — holding ${gatedBatch.length} eligible job(s)`;
|
|
5270
|
+
if (load.escalate) {
|
|
5271
|
+
const top = topCpuConsumers(3);
|
|
5272
|
+
console.warn(`${line} for ${Math.round(load.gatedSinceMs / 60_000)}m; top CPU: ${top.length ? top.join(' | ') : 'n/a'}`);
|
|
5273
|
+
} else {
|
|
5274
|
+
console.log(line);
|
|
5275
|
+
}
|
|
5276
|
+
if (load.shouldAudit) {
|
|
5277
|
+
appendAuditEvent('launch_load_gated', {
|
|
5278
|
+
loadavg1: load.loadavg1, cores: load.cores, ratio: load.ratio, threshold: load.threshold,
|
|
5279
|
+
held: gatedBatch.map((j) => j.slug), gatedSinceMs: load.gatedSinceMs,
|
|
5280
|
+
});
|
|
5281
|
+
}
|
|
5282
|
+
return recordTick(
|
|
5283
|
+
{ fired: false, reason: 'load-deferred', deferredCount: gatedBatch.length, ratio: load.ratio, threshold: load.threshold },
|
|
5284
|
+
{ detail: `load gate: ${load.loadavg1} / ${load.cores} cores = ${load.ratio} > ${load.threshold}`, holds },
|
|
5285
|
+
);
|
|
5286
|
+
}
|
|
5287
|
+
|
|
4028
5288
|
await mutate((s) => { s.lastRunAt = new Date().toISOString(); });
|
|
4029
5289
|
await broadcast();
|
|
4030
5290
|
|
|
@@ -4072,7 +5332,7 @@ function forceTickOutcome(result) {
|
|
|
4072
5332
|
}
|
|
4073
5333
|
}
|
|
4074
5334
|
|
|
4075
|
-
async function runDueJobs() {
|
|
5335
|
+
async function runDueJobs({ bypassLoadGate = false } = {}) {
|
|
4076
5336
|
const state = await readQueue();
|
|
4077
5337
|
if (state.unreadable) {
|
|
4078
5338
|
console.error('[scheduler] runDueJobs skipped: queue.json unreadable');
|
|
@@ -4083,7 +5343,7 @@ async function runDueJobs() {
|
|
|
4083
5343
|
return { fired: false, reason: 'paused' };
|
|
4084
5344
|
}
|
|
4085
5345
|
cancelToken = { cancelled: false };
|
|
4086
|
-
const result = await tickQueue();
|
|
5346
|
+
const result = await tickQueue({ bypassLoadGate });
|
|
4087
5347
|
// Clear the one-shot scheduledFor without waiting for jobs to settle.
|
|
4088
5348
|
await mutate((s) => { s.scheduledFor = null; });
|
|
4089
5349
|
await broadcast();
|
|
@@ -4108,12 +5368,38 @@ async function maybeLaunchWhenAvailable(state) {
|
|
|
4108
5368
|
|
|
4109
5369
|
// ---------- dead-process reaper ----------
|
|
4110
5370
|
|
|
5371
|
+
// Queue-health sweep cadence: hangs off reapDeadRunningJobs's own cycle
|
|
5372
|
+
// counter (it already runs once per poll tick) rather than a second timer,
|
|
5373
|
+
// so its cadence can never drift from the poll cadence or double-fire
|
|
5374
|
+
// across a backoff reset.
|
|
5375
|
+
let queueHealthSweepCycle = 0;
|
|
5376
|
+
const QUEUE_HEALTH_SWEEP_EVERY_N_CYCLES = 20;
|
|
5377
|
+
|
|
5378
|
+
/**
|
|
5379
|
+
* runQueueHealthSweep(jobs) — read-only reporting pass over the queue
|
|
5380
|
+
* snapshot reapDeadRunningJobs already read this cycle. Never transitions a
|
|
5381
|
+
* job, never archives a PRD, never spawns anything; only logs and appends
|
|
5382
|
+
* an audit event for any project with drift worth a human glance.
|
|
5383
|
+
*/
|
|
5384
|
+
function runQueueHealthSweep(jobs) {
|
|
5385
|
+
try {
|
|
5386
|
+
for (const { cwd, neverRan, looksDone, stuck } of computeQueueHealth(jobs)) {
|
|
5387
|
+
console.log(`[scheduler] queue-health ${cwd}: ${neverRan} never_ran, ${looksDone} looks-done, ${stuck} stuck`);
|
|
5388
|
+
appendAuditEvent('scheduler_queue_health', { cwd, neverRan, looksDone, stuck });
|
|
5389
|
+
}
|
|
5390
|
+
} catch (e) {
|
|
5391
|
+
console.warn('[scheduler] queue-health sweep error', e?.message);
|
|
5392
|
+
}
|
|
5393
|
+
}
|
|
5394
|
+
|
|
4111
5395
|
/**
|
|
4112
|
-
* Scan running jobs, identify those whose claude process is provably dead
|
|
4113
|
-
*
|
|
4114
|
-
*
|
|
4115
|
-
*
|
|
4116
|
-
*
|
|
5396
|
+
* Scan running jobs, identify those whose claude process is provably dead OR
|
|
5397
|
+
* whose spawn never got far enough to record a runtime.pid in the first
|
|
5398
|
+
* place, and finalize them to completed/failed by reading the run log.
|
|
5399
|
+
* Called once per poll cycle. A job whose pid is alive (claudePidAlive) is
|
|
5400
|
+
* always skipped. A pidless job younger than PIDLESS_SPAWN_GRACE_MS is
|
|
5401
|
+
* skipped too (spawn may still be mid-flight) — see selectReapableJobs for
|
|
5402
|
+
* the full predicate. Exported so unit tests can invoke it directly.
|
|
4117
5403
|
*/
|
|
4118
5404
|
async function reapDeadRunningJobs() {
|
|
4119
5405
|
try {
|
|
@@ -4123,32 +5409,102 @@ async function reapDeadRunningJobs() {
|
|
|
4123
5409
|
// status:"running" with no slug left in runningSet to trigger reconciliation.
|
|
4124
5410
|
// queue.json is the source of truth for which jobs are actually running.
|
|
4125
5411
|
const state = await readQueue();
|
|
5412
|
+
const { reapable, warnings } = selectReapableJobs(state.jobs, Date.now(), {
|
|
5413
|
+
pidAlive: claudePidAlive,
|
|
5414
|
+
grace: PIDLESS_SPAWN_GRACE_MS,
|
|
5415
|
+
});
|
|
5416
|
+
for (const w of warnings) {
|
|
5417
|
+
console.warn(`[scheduler] reapDeadRunningJobs: ${w.reason} slug=${w.slug} — leaving row alone`);
|
|
5418
|
+
}
|
|
5419
|
+
|
|
4126
5420
|
const dead = [];
|
|
4127
|
-
for (const
|
|
4128
|
-
|
|
4129
|
-
const
|
|
4130
|
-
if (!pid) continue; // spawn may be mid-flight; give it a cycle
|
|
4131
|
-
if (claudePidAlive(pid)) continue;
|
|
4132
|
-
const logPath = j.runId
|
|
5421
|
+
for (const { slug, pid, pidless, reason } of reapable) {
|
|
5422
|
+
const j = state.jobs.find((x) => x.slug === slug);
|
|
5423
|
+
const logPath = j?.runId
|
|
4133
5424
|
? path.join(RUNS_DIR, j.runId, `${j.slug}.log`)
|
|
4134
5425
|
: null;
|
|
5426
|
+
// Absent/empty run dir → classifyRunOutcome finds no result event →
|
|
5427
|
+
// 'no_result' → non-success below → filed as failed, never completed.
|
|
4135
5428
|
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
4136
|
-
|
|
5429
|
+
// A pidless reap means the spawn never got far enough to record a
|
|
5430
|
+
// pid — the gate could not possibly have run, regardless of what
|
|
5431
|
+
// classifyRunOutcome makes of an absent/empty log.
|
|
5432
|
+
const gateOutcome = pidless ? 'never_ran' : mapOutcomeToGateOutcome(outcome);
|
|
5433
|
+
dead.push({ slug, pid, outcome, gateOutcome, pidless, reason });
|
|
4137
5434
|
}
|
|
5435
|
+
|
|
5436
|
+
queueHealthSweepCycle += 1;
|
|
5437
|
+
if (queueHealthSweepCycle % QUEUE_HEALTH_SWEEP_EVERY_N_CYCLES === 0) {
|
|
5438
|
+
runQueueHealthSweep(state.jobs);
|
|
5439
|
+
}
|
|
5440
|
+
|
|
4138
5441
|
if (dead.length === 0) return;
|
|
4139
5442
|
|
|
4140
|
-
await mutate((s) => {
|
|
4141
|
-
for (const { slug, pid, outcome } of dead) {
|
|
5443
|
+
await mutate(async (s) => {
|
|
5444
|
+
for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
|
|
4142
5445
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
4143
5446
|
if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
|
|
4144
5447
|
const success = outcome === 'success';
|
|
4145
|
-
|
|
5448
|
+
|
|
5449
|
+
// Best-effort in-place leftover computation: a job whose owning
|
|
5450
|
+
// process vanished without spawnJob()'s own finally block ever
|
|
5451
|
+
// running (the exact case this reaper exists for) never got that
|
|
5452
|
+
// block's salvage OR leftover-attribution pass either. Only
|
|
5453
|
+
// attempted when the row carries a persisted pre-run baseline
|
|
5454
|
+
// (guardBaseline, persisted by spawnJob at dispatch — see there).
|
|
5455
|
+
// With no baseline there is no safe way to tell this job's own dirt
|
|
5456
|
+
// from a human's or a sibling's pre-existing WIP, so this skips
|
|
5457
|
+
// rather than ever dumping/attributing the whole tree.
|
|
5458
|
+
let deltaPaths = null;
|
|
5459
|
+
if (Array.isArray(s.jobs[idx].guardBaseline) && s.jobs[idx].runId) {
|
|
5460
|
+
try {
|
|
5461
|
+
const rowCwd = s.jobs[idx].cwd || s.config?.defaultCwd || DEFAULT_PROJECT_CWD;
|
|
5462
|
+
const after = await uncommittedChanges(rowCwd);
|
|
5463
|
+
if (after) {
|
|
5464
|
+
const baseSet = new Set(s.jobs[idx].guardBaseline);
|
|
5465
|
+
deltaPaths = after.filter((p) => !baseSet.has(p));
|
|
5466
|
+
if (deltaPaths.length) {
|
|
5467
|
+
const salvagePath = path.join(RUNS_DIR, s.jobs[idx].runId, `${slug}.uncommitted.patch`);
|
|
5468
|
+
const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: rowCwd, paths: deltaPaths, outFile: salvagePath });
|
|
5469
|
+
if (salvage && salvage.ok) {
|
|
5470
|
+
s.jobs[idx].salvagePatch = salvagePath;
|
|
5471
|
+
console.log(`[scheduler] reapDeadRunningJobs: salvaged ${salvage.bytes} byte(s) of uncommitted in-place diff for ${slug} to ${salvagePath}`);
|
|
5472
|
+
}
|
|
5473
|
+
}
|
|
5474
|
+
}
|
|
5475
|
+
} catch (e) {
|
|
5476
|
+
console.error(`[scheduler] reapDeadRunningJobs: in-place salvage failed for ${slug}`, e);
|
|
5477
|
+
}
|
|
5478
|
+
}
|
|
5479
|
+
const leftoverSuffix = deltaPaths && deltaPaths.length
|
|
5480
|
+
? ` — left ${deltaPaths.length} files uncommitted`
|
|
5481
|
+
: '';
|
|
5482
|
+
const transitionReason = (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
|
|
5483
|
+
|
|
5484
|
+
transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
|
|
4146
5485
|
s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
|
|
4147
5486
|
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
4148
|
-
s.jobs[idx].error = success ? null :
|
|
5487
|
+
s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
|
|
5488
|
+
s.jobs[idx].gateOutcome = gateOutcome;
|
|
4149
5489
|
delete s.jobs[idx].runtime;
|
|
5490
|
+
delete s.jobs[idx].guardBaseline;
|
|
5491
|
+
delete s.jobs[idx].guardHeadBefore;
|
|
5492
|
+
applyLeftoverFields(s.jobs[idx], deltaPaths);
|
|
4150
5493
|
runningSet.delete(slug);
|
|
4151
|
-
|
|
5494
|
+
// A dead job reaped here never reached spawnJob's own finally block
|
|
5495
|
+
// (that's this reaper's whole reason to exist — see its header
|
|
5496
|
+
// comment) — so if it held the quiet-machine lease, spawnJob never
|
|
5497
|
+
// got the chance to release it. Release it here too, or a
|
|
5498
|
+
// quietMachine job whose process silently vanished (OOM, a crash
|
|
5499
|
+
// with no exit event) wedges the lease held forever and stalls
|
|
5500
|
+
// dispatch for every project until the app restarts.
|
|
5501
|
+
if (s.jobs[idx].quietMachine === true) quietMachineLease.release(slug);
|
|
5502
|
+
if (pidless) {
|
|
5503
|
+
console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
|
|
5504
|
+
appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
|
|
5505
|
+
} else {
|
|
5506
|
+
console.log(`[scheduler] reaped dead job slug=${slug} pid=${pid} outcome=${outcome}`);
|
|
5507
|
+
}
|
|
4152
5508
|
}
|
|
4153
5509
|
});
|
|
4154
5510
|
|
|
@@ -4335,14 +5691,19 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
|
|
|
4335
5691
|
// investigation jobs correctly found "nothing to fix" but were flagged
|
|
4336
5692
|
// anyway). For non-fix-plan jobs the exemption never applies, so rescanning
|
|
4337
5693
|
// their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
|
|
4338
|
-
const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'pass_no_commit', 'pass_no_commit_already_shipped']);
|
|
4339
|
-
|
|
4340
|
-
// Bounds fix-plan recursion:
|
|
4341
|
-
// (
|
|
4342
|
-
//
|
|
4343
|
-
//
|
|
4344
|
-
//
|
|
4345
|
-
|
|
5694
|
+
const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'abandoned_background_task', 'pass_no_commit', 'pass_no_commit_already_shipped']);
|
|
5695
|
+
|
|
5696
|
+
// Bounds fix-plan recursion: cap N permits at most N+1 fix jobs per original
|
|
5697
|
+
// slug (depth 1 = the original job, depth 2 = its `-fix`, depth 3+ is
|
|
5698
|
+
// excluded). With N=1 that's `<slug>-fix` and `<slug>-fix-fix`, never a third
|
|
5699
|
+
// `-fix-fix-fix`. Lowered from 2 to 1 on 2026-08-31 (starry-night-ships):
|
|
5700
|
+
// three concurrent chains (115-fix-fix, 113-fix-fix, 111-fix-fix-fix) were
|
|
5701
|
+
// riding the old cap, and 115-fix-fix's own root-cause section read "The
|
|
5702
|
+
// code was already CORRECT. Only verification and commit failed." — a third
|
|
5703
|
+
// auto-retry re-runs an entire PRD and test battery to redo a `git commit`,
|
|
5704
|
+
// at near-zero marginal success probability. Shared by selectAutoFixTargets
|
|
5705
|
+
// and spawnInvestigation so both call sites agree on one threshold.
|
|
5706
|
+
const MAX_INVESTIGATION_DEPTH = 1;
|
|
4346
5707
|
|
|
4347
5708
|
/**
|
|
4348
5709
|
* True when a fix-plan job's investigationDepth is at or past the recursion
|
|
@@ -4467,22 +5828,48 @@ function isPlanUnqueued(job, queuedSlugs) {
|
|
|
4467
5828
|
* Bias to needs_review: a false yellow costs a human glance, a false green
|
|
4468
5829
|
* costs a silently-unfixed bug — which is exactly what happened.
|
|
4469
5830
|
*/
|
|
5831
|
+
// abandoned_background_task shares no_verdict_sentinel's exact rescan path
|
|
5832
|
+
// (same "sentinel === null && !commitEvidence" gate in runVerify, same
|
|
5833
|
+
// committedDuringRun repo-wide-not-per-job attribution problem) — the PRD 983
|
|
5834
|
+
// incident mechanism above applies identically, so it gets the same guard
|
|
5835
|
+
// rather than a carve-out that would silently reopen the same false-heal hole.
|
|
5836
|
+
const NO_ATTRIBUTABLE_COMMIT_VERDICTS = new Set(['no_verdict_sentinel', 'abandoned_background_task']);
|
|
5837
|
+
|
|
4470
5838
|
function healRefusalReason(job, verdict, committedDuringRun) {
|
|
4471
5839
|
if (!job || !verdict) return null;
|
|
4472
5840
|
if (!COMPLETED_EQUIVALENT_VERDICTS.has(verdict.verdict)) return null;
|
|
4473
|
-
if (job.verifierVerdict
|
|
5841
|
+
if (!NO_ATTRIBUTABLE_COMMIT_VERDICTS.has(job.verifierVerdict)) return null;
|
|
4474
5842
|
// A commit this job actually recorded as its own is real evidence; the
|
|
4475
5843
|
// repo-wide window scan is not.
|
|
4476
5844
|
if (job.landedCommit) return null;
|
|
4477
|
-
return
|
|
5845
|
+
return `${job.verifierVerdict} with no job-attributable commit — refusing to heal`
|
|
4478
5846
|
+ ` (committedInWindow=${committedDuringRun === true} is repo-wide, not proof this job delivered)`;
|
|
4479
5847
|
}
|
|
4480
5848
|
|
|
5849
|
+
/**
|
|
5850
|
+
* True when a `failed` job's failure is unverified-shaped — no result event
|
|
5851
|
+
* was ever recorded for its run (classifyRunOutcome === 'no_result'), so no
|
|
5852
|
+
* SCHEDULER_VERDICT sentinel could have been parsed either, OR it already
|
|
5853
|
+
* carries a RESCANNABLE_VERDICTS verifierVerdict. A row that failed with a
|
|
5854
|
+
* real result event (classifyRunOutcome === 'failed', i.e. a genuine red
|
|
5855
|
+
* gate or a real non-zero-exit error) is excluded — that failure is
|
|
5856
|
+
* evidence, not silence, and must never become a heal candidate (PRD 1102).
|
|
5857
|
+
*/
|
|
5858
|
+
function isFailedUnverifiedShaped(job) {
|
|
5859
|
+
if (!job || job.status !== 'failed') return false;
|
|
5860
|
+
if (job.verifierVerdict && RESCANNABLE_VERDICTS.has(job.verifierVerdict)) return true;
|
|
5861
|
+
const runId = job.runId || resolveRunId(job);
|
|
5862
|
+
if (!runId) return false;
|
|
5863
|
+
const logPath = path.join(RUNS_DIR, runId, `${job.slug}.log`);
|
|
5864
|
+
return classifyRunOutcome(logPath) === 'no_result';
|
|
5865
|
+
}
|
|
5866
|
+
|
|
4481
5867
|
function isRescanCandidate(job) {
|
|
4482
|
-
return
|
|
4483
|
-
|
|
4484
|
-
|
|
4485
|
-
|
|
5868
|
+
if (!job) return false;
|
|
5869
|
+
if (!(job.runId || resolveRunId(job))) return false;
|
|
5870
|
+
if (job.status === 'needs_review') return RESCANNABLE_VERDICTS.has(job.verifierVerdict);
|
|
5871
|
+
if (job.status === 'failed') return isFailedUnverifiedShaped(job);
|
|
5872
|
+
return false;
|
|
4486
5873
|
}
|
|
4487
5874
|
|
|
4488
5875
|
/**
|
|
@@ -4516,10 +5903,36 @@ function isRescanCandidate(job) {
|
|
|
4516
5903
|
* exhausted retry is excluded
|
|
4517
5904
|
* - no fix sibling on disk (fixSlugExists) or already in the queue
|
|
4518
5905
|
*/
|
|
5906
|
+
/**
|
|
5907
|
+
* Persist a writeRcaReport() result onto its job row — job.rcaFailureClass /
|
|
5908
|
+
* job.rcaRecoveryAction — so selectAutoFixTargets and future routing can read
|
|
5909
|
+
* the classification straight off the queue row instead of re-parsing the RCA
|
|
5910
|
+
* markdown. Pure mutation of the passed-in job object; no I/O. A no-op when
|
|
5911
|
+
* the job is missing, has moved off needs_review (e.g. resumed and completed
|
|
5912
|
+
* before this async write landed), or the report was never filed (disabled,
|
|
5913
|
+
* error, etc). Returns whether it applied, for callers/tests that want to
|
|
5914
|
+
* assert on it.
|
|
5915
|
+
*/
|
|
5916
|
+
function applyRcaClassification(job, report) {
|
|
5917
|
+
if (!job || job.status !== 'needs_review' || !report?.filed) return false;
|
|
5918
|
+
job.rcaFailureClass = report.failureClass;
|
|
5919
|
+
job.rcaRecoveryAction = report.recoveryAction;
|
|
5920
|
+
return true;
|
|
5921
|
+
}
|
|
5922
|
+
|
|
4519
5923
|
function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRunId }) {
|
|
4520
5924
|
const slugsInQueue = new Set(jobs.map((j) => j.slug));
|
|
4521
5925
|
return jobs.filter((job) => {
|
|
4522
5926
|
if (job.status !== 'needs_review') return false;
|
|
5927
|
+
// A stale re-run whose work already shipped (rcaReport's 'already-shipped'
|
|
5928
|
+
// class) must never buy a fix-plan PRD — there is nothing to fix, and the
|
|
5929
|
+
// correct recovery (archiving the PRD) is a human/reconcile action, not
|
|
5930
|
+
// an investigation.
|
|
5931
|
+
if (job.rcaRecoveryAction === 'archive') return false;
|
|
5932
|
+
// Resume-first recovery (PRD 1111): a job still eligible for its one
|
|
5933
|
+
// bounded `--resume` attempt must never also become a fix-plan target
|
|
5934
|
+
// in the same pass — see spawnInvestigation's own identical guard.
|
|
5935
|
+
if (selectResumeRecoveryTarget(job)) return false;
|
|
4523
5936
|
const runId = job.runId || resolveJobRunId(job);
|
|
4524
5937
|
if (!runId) return false;
|
|
4525
5938
|
if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth)) return false;
|
|
@@ -4555,12 +5968,52 @@ function isEligibleForImmediateAutoFix(job, allJobs, fixSlugExists) {
|
|
|
4555
5968
|
return targets.some((t) => t.slug === job.slug);
|
|
4556
5969
|
}
|
|
4557
5970
|
|
|
5971
|
+
/**
|
|
5972
|
+
* Widened evidence check (PRD 1102): does at least one commit land AFTER
|
|
5973
|
+
* this job's run window that touches a path the PRD itself declares? Scoped
|
|
5974
|
+
* to the PRD's own declared paths (never the whole repo) so a sibling job's
|
|
5975
|
+
* unrelated commit is not credited to this one — see healRefusalReason's own
|
|
5976
|
+
* rationale for why unscoped, repo-wide evidence is not attribution.
|
|
5977
|
+
*
|
|
5978
|
+
* Returns null (no annotation, never fabricated) when the PRD names no
|
|
5979
|
+
* paths — the caller then has only the existing, already-computed
|
|
5980
|
+
* committedInWindow signal to go on, same as before this PRD.
|
|
5981
|
+
*
|
|
5982
|
+
* @returns {Promise<{commits: string[], paths: string[], detectedAt: string} | null>}
|
|
5983
|
+
*/
|
|
5984
|
+
async function computeLooksDone(job) {
|
|
5985
|
+
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
5986
|
+
const paths = declaredPathsForPrd(prdPath);
|
|
5987
|
+
if (!paths.length) return null;
|
|
5988
|
+
await fetchAllRefs(job.cwd);
|
|
5989
|
+
const commits = await landedSinceRun(job.cwd, job.startedAt, paths);
|
|
5990
|
+
if (!commits.length) return null;
|
|
5991
|
+
return { commits, paths, detectedAt: new Date().toISOString() };
|
|
5992
|
+
}
|
|
5993
|
+
|
|
4558
5994
|
async function reverifyNeedsReview() {
|
|
4559
5995
|
const snap = await readQueue();
|
|
4560
5996
|
const candidates = snap.jobs.filter(isRescanCandidate);
|
|
4561
5997
|
const healed = [];
|
|
4562
5998
|
const leftForReview = [];
|
|
5999
|
+
const looksDoneUpdates = [];
|
|
4563
6000
|
for (const job of candidates) {
|
|
6001
|
+
if (job.status === 'failed') {
|
|
6002
|
+
// A failed row never runs the transcript-verifier rescan below — that
|
|
6003
|
+
// machinery (verifyRun/COMPLETED_EQUIVALENT_VERDICTS) exists to
|
|
6004
|
+
// auto-COMPLETE a stale needs_review row, and a failed row must never
|
|
6005
|
+
// auto-complete through this pass (see the AC's conservative-in-the-
|
|
6006
|
+
// completing-direction constraint). The only thing a failed candidate
|
|
6007
|
+
// can gain here is a looksDone annotation + a failed → needs_review
|
|
6008
|
+
// transition, for a human to confirm.
|
|
6009
|
+
const looksDone = await computeLooksDone(job);
|
|
6010
|
+
if (looksDone) {
|
|
6011
|
+
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: true });
|
|
6012
|
+
} else {
|
|
6013
|
+
leftForReview.push({ slug: job.slug, reason: 'failed, unverified-shaped run — no post-window evidence on declared paths' });
|
|
6014
|
+
}
|
|
6015
|
+
continue;
|
|
6016
|
+
}
|
|
4564
6017
|
const runDir = path.join(RUNS_DIR, job.runId || resolveRunId(job));
|
|
4565
6018
|
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
4566
6019
|
// Derive committedDuringRun from the recorded run window. The live
|
|
@@ -4587,13 +6040,45 @@ async function reverifyNeedsReview() {
|
|
|
4587
6040
|
});
|
|
4588
6041
|
} catch { leftForReview.push({ slug: job.slug, reason: 'verifyRun threw' }); continue; }
|
|
4589
6042
|
const refusal = healRefusalReason(job, v, committedDuringRun);
|
|
6043
|
+
let stillOpen = true;
|
|
4590
6044
|
if (refusal) {
|
|
4591
6045
|
leftForReview.push({ slug: job.slug, reason: refusal });
|
|
4592
6046
|
} else if (v && COMPLETED_EQUIVALENT_VERDICTS.has(v.verdict)) {
|
|
4593
6047
|
healed.push(job.slug);
|
|
6048
|
+
stillOpen = false;
|
|
4594
6049
|
} else {
|
|
4595
6050
|
leftForReview.push({ slug: job.slug, reason: v ? `${v.verdict}: ${v.reason}` : 'null verdict' });
|
|
4596
6051
|
}
|
|
6052
|
+
// Still needs_review after the existing heal pass — widen the evidence
|
|
6053
|
+
// window before giving up on it entirely (unchanged heal semantics for
|
|
6054
|
+
// rows that already qualified above; this only adds an annotation).
|
|
6055
|
+
if (stillOpen) {
|
|
6056
|
+
const looksDone = await computeLooksDone(job);
|
|
6057
|
+
if (looksDone) {
|
|
6058
|
+
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
|
|
6059
|
+
}
|
|
6060
|
+
}
|
|
6061
|
+
}
|
|
6062
|
+
if (looksDoneUpdates.length) {
|
|
6063
|
+
const bySlug = new Map(looksDoneUpdates.map((u) => [u.slug, u]));
|
|
6064
|
+
await mutate((s) => {
|
|
6065
|
+
for (const j of s.jobs) {
|
|
6066
|
+
const u = bySlug.get(j.slug);
|
|
6067
|
+
if (!u) continue;
|
|
6068
|
+
if (u.fromFailed) {
|
|
6069
|
+
transitionJob(j, 'needs_review', {
|
|
6070
|
+
reason: 'looks done — commit(s) since this run touch this PRD\'s declared paths; confirm before archiving',
|
|
6071
|
+
source: 'reverifyNeedsReview:looksDone',
|
|
6072
|
+
});
|
|
6073
|
+
}
|
|
6074
|
+
if (j.status !== 'needs_review') continue;
|
|
6075
|
+
j.looksDone = u.looksDone;
|
|
6076
|
+
const shaList = u.looksDone.commits.slice(0, 5).map((c) => c.slice(0, 7)).join(', ');
|
|
6077
|
+
j.error = `looks done — ${u.looksDone.commits.length} commit(s) since this run touch this PRD's paths (${shaList}); confirm before archiving`;
|
|
6078
|
+
}
|
|
6079
|
+
});
|
|
6080
|
+
console.log(`[scheduler] boot reverify: looksDone annotated for ${looksDoneUpdates.length} row(s): ${looksDoneUpdates.map((u) => u.slug).join(', ')}`);
|
|
6081
|
+
await broadcast();
|
|
4597
6082
|
}
|
|
4598
6083
|
if (healed.length) {
|
|
4599
6084
|
const healSet = new Set(healed);
|
|
@@ -4604,6 +6089,7 @@ async function reverifyNeedsReview() {
|
|
|
4604
6089
|
transitionJob(j, 'completed', { reason: 'boot reverify: stale needs_review healed', source: 'reverifyNeedsReview:heal' });
|
|
4605
6090
|
j.error = null;
|
|
4606
6091
|
delete j.verifierVerdict;
|
|
6092
|
+
delete j.looksDone;
|
|
4607
6093
|
healedPrds.push({ slug: j.slug, cwd: j.cwd });
|
|
4608
6094
|
}
|
|
4609
6095
|
}
|
|
@@ -4643,6 +6129,7 @@ async function reverifyNeedsReview() {
|
|
|
4643
6129
|
orig.exitCode = 0;
|
|
4644
6130
|
orig.error = null;
|
|
4645
6131
|
orig.completedBy = job.slug;
|
|
6132
|
+
delete orig.looksDone;
|
|
4646
6133
|
if (priorStatus === 'needs_review') delete orig.verifierVerdict;
|
|
4647
6134
|
promoted.push(`${orig.slug} (was ${priorStatus}, via ${job.slug})`);
|
|
4648
6135
|
promotedPrds.push({ slug: orig.slug, cwd: orig.cwd });
|
|
@@ -4709,14 +6196,40 @@ async function reverifyNeedsReview() {
|
|
|
4709
6196
|
await broadcast();
|
|
4710
6197
|
}
|
|
4711
6198
|
|
|
6199
|
+
// The annotate mutate above only runs conditionally — when it didn't fire,
|
|
6200
|
+
// afterHealForAnnotate is still the current on-disk state, so reuse it
|
|
6201
|
+
// instead of re-reading queue.json twice more back-to-back for the
|
|
6202
|
+
// resume-recovery and auto-fix passes below (neither of which mutates
|
|
6203
|
+
// synchronously: spawnResumeRecovery/spawnJob's own writes land later).
|
|
6204
|
+
const queueForResumeAndAutofix = (unresolvable.length || exhaustedAutoFix.length || planUnqueued.length)
|
|
6205
|
+
? await readQueue()
|
|
6206
|
+
: afterHealForAnnotate;
|
|
6207
|
+
|
|
6208
|
+
// Resume-first recovery (PRD 1111): before any fix-plan investigation is
|
|
6209
|
+
// authored below, offer the bounded one-attempt `--resume` dispatch to any
|
|
6210
|
+
// needs_review job this periodic pass finds still eligible — e.g. one the
|
|
6211
|
+
// same-tick check in spawnJob missed because the app restarted between
|
|
6212
|
+
// that job parking and this pass running. selectAutoFixTargets below
|
|
6213
|
+
// already excludes every job this loop dispatches, so a resumable job
|
|
6214
|
+
// never also gets a fix-plan PRD authored in the same pass.
|
|
6215
|
+
{
|
|
6216
|
+
for (const job of queueForResumeAndAutofix.jobs) {
|
|
6217
|
+
const target = selectResumeRecoveryTarget(job);
|
|
6218
|
+
if (!target) continue;
|
|
6219
|
+
console.log(`[scheduler] resume-recovery: needs_review ${job.slug} → resuming session ${target.sessionId}`);
|
|
6220
|
+
spawnResumeRecovery(job, target).catch((e) => {
|
|
6221
|
+
console.error('[scheduler] spawnResumeRecovery error', job.slug, e);
|
|
6222
|
+
});
|
|
6223
|
+
}
|
|
6224
|
+
}
|
|
6225
|
+
|
|
4712
6226
|
// Auto-fix: spawn a fix-plan investigation for each job still in
|
|
4713
6227
|
// needs_review after the heal pass (kill-switch: SM_AUTOFIX_DISABLE=1).
|
|
4714
6228
|
// spawnInvestigation early-returns once investigationsInFlight reaches
|
|
4715
6229
|
// MAX_CONCURRENT_INVESTIGATIONS (queues the rest for retry), so this loop
|
|
4716
6230
|
// cannot fan out past the cap regardless of how many targets are selected.
|
|
4717
6231
|
if (process.env.SM_AUTOFIX_DISABLE !== '1') {
|
|
4718
|
-
const
|
|
4719
|
-
const targets = selectAutoFixTargets(afterHeal.jobs, {
|
|
6232
|
+
const targets = selectAutoFixTargets(queueForResumeAndAutofix.jobs, {
|
|
4720
6233
|
fixSlugExists: (s) => candidatePrdsDirs().some((dir) => fs.existsSync(path.join(dir, `${s}.md`))),
|
|
4721
6234
|
});
|
|
4722
6235
|
for (const job of targets) {
|
|
@@ -4745,7 +6258,7 @@ async function reverifyNeedsReview() {
|
|
|
4745
6258
|
}
|
|
4746
6259
|
}
|
|
4747
6260
|
|
|
4748
|
-
return { rescanned: candidates.length, healed, leftForReview };
|
|
6261
|
+
return { rescanned: candidates.length, healed, leftForReview, looksDone: looksDoneUpdates.map((u) => u.slug) };
|
|
4749
6262
|
}
|
|
4750
6263
|
|
|
4751
6264
|
/**
|
|
@@ -4841,7 +6354,7 @@ function registerScheduleHandlers() {
|
|
|
4841
6354
|
// Clears any existing pause first (same semantics as run-now).
|
|
4842
6355
|
await clearPause('run-now');
|
|
4843
6356
|
try {
|
|
4844
|
-
const result = await runDueJobs();
|
|
6357
|
+
const result = await runDueJobs({ bypassLoadGate: true });
|
|
4845
6358
|
return forceTickOutcome(result);
|
|
4846
6359
|
} catch (e) {
|
|
4847
6360
|
logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (force-tick)', meta: { error: e?.message } });
|
|
@@ -4917,7 +6430,7 @@ function registerScheduleHandlers() {
|
|
|
4917
6430
|
ipcMain.handle('schedule:run-now', async () => {
|
|
4918
6431
|
// Manual run-now overrides any auto-pause. Clear it first.
|
|
4919
6432
|
await clearPause('run-now');
|
|
4920
|
-
runDueJobs().catch((e) => logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (run-now)', meta: { error: e?.message } }));
|
|
6433
|
+
runDueJobs({ bypassLoadGate: true }).catch((e) => logs.writeLine({ level: 'error', scope: 'scheduler', message: 'runDueJobs error (run-now)', meta: { error: e?.message } }));
|
|
4921
6434
|
return { ok: true };
|
|
4922
6435
|
});
|
|
4923
6436
|
|
|
@@ -5286,6 +6799,49 @@ async function init() {
|
|
|
5286
6799
|
slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
|
|
5287
6800
|
});
|
|
5288
6801
|
}
|
|
6802
|
+
|
|
6803
|
+
// Stranded-investigation restore. Unlike the two escalations above, this
|
|
6804
|
+
// one ACTS: 'investigating' is a transient status whose restore
|
|
6805
|
+
// (spawnInvestigation's onExit/catch) only runs inside the process that
|
|
6806
|
+
// spawned the probe, so an app restart mid-probe leaves the row frozen
|
|
6807
|
+
// there forever (see findStrandedInvestigations' header, and the
|
|
6808
|
+
// "'investigating' must never be the job's resting state" comment at
|
|
6809
|
+
// spawnInvestigation's onExit). This restores each stranded row to the
|
|
6810
|
+
// exact terminal status it already carried before the probe was
|
|
6811
|
+
// spawned — it never re-runs or re-investigates anything.
|
|
6812
|
+
const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
|
|
6813
|
+
if (stranded.length > 0) {
|
|
6814
|
+
mutate((ms) => {
|
|
6815
|
+
for (const st of stranded) {
|
|
6816
|
+
const j = ms.jobs.find((x) => x.slug === st.slug);
|
|
6817
|
+
if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
|
|
6818
|
+
transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
|
|
6819
|
+
delete j.runtime;
|
|
6820
|
+
console.warn(
|
|
6821
|
+
`[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
|
|
6822
|
+
+ `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
|
|
6823
|
+
+ `restored to '${st.restoreStatus}'`,
|
|
6824
|
+
);
|
|
6825
|
+
appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
|
|
6826
|
+
}
|
|
6827
|
+
})
|
|
6828
|
+
.then(() => broadcast({ flush: true }))
|
|
6829
|
+
.catch(() => {});
|
|
6830
|
+
}
|
|
6831
|
+
|
|
6832
|
+
// Per-project starvation (PRD 1087): a project with pending work that has
|
|
6833
|
+
// been passed over on every tick while OTHER projects dispatch. Nothing
|
|
6834
|
+
// else distinguishes "no pending work" from "pending work, never
|
|
6835
|
+
// started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
|
|
6836
|
+
// Escalation only, same shape as the quarantine/overrun warnings above.
|
|
6837
|
+
for (const sp of findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS)) {
|
|
6838
|
+
console.warn(
|
|
6839
|
+
`[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
|
|
6840
|
+
+ `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
|
|
6841
|
+
+ `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
|
|
6842
|
+
);
|
|
6843
|
+
appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
|
|
6844
|
+
}
|
|
5289
6845
|
}, 10 * 60_000);
|
|
5290
6846
|
|
|
5291
6847
|
// Self-rescheduling poll loop with exponential backoff. Replaces the
|
|
@@ -5412,8 +6968,8 @@ async function init() {
|
|
|
5412
6968
|
// there" (parallelGroup/estimateMinutes/sourcePromptId/epicId/
|
|
5413
6969
|
// archivedStatus); `fields=full` restores them.
|
|
5414
6970
|
function toCompactPrdEntry(entry) {
|
|
5415
|
-
const { slug, title, cwd, mtimeMs, archived, status } = entry;
|
|
5416
|
-
return { slug, title, cwd, mtimeMs, archived, status };
|
|
6971
|
+
const { slug, title, cwd, mtimeMs, archived, status, agentType } = entry;
|
|
6972
|
+
return { slug, title, cwd, mtimeMs, archived, status, agentType };
|
|
5417
6973
|
}
|
|
5418
6974
|
|
|
5419
6975
|
/**
|
|
@@ -5460,6 +7016,7 @@ async function listPrdsInternal() {
|
|
|
5460
7016
|
estimateMinutes: parsed.estimateMinutes,
|
|
5461
7017
|
sourcePromptId: parsed.sourcePromptId,
|
|
5462
7018
|
epicId: parsed.epicId ?? null,
|
|
7019
|
+
agentType: parsed.agentType ?? null,
|
|
5463
7020
|
mtimeMs: stat.mtimeMs,
|
|
5464
7021
|
archived,
|
|
5465
7022
|
};
|
|
@@ -5638,9 +7195,16 @@ const remote = {
|
|
|
5638
7195
|
},
|
|
5639
7196
|
|
|
5640
7197
|
async resetJob(slug, opts = {}) {
|
|
5641
|
-
|
|
7198
|
+
const resolved = await resolveSlugOrReason(slug, opts.cwd);
|
|
7199
|
+
if (!resolved.ok) {
|
|
7200
|
+
return { ok: false, error: resolved.reason === 'invalid-slug' ? 'invalid slug' : unknownSlugMessage(slug) };
|
|
7201
|
+
}
|
|
5642
7202
|
const outcome = await mutate((state) => {
|
|
5643
|
-
|
|
7203
|
+
// Same cwd filter as resolveSlugOrReason's file lookup above — slugs are
|
|
7204
|
+
// derived from title text with no cwd salt, so two different projects
|
|
7205
|
+
// can independently produce the identical slug; an opts.cwd caller must
|
|
7206
|
+
// reset THAT project's job, not just any queue row matching the string.
|
|
7207
|
+
const idx = state.jobs.findIndex((j) => j.slug === slug && (!opts.cwd || j.cwd === opts.cwd));
|
|
5644
7208
|
if (idx < 0) return { kind: 'not-found' };
|
|
5645
7209
|
// Terminal-status guard lives in resetJobFields itself; force:true
|
|
5646
7210
|
// threads through to override it.
|
|
@@ -5662,7 +7226,7 @@ const remote = {
|
|
|
5662
7226
|
|
|
5663
7227
|
async listJobs() {
|
|
5664
7228
|
const state = await readQueue();
|
|
5665
|
-
return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd }));
|
|
7229
|
+
return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd, agentType: j.agentType ?? null }));
|
|
5666
7230
|
},
|
|
5667
7231
|
|
|
5668
7232
|
// Single queue row lookup, used by cancelJob/updatePrd's status guards and
|
|
@@ -5828,10 +7392,16 @@ const remote = {
|
|
|
5828
7392
|
// cancelled job lands in 'failed' with an error naming the cause,
|
|
5829
7393
|
// consistent with every other non-success terminal outcome. Refuses a
|
|
5830
7394
|
// slug that's already terminal — nothing left to cancel.
|
|
5831
|
-
async cancelJob(slug) {
|
|
7395
|
+
async cancelJob(slug, opts = {}) {
|
|
7396
|
+
if (!SCHEDULE_SLUG_RE.test(slug)) return { ok: false, error: 'invalid slug' };
|
|
5832
7397
|
const state = await readQueue();
|
|
5833
|
-
const job = state.jobs.find((j) => j.slug === slug);
|
|
5834
|
-
if (!job)
|
|
7398
|
+
const job = state.jobs.find((j) => j.slug === slug && (!opts.cwd || j.cwd === opts.cwd));
|
|
7399
|
+
if (!job) {
|
|
7400
|
+
return {
|
|
7401
|
+
ok: false,
|
|
7402
|
+
error: `unknown slug "${slug}": no queued job with that name${opts.cwd ? ` in cwd ${opts.cwd}` : ''} — call scheduler_list_jobs to see what exists`,
|
|
7403
|
+
};
|
|
7404
|
+
}
|
|
5835
7405
|
if (job.status === 'completed' || job.status === 'failed' || job.status === 'needs_review' || job.status === 'skipped') {
|
|
5836
7406
|
return { ok: false, error: `job already terminal (status: "${job.status}") — nothing to cancel` };
|
|
5837
7407
|
}
|
|
@@ -5889,9 +7459,10 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
5889
7459
|
return;
|
|
5890
7460
|
}
|
|
5891
7461
|
const force = parsed.force === true;
|
|
5892
|
-
const
|
|
7462
|
+
const cwd = typeof parsed.cwd === 'string' ? parsed.cwd : undefined;
|
|
7463
|
+
const result = await remoteObj.resetJob(slug, { force, cwd });
|
|
5893
7464
|
sendJson(res, 200, result);
|
|
5894
7465
|
});
|
|
5895
7466
|
}
|
|
5896
7467
|
|
|
5897
|
-
module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims };
|
|
7468
|
+
module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure };
|