claude-code-session-manager 0.76.0 → 0.77.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-CBx9l4zN.js → AgentLibrary-B2ie8bbw.js} +2 -2
- package/dist/assets/{DataModel-Bf0EIE_t.js → DataModel-BIJPYw32.js} +1 -1
- package/dist/assets/{History-CpdtWhC8.js → History-CeY6dk9S.js} +2 -2
- package/dist/assets/{Hooks-DyUbMDmg.js → Hooks-BFH2ocKg.js} +2 -2
- package/dist/assets/{HostBilko-By-wIpry.js → HostBilko-36gj9wLz.js} +1 -1
- package/dist/assets/{Library-CQmo4QVC.js → Library-C-hBct39.js} +1 -1
- package/dist/assets/{ListDetail-BQMd6NOm.js → ListDetail-CNq64VWV.js} +1 -1
- package/dist/assets/{MarkdownEditor-DEp43FXX.js → MarkdownEditor-Bh3qt5-1.js} +1 -1
- package/dist/assets/{McpServers-CLarzwqA.js → McpServers-DpGN0oyz.js} +1 -1
- package/dist/assets/{Memory-B0sCdIy1.js → Memory-D59hUjC4.js} +6 -6
- package/dist/assets/{Panel-BhWPVOCD.js → Panel-DCgbaoci.js} +1 -1
- package/dist/assets/{Permissions-Ddlq8T_O.js → Permissions-DAmQ0DYV.js} +2 -2
- package/dist/assets/{Plugins-D2oA_2Jl.js → Plugins-Dyfgn6Is.js} +2 -2
- package/dist/assets/{ProvenanceBadge-DgAgavUM.js → ProvenanceBadge-BiYhPO1U.js} +1 -1
- package/dist/assets/SaveBar-RV7B6sOh.js +1 -0
- package/dist/assets/Scheduler-BPaNqx1b.js +14 -0
- package/dist/assets/{ScopeSwitcher-C_zWEtIl.js → ScopeSwitcher-P4mdLGNU.js} +1 -1
- package/dist/assets/{Settings-2Vx3X5SI.js → Settings-BL4vf5aX.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-BDEUjlTQ.js → SkillReferenceGraph-BRBDyi1_.js} +1 -1
- package/dist/assets/{Skills-Cmrz_LeN.js → Skills-BV08gDUH.js} +2 -2
- package/dist/assets/{SystemPrompt-DVA1eYDP.js → SystemPrompt-CLftSsDw.js} +1 -1
- package/dist/assets/TagLibrary-Bp8jGsd5.js +1 -0
- package/dist/assets/{TiptapBody-DmPc3amD.js → TiptapBody-jCpuB6E5.js} +1 -1
- package/dist/assets/{Toggle-zfd5LJkK.js → Toggle-D2paA1xf.js} +1 -1
- package/dist/assets/{index-B_4PNh9T.js → index-BDRSqBl3.js} +175 -175
- package/dist/assets/{index-DIjnPkRN.css → index-CYhdtisq.css} +1 -1
- package/dist/assets/{settingsSchema-B9es6fdA.js → settingsSchema-6IOLjZZN.js} +1 -1
- package/dist/index.html +2 -2
- package/package.json +8 -2
- package/plugins/session-manager-dev/skills/develop/standards.md +1 -1
- package/scripts/project-pages-logic/dist/logic.cjs +4709 -0
- package/scripts/render-project-pages/dist/renderer.cjs +18900 -0
- package/scripts/render-project-pages.cjs +70 -0
- package/scripts/scheduler-mcp-server.cjs +115 -1
- package/scripts/validate-project-pages-summary.cjs +62 -0
- package/src/main/__tests__/agentModelResolve.test.cjs +66 -0
- package/src/main/__tests__/health-delegation-chain.test.cjs +2 -1
- package/src/main/__tests__/prdAgentType.test.cjs +103 -0
- package/src/main/__tests__/prdCreate.test.cjs +138 -0
- package/src/main/__tests__/prdFrontmatterAgentType.test.cjs +117 -0
- package/src/main/__tests__/prdFrontmatterQuietMachine.test.cjs +108 -0
- package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +485 -0
- package/src/main/__tests__/projectPages.test.cjs +73 -1
- package/src/main/__tests__/rcaReport.test.cjs +54 -0
- package/src/main/__tests__/runVerify.test.cjs +94 -0
- package/src/main/__tests__/scheduler-autofix-select.test.cjs +43 -0
- package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +103 -0
- package/src/main/__tests__/scheduler-effective-concurrency.test.cjs +10 -0
- package/src/main/__tests__/scheduler-foreign-wip-manifest.test.cjs +78 -0
- package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +242 -0
- package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +31 -0
- package/src/main/__tests__/scheduler-launch-failure.test.cjs +201 -0
- package/src/main/__tests__/scheduler-leftover-fields.test.cjs +52 -0
- package/src/main/__tests__/scheduler-looks-done.test.cjs +241 -0
- package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +135 -0
- package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +222 -0
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +147 -0
- package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +212 -0
- package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +194 -0
- package/src/main/__tests__/seedAgentPersonas.test.cjs +75 -14
- package/src/main/config.cjs +4 -1
- package/src/main/index.cjs +8 -1
- package/src/main/ipcSchemas.cjs +51 -0
- package/src/main/lib/__tests__/childWithLog.test.cjs +78 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +152 -2
- package/src/main/lib/__tests__/epicWorktreeMint.test.cjs +4 -2
- package/src/main/lib/__tests__/fixChainDepth.test.cjs +40 -0
- package/src/main/lib/__tests__/gitWorktree.test.cjs +277 -4
- package/src/main/lib/__tests__/gitWorktreeSalvageDelta.test.cjs +153 -0
- package/src/main/lib/__tests__/jobWorktree.test.cjs +5 -3
- package/src/main/lib/__tests__/landedSinceRun.test.cjs +73 -0
- package/src/main/lib/__tests__/launchFailure.test.cjs +220 -0
- package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +1 -0
- package/src/main/lib/__tests__/opsOwnership.test.cjs +7 -0
- package/src/main/lib/__tests__/prdDeclaredPaths.test.cjs +82 -0
- package/src/main/lib/__tests__/queueHealth.test.cjs +58 -0
- package/src/main/lib/__tests__/quietMachineLease.test.cjs +39 -0
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +22 -1
- package/src/main/lib/__tests__/schedulerBatchLaunchHold.test.cjs +125 -0
- package/src/main/lib/__tests__/schedulerBatchQuietMachine.test.cjs +109 -0
- package/src/main/lib/__tests__/schedulerMcpServerHeadlessRefusal.test.cjs +71 -0
- package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +350 -0
- package/src/main/lib/agentModelResolve.cjs +58 -0
- package/src/main/lib/childWithLog.cjs +40 -5
- package/src/main/lib/claudeBin.cjs +54 -1
- package/src/main/lib/definitionOfDone.cjs +3 -2
- package/src/main/lib/delegationReadiness.cjs +115 -9
- package/src/main/lib/epicWorktreeMint.cjs +5 -2
- package/src/main/lib/fixChainDepth.cjs +45 -0
- package/src/main/lib/gitWorktree.cjs +464 -19
- package/src/main/lib/jobWorktree.cjs +1 -0
- package/src/main/lib/landedSinceRun.cjs +55 -0
- package/src/main/lib/launchFailure.cjs +357 -0
- package/src/main/lib/mcpToolCatalog.cjs +87 -2
- package/src/main/lib/opsOwnership.cjs +12 -0
- package/src/main/lib/prdAgentType.cjs +84 -0
- package/src/main/lib/prdCreate.cjs +57 -1
- package/src/main/lib/prdDeclaredPaths.cjs +70 -0
- package/src/main/lib/prdFrontmatter.cjs +17 -3
- package/src/main/lib/projectHomeAdminRoutes.cjs +402 -0
- package/src/main/lib/projectPageSummarySchema.cjs +181 -0
- package/src/main/lib/queueHealth.cjs +38 -0
- package/src/main/lib/queueStore.cjs +9 -2
- package/src/main/lib/quietMachineLease.cjs +48 -0
- package/src/main/lib/rcaReport.cjs +53 -3
- package/src/main/lib/reaperHelpers.cjs +18 -1
- package/src/main/lib/scheduleJobSchema.cjs +31 -0
- package/src/main/lib/scheduleJobTransitions.cjs +6 -2
- package/src/main/lib/schedulerBatch.cjs +133 -29
- package/src/main/lib/schedulerConfig.cjs +19 -0
- package/src/main/projectPages.cjs +160 -2
- package/src/main/runVerify.cjs +50 -9
- package/src/main/scheduler/prdParser.cjs +18 -1
- package/src/main/scheduler.cjs +1371 -97
- package/src/main/seedAgentPersonas.cjs +62 -21
- package/src/main/templates/project-pages-catalog.json +741 -0
- package/src/main/templates/project-pages-pipeline.md +417 -0
- package/src/preload/api.d.ts +118 -2
- package/src/preload/index.cjs +7 -0
- package/src/seed/agents/project-home-builder.md +59 -0
- package/dist/assets/SaveBar-Qvc4Ek-H.js +0 -1
- package/dist/assets/Scheduler-BmYJvNzK.js +0 -14
- package/dist/assets/TagLibrary-DYJGAKZu.js +0 -1
package/src/main/scheduler.cjs
CHANGED
|
@@ -53,9 +53,12 @@ const { ipcMain } = require('electron');
|
|
|
53
53
|
const billing = require('./usage.cjs');
|
|
54
54
|
const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
|
|
55
55
|
const supervisor = require('./supervisor.cjs');
|
|
56
|
-
const { resolveClaudeBin } = require('./lib/claudeBin.cjs');
|
|
56
|
+
const { resolveClaudeBin, probeClaudeVersion } = require('./lib/claudeBin.cjs');
|
|
57
|
+
const launchFailure = require('./lib/launchFailure.cjs');
|
|
58
|
+
const { appendError } = require('./lib/opsErrorLog.cjs');
|
|
57
59
|
const { readTail } = require('./lib/fileTail.cjs');
|
|
58
|
-
const { claudePidAlive, classifyRunOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
|
|
60
|
+
const { claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
|
|
61
|
+
const { computeQueueHealth } = require('./lib/queueHealth.cjs');
|
|
59
62
|
const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
|
|
60
63
|
const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
|
|
61
64
|
const { sendIfAlive } = require('./lib/sendToRenderer.cjs');
|
|
@@ -68,6 +71,8 @@ const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs')
|
|
|
68
71
|
const promptSessionTranscript = require('./promptSessionTranscript.cjs');
|
|
69
72
|
const { verifyRun } = require('./runVerify.cjs');
|
|
70
73
|
const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
|
|
74
|
+
const { landedSinceRun } = require('./lib/landedSinceRun.cjs');
|
|
75
|
+
const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
|
|
71
76
|
const logs = require('./logs.cjs');
|
|
72
77
|
const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
|
|
73
78
|
const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
|
|
@@ -104,6 +109,7 @@ const queueOps = require('./queueOps.cjs');
|
|
|
104
109
|
// home-dir layout.
|
|
105
110
|
const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
|
|
106
111
|
const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
|
|
112
|
+
const agentModelResolve = require('./lib/agentModelResolve.cjs');
|
|
107
113
|
const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
|
|
108
114
|
const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
|
|
109
115
|
const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
|
|
@@ -127,6 +133,7 @@ function resolveOriginSessionId(cwd, epicId) {
|
|
|
127
133
|
return session && typeof session.claudeSessionId === 'string' ? session.claudeSessionId : null;
|
|
128
134
|
}
|
|
129
135
|
const sessionSlots = require('./lib/sessionSlots.cjs');
|
|
136
|
+
const quietMachineLease = require('./lib/quietMachineLease.cjs');
|
|
130
137
|
const jobWorktree = require('./lib/jobWorktree.cjs');
|
|
131
138
|
const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
|
|
132
139
|
const queueStore = require('./lib/queueStore.cjs');
|
|
@@ -184,6 +191,21 @@ const RESULT_TEXT_TAIL_BYTES = 64 * 1024;
|
|
|
184
191
|
const IDLE_OUTPUT_KILL_MS = 20 * 60_000;
|
|
185
192
|
const IDLE_CHECK_INTERVAL_MS = 60_000;
|
|
186
193
|
|
|
194
|
+
// Foreground Bash budget for every spawned `claude -p` job (executor +
|
|
195
|
+
// investigation). The Claude Code harness auto-backgrounds any foreground
|
|
196
|
+
// Bash command past its own default (120s) or max (600s) timeout and returns
|
|
197
|
+
// a tool result promising a later notification — but a headless single-shot
|
|
198
|
+
// run has no later turn, so that notification can never arrive and the run
|
|
199
|
+
// dead-ends mid-verification with no commit and no verdict. Raising these
|
|
200
|
+
// via the child's env moves that trap out of reach of normal gate commands
|
|
201
|
+
// (test suites, builds). BASH_MAX_TIMEOUT_MS MUST stay strictly below
|
|
202
|
+
// IDLE_OUTPUT_KILL_MS with real margin: a long foreground Bash emits no
|
|
203
|
+
// stream-json events while it runs, so the log mtime stalls and the
|
|
204
|
+
// idle-tail watchdog above would SIGTERM the job mid-gate if the two ever
|
|
205
|
+
// crossed — trading one silent failure for another.
|
|
206
|
+
const BASH_DEFAULT_TIMEOUT_MS = 600_000; // 10 min
|
|
207
|
+
const BASH_MAX_TIMEOUT_MS = 900_000; // 15 min — must stay below IDLE_OUTPUT_KILL_MS
|
|
208
|
+
|
|
187
209
|
// Boot reconciliation: a job left 'running' by an app restart/crash whose log
|
|
188
210
|
// shows neither success nor a real failure result was merely interrupted — the
|
|
189
211
|
// host died, the PRD didn't. Re-queue it up to this many times before giving up
|
|
@@ -216,10 +238,12 @@ for it to return. Never start a verification command as a background task
|
|
|
216
238
|
(no background Bash) and then call Monitor, TaskOutput, or ScheduleWakeup to
|
|
217
239
|
pick up its result later — a headless \`claude -p\` run has no later turn, so
|
|
218
240
|
nothing ever delivers that notification and the run dies mid-verification with
|
|
219
|
-
no commit and no verdict.
|
|
220
|
-
|
|
221
|
-
timeout
|
|
222
|
-
|
|
241
|
+
no commit and no verdict. Your foreground Bash budget for this run is
|
|
242
|
+
${BASH_DEFAULT_TIMEOUT_MS / 1000}s by default, up to ${BASH_MAX_TIMEOUT_MS / 1000}s max
|
|
243
|
+
— size your own \`timeout <n>\` wrapper (e.g. \`timeout ${Math.floor(BASH_MAX_TIMEOUT_MS / 1000)} npm test\`)
|
|
244
|
+
to fit inside that ceiling; if a gate command still cannot finish inside
|
|
245
|
+
budget, stop and emit SCHEDULER_VERDICT: FAIL with the reason instead of
|
|
246
|
+
deferring it.
|
|
223
247
|
|
|
224
248
|
1. CODE REVIEW — run \`/code-review --fix\` on your changes and apply the fixes it
|
|
225
249
|
surfaces (correctness first). For any finding you judge a false positive, say
|
|
@@ -303,6 +327,156 @@ function gitHead(cwd) {
|
|
|
303
327
|
});
|
|
304
328
|
}
|
|
305
329
|
|
|
330
|
+
// Return the current `git stash list` entries in cwd as raw lines
|
|
331
|
+
// "<hash> <ref> <subject>" (hash is stable even as ref indices shift when a
|
|
332
|
+
// new entry is pushed on top), or null when the guard does not apply (cwd is
|
|
333
|
+
// not a git work tree, git is missing, or the call errors). Never throws.
|
|
334
|
+
function stashList(cwd) {
|
|
335
|
+
return new Promise((resolve) => {
|
|
336
|
+
if (!cwd) { resolve(null); return; }
|
|
337
|
+
execFile(
|
|
338
|
+
'git',
|
|
339
|
+
['-C', cwd, 'stash', 'list', '--format=%H %gd %gs'],
|
|
340
|
+
{ timeout: 10_000, windowsHide: true },
|
|
341
|
+
(err, stdout) => {
|
|
342
|
+
if (err) { resolve(null); return; }
|
|
343
|
+
resolve(String(stdout || '').split('\n').filter(Boolean));
|
|
344
|
+
},
|
|
345
|
+
);
|
|
346
|
+
});
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
// Parse one `stashList()` line into { hash, ref, subject }. Pure, exported
|
|
350
|
+
// for unit testing. Returns null for a malformed line.
|
|
351
|
+
function parseStashLine(line) {
|
|
352
|
+
const m = /^(\S+)\s+(\S+)\s+(.*)$/.exec(String(line || ''));
|
|
353
|
+
return m ? { hash: m[1], ref: m[2], subject: m[3] } : null;
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
// Paths touched by any commit landed in cwd strictly between headBefore and
|
|
357
|
+
// headAfter. Returns [] when no commit landed (headBefore === headAfter, or
|
|
358
|
+
// either is missing) — used by the shared-tree guard below to tell a path
|
|
359
|
+
// the job legitimately committed apart from a path that just silently went
|
|
360
|
+
// quiet with nothing to explain it. Never throws.
|
|
361
|
+
function pathsChangedSince(cwd, headBefore, headAfter) {
|
|
362
|
+
return new Promise((resolve) => {
|
|
363
|
+
if (!cwd || !headBefore || !headAfter || headBefore === headAfter) { resolve([]); return; }
|
|
364
|
+
execFile(
|
|
365
|
+
'git',
|
|
366
|
+
['-C', cwd, 'diff', '--name-only', `${headBefore}..${headAfter}`],
|
|
367
|
+
{ timeout: 10_000, windowsHide: true },
|
|
368
|
+
(err, stdout) => { resolve(err ? [] : String(stdout || '').split('\n').filter(Boolean)); },
|
|
369
|
+
);
|
|
370
|
+
});
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
// Restore ONE specific stash ref (never a blanket pop of "whatever is on
|
|
374
|
+
// top") into cwd: apply, then drop only on a clean apply. On conflict the
|
|
375
|
+
// entry is left in place — never dropped, never forced — so the operator's
|
|
376
|
+
// own `git stash pop`/`apply` still works afterward. Never throws.
|
|
377
|
+
function restoreSpecificStash(cwd, ref) {
|
|
378
|
+
return new Promise((resolve) => {
|
|
379
|
+
execFile('git', ['-C', cwd, 'stash', 'apply', ref], { timeout: 10_000, windowsHide: true }, (applyErr, _stdout, applyStderr) => {
|
|
380
|
+
if (applyErr) {
|
|
381
|
+
resolve({ ok: false, error: String(applyStderr || applyErr.message || applyErr).trim().split('\n')[0] });
|
|
382
|
+
return;
|
|
383
|
+
}
|
|
384
|
+
execFile('git', ['-C', cwd, 'stash', 'drop', ref], { timeout: 10_000, windowsHide: true }, () => {
|
|
385
|
+
resolve({ ok: true });
|
|
386
|
+
});
|
|
387
|
+
});
|
|
388
|
+
});
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
// Diff a before/after `stashList()` pair plus a before/after dirty-path pair
|
|
392
|
+
// to find what an in-place job silently discarded from a tree it shares with
|
|
393
|
+
// something else (Incident: social-signals-trader 2026-09-01, a blanket
|
|
394
|
+
// `git stash` reverted a live operator config edit with no error anywhere).
|
|
395
|
+
// Two independent signals, either of which means the job discarded state it
|
|
396
|
+
// did not create:
|
|
397
|
+
// - newStashes: a stash entry now present that wasn't in the baseline —
|
|
398
|
+
// the job ran `git stash` itself.
|
|
399
|
+
// - reverted: a path that was dirty in the baseline, is clean now, and was
|
|
400
|
+
// not touched by any commit landed during the run — the job reset/
|
|
401
|
+
// checked-out over pre-existing uncommitted work without stashing it.
|
|
402
|
+
// Pure/no I/O — the guard's git calls happen at the call site
|
|
403
|
+
// (checkSharedTreeGuard). Exported for unit testing.
|
|
404
|
+
function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAfter, pathsCommittedDuringRun }) {
|
|
405
|
+
const beforeHashes = new Set((stashBefore || []).map((l) => parseStashLine(l)?.hash).filter(Boolean));
|
|
406
|
+
const newStashes = (stashAfter || [])
|
|
407
|
+
.map(parseStashLine)
|
|
408
|
+
.filter((e) => e && !beforeHashes.has(e.hash));
|
|
409
|
+
const dirtyAfterSet = new Set(dirtyAfter || []);
|
|
410
|
+
const committedSet = new Set(pathsCommittedDuringRun || []);
|
|
411
|
+
const reverted = (dirtyBefore || []).filter((p) => !dirtyAfterSet.has(p) && !committedSet.has(p));
|
|
412
|
+
return { newStashes, reverted };
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// Post-run shared-tree guard for an IN-PLACE job (worktree.ok === false —
|
|
416
|
+
// callers must gate on that; a worktree-isolated run's git state can never
|
|
417
|
+
// leak into guardCwd, so there is nothing here to check). Best-effort: never
|
|
418
|
+
// throws, never changes the job's exit code. Restores exactly one
|
|
419
|
+
// executor-created stash (never guesses when there are 2+); reports anything
|
|
420
|
+
// it can't safely resolve on the returned object so the caller can surface it
|
|
421
|
+
// on the job row instead of finishing silently green.
|
|
422
|
+
async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug }) {
|
|
423
|
+
try {
|
|
424
|
+
const [stashAfter, headAfter] = await Promise.all([
|
|
425
|
+
module.exports.stashList(cwd),
|
|
426
|
+
module.exports.gitHead(cwd),
|
|
427
|
+
]);
|
|
428
|
+
const pathsCommittedDuringRun = await module.exports.pathsChangedSince(cwd, headBefore, headAfter);
|
|
429
|
+
// First pass: which stashes are new. Decided before charging anything
|
|
430
|
+
// against dirtyBaseline — a path this run's own stash covers must not be
|
|
431
|
+
// judged "reverted" using dirty state captured before the restore below
|
|
432
|
+
// has had a chance to bring it back.
|
|
433
|
+
const { newStashes } = module.exports.evaluateSharedTreeGuard({
|
|
434
|
+
stashBefore: stashBaseline,
|
|
435
|
+
stashAfter,
|
|
436
|
+
dirtyBefore: [],
|
|
437
|
+
dirtyAfter: [],
|
|
438
|
+
pathsCommittedDuringRun,
|
|
439
|
+
});
|
|
440
|
+
|
|
441
|
+
const result = {};
|
|
442
|
+
if (newStashes.length === 1) {
|
|
443
|
+
const [entry] = newStashes;
|
|
444
|
+
const restore = await module.exports.restoreSpecificStash(cwd, entry.ref);
|
|
445
|
+
if (restore.ok) {
|
|
446
|
+
result.restoredStash = entry.ref;
|
|
447
|
+
console.log(`[scheduler] ${slug}: restored a stash the job created in the shared tree (${entry.ref})`);
|
|
448
|
+
} else {
|
|
449
|
+
result.restoreFailed = `${entry.ref}: ${restore.error || 'apply failed'}`;
|
|
450
|
+
console.error(`[scheduler] ${slug}: shared-tree guard could not restore ${entry.ref}: ${restore.error}`);
|
|
451
|
+
}
|
|
452
|
+
} else if (newStashes.length > 1) {
|
|
453
|
+
result.ambiguousStashes = newStashes.map((e) => e.ref);
|
|
454
|
+
console.error(`[scheduler] ${slug}: shared-tree guard found ${newStashes.length} stashes the job created — ambiguous, not auto-restoring (${result.ambiguousStashes.join(', ')})`);
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
// Second pass: recompute "reverted" against the tree's dirty state AFTER
|
|
458
|
+
// any restore attempt above, so a path that came back via a successfully
|
|
459
|
+
// restored stash is not ALSO reported as an unexplained revert (it was
|
|
460
|
+
// explained — by the stash this guard just restored).
|
|
461
|
+
const dirtyAfter = await module.exports.uncommittedChanges(cwd);
|
|
462
|
+
const { reverted } = module.exports.evaluateSharedTreeGuard({
|
|
463
|
+
stashBefore: stashBaseline,
|
|
464
|
+
stashAfter,
|
|
465
|
+
dirtyBefore: dirtyBaseline,
|
|
466
|
+
dirtyAfter,
|
|
467
|
+
pathsCommittedDuringRun,
|
|
468
|
+
});
|
|
469
|
+
if (reverted.length) {
|
|
470
|
+
result.reverted = reverted;
|
|
471
|
+
console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
|
|
472
|
+
}
|
|
473
|
+
return (result.restoredStash || result.restoreFailed || result.ambiguousStashes || result.reverted) ? result : null;
|
|
474
|
+
} catch (e) {
|
|
475
|
+
console.error(`[scheduler] ${slug}: shared-tree guard error`, e);
|
|
476
|
+
return null;
|
|
477
|
+
}
|
|
478
|
+
}
|
|
479
|
+
|
|
306
480
|
// True when cwd is inside a git repository. Used to keep a non-git cwd (e.g.
|
|
307
481
|
// a scratch dir like /tmp) from ever being handed to an investigation's
|
|
308
482
|
// fix-plan as its cwd — the commit guard, worktree isolation, and
|
|
@@ -1563,9 +1737,11 @@ async function reconcile(state) {
|
|
|
1563
1737
|
// membership, so moving the file between Epic dirs must re-point the row.
|
|
1564
1738
|
epicId: p.epicId ?? job.epicId ?? null,
|
|
1565
1739
|
dependsOn: p.dependsOn,
|
|
1740
|
+
quietMachine: p.quietMachine === true,
|
|
1566
1741
|
originSessionId: job.originSessionId
|
|
1567
1742
|
?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
|
|
1568
1743
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1744
|
+
agentType: p.agentType ?? job.agentType ?? null,
|
|
1569
1745
|
};
|
|
1570
1746
|
// Adopt path: a row parked 'quarantined' (no createdVia provenance when
|
|
1571
1747
|
// discovered) whose PRD file now carries a stamp — written via the
|
|
@@ -1674,8 +1850,10 @@ async function reconcile(state) {
|
|
|
1674
1850
|
sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
|
|
1675
1851
|
epicId: p.epicId ?? inv.row?.epicId ?? null,
|
|
1676
1852
|
dependsOn: p.dependsOn,
|
|
1853
|
+
quietMachine: p.quietMachine === true,
|
|
1677
1854
|
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
1678
1855
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1856
|
+
agentType: p.agentType ?? inv.row?.agentType ?? null,
|
|
1679
1857
|
};
|
|
1680
1858
|
const reason = `reconcile: repaired invalid status ${JSON.stringify(oldStatus)}`;
|
|
1681
1859
|
// A repair is not a lifecycle transition — the corrupted `status` was
|
|
@@ -1794,8 +1972,10 @@ async function reconcile(state) {
|
|
|
1794
1972
|
sourceTabId: p.sourceTabId,
|
|
1795
1973
|
epicId: p.epicId ?? null,
|
|
1796
1974
|
dependsOn: p.dependsOn,
|
|
1975
|
+
quietMachine: p.quietMachine === true,
|
|
1797
1976
|
originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
1798
1977
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1978
|
+
agentType: p.agentType ?? null,
|
|
1799
1979
|
status: 'pending',
|
|
1800
1980
|
// Enqueue time (PRD 1086/1087): the cross-project fairness tiebreak and
|
|
1801
1981
|
// the starvation escalation both need a provable age for a pending row;
|
|
@@ -2051,6 +2231,10 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
|
|
|
2051
2231
|
lastRunAt: state.lastRunAt,
|
|
2052
2232
|
nextReset: getNextResetCached(),
|
|
2053
2233
|
paused: state.paused,
|
|
2234
|
+
// Launch circuit breaker (issue #11): which personas cannot launch right
|
|
2235
|
+
// now and why, plus any degraded-mode env in force. Empty objects when healthy.
|
|
2236
|
+
launchBlocks: state.launchBlocks ?? {},
|
|
2237
|
+
launchMitigations: state.launchMitigations ?? {},
|
|
2054
2238
|
utilization: cachedUtilization,
|
|
2055
2239
|
pollHealth: {
|
|
2056
2240
|
lastPollAt,
|
|
@@ -2194,7 +2378,17 @@ async function setPaused(reason, resumeAtIso) {
|
|
|
2194
2378
|
|
|
2195
2379
|
async function clearPause(source) {
|
|
2196
2380
|
if (resumeTimer) { clearTimeout(resumeTimer); resumeTimer = null; }
|
|
2381
|
+
const humanOverride = source === 'manual' || source === 'run-now';
|
|
2197
2382
|
const wasPaused = await mutate((s) => {
|
|
2383
|
+
// A human Resume / Run now also re-closes every launch circuit breaker:
|
|
2384
|
+
// the operator is asserting the environment is fixed (CLI updated,
|
|
2385
|
+
// re-logged-in). The next dispatch of each persona is its probe; if the
|
|
2386
|
+
// environment is still broken the breaker simply re-arms.
|
|
2387
|
+
if (humanOverride && s.launchBlocks && Object.keys(s.launchBlocks).length) {
|
|
2388
|
+
console.log(`[scheduler] clearPause (${source}): clearing launch blocks [${Object.keys(s.launchBlocks).join(', ')}]`);
|
|
2389
|
+
for (const j of s.jobs) if (j.status === 'pending' && j.heldReason && /^launch blocked/.test(j.heldReason)) delete j.heldReason;
|
|
2390
|
+
s.launchBlocks = {};
|
|
2391
|
+
}
|
|
2198
2392
|
if (!s.paused) return false;
|
|
2199
2393
|
console.log(`[scheduler] clearPause (${source || 'manual'})`);
|
|
2200
2394
|
s.paused = null;
|
|
@@ -2242,6 +2436,29 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
2242
2436
|
job.error = errorMsg ?? null;
|
|
2243
2437
|
delete job.runtime;
|
|
2244
2438
|
delete job.verifierVerdict;
|
|
2439
|
+
delete job.uncommittedPaths;
|
|
2440
|
+
delete job.resumeRecoveryAttempted;
|
|
2441
|
+
// Same "this run's outcome, not durable across a reset" category as the
|
|
2442
|
+
// fields above — a stale 'archive' recoveryAction from a prior life of this
|
|
2443
|
+
// slug must never survive a reset and silently exclude a genuinely-new
|
|
2444
|
+
// needs_review episode from selectAutoFixTargets (applyRcaClassification
|
|
2445
|
+
// only overwrites these on a successful RCA write, so without this they
|
|
2446
|
+
// can otherwise linger forever when RCA is disabled or errors).
|
|
2447
|
+
delete job.rcaFailureClass;
|
|
2448
|
+
delete job.rcaRecoveryAction;
|
|
2449
|
+
// Like exitCode: this run's outcome, not durable across a reset — a stale
|
|
2450
|
+
// leak badge from a prior attempt must not linger once the job re-fires.
|
|
2451
|
+
delete job.leakedDescendants;
|
|
2452
|
+
// A pending row is about to re-run fresh — a stale leftover badge or a
|
|
2453
|
+
// stale pre-run baseline from the attempt that just ended must not linger
|
|
2454
|
+
// and be mistaken for THIS (not-yet-run) attempt's own output. spawnJob
|
|
2455
|
+
// persists a brand-new guardBaseline at the next dispatch.
|
|
2456
|
+
delete job.guardBaseline;
|
|
2457
|
+
delete job.guardHeadBefore;
|
|
2458
|
+
delete job.leftoverPaths;
|
|
2459
|
+
delete job.leftoverCount;
|
|
2460
|
+
delete job.leftoverPathsTruncated;
|
|
2461
|
+
delete job.preRunDirtyPaths;
|
|
2245
2462
|
// Deliberately NOT deleting job.landedCommit: it must outlive a reset so a
|
|
2246
2463
|
// re-fired run of this same slug can pass it to verifyRun as
|
|
2247
2464
|
// priorLandedCommit (pass_no_commit_prior_run_verified exemption).
|
|
@@ -2792,7 +3009,190 @@ function commitGuardVerdict({ newlyDirty, siblingRunning, ranInWorktree, jobSelf
|
|
|
2792
3009
|
reason: `finish protocol incomplete: ${dirty.length} uncommitted file(s) left in working tree (e.g. ${sample})${salvageNote}`,
|
|
2793
3010
|
downgradeTo: 'needs_review',
|
|
2794
3011
|
annotations: carried.length ? carried : undefined,
|
|
3012
|
+
// The exact dirty-path list, persisted on the job row (see the
|
|
3013
|
+
// commit-guard call site) so a later resume-recovery attempt
|
|
3014
|
+
// (selectResumeRecoveryTarget) can name these paths without re-running
|
|
3015
|
+
// `git status` against a tree that may have moved on since.
|
|
3016
|
+
dirtyPaths: dirty,
|
|
3017
|
+
};
|
|
3018
|
+
}
|
|
3019
|
+
|
|
3020
|
+
// Every path list this job leaves attributed on the row is capped here so a
|
|
3021
|
+
// pathological run (thousands of newly-dirty files) never bloats queue.json
|
|
3022
|
+
// or history.jsonl — the count is still recorded in full via leftoverCount,
|
|
3023
|
+
// only the displayed sample is capped.
|
|
3024
|
+
const LEFTOVER_PATHS_CAP = 50;
|
|
3025
|
+
|
|
3026
|
+
/**
|
|
3027
|
+
* Pure: turn a newly-dirty path list (or null, meaning "couldn't tell" —
|
|
3028
|
+
* never "left nothing") into the `leftoverPaths`/`leftoverCount`/
|
|
3029
|
+
* `leftoverPathsTruncated` triple stamped on a terminal job row, or null when
|
|
3030
|
+
* there is nothing to attribute (empty list, or the list itself is
|
|
3031
|
+
* unavailable). One shape for both the worktree-leftover path and the
|
|
3032
|
+
* in-place baseline-delta path — see this function's callers in spawnJob and
|
|
3033
|
+
* reapDeadRunningJobs, both of which diff against a persisted pre-run
|
|
3034
|
+
* baseline so a human's or a sibling's pre-existing WIP is never
|
|
3035
|
+
* misattributed to this job.
|
|
3036
|
+
*/
|
|
3037
|
+
function leftoverFieldsFrom(paths) {
|
|
3038
|
+
if (!Array.isArray(paths) || paths.length === 0) return null;
|
|
3039
|
+
const fields = {
|
|
3040
|
+
leftoverPaths: paths.slice(0, LEFTOVER_PATHS_CAP),
|
|
3041
|
+
leftoverCount: paths.length,
|
|
2795
3042
|
};
|
|
3043
|
+
if (paths.length > LEFTOVER_PATHS_CAP) fields.leftoverPathsTruncated = true;
|
|
3044
|
+
return fields;
|
|
3045
|
+
}
|
|
3046
|
+
|
|
3047
|
+
/** Stamps (or clears) the leftover-attribution fields on a job row in place. */
|
|
3048
|
+
function applyLeftoverFields(row, paths) {
|
|
3049
|
+
delete row.leftoverPaths;
|
|
3050
|
+
delete row.leftoverCount;
|
|
3051
|
+
delete row.leftoverPathsTruncated;
|
|
3052
|
+
const fields = leftoverFieldsFrom(paths);
|
|
3053
|
+
if (fields) Object.assign(row, fields);
|
|
3054
|
+
}
|
|
3055
|
+
|
|
3056
|
+
// Same bloat concern as LEFTOVER_PATHS_CAP, applied to the PRE-run dirty
|
|
3057
|
+
// snapshot (foreign WIP the job did not create) instead of the post-run
|
|
3058
|
+
// leftover delta.
|
|
3059
|
+
const PRE_RUN_DIRTY_PATHS_CAP = 200;
|
|
3060
|
+
|
|
3061
|
+
/**
|
|
3062
|
+
* Pure: cap a dirty-path list at PRE_RUN_DIRTY_PATHS_CAP, appending a
|
|
3063
|
+
* `+N more` marker entry when truncated, so queue.json/history.jsonl never
|
|
3064
|
+
* take on an unbounded row for a pathologically dirty shared tree. Returns
|
|
3065
|
+
* [] for null/empty input (never null) — callers gate storage/prompt
|
|
3066
|
+
* injection on `.length` the same way carriedPaths already does.
|
|
3067
|
+
*/
|
|
3068
|
+
function capDirtyPaths(paths, cap = PRE_RUN_DIRTY_PATHS_CAP) {
|
|
3069
|
+
if (!Array.isArray(paths) || paths.length === 0) return [];
|
|
3070
|
+
if (paths.length <= cap) return paths.slice();
|
|
3071
|
+
return [...paths.slice(0, cap), `+${paths.length - cap} more`];
|
|
3072
|
+
}
|
|
3073
|
+
|
|
3074
|
+
// Stable, machine-greppable delimiter — a downstream PRD (verifier scoring
|
|
3075
|
+
// foreign-WIP test failures separately) greps the executor log for this
|
|
3076
|
+
// exact marker, so its text must never be reworded casually.
|
|
3077
|
+
const FOREIGN_WIP_DELIMITER = '--- FOREIGN WORKING-TREE STATE (not your work) ---';
|
|
3078
|
+
const FOREIGN_WIP_END_DELIMITER = '--- END FOREIGN WORKING-TREE STATE ---';
|
|
3079
|
+
|
|
3080
|
+
/**
|
|
3081
|
+
* Pure: build the executor-prompt section warning about pre-existing dirty
|
|
3082
|
+
* paths this job does not own — either base WIP carried into an isolated
|
|
3083
|
+
* worktree (PRD 1094's carriedPaths, checked first since it's the more
|
|
3084
|
+
* specific/authoritative case) or the raw pre-run dirty snapshot of a shared
|
|
3085
|
+
* (non-isolated) tree. Returns '' when both lists are empty so a clean spawn
|
|
3086
|
+
* produces a byte-identical prompt to before this section existed.
|
|
3087
|
+
*/
|
|
3088
|
+
function buildForeignWipSection({ preRunDirtyPaths, carriedPaths } = {}) {
|
|
3089
|
+
const carried = Array.isArray(carriedPaths) ? carriedPaths.filter(Boolean) : [];
|
|
3090
|
+
if (carried.length) {
|
|
3091
|
+
return [
|
|
3092
|
+
FOREIGN_WIP_DELIMITER,
|
|
3093
|
+
'This job is running in an isolated git worktree, but the following paths carry uncommitted base-tree work-in-progress that was carried into this checkout so the tree is self-consistent. The authoritative copy of these files lives in the MAIN tree, not this worktree.',
|
|
3094
|
+
'These files were already modified before this job started. They are NOT this job\'s work:',
|
|
3095
|
+
...carried.map((p) => ` ${p}`),
|
|
3096
|
+
'Do not stage, commit, revert, or stash these paths. A test failure confined to these paths is not this job\'s regression.',
|
|
3097
|
+
FOREIGN_WIP_END_DELIMITER,
|
|
3098
|
+
].join('\n');
|
|
3099
|
+
}
|
|
3100
|
+
const dirty = Array.isArray(preRunDirtyPaths) ? preRunDirtyPaths.filter(Boolean) : [];
|
|
3101
|
+
if (dirty.length) {
|
|
3102
|
+
return [
|
|
3103
|
+
FOREIGN_WIP_DELIMITER,
|
|
3104
|
+
'This job is running in a SHARED working tree (not isolated in its own worktree). The following paths were already modified when this job started:',
|
|
3105
|
+
...dirty.map((p) => ` ${p}`),
|
|
3106
|
+
'These files are NOT this job\'s work. Do not stage, commit, revert, or stash them. A test failure confined to these paths is not this job\'s regression.',
|
|
3107
|
+
FOREIGN_WIP_END_DELIMITER,
|
|
3108
|
+
].join('\n');
|
|
3109
|
+
}
|
|
3110
|
+
return '';
|
|
3111
|
+
}
|
|
3112
|
+
|
|
3113
|
+
/**
|
|
3114
|
+
* Resume-first recovery (PRD 1111). A job parked in needs_review with verdict
|
|
3115
|
+
* 'uncommitted_changes' has a live claude session (job.sessionId, minted by
|
|
3116
|
+
* spawnJob's `--session-id`) that already has full context of the work it
|
|
3117
|
+
* left uncommitted — resuming it via `claude -p --resume <sessionId>` lets it
|
|
3118
|
+
* finish its own finish-protocol COMMIT step, instead of spawnInvestigation
|
|
3119
|
+
* cold-reading the log to author a fix-plan PRD that a FRESH session then has
|
|
3120
|
+
* to re-derive that same context for. Pure/no I/O so the eligibility rule can
|
|
3121
|
+
* be unit-tested directly, matching classifyFailureOutcome/commitGuardVerdict.
|
|
3122
|
+
*
|
|
3123
|
+
* Bounded to exactly one attempt via job.resumeRecoveryAttempted, stamped
|
|
3124
|
+
* atomically with the 'running' transition inside spawnJob's own dispatch
|
|
3125
|
+
* mutate (see spawnJob) — never here — so a crash between this function
|
|
3126
|
+
* returning a target and the resume child actually spawning cannot leave the
|
|
3127
|
+
* job re-eligible.
|
|
3128
|
+
*
|
|
3129
|
+
* Kill-switch: SM_RESUME_RECOVERY_DISABLE=1 restores today's behaviour
|
|
3130
|
+
* exactly (always returns null), mirroring SM_RCA_DISABLE/SM_DOD_DISABLE.
|
|
3131
|
+
*/
|
|
3132
|
+
function selectResumeRecoveryTarget(job) {
|
|
3133
|
+
if (process.env.SM_RESUME_RECOVERY_DISABLE === '1') return null;
|
|
3134
|
+
if (!job || job.status !== 'needs_review') return null;
|
|
3135
|
+
if (job.verifierVerdict !== 'uncommitted_changes') return null;
|
|
3136
|
+
if (typeof job.sessionId !== 'string' || job.sessionId.length === 0) return null;
|
|
3137
|
+
if (job.resumeRecoveryAttempted === true) return null;
|
|
3138
|
+
const dirtyPaths = Array.isArray(job.uncommittedPaths)
|
|
3139
|
+
? job.uncommittedPaths.filter((p) => typeof p === 'string' && p.length > 0)
|
|
3140
|
+
: [];
|
|
3141
|
+
if (!dirtyPaths.length) return null;
|
|
3142
|
+
return { slug: job.slug, sessionId: job.sessionId, dirtyPaths, salvagePatch: job.salvagePatch || null };
|
|
3143
|
+
}
|
|
3144
|
+
|
|
3145
|
+
/**
|
|
3146
|
+
* Short deterministic preamble for a resume-recovery dispatch — NEVER the
|
|
3147
|
+
* original PRD body (the resumed session already has that in its own
|
|
3148
|
+
* conversation history; re-embedding it would just waste context and risk
|
|
3149
|
+
* contradicting whatever state the session actually left behind). Names the
|
|
3150
|
+
* exact paths recorded on the parked job row so the resumed run can verify
|
|
3151
|
+
* them on disk before trusting them, rather than re-deriving them itself.
|
|
3152
|
+
*/
|
|
3153
|
+
function buildResumeRecoveryPreamble({ dirtyPaths, salvagePatch }) {
|
|
3154
|
+
const pathList = dirtyPaths.map((p) => `- ${p}`).join('\n');
|
|
3155
|
+
const salvageLine = salvagePatch
|
|
3156
|
+
? `\nA salvage patch of this work was also captured at: ${salvagePatch} — apply it if any of the paths above are missing from the working tree.\n`
|
|
3157
|
+
: '';
|
|
3158
|
+
return `RESUME RECOVERY: your previous run in this same session left uncommitted work on disk and exited before the finish protocol's COMMIT step ran. This is a continuation of that same session, not a new task — do not restart from scratch.
|
|
3159
|
+
|
|
3160
|
+
The following path(s) were recorded as uncommitted when this job was parked for review:
|
|
3161
|
+
${pathList}
|
|
3162
|
+
${salvageLine}
|
|
3163
|
+
Do the following now:
|
|
3164
|
+
1. Run \`git status\` and verify each path above is present on disk and reflects your intended work. If a path is missing, investigate before recreating it — don't blindly redo work that may already be committed or salvaged elsewhere.
|
|
3165
|
+
2. Run the project's verification gate (typecheck/lint/tests) in the FOREGROUND — wait for it to finish and read its real exit code before proceeding. Do not background it.
|
|
3166
|
+
3. If the gate is green, stage exactly the paths you created or modified for this work and commit them: \`git add <path> [<path>...] && git commit -m "<type>(<scope>): <summary>"\`.
|
|
3167
|
+
4. If the gate is red, fix it, then commit.
|
|
3168
|
+
|
|
3169
|
+
As the LAST LINE of your final result text, emit exactly one of:
|
|
3170
|
+
SCHEDULER_VERDICT: PASS
|
|
3171
|
+
SCHEDULER_VERDICT: FAIL <one-line reason>
|
|
3172
|
+
Print PASS only once the commit above has actually landed.`;
|
|
3173
|
+
}
|
|
3174
|
+
|
|
3175
|
+
/**
|
|
3176
|
+
* Pure argv builder for a `claude -p` child spawn, shared so the
|
|
3177
|
+
* resume-vs-fresh-session choice is made in exactly one place. `resume`
|
|
3178
|
+
* selects `--resume <sessionId>` (reconnect) INSTEAD of `--session-id
|
|
3179
|
+
* <sessionId>` (mint) — the two flags are mutually exclusive, never both.
|
|
3180
|
+
* `--model` is always explicit (never left to the CLI's drifting default —
|
|
3181
|
+
* see conventions.md). `systemPrompt`, when given (the PRD's `agentType`
|
|
3182
|
+
* persona body, resolved by agentModelResolve.cjs's resolvePrdPersonaForSpawn),
|
|
3183
|
+
* is passed as `--append-system-prompt` so the executor IS that persona at
|
|
3184
|
+
* launch rather than being asked in prose to adopt one.
|
|
3185
|
+
*/
|
|
3186
|
+
function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }) {
|
|
3187
|
+
return [
|
|
3188
|
+
'-p', prompt,
|
|
3189
|
+
'--model', model,
|
|
3190
|
+
...(systemPrompt ? ['--append-system-prompt', systemPrompt] : []),
|
|
3191
|
+
'--dangerously-skip-permissions',
|
|
3192
|
+
'--output-format', 'stream-json',
|
|
3193
|
+
'--verbose',
|
|
3194
|
+
...(resume ? ['--resume', sessionId] : ['--session-id', sessionId]),
|
|
3195
|
+
];
|
|
2796
3196
|
}
|
|
2797
3197
|
|
|
2798
3198
|
// ---------- execution ----------
|
|
@@ -2813,7 +3213,7 @@ function pickRunDir() {
|
|
|
2813
3213
|
* Watchdogs are declared as an array; the result-tailer's exit-code mapping
|
|
2814
3214
|
* (success+killedBySignal → 0) is scheduler-specific and lives in onExit.
|
|
2815
3215
|
*/
|
|
2816
|
-
async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
3216
|
+
async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget = null, foreignWip = null, launchEnv = null) {
|
|
2817
3217
|
const logPath = path.join(runDir, `${job.slug}.log`);
|
|
2818
3218
|
const metaPath = path.join(runDir, `${job.slug}.meta.json`);
|
|
2819
3219
|
// `cwd` stays the MAIN tree throughout — PRD lookup (findPrdDir/prdPathForJob)
|
|
@@ -2823,7 +3223,10 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2823
3223
|
const cwd = job.cwd || defaultCwd;
|
|
2824
3224
|
const spawnCwd = execCwd || cwd;
|
|
2825
3225
|
const startedAt = Date.now();
|
|
2826
|
-
|
|
3226
|
+
// Resume mode (PRD 1111) reconnects to the SAME session that left the
|
|
3227
|
+
// uncommitted work — reusing its id via `--resume` instead of minting a
|
|
3228
|
+
// fresh one via `--session-id` is the entire point of the recovery.
|
|
3229
|
+
const sessionId = resumeTarget ? resumeTarget.sessionId : randomUUID();
|
|
2827
3230
|
|
|
2828
3231
|
// Phase 1: open log fd so we can emit pre-spawn diagnostics (early-exit
|
|
2829
3232
|
// error paths) before the child is created. withChildAndLog takes ownership
|
|
@@ -2848,14 +3251,23 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2848
3251
|
return { exitCode: -1, durationMs: 0, error: errMsg, sessionId };
|
|
2849
3252
|
}
|
|
2850
3253
|
|
|
3254
|
+
let prompt;
|
|
3255
|
+
let prdPath = null;
|
|
3256
|
+
if (resumeTarget) {
|
|
3257
|
+
// Resume mode (PRD 1111): a short deterministic preamble naming the
|
|
3258
|
+
// recorded dirty paths, NEVER the original PRD body — the resumed
|
|
3259
|
+
// session already has that in its own conversation history via
|
|
3260
|
+
// --resume, and re-embedding it here would just contradict whatever
|
|
3261
|
+
// state the session actually left on disk.
|
|
3262
|
+
prompt = buildResumeRecoveryPreamble({ dirtyPaths: resumeTarget.dirtyPaths, salvagePatch: resumeTarget.salvagePatch });
|
|
3263
|
+
} else {
|
|
2851
3264
|
// Read full PRD body fresh from disk (queue stored only the preview).
|
|
2852
3265
|
// Resolve through findPrdDir's full candidate search (legacy flat dir +
|
|
2853
3266
|
// every project's Epic-scoped dirs) first, so the common case — a live
|
|
2854
3267
|
// Epic-scoped PRD — is a first-try hit instead of probing the retired flat
|
|
2855
3268
|
// dir and only then falling back.
|
|
2856
|
-
let prompt;
|
|
2857
3269
|
const resolvedDir = await findPrdDir(job.slug);
|
|
2858
|
-
|
|
3270
|
+
prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
|
|
2859
3271
|
try {
|
|
2860
3272
|
const parsed = await parsePrd(prdPath);
|
|
2861
3273
|
// The review → security-review → verify → commit finish sequence is
|
|
@@ -2911,14 +3323,17 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2911
3323
|
return { exitCode: -1, durationMs: 0, error: e?.message };
|
|
2912
3324
|
}
|
|
2913
3325
|
}
|
|
3326
|
+
} // end resumeTarget ? preamble : normal-PRD-read
|
|
2914
3327
|
|
|
3328
|
+
let contextDigestApplied = false;
|
|
3329
|
+
let originSessionId = null;
|
|
3330
|
+
if (!resumeTarget) {
|
|
2915
3331
|
// Prepend the Epic's own session digest (PRD 950/958) when this job traces
|
|
2916
3332
|
// back to a known Epic — additive only, never mutates the PRD body itself.
|
|
2917
3333
|
// A missing/unresolved epicId or a digest build failure is a silent no-op:
|
|
2918
3334
|
// the PRD's own body must remain sufficient to complete the job on its own.
|
|
2919
3335
|
const digestEpicId = job.epicId ?? job.sourcePromptId ?? null;
|
|
2920
|
-
|
|
2921
|
-
let contextDigestApplied = false;
|
|
3336
|
+
originSessionId = resolveOriginSessionId(cwd, digestEpicId);
|
|
2922
3337
|
let digestText = '';
|
|
2923
3338
|
if (originSessionId) {
|
|
2924
3339
|
try {
|
|
@@ -2929,12 +3344,39 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2929
3344
|
digestText = '';
|
|
2930
3345
|
}
|
|
2931
3346
|
}
|
|
3347
|
+
// Quiet-machine degraded dispatch (PRD 1107): this job opted into
|
|
3348
|
+
// `quietMachine: true` but waited past quietMachineWaitMs() without the
|
|
3349
|
+
// machine ever going quiet, so pickNextBatch dispatched it anyway rather
|
|
3350
|
+
// than wedge the queue forever. Told to the executor as a plain prompt
|
|
3351
|
+
// line — its own wall-clock/timing acceptance criteria were measured (or
|
|
3352
|
+
// will be measured) under CPU contention from sibling jobs, not on a
|
|
3353
|
+
// quiet machine, so it should not report a timing result as trustworthy
|
|
3354
|
+
// without saying so.
|
|
3355
|
+
if (job.quietLeaseDegraded === true) {
|
|
3356
|
+
prompt = `NOTE: this job requested \`quietMachine: true\` but the machine never went idle within the `
|
|
3357
|
+
+ `configured wait window, so it was dispatched anyway (degraded). Any timing/frame-rate/performance `
|
|
3358
|
+
+ `measurement in this run may be affected by CPU contention from other concurrent jobs — say so explicitly `
|
|
3359
|
+
+ `in your result rather than reporting it as a clean measurement.\n\n${prompt}`;
|
|
3360
|
+
}
|
|
2932
3361
|
// Always route through composeExecutorPrompt (even with an empty digest)
|
|
2933
3362
|
// so the finish protocol is appended in the prompt's tail exactly once,
|
|
2934
3363
|
// after any digest fence rather than concatenated ahead of it.
|
|
2935
3364
|
prompt = composeExecutorPrompt({ prdBody: prompt, digestText, finishProtocol: FINISH_PROTOCOL });
|
|
2936
3365
|
|
|
2937
|
-
|
|
3366
|
+
// Foreign-WIP manifest (starry-night-ships PRD 148 postmortem): the
|
|
3367
|
+
// scheduler already knows, at spawn time, which dirty paths this job did
|
|
3368
|
+
// not create — either a shared tree's pre-existing dirty set or worktree
|
|
3369
|
+
// WIP carried in from the base tree (PRD 1094). Telling the executor
|
|
3370
|
+
// explicitly here means it never has to bisect by content to prove a test
|
|
3371
|
+
// failure isn't its own regression. '' (clean spawn) leaves prompt
|
|
3372
|
+
// byte-identical to before this section existed.
|
|
3373
|
+
const foreignWipSection = buildForeignWipSection(foreignWip || {});
|
|
3374
|
+
if (foreignWipSection) {
|
|
3375
|
+
prompt = `${prompt}\n\n${foreignWipSection}`;
|
|
3376
|
+
}
|
|
3377
|
+
} // end !resumeTarget digest/finish-protocol composition
|
|
3378
|
+
|
|
3379
|
+
const promptCheck = validatePromptForSpawn(prompt, resumeTarget ? `<resume recovery preamble for ${job.slug}>` : prdPath);
|
|
2938
3380
|
if (!promptCheck.ok) {
|
|
2939
3381
|
safeLog(`[scheduler] ${promptCheck.error}\n`);
|
|
2940
3382
|
closeFd();
|
|
@@ -2942,6 +3384,15 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2942
3384
|
return { exitCode: -1, durationMs: 0, error: promptCheck.error, sessionId };
|
|
2943
3385
|
}
|
|
2944
3386
|
|
|
3387
|
+
// PRD agentType → persona + model (PRD 1115): resolved for both fresh and
|
|
3388
|
+
// resume dispatches, keyed off job.agentType (persisted on the queue row
|
|
3389
|
+
// by reconcile()) rather than re-reading the PRD file — a resumed session
|
|
3390
|
+
// must keep launching as the SAME persona it started as. Never throws;
|
|
3391
|
+
// a dangling/absent agentType falls back to no persona + FALLBACK_MODEL
|
|
3392
|
+
// and is logged once by resolvePrdPersonaForSpawn itself.
|
|
3393
|
+
const personaResolution = await agentModelResolve.resolvePrdPersonaForSpawn({ cwd, agentType: job.agentType });
|
|
3394
|
+
safeLog(`[scheduler] agentType=${job.agentType || '(none)'} persona=${personaResolution.personaPath || '(fallback — no persona applied)'} model=${personaResolution.model}\n`);
|
|
3395
|
+
|
|
2945
3396
|
return await new Promise((resolve) => {
|
|
2946
3397
|
const claudeBin = resolveClaudeBin();
|
|
2947
3398
|
// Strip Claude Code env and secrets that leak in when session-manager is
|
|
@@ -2954,7 +3405,26 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
2954
3405
|
// originProjectRoot so a job running inside its own worktree can still
|
|
2955
3406
|
// resolve the real project for create-prd/open-session/readiness. See
|
|
2956
3407
|
// projectRootResolve.cjs.
|
|
2957
|
-
|
|
3408
|
+
// SM_SCHEDULER_JOB_SLUG marks the child (and the MCP servers it
|
|
3409
|
+
// inherits its env to) as a headless scheduled executor, so
|
|
3410
|
+
// scheduler-mcp-server.cjs can refuse scheduler_create_prd from inside a
|
|
3411
|
+
// run (issue #11 list C1 — the PRD 460 self-queue incident). Only a
|
|
3412
|
+
// persona whose whole job is decomposition may still queue.
|
|
3413
|
+
// `launchEnv` is the launch circuit breaker's degraded-mode env (e.g.
|
|
3414
|
+
// MAX_THINKING_TOKENS=0 while an outdated CLI's thinking parameter is
|
|
3415
|
+
// being rejected — lib/launchFailure.cjs); applied last so it wins.
|
|
3416
|
+
const childEnv = cleanChildEnv({
|
|
3417
|
+
PATH: pathWithUserBins(),
|
|
3418
|
+
SM_PROJECT_ROOT: cwd,
|
|
3419
|
+
SM_SCHEDULER_JOB_SLUG: job.slug,
|
|
3420
|
+
SM_SCHEDULER_JOB_MAY_QUEUE: job.agentType === 'architect' ? '1' : '0',
|
|
3421
|
+
BASH_DEFAULT_TIMEOUT_MS: String(BASH_DEFAULT_TIMEOUT_MS),
|
|
3422
|
+
BASH_MAX_TIMEOUT_MS: String(BASH_MAX_TIMEOUT_MS),
|
|
3423
|
+
...(launchEnv && typeof launchEnv === 'object' ? launchEnv : {}),
|
|
3424
|
+
});
|
|
3425
|
+
if (launchEnv && Object.keys(launchEnv).length) {
|
|
3426
|
+
safeLog(`[scheduler] launch mitigation env applied: ${Object.entries(launchEnv).map(([k, v]) => `${k}=${v}`).join(' ')}\n`);
|
|
3427
|
+
}
|
|
2958
3428
|
|
|
2959
3429
|
// Track whether the agent has emitted a `result` event in its JSONL stream.
|
|
2960
3430
|
// null until seen; then one of "success" | "error_max_turns" | … per the
|
|
@@ -3061,14 +3531,16 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
3061
3531
|
closeFd,
|
|
3062
3532
|
spawn: {
|
|
3063
3533
|
command: claudeBin,
|
|
3064
|
-
|
|
3065
|
-
|
|
3066
|
-
|
|
3067
|
-
|
|
3068
|
-
|
|
3069
|
-
|
|
3070
|
-
|
|
3071
|
-
|
|
3534
|
+
// Resume mode passes `--resume <sessionId>` (reconnect to the SAME
|
|
3535
|
+
// session) INSTEAD of `--session-id <sessionId>` (mint a new one) —
|
|
3536
|
+
// never both, see buildClaudeSpawnArgs.
|
|
3537
|
+
args: buildClaudeSpawnArgs({
|
|
3538
|
+
prompt,
|
|
3539
|
+
model: personaResolution.model,
|
|
3540
|
+
sessionId,
|
|
3541
|
+
resume: !!resumeTarget,
|
|
3542
|
+
systemPrompt: personaResolution.systemPrompt,
|
|
3543
|
+
}),
|
|
3072
3544
|
options: {
|
|
3073
3545
|
cwd: spawnCwd,
|
|
3074
3546
|
env: childEnv,
|
|
@@ -3081,8 +3553,13 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
3081
3553
|
},
|
|
3082
3554
|
},
|
|
3083
3555
|
watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog],
|
|
3084
|
-
onExit({ exitCode, signal, killedByWatchdog: _kbw, error, spawnFailed, safeLog: sl }) {
|
|
3556
|
+
onExit({ exitCode, signal, killedByWatchdog: _kbw, error, spawnFailed, leakedDescendants, safeLog: sl }) {
|
|
3085
3557
|
const durationMs = Date.now() - startedAt;
|
|
3558
|
+
const leaked = leakedDescendants ?? [];
|
|
3559
|
+
if (leaked.length > 0) {
|
|
3560
|
+
sl(`\n[scheduler] leaked ${leaked.length} descendant(s) swept from job process group: ` +
|
|
3561
|
+
`${leaked.map((p) => `pid=${p.pid} comm=${p.comm} pcpu=${p.pcpu} etimes=${p.etimes}s`).join(', ')}\n`);
|
|
3562
|
+
}
|
|
3086
3563
|
|
|
3087
3564
|
if (error) {
|
|
3088
3565
|
// Covers both synchronous spawn failure and child 'error' events.
|
|
@@ -3092,8 +3569,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
3092
3569
|
sl(`\n[scheduler] ${errMsg}\n`);
|
|
3093
3570
|
// Sync write: inside a Promise executor callback; must flush meta
|
|
3094
3571
|
// before resolve() so the spawnJob mutate() that follows sees it.
|
|
3095
|
-
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA, originSessionId, contextDigestApplied });
|
|
3096
|
-
resolve({ exitCode: -1, durationMs, error: errMsg, sessionId });
|
|
3572
|
+
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA, originSessionId, contextDigestApplied });
|
|
3573
|
+
resolve({ exitCode: -1, durationMs, error: errMsg, leakedDescendants: leaked, sessionId });
|
|
3097
3574
|
return;
|
|
3098
3575
|
}
|
|
3099
3576
|
|
|
@@ -3116,16 +3593,33 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd) {
|
|
|
3116
3593
|
`duration=${Math.round(durationMs / 1000)}s\n`);
|
|
3117
3594
|
const rateLimited = effectiveCode !== 0 && detectRateLimitInLog(logPath);
|
|
3118
3595
|
const networkError = effectiveCode !== 0 && !rateLimited && detectNetworkErrorInLog(logPath);
|
|
3596
|
+
// Non-run detection (issue #11 lists A1–A3): the harness's `result`
|
|
3597
|
+
// event tells us whether the model ever got a turn. A first-request
|
|
3598
|
+
// API rejection (num_turns ≤ 1, output_tokens 0, `API Error:` text)
|
|
3599
|
+
// is a broken ENVIRONMENT, not a failed PRD — spawnJob routes it to
|
|
3600
|
+
// the launch circuit breaker instead of failed/investigation.
|
|
3601
|
+
const resultStats = launchFailure.readResultEvent(logPath);
|
|
3602
|
+
const launchFailed = (effectiveCode !== 0 && !rateLimited && !networkError)
|
|
3603
|
+
? launchFailure.classifyLaunchFailure(resultStats)
|
|
3604
|
+
: null;
|
|
3605
|
+
if (launchFailed) {
|
|
3606
|
+
sl(`\n[scheduler] LAUNCH FAILURE (${launchFailed.kind}${launchFailed.httpStatus ? ` HTTP ${launchFailed.httpStatus}` : ''}): ` +
|
|
3607
|
+
`${launchFailed.message} — no turn was taken; this is not a PRD failure\n`);
|
|
3608
|
+
}
|
|
3119
3609
|
// Sync write: child 'exit' handler must flush meta before resolve()
|
|
3120
3610
|
// so the spawnJob mutate() that follows sees the persisted exit code.
|
|
3121
3611
|
config.writeJsonSync(metaPath, {
|
|
3122
3612
|
slug: job.slug, cwd, sessionId, exitCode: effectiveCode, rateLimited, networkError,
|
|
3123
|
-
|
|
3613
|
+
launchFailure: launchFailed,
|
|
3614
|
+
numTurns: resultStats?.numTurns ?? null, outputTokens: resultStats?.outputTokens ?? null,
|
|
3615
|
+
totalCostUsd: resultStats?.totalCostUsd ?? null, terminalReasonFromHarness: resultStats?.terminalReason ?? null,
|
|
3616
|
+
launchEnvApplied: launchEnv && Object.keys(launchEnv).length ? Object.keys(launchEnv) : [],
|
|
3617
|
+
startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
|
|
3124
3618
|
agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
|
|
3125
3619
|
schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
|
|
3126
3620
|
originSessionId, contextDigestApplied,
|
|
3127
3621
|
});
|
|
3128
|
-
resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, sessionId });
|
|
3622
|
+
resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, launchFailure: launchFailed, resultStats, leakedDescendants: leaked, sessionId });
|
|
3129
3623
|
},
|
|
3130
3624
|
});
|
|
3131
3625
|
|
|
@@ -3190,7 +3684,27 @@ function healTargetForFix(fixSlug, jobs) {
|
|
|
3190
3684
|
* spawnInvestigation computes.
|
|
3191
3685
|
*/
|
|
3192
3686
|
function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group }) {
|
|
3193
|
-
|
|
3687
|
+
const abandonedBackgroundTaskNote = failedJob.verifierVerdict === 'abandoned_background_task' ? `
|
|
3688
|
+
|
|
3689
|
+
# Known failure class: abandoned background task
|
|
3690
|
+
This job's verifier verdict is \`abandoned_background_task\`: the transcript shows a Bash command
|
|
3691
|
+
auto-backgrounded past its foreground timeout, and the run ended waiting for a "you will be
|
|
3692
|
+
notified when it completes" callback a headless run structurally cannot receive. This is NOT
|
|
3693
|
+
evidence the work failed — it is evidence the run stopped short of its finish protocol. The work is
|
|
3694
|
+
usually already written and correct; only the commit is missing.
|
|
3695
|
+
|
|
3696
|
+
By the time this investigation runs, the failed job's isolated worktree (if it ran in one) has
|
|
3697
|
+
already been cleaned up — \`${cwd}\` is the BASE repo, not that worktree, so a plain \`git status\`/
|
|
3698
|
+
\`git diff\` there will usually show nothing even though real work was produced. The scheduler
|
|
3699
|
+
salvages any uncommitted diff from a killed job's worktree BEFORE deleting it${
|
|
3700
|
+
failedJob.salvagePatch ? `, and this job's salvage patch was captured at:\n\n ${failedJob.salvagePatch}` : ', to a `.uncommitted.patch` file next to the run log — check the run dir for one'
|
|
3701
|
+
}.
|
|
3702
|
+
|
|
3703
|
+
The fix-plan PRD you write for this MUST instruct its executor to, in order:
|
|
3704
|
+
1. Check for a salvage patch (named \`<slug>.uncommitted.patch\` in the run directory${failedJob.salvagePatch ? `, e.g. \`${failedJob.salvagePatch}\`` : ''}) and, if found, apply it to the working tree BEFORE inspecting \`git status\`/\`git diff\` in ${cwd} for uncommitted changes matching the original PRD's acceptance criteria.
|
|
3705
|
+
2. If the work is present (via the applied patch or already in the tree) and satisfies the acceptance criteria, run the project's verify commands and COMMIT it — do not re-implement or re-plan the PRD from scratch.
|
|
3706
|
+
3. Only fall back to re-implementing whatever acceptance criteria are genuinely missing after applying any salvage patch, not the whole PRD.` : '';
|
|
3707
|
+
return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${abandonedBackgroundTaskNote}
|
|
3194
3708
|
|
|
3195
3709
|
# Failed job
|
|
3196
3710
|
- Slug: ${failedJob.slug}
|
|
@@ -3308,7 +3822,37 @@ function readRunOutcomeSidecars(runDir, slug) {
|
|
|
3308
3822
|
* or if the run being investigated actually verified clean (nothing to fix —
|
|
3309
3823
|
* see shouldSkipInvestigationForCleanRun).
|
|
3310
3824
|
*/
|
|
3825
|
+
const INVESTIGATION_LAUNCH_KEY = 'investigation';
|
|
3826
|
+
|
|
3311
3827
|
async function spawnInvestigation(failedJob, runDir) {
|
|
3828
|
+
// The probe launches with the same CLI as the job it diagnoses. While
|
|
3829
|
+
// that CLI cannot launch at all (launch circuit breaker, issue #11 list
|
|
3830
|
+
// B1: probes e4f82da2/d374e6bf died on the same HTTP 400 as the runs
|
|
3831
|
+
// they were investigating) there is nothing to diagnose — skip, loudly.
|
|
3832
|
+
{
|
|
3833
|
+
const state = await readQueue().catch(() => null);
|
|
3834
|
+
const block = state?.launchBlocks?.[INVESTIGATION_LAUNCH_KEY];
|
|
3835
|
+
const jobBlock = state?.launchBlocks?.[launchFailure.launchBlockKeyFor(failedJob)];
|
|
3836
|
+
const gate = launchFailure.evaluateLaunchGate(block || jobBlock, { now: Date.now(), claudeVersion: await probeClaudeVersion() });
|
|
3837
|
+
if (gate.state === 'blocked') {
|
|
3838
|
+
console.log(`[scheduler] skip investigation: ${failedJob.slug} — ${gate.reason}`);
|
|
3839
|
+
await mutate((s) => {
|
|
3840
|
+
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
3841
|
+
if (j) { j.autoFixOutcome = 'launch-blocked'; j.autoFixNote = gate.reason; }
|
|
3842
|
+
}).catch(() => {});
|
|
3843
|
+
return { deferred: false };
|
|
3844
|
+
}
|
|
3845
|
+
}
|
|
3846
|
+
// Resume-first recovery (PRD 1111) always gets first refusal — a job
|
|
3847
|
+
// eligible for a bounded `--resume` dispatch must never also get a
|
|
3848
|
+
// cold-read fix-plan PRD authored in the same pass. selectResumeRecoveryTarget
|
|
3849
|
+
// returns null for every job shape spawnInvestigation is normally called
|
|
3850
|
+
// with (e.g. plain 'failed' jobs never carry verifierVerdict
|
|
3851
|
+
// 'uncommitted_changes'), so this is a no-op for the common case.
|
|
3852
|
+
if (selectResumeRecoveryTarget(failedJob)) {
|
|
3853
|
+
console.log(`[scheduler] skip investigation: ${failedJob.slug} is resume-recovery eligible`);
|
|
3854
|
+
return { deferred: false };
|
|
3855
|
+
}
|
|
3312
3856
|
if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth)) {
|
|
3313
3857
|
console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
|
|
3314
3858
|
return { deferred: false };
|
|
@@ -3418,7 +3962,11 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3418
3962
|
await broadcast({ flush: true });
|
|
3419
3963
|
|
|
3420
3964
|
const claudeBin = resolveClaudeBin();
|
|
3421
|
-
const childEnv = cleanChildEnv({
|
|
3965
|
+
const childEnv = cleanChildEnv({
|
|
3966
|
+
PATH: pathWithUserBins(), // Homebrew/user bins for macOS
|
|
3967
|
+
BASH_DEFAULT_TIMEOUT_MS: String(BASH_DEFAULT_TIMEOUT_MS),
|
|
3968
|
+
BASH_MAX_TIMEOUT_MS: String(BASH_MAX_TIMEOUT_MS),
|
|
3969
|
+
});
|
|
3422
3970
|
|
|
3423
3971
|
// Investigation needs only a deadman watchdog — no idle-tail or result-tail
|
|
3424
3972
|
// since investigations are short-running Opus probes with a hard ceiling.
|
|
@@ -3480,6 +4028,23 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3480
4028
|
return;
|
|
3481
4029
|
}
|
|
3482
4030
|
sl(`\n[scheduler] investigation exit code=${exitCode}\n`);
|
|
4031
|
+
if (exitCode !== 0) {
|
|
4032
|
+
const probeResult = launchFailure.readResultEvent(investigationLogPath);
|
|
4033
|
+
const probeLaunchFailure = launchFailure.classifyLaunchFailure(probeResult);
|
|
4034
|
+
if (probeLaunchFailure) {
|
|
4035
|
+
sl(`\n[scheduler] investigation LAUNCH FAILURE (${probeLaunchFailure.kind}): ${probeLaunchFailure.message} — arming '${INVESTIGATION_LAUNCH_KEY}' launch block\n`);
|
|
4036
|
+
probeClaudeVersion().then((claudeVersion) => mutate((s) => {
|
|
4037
|
+
s.launchBlocks = s.launchBlocks || {};
|
|
4038
|
+
s.launchBlocks[INVESTIGATION_LAUNCH_KEY] = launchFailure.armLaunchBlock(s.launchBlocks[INVESTIGATION_LAUNCH_KEY] || null, {
|
|
4039
|
+
kind: probeLaunchFailure.kind, httpStatus: probeLaunchFailure.httpStatus, message: probeLaunchFailure.message,
|
|
4040
|
+
now: Date.now(), claudeVersion, slug: failedJob.slug, runId: failedJob.runId ?? null,
|
|
4041
|
+
});
|
|
4042
|
+
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
4043
|
+
if (j) { j.autoFixOutcome = 'launch-blocked'; j.autoFixNote = `investigation probe never ran: ${probeLaunchFailure.message}`; }
|
|
4044
|
+
})).catch(() => {});
|
|
4045
|
+
return;
|
|
4046
|
+
}
|
|
4047
|
+
}
|
|
3483
4048
|
// Fold the investigation's <RCA> summary into the root-cause report already
|
|
3484
4049
|
// written for this job (needs_review jobs only — writeRcaReport no-ops when
|
|
3485
4050
|
// failedJob has no verifierVerdict, e.g. plain 'failed' jobs never got one).
|
|
@@ -3547,7 +4112,122 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3547
4112
|
}
|
|
3548
4113
|
}
|
|
3549
4114
|
|
|
3550
|
-
|
|
4115
|
+
/**
|
|
4116
|
+
* computeLaunchHolds(state) → Map<slug, reason>
|
|
4117
|
+
*
|
|
4118
|
+
* The launch circuit breaker's per-tick view (lib/launchFailure.cjs, issue
|
|
4119
|
+
* #11): every pending row whose persona is blocked is held with its reason;
|
|
4120
|
+
* when a persona's backoff has elapsed exactly ONE of its pending rows is
|
|
4121
|
+
* left pickable (the half-open probe) and the rest are held behind it. A
|
|
4122
|
+
* CLI version change drops the block outright — that is the incident's real
|
|
4123
|
+
* fix (`claude update`) and the queue must resume on the next tick.
|
|
4124
|
+
* Mutates nothing; spawnJob makes the durable decision at dispatch.
|
|
4125
|
+
*/
|
|
4126
|
+
async function computeLaunchHolds(state, { now = Date.now(), claudeVersion } = {}) {
|
|
4127
|
+
const held = new Map();
|
|
4128
|
+
const blocks = state?.launchBlocks;
|
|
4129
|
+
if (!blocks || typeof blocks !== 'object' || !Object.keys(blocks).length) return held;
|
|
4130
|
+
const version = claudeVersion === undefined ? await probeClaudeVersion() : claudeVersion;
|
|
4131
|
+
const probeAllowed = new Set();
|
|
4132
|
+
for (const j of state.jobs || []) {
|
|
4133
|
+
if (j.status !== 'pending') continue;
|
|
4134
|
+
const key = launchFailure.launchBlockKeyFor(j);
|
|
4135
|
+
const block = blocks[key];
|
|
4136
|
+
if (!block) continue;
|
|
4137
|
+
const gate = launchFailure.evaluateLaunchGate(block, { now, claudeVersion: version });
|
|
4138
|
+
if (gate.state === 'open') continue;
|
|
4139
|
+
if (gate.state === 'probe' && !probeAllowed.has(key)) {
|
|
4140
|
+
probeAllowed.add(key);
|
|
4141
|
+
continue;
|
|
4142
|
+
}
|
|
4143
|
+
held.set(j.slug, gate.state === 'probe'
|
|
4144
|
+
? `launch blocked (${block.kind}) — waiting for this tick's probe of '${key}'`
|
|
4145
|
+
: gate.reason);
|
|
4146
|
+
}
|
|
4147
|
+
return held;
|
|
4148
|
+
}
|
|
4149
|
+
|
|
4150
|
+
/**
|
|
4151
|
+
* A run that never got a turn (res.launchFailure — see executeJob's onExit)
|
|
4152
|
+
* is routed here instead of the failed/investigation path (issue #11 lists
|
|
4153
|
+
* A1–A3, B1): the row goes back to `pending` carrying the API's own message
|
|
4154
|
+
* as its error, no retry budget is consumed, no auto-fix probe is spawned
|
|
4155
|
+
* (it would die the same way), and the persona's launch circuit breaker is
|
|
4156
|
+
* armed so the queue stops re-dispatching identical doomed launches while
|
|
4157
|
+
* still self-healing on backoff / CLI update / human Retry.
|
|
4158
|
+
*/
|
|
4159
|
+
/**
|
|
4160
|
+
* Pure state mutation behind handleLaunchFailure (exported for tests): arms
|
|
4161
|
+
* the persona's breaker and returns the job's `running` row to `pending`
|
|
4162
|
+
* carrying the API message. Returns the armed block.
|
|
4163
|
+
*/
|
|
4164
|
+
function applyLaunchFailure(s, { job, lf, runId, launchKey, mitigationApplied, claudeVersion, now = Date.now() }) {
|
|
4165
|
+
s.launchBlocks = s.launchBlocks || {};
|
|
4166
|
+
s.launchMitigations = s.launchMitigations || {};
|
|
4167
|
+
const prev = s.launchBlocks[launchKey] || null;
|
|
4168
|
+
const armed = launchFailure.armLaunchBlock(prev, {
|
|
4169
|
+
kind: lf.kind, httpStatus: lf.httpStatus, message: lf.message, now, claudeVersion,
|
|
4170
|
+
slug: job.slug, runId, mitigationApplied,
|
|
4171
|
+
});
|
|
4172
|
+
s.launchBlocks[launchKey] = armed;
|
|
4173
|
+
// A mitigation that was in force and still failed is no longer proven —
|
|
4174
|
+
// drop it so the hint and the next probe are honest.
|
|
4175
|
+
if (mitigationApplied && s.launchMitigations[launchKey]) delete s.launchMitigations[launchKey];
|
|
4176
|
+
const i = (s.jobs || []).findIndex((x) => x.slug === job.slug);
|
|
4177
|
+
if (i >= 0 && s.jobs[i].status === 'running') {
|
|
4178
|
+
const prevCount = s.jobs[i].launchFailure?.count ?? 0;
|
|
4179
|
+
const msg = `launch failure (${lf.kind}${lf.httpStatus ? ` HTTP ${lf.httpStatus}` : ''}): ${lf.message}`;
|
|
4180
|
+
resetJobFields(s.jobs[i], msg, { source: 'spawnJob:launch-failure' });
|
|
4181
|
+
s.jobs[i].launchFailure = {
|
|
4182
|
+
kind: lf.kind, httpStatus: lf.httpStatus ?? null, message: lf.message,
|
|
4183
|
+
at: new Date(now).toISOString(), runId, count: prevCount + 1, mitigationApplied,
|
|
4184
|
+
};
|
|
4185
|
+
s.jobs[i].terminalReason = `launch_failure:${lf.kind}`;
|
|
4186
|
+
s.jobs[i].heldReason = armed.exhausted
|
|
4187
|
+
? `launch blocked (${lf.kind}) after ${armed.attempts} failed probe(s) — ${armed.hint}`
|
|
4188
|
+
: `launch blocked (${lf.kind}) — re-probe at ${armed.until}. ${armed.hint}`;
|
|
4189
|
+
}
|
|
4190
|
+
return armed;
|
|
4191
|
+
}
|
|
4192
|
+
|
|
4193
|
+
async function handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion }) {
|
|
4194
|
+
const lf = res.launchFailure;
|
|
4195
|
+
const now = Date.now();
|
|
4196
|
+
const mitigationApplied = !!(launchEnv && Object.keys(launchEnv).length);
|
|
4197
|
+
let armed = null;
|
|
4198
|
+
await mutate((s) => {
|
|
4199
|
+
armed = applyLaunchFailure(s, { job, lf, runId, launchKey, mitigationApplied, claudeVersion, now });
|
|
4200
|
+
});
|
|
4201
|
+
launchFailure.writeOutcomeSidecar(runDir, job.slug, {
|
|
4202
|
+
runId,
|
|
4203
|
+
exitCode: res.exitCode,
|
|
4204
|
+
durationMs: res.durationMs ?? null,
|
|
4205
|
+
numTurns: res.resultStats?.numTurns ?? null,
|
|
4206
|
+
outputTokens: res.resultStats?.outputTokens ?? null,
|
|
4207
|
+
totalCostUsd: res.resultStats?.totalCostUsd ?? null,
|
|
4208
|
+
verdict: null,
|
|
4209
|
+
status: 'pending',
|
|
4210
|
+
terminalReason: `launch_failure:${lf.kind}`,
|
|
4211
|
+
launchFailure: { kind: lf.kind, httpStatus: lf.httpStatus ?? null, message: lf.message },
|
|
4212
|
+
launchEnvApplied: launchEnv ? Object.keys(launchEnv) : [],
|
|
4213
|
+
filesChanged: 0,
|
|
4214
|
+
landedCommit: null,
|
|
4215
|
+
});
|
|
4216
|
+
try {
|
|
4217
|
+
appendError({
|
|
4218
|
+
cwd: job.cwd || DEFAULT_PROJECT_CWD,
|
|
4219
|
+
scope: 'scheduler',
|
|
4220
|
+
level: 'error',
|
|
4221
|
+
message: `launch failure (${lf.kind}) for ${job.slug}: ${lf.message} — persona '${launchKey}' blocked, attempt ${armed?.attempts}${armed?.exhausted ? ' (exhausted; needs CLI update or Retry)' : ''}`,
|
|
4222
|
+
meta: { slug: job.slug, runId, kind: lf.kind, httpStatus: lf.httpStatus ?? null, claudeVersion: claudeVersion ?? null, mitigationApplied, hint: armed?.hint },
|
|
4223
|
+
});
|
|
4224
|
+
} catch { /* durable logging must never break the queue */ }
|
|
4225
|
+
console.error(`[scheduler] ${job.slug}: LAUNCH FAILURE (${lf.kind}${lf.httpStatus ? ` HTTP ${lf.httpStatus}` : ''}) — ${lf.message}. ` +
|
|
4226
|
+
`Persona '${launchKey}' blocked (attempt ${armed?.attempts}${armed?.until ? `, re-probe at ${armed.until}` : ', exhausted'}). ${armed?.hint}`);
|
|
4227
|
+
await broadcast({ flush: true });
|
|
4228
|
+
}
|
|
4229
|
+
|
|
4230
|
+
async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
3551
4231
|
// Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
|
|
3552
4232
|
// — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
|
|
3553
4233
|
// leaves the job pending; the next tick retries when a slot frees up.
|
|
@@ -3557,13 +4237,108 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3557
4237
|
return;
|
|
3558
4238
|
}
|
|
3559
4239
|
runningSet.add(job.slug);
|
|
4240
|
+
// Exclusive quiet-machine lease (PRD 1107) — acquired here, in the same
|
|
4241
|
+
// slot-acquire/dispatch step as sessionSlots, and released in this
|
|
4242
|
+
// function's own finally below alongside sessionSlots.release, so every
|
|
4243
|
+
// exit path (normal exit, timeout, SIGTERM, crash) that already frees the
|
|
4244
|
+
// session slot also frees the lease. pickNextBatch only ever hands this
|
|
4245
|
+
// function a quietMachine job when the lease was free at pick time, so
|
|
4246
|
+
// acquire() here should never fail in practice — but check anyway rather
|
|
4247
|
+
// than assume, since a lease held by a stale slug would otherwise wedge
|
|
4248
|
+
// silently.
|
|
4249
|
+
const quietLeaseAcquired = job.quietMachine === true && quietMachineLease.acquire(job.slug);
|
|
3560
4250
|
try {
|
|
4251
|
+
// Worktree isolation cap check (PRD 1112) — probed BEFORE the job is
|
|
4252
|
+
// marked 'running', so a job that can't get isolation right now is a
|
|
4253
|
+
// DEFERRAL, not a fallback: it stays 'pending' and is retried on the
|
|
4254
|
+
// next dispatch pass, exactly like the sessionSlots miss above, instead
|
|
4255
|
+
// of degrading into an in-place run in a tree a sibling job may be
|
|
4256
|
+
// actively writing to (the shared-tree collision this cap exists to
|
|
4257
|
+
// prevent). Every OTHER worktree.ok===false reason (not a git repo,
|
|
4258
|
+
// disabled, carry-over failure) keeps the existing in-place fallback —
|
|
4259
|
+
// only the cap-reached reason is a deferral, checked here via
|
|
4260
|
+
// createJobWorktree's own reason string so the two paths never
|
|
4261
|
+
// silently drift out of sync with gitWorktree.cjs's actual wording.
|
|
4262
|
+
const preflightWorktree = resumeTarget
|
|
4263
|
+
? { ok: false, reason: 'resume-recovery: running in place to reuse the session\'s prior working tree' }
|
|
4264
|
+
: await jobWorktree.createJobWorktree({ cwd: job.cwd || defaultCwd, slug: job.slug });
|
|
4265
|
+
if (!preflightWorktree.ok && /^worktree cap reached\b/.test(preflightWorktree.reason || '')) {
|
|
4266
|
+
console.log(`[scheduler] ${job.slug}: deferring — ${preflightWorktree.reason}`);
|
|
4267
|
+
await mutate((s) => {
|
|
4268
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4269
|
+
if (idx >= 0) s.jobs[idx].heldReason = preflightWorktree.reason;
|
|
4270
|
+
});
|
|
4271
|
+
await broadcast({ flush: true });
|
|
4272
|
+
return;
|
|
4273
|
+
}
|
|
4274
|
+
// Launch circuit breaker (lib/launchFailure.cjs, issue #11). Re-evaluated
|
|
4275
|
+
// here, not just in tickQueue, because the block can change between the
|
|
4276
|
+
// pick and this dispatch (another job's probe just failed). 'blocked' →
|
|
4277
|
+
// hold the row; 'probe' → this job is the single half-open probe and is
|
|
4278
|
+
// stamped as such so no sibling probes the same broken persona at once.
|
|
4279
|
+
const launchKey = launchFailure.launchBlockKeyFor(job);
|
|
4280
|
+
const claudeVersionNow = await probeClaudeVersion();
|
|
4281
|
+
let launchEnv = null;
|
|
4282
|
+
let launchProbe = false;
|
|
4283
|
+
const launchGate = await mutate((s) => {
|
|
4284
|
+
s.launchBlocks = s.launchBlocks || {};
|
|
4285
|
+
s.launchMitigations = s.launchMitigations || {};
|
|
4286
|
+
const mitigation = s.launchMitigations[launchKey];
|
|
4287
|
+
if (mitigation && claudeVersionNow && mitigation.claudeVersion && mitigation.claudeVersion !== claudeVersionNow) {
|
|
4288
|
+
console.log(`[scheduler] launch gate: CLI version changed (${mitigation.claudeVersion} → ${claudeVersionNow}) — dropping ${launchKey} mitigation to retry a clean launch`);
|
|
4289
|
+
delete s.launchMitigations[launchKey];
|
|
4290
|
+
}
|
|
4291
|
+
const block = s.launchBlocks[launchKey];
|
|
4292
|
+
const gate = launchFailure.evaluateLaunchGate(block, { now: Date.now(), claudeVersion: claudeVersionNow });
|
|
4293
|
+
if (gate.state === 'open' && block) {
|
|
4294
|
+
console.log(`[scheduler] launch gate: clearing ${launchKey} block (${gate.reason})`);
|
|
4295
|
+
delete s.launchBlocks[launchKey];
|
|
4296
|
+
}
|
|
4297
|
+
if (gate.state === 'blocked') {
|
|
4298
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4299
|
+
if (idx >= 0) s.jobs[idx].heldReason = gate.reason;
|
|
4300
|
+
return gate;
|
|
4301
|
+
}
|
|
4302
|
+
if (gate.state === 'probe') {
|
|
4303
|
+
block.probing = { slug: job.slug, at: new Date().toISOString() };
|
|
4304
|
+
launchProbe = true;
|
|
4305
|
+
launchEnv = block.mitigationEnv || null;
|
|
4306
|
+
} else if (s.launchMitigations[launchKey]?.env) {
|
|
4307
|
+
launchEnv = { ...s.launchMitigations[launchKey].env };
|
|
4308
|
+
}
|
|
4309
|
+
return gate;
|
|
4310
|
+
});
|
|
4311
|
+
if (launchGate.state === 'blocked') {
|
|
4312
|
+
console.log(`[scheduler] ${job.slug}: deferring — ${launchGate.reason}`);
|
|
4313
|
+
await broadcast({ flush: true });
|
|
4314
|
+
return;
|
|
4315
|
+
}
|
|
4316
|
+
if (launchProbe) {
|
|
4317
|
+
console.log(`[scheduler] ${job.slug}: dispatching as launch probe for '${launchKey}'${launchEnv ? ` with mitigation ${JSON.stringify(launchEnv)}` : ''}`);
|
|
4318
|
+
}
|
|
4319
|
+
|
|
3561
4320
|
await mutate((s) => {
|
|
3562
4321
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3563
4322
|
if (idx >= 0) {
|
|
3564
|
-
transitionJob(s.jobs[idx], 'running', {
|
|
4323
|
+
transitionJob(s.jobs[idx], 'running', {
|
|
4324
|
+
reason: resumeTarget ? 'dispatched for resume-recovery' : 'dispatched for execution',
|
|
4325
|
+
source: 'spawnJob:dispatch',
|
|
4326
|
+
});
|
|
4327
|
+
delete s.jobs[idx].heldReason;
|
|
3565
4328
|
s.jobs[idx].runId = runId;
|
|
3566
4329
|
s.jobs[idx].startedAt = new Date().toISOString();
|
|
4330
|
+
if (job.quietMachine === true) {
|
|
4331
|
+
s.jobs[idx].quietMachine = true;
|
|
4332
|
+
s.jobs[idx].quietLeaseDegraded = job.quietLeaseDegraded === true;
|
|
4333
|
+
}
|
|
4334
|
+
// Stamp the bounded one-attempt marker BEFORE the resume spawn, in
|
|
4335
|
+
// the SAME mutate as the 'running' transition, so an app crash
|
|
4336
|
+
// between here and the child actually spawning still leaves this
|
|
4337
|
+
// job un-retriable (selectResumeRecoveryTarget returns null once
|
|
4338
|
+
// this is true) rather than silently re-firing forever.
|
|
4339
|
+
if (resumeTarget) {
|
|
4340
|
+
s.jobs[idx].resumeRecoveryAttempted = true;
|
|
4341
|
+
}
|
|
3567
4342
|
}
|
|
3568
4343
|
});
|
|
3569
4344
|
await broadcast({ flush: true });
|
|
@@ -3573,15 +4348,50 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3573
4348
|
const guardCwd = job.cwd || defaultCwd;
|
|
3574
4349
|
const guardBaseline = await uncommittedChanges(guardCwd);
|
|
3575
4350
|
const guardHeadBefore = await gitHead(guardCwd);
|
|
4351
|
+
// Shared-tree stash guard baseline (incident 2026-09-01): captured
|
|
4352
|
+
// unconditionally, before worktree isolation is even attempted, so an
|
|
4353
|
+
// in-place run always has a true pre-run snapshot to diff against. See
|
|
4354
|
+
// checkSharedTreeGuard below, gated to in-place runs only.
|
|
4355
|
+
const stashBaseline = await stashList(guardCwd);
|
|
4356
|
+
|
|
4357
|
+
// Persist the pre-run baseline onto the row itself (not just the local
|
|
4358
|
+
// variable) so a finalizer that never reaches the rest of THIS function
|
|
4359
|
+
// — namely reapDeadRunningJobs, when the process vanishes mid-run — can
|
|
4360
|
+
// still compute a truthful newly-dirty delta instead of having no
|
|
4361
|
+
// baseline at all. `runtime` (unlike this) is deleted on finalize; this
|
|
4362
|
+
// survives until the finalize mutate below explicitly clears it.
|
|
4363
|
+
//
|
|
4364
|
+
// preRunDirtyPaths is the SAME snapshot, capped and reworked into the
|
|
4365
|
+
// executor-facing manifest (buildForeignWipSection) telling the job which
|
|
4366
|
+
// paths it does not own — unlike guardBaseline/guardHeadBefore, it is
|
|
4367
|
+
// deliberately left on the row through to history.jsonl (not deleted at
|
|
4368
|
+
// finalize) so a post-hoc reader can tell whether a completed job ran
|
|
4369
|
+
// against foreign WIP.
|
|
4370
|
+
const preRunDirtyPaths = capDirtyPaths(guardBaseline);
|
|
4371
|
+
await mutate((s) => {
|
|
4372
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4373
|
+
if (idx >= 0) {
|
|
4374
|
+
s.jobs[idx].guardBaseline = guardBaseline || [];
|
|
4375
|
+
s.jobs[idx].guardHeadBefore = guardHeadBefore || null;
|
|
4376
|
+
if (preRunDirtyPaths.length) s.jobs[idx].preRunDirtyPaths = preRunDirtyPaths;
|
|
4377
|
+
else delete s.jobs[idx].preRunDirtyPaths;
|
|
4378
|
+
}
|
|
4379
|
+
});
|
|
3576
4380
|
|
|
3577
4381
|
// Worktree isolation (PRD 994): give this job its own linked `git worktree`
|
|
3578
4382
|
// checkout so its edits/tests/commit never collide with a sibling job or
|
|
3579
4383
|
// an interactive session in the SAME repo. `worktree.ok` is false (with a
|
|
3580
|
-
// logged reason) for a non-git cwd, a dirty base tree,
|
|
3581
|
-
//
|
|
3582
|
-
// never a hard failure.
|
|
3583
|
-
//
|
|
3584
|
-
|
|
4384
|
+
// logged reason) for a non-git cwd, a dirty base tree, or
|
|
4385
|
+
// SM_JOB_WORKTREE_DISABLE=1 — every case falls back to running in place,
|
|
4386
|
+
// never a hard failure. (The cap-reached reason was already handled above
|
|
4387
|
+
// as a pre-dispatch DEFERRAL — a job never reaches this point with that
|
|
4388
|
+
// reason.) See jobWorktree.cjs's header comment for why job.cwd
|
|
4389
|
+
// (guardCwd) itself is NEVER repointed at the worktree dir. Reuses
|
|
4390
|
+
// `preflightWorktree` computed above the 'running' transition — it
|
|
4391
|
+
// already IS this job's worktree attempt (or already-created checkout),
|
|
4392
|
+
// so calling createJobWorktree a second time here would double-create
|
|
4393
|
+
// (or double-count the cap) for the exact same job.
|
|
4394
|
+
const worktree = preflightWorktree;
|
|
3585
4395
|
if (worktree.ok) {
|
|
3586
4396
|
console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
|
|
3587
4397
|
} else {
|
|
@@ -3597,6 +4407,16 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3597
4407
|
});
|
|
3598
4408
|
}
|
|
3599
4409
|
}
|
|
4410
|
+
// Base-tree WIP carried into the worktree (createWorktree, PRD 1094) —
|
|
4411
|
+
// recorded on the job row so integration can exclude these paths from
|
|
4412
|
+
// the branch diff below, and so it's queryable from the queue.
|
|
4413
|
+
const carriedPaths = (worktree.ok && Array.isArray(worktree.carriedPaths)) ? worktree.carriedPaths : [];
|
|
4414
|
+
if (carriedPaths.length) {
|
|
4415
|
+
await mutate((s) => {
|
|
4416
|
+
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4417
|
+
if (idx >= 0) s.jobs[idx].carriedPaths = carriedPaths;
|
|
4418
|
+
});
|
|
4419
|
+
}
|
|
3600
4420
|
|
|
3601
4421
|
// Integrate the job's branch back into guardCwd's own HEAD, THEN tear the
|
|
3602
4422
|
// worktree checkout down — both must happen BEFORE any git read below
|
|
@@ -3614,7 +4434,17 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3614
4434
|
let res;
|
|
3615
4435
|
let worktreeLeftoverDirty = [];
|
|
3616
4436
|
let worktreeIntegrationFailure = null;
|
|
3617
|
-
|
|
4437
|
+
// A job's uncommitted-work patch, whichever isolation mode produced it —
|
|
4438
|
+
// set by EITHER branch below, never both (worktree.ok picks exactly one
|
|
4439
|
+
// shape for the whole run). Named generically (not "worktree...") because
|
|
4440
|
+
// an in-place run salvages one too (PRD 1098).
|
|
4441
|
+
let salvagePatch = null;
|
|
4442
|
+
// Which foreign-WIP shape applies to THIS run: an isolated worktree only
|
|
4443
|
+
// ever needs to disclose carriedPaths (its checkout starts clean apart
|
|
4444
|
+
// from those carried paths); an in-place/shared-tree run discloses the
|
|
4445
|
+
// raw pre-run dirty snapshot instead. Never both — see
|
|
4446
|
+
// buildForeignWipSection.
|
|
4447
|
+
const foreignWip = worktree.ok ? { carriedPaths } : { preRunDirtyPaths };
|
|
3618
4448
|
try {
|
|
3619
4449
|
res = await executeJob(job, runDir, defaultCwd, async (pid, sessionId, cwd) => {
|
|
3620
4450
|
await mutate((s) => {
|
|
@@ -3625,7 +4455,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3625
4455
|
}
|
|
3626
4456
|
});
|
|
3627
4457
|
await broadcast({ flush: true });
|
|
3628
|
-
}, worktree.ok ? worktree.dir : undefined);
|
|
4458
|
+
}, worktree.ok ? worktree.dir : undefined, resumeTarget, foreignWip, launchEnv);
|
|
3629
4459
|
} finally {
|
|
3630
4460
|
if (worktree.ok) {
|
|
3631
4461
|
worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
|
|
@@ -3638,11 +4468,14 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3638
4468
|
const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
|
|
3639
4469
|
const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
|
|
3640
4470
|
if (salvage && salvage.ok) {
|
|
3641
|
-
|
|
4471
|
+
salvagePatch = salvagePath;
|
|
3642
4472
|
console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
|
|
3643
4473
|
}
|
|
3644
4474
|
}
|
|
3645
|
-
const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug });
|
|
4475
|
+
const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
|
|
4476
|
+
if (integration.ok && integration.reason === 'carried-wip-only') {
|
|
4477
|
+
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
|
|
4478
|
+
}
|
|
3646
4479
|
if (!integration.ok) {
|
|
3647
4480
|
worktreeIntegrationFailure = integration.reason;
|
|
3648
4481
|
console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
|
|
@@ -3655,9 +4488,86 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3655
4488
|
branch: worktree.branch,
|
|
3656
4489
|
keepBranch: !integration.ok,
|
|
3657
4490
|
});
|
|
4491
|
+
} else {
|
|
4492
|
+
// In-place run (non-git cwd, cap reached, env-disabled, or a carry-over
|
|
4493
|
+
// failure) — there is no throwaway checkout to diff, so salvage only
|
|
4494
|
+
// the DELTA this job itself dirtied: paths in guardBaseline are a
|
|
4495
|
+
// human's or a sibling job's pre-existing WIP and must never appear in
|
|
4496
|
+
// this job's patch. Runs for every exit code (finally always fires
|
|
4497
|
+
// once `res` resolves, success or not) including signal deaths and the
|
|
4498
|
+
// rate-limited/halt path — a killed in-place run is exactly the case
|
|
4499
|
+
// this exists to cover. Never mutates guardCwd's index or stashes:
|
|
4500
|
+
// salvageDirtyDelta is read-only (git status + git diff only).
|
|
4501
|
+
try {
|
|
4502
|
+
const after = await uncommittedChanges(guardCwd);
|
|
4503
|
+
if (after) {
|
|
4504
|
+
const baseSet = new Set(guardBaseline || []);
|
|
4505
|
+
const deltaPaths = after.filter((p) => !baseSet.has(p));
|
|
4506
|
+
if (deltaPaths.length) {
|
|
4507
|
+
const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
|
|
4508
|
+
const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: guardCwd, paths: deltaPaths, outFile: salvagePath });
|
|
4509
|
+
if (salvage && salvage.ok) {
|
|
4510
|
+
salvagePatch = salvagePath;
|
|
4511
|
+
console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted in-place diff (${deltaPaths.length} path(s)) to ${salvagePath}`);
|
|
4512
|
+
}
|
|
4513
|
+
}
|
|
4514
|
+
}
|
|
4515
|
+
} catch (e) {
|
|
4516
|
+
console.error(`[scheduler] ${job.slug}: in-place salvage failed`, e);
|
|
4517
|
+
}
|
|
3658
4518
|
}
|
|
3659
4519
|
}
|
|
3660
4520
|
|
|
4521
|
+
// Newly-dirty leftover computation — hoisted OUT of the exit===0 branch
|
|
4522
|
+
// (below) so it runs for every terminal outcome: exit 0, any non-zero
|
|
4523
|
+
// exit including 137/143, and the rate-limited/halt path alike. This is
|
|
4524
|
+
// the exact same shape the exit=0 commit-guard and the transient-failure
|
|
4525
|
+
// classifier each used to compute independently (guardCwd's own
|
|
4526
|
+
// baseline-delta UNION worktreeLeftoverDirty, which is already
|
|
4527
|
+
// inherently-new since it came from a fresh worktree checkout with no
|
|
4528
|
+
// baseline to diff against) — computed once here and reused by both
|
|
4529
|
+
// below, plus by the terminal-finalize mutate for leftoverPaths/
|
|
4530
|
+
// leftoverCount. null only when git-status itself is unavailable
|
|
4531
|
+
// (non-git cwd / git errored) — NEVER treated as "left nothing", exactly
|
|
4532
|
+
// like every other best-effort git-state check in this function.
|
|
4533
|
+
const afterGuardCwd = await uncommittedChanges(guardCwd);
|
|
4534
|
+
const newlyDirtyAll = afterGuardCwd === null
|
|
4535
|
+
? null
|
|
4536
|
+
: [...new Set([
|
|
4537
|
+
...afterGuardCwd.filter((p) => !new Set(guardBaseline || []).has(p)),
|
|
4538
|
+
...worktreeLeftoverDirty,
|
|
4539
|
+
])];
|
|
4540
|
+
|
|
4541
|
+
if (res.launchFailure) {
|
|
4542
|
+
await handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion: claudeVersionNow });
|
|
4543
|
+
return;
|
|
4544
|
+
}
|
|
4545
|
+
if (launchFailure.resultShowsRealTurn(res.resultStats)) {
|
|
4546
|
+
// The launch worked (whatever happens to the run next) — close the
|
|
4547
|
+
// breaker for this persona. If the probe only got through thanks to a
|
|
4548
|
+
// mitigation env, keep applying that env to every later launch of the
|
|
4549
|
+
// persona until the CLI version changes; otherwise the very next job
|
|
4550
|
+
// would fail the same way and re-arm the block (a flap per job).
|
|
4551
|
+
await mutate((s) => {
|
|
4552
|
+
const block = s.launchBlocks?.[launchKey];
|
|
4553
|
+
if (!block) return;
|
|
4554
|
+
delete s.launchBlocks[launchKey];
|
|
4555
|
+
if (launchEnv && Object.keys(launchEnv).length) {
|
|
4556
|
+
s.launchMitigations = s.launchMitigations || {};
|
|
4557
|
+
s.launchMitigations[launchKey] = {
|
|
4558
|
+
kind: block.kind,
|
|
4559
|
+
env: { ...launchEnv },
|
|
4560
|
+
since: new Date().toISOString(),
|
|
4561
|
+
claudeVersion: claudeVersionNow ?? block.claudeVersion ?? null,
|
|
4562
|
+
hint: launchFailure.launchFailureHint(block.kind, { claudeVersion: claudeVersionNow ?? block.claudeVersion }),
|
|
4563
|
+
};
|
|
4564
|
+
console.log(`[scheduler] launch gate: ${launchKey} recovered via mitigation ${JSON.stringify(launchEnv)} — kept in force until the CLI version changes`);
|
|
4565
|
+
} else {
|
|
4566
|
+
console.log(`[scheduler] launch gate: ${launchKey} recovered — block cleared after ${block.attempts} failed probe(s)`);
|
|
4567
|
+
}
|
|
4568
|
+
});
|
|
4569
|
+
}
|
|
4570
|
+
|
|
3661
4571
|
if (res.rateLimited) {
|
|
3662
4572
|
const resetIso = await refreshNextReset().catch(() => cachedNextReset);
|
|
3663
4573
|
await setPaused('rate_limit', resetIso);
|
|
@@ -3699,6 +4609,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3699
4609
|
// pass it back into verifyRun as priorLandedCommit (see the
|
|
3700
4610
|
// pass_no_commit_prior_run_verified exemption in runVerify.cjs).
|
|
3701
4611
|
let jobLandedCommitThisRun = null;
|
|
4612
|
+
let sharedTreeGuard = null;
|
|
3702
4613
|
if (res.exitCode === 0 && !res.rateLimited) {
|
|
3703
4614
|
// Detect whether the job self-committed by comparing HEAD before/after.
|
|
3704
4615
|
// Used by the sentinel override: SCHEDULER_VERDICT: PASS + a landed
|
|
@@ -3771,20 +4682,12 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3771
4682
|
const guardWillRefire = verifyResult && verifyResult.downgradeTo === 'pending';
|
|
3772
4683
|
const guardIsLegitimateNoOp = verifyResult && COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict);
|
|
3773
4684
|
if (res.exitCode === 0 && !res.rateLimited && !guardWillRefire && !guardIsLegitimateNoOp) {
|
|
3774
|
-
|
|
3775
|
-
//
|
|
3776
|
-
//
|
|
3777
|
-
//
|
|
3778
|
-
if (
|
|
3779
|
-
const
|
|
3780
|
-
// worktreeLeftoverDirty was captured from a FRESH checkout (no baseline
|
|
3781
|
-
// to diff against — every path in it is inherently new) right before
|
|
3782
|
-
// the worktree was torn down, so it must be counted here or a job's
|
|
3783
|
-
// uncommitted leftovers silently vanish with the worktree.
|
|
3784
|
-
const newlyDirty = [...new Set([
|
|
3785
|
-
...after.filter((p) => !baseSet.has(p)),
|
|
3786
|
-
...worktreeLeftoverDirty,
|
|
3787
|
-
])];
|
|
4685
|
+
// afterGuardCwd === null means non-git cwd (or git errored) —
|
|
4686
|
+
// best-effort skip, same as always; only a git-status result (even an
|
|
4687
|
+
// empty one) counts as evidence for the zero-edit path. newlyDirtyAll
|
|
4688
|
+
// was computed once, above, right after the try/finally.
|
|
4689
|
+
if (afterGuardCwd !== null) {
|
|
4690
|
+
const newlyDirty = newlyDirtyAll;
|
|
3788
4691
|
const guardState = await readQueue().catch(() => ({ jobs: [] }));
|
|
3789
4692
|
const siblingRunning = (guardState.jobs || []).some(
|
|
3790
4693
|
(j) => j.slug !== job.slug && j.status === 'running' && (j.cwd || defaultCwd) === guardCwd,
|
|
@@ -3799,7 +4702,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3799
4702
|
legitimateNoOp: guardIsLegitimateNoOp,
|
|
3800
4703
|
isFixPlanJob: isFixPlanSlug(job.slug),
|
|
3801
4704
|
verifyResult,
|
|
3802
|
-
salvagePatch
|
|
4705
|
+
salvagePatch,
|
|
3803
4706
|
});
|
|
3804
4707
|
if (guardVerdict) {
|
|
3805
4708
|
verifyResult = guardVerdict;
|
|
@@ -3823,6 +4726,36 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3823
4726
|
};
|
|
3824
4727
|
}
|
|
3825
4728
|
|
|
4729
|
+
// Shared-tree stash guard (incident 2026-09-01): only meaningful for an
|
|
4730
|
+
// IN-PLACE run — worktree.ok isolates the job's git state into its own
|
|
4731
|
+
// checkout, so nothing there can leak into guardCwd. Best-effort and run
|
|
4732
|
+
// regardless of exit code: a job can discard shared state on its way to
|
|
4733
|
+
// a non-zero exit just as easily as on a clean one.
|
|
4734
|
+
if (!worktree.ok) {
|
|
4735
|
+
sharedTreeGuard = await module.exports.checkSharedTreeGuard({
|
|
4736
|
+
cwd: guardCwd,
|
|
4737
|
+
stashBaseline,
|
|
4738
|
+
dirtyBaseline: guardBaseline,
|
|
4739
|
+
headBefore: guardHeadBefore,
|
|
4740
|
+
slug: job.slug,
|
|
4741
|
+
});
|
|
4742
|
+
// A restored stash alone isn't silence — it's logged loudly above and
|
|
4743
|
+
// surfaced on the job row below — but a path that's still missing
|
|
4744
|
+
// (restore failed, or two-plus stashes we refused to guess between, or
|
|
4745
|
+
// a revert with no stash to restore at all) must not finish green.
|
|
4746
|
+
if (sharedTreeGuard && (sharedTreeGuard.restoreFailed || sharedTreeGuard.ambiguousStashes || sharedTreeGuard.reverted)) {
|
|
4747
|
+
verifyResult = {
|
|
4748
|
+
verdict: 'shared_tree_reverted',
|
|
4749
|
+
reason: sharedTreeGuard.reverted
|
|
4750
|
+
? `job discarded pre-existing state in the shared tree: ${sharedTreeGuard.reverted.length} path(s) reverted with no commit to explain it (${sharedTreeGuard.reverted.slice(0, 3).join(', ')})`
|
|
4751
|
+
: sharedTreeGuard.restoreFailed
|
|
4752
|
+
? `job stashed the shared tree and the stash could not be auto-restored: ${sharedTreeGuard.restoreFailed}`
|
|
4753
|
+
: `job created ${sharedTreeGuard.ambiguousStashes.length} stashes in the shared tree — ambiguous, not auto-restored (${sharedTreeGuard.ambiguousStashes.join(', ')})`,
|
|
4754
|
+
downgradeTo: 'needs_review',
|
|
4755
|
+
};
|
|
4756
|
+
}
|
|
4757
|
+
}
|
|
4758
|
+
|
|
3826
4759
|
// SIGTERM commit check: reuse the same commit-window scan the exit=0
|
|
3827
4760
|
// guard uses above (one commit-detection path, not two) to see whether a
|
|
3828
4761
|
// 143 (SIGTERM) run still landed a deliverable before it died. Scoped
|
|
@@ -3845,6 +4778,8 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3845
4778
|
let needsInvestigationNow = false;
|
|
3846
4779
|
let investigationJobSnapshot = null;
|
|
3847
4780
|
let needsReviewRcaSnapshot = null;
|
|
4781
|
+
let resumeRecoveryJob = null;
|
|
4782
|
+
let resumeRecoveryTarget = null;
|
|
3848
4783
|
let terminalNotifySnapshot = null;
|
|
3849
4784
|
const newlyCompletedPrds = [];
|
|
3850
4785
|
await mutate((s) => {
|
|
@@ -3891,10 +4826,11 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3891
4826
|
transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
|
|
3892
4827
|
s.jobs[i2].finishedAt = new Date().toISOString();
|
|
3893
4828
|
s.jobs[i2].exitCode = res.exitCode;
|
|
3894
|
-
|
|
3895
|
-
|
|
4829
|
+
s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
|
|
4830
|
+
if (salvagePatch) {
|
|
4831
|
+
s.jobs[i2].salvagePatch = salvagePatch;
|
|
3896
4832
|
} else {
|
|
3897
|
-
delete s.jobs[i2].
|
|
4833
|
+
delete s.jobs[i2].salvagePatch;
|
|
3898
4834
|
}
|
|
3899
4835
|
s.jobs[i2].error = effectiveStatus === 'needs_review'
|
|
3900
4836
|
? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
|
|
@@ -3918,6 +4854,26 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3918
4854
|
} else {
|
|
3919
4855
|
delete s.jobs[i2].verifierVerdict;
|
|
3920
4856
|
}
|
|
4857
|
+
// Closed-set outcome taxonomy (issue #11 list A2) so a queue row
|
|
4858
|
+
// says WHY it ended without anyone opening the transcript.
|
|
4859
|
+
s.jobs[i2].terminalReason = launchFailure.deriveTerminalReason({
|
|
4860
|
+
effectiveStatus, exitCode: res.exitCode, verifyResult, sigtermOverride, worktreeIntegrationFailure,
|
|
4861
|
+
});
|
|
4862
|
+
delete s.jobs[i2].launchFailure;
|
|
4863
|
+
delete s.jobs[i2].heldReason;
|
|
4864
|
+
// Persist the commit-guard's exact dirty-path list (verdict
|
|
4865
|
+
// 'uncommitted_changes' only) so a later resume-recovery attempt
|
|
4866
|
+
// (selectResumeRecoveryTarget) can name these paths without
|
|
4867
|
+
// re-running `git status` against a tree that may have moved on.
|
|
4868
|
+
if (verifyResult?.verdict === 'uncommitted_changes' && Array.isArray(verifyResult.dirtyPaths)) {
|
|
4869
|
+
// Capped the same way preRunDirtyPaths/leftoverPaths are — an
|
|
4870
|
+
// uncapped list here would let a pathologically dirty tree bloat
|
|
4871
|
+
// queue.json/history.jsonl and the resume-recovery prompt built
|
|
4872
|
+
// from it (buildResumeRecoveryPreamble/selectResumeRecoveryTarget).
|
|
4873
|
+
s.jobs[i2].uncommittedPaths = capDirtyPaths(verifyResult.dirtyPaths);
|
|
4874
|
+
} else {
|
|
4875
|
+
delete s.jobs[i2].uncommittedPaths;
|
|
4876
|
+
}
|
|
3921
4877
|
// Non-blocking notes (e.g. a recovered missing-dependency probe, or a
|
|
3922
4878
|
// pattern hit demoted because a materially-checkable verdict outranked
|
|
3923
4879
|
// it) — surfaced even on completed jobs so the signal isn't lost.
|
|
@@ -3928,7 +4884,29 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3928
4884
|
} else {
|
|
3929
4885
|
delete s.jobs[i2].verifierAnnotations;
|
|
3930
4886
|
}
|
|
4887
|
+
// Shared-tree guard outcome (restored stash / unresolved revert /
|
|
4888
|
+
// ambiguous stashes) — visible on the row even when a restored
|
|
4889
|
+
// stash left the run otherwise green, so it's never silent.
|
|
4890
|
+
if (sharedTreeGuard) {
|
|
4891
|
+
s.jobs[i2].sharedTreeGuard = sharedTreeGuard;
|
|
4892
|
+
} else {
|
|
4893
|
+
delete s.jobs[i2].sharedTreeGuard;
|
|
4894
|
+
}
|
|
3931
4895
|
delete s.jobs[i2].runtime;
|
|
4896
|
+
// Pre-run baseline no longer needed once this run has finalized —
|
|
4897
|
+
// its whole purpose (letting THIS finalize compute a truthful
|
|
4898
|
+
// delta) is done; a fresh one is captured at the next dispatch.
|
|
4899
|
+
delete s.jobs[i2].guardBaseline;
|
|
4900
|
+
delete s.jobs[i2].guardHeadBefore;
|
|
4901
|
+
// Leftover-attribution fields (PRD: capture+surface uncommitted
|
|
4902
|
+
// work on every terminal path, not just exit=0) — set for EVERY
|
|
4903
|
+
// terminal outcome above (completed/failed/needs_review alike),
|
|
4904
|
+
// not just the exit=0 commit-guard branch, so a bare `failed` row
|
|
4905
|
+
// is visually distinguishable from one that quietly left work
|
|
4906
|
+
// behind. newlyDirtyAll is null when git-status was unavailable
|
|
4907
|
+
// (non-git cwd) — applyLeftoverFields treats null like "nothing to
|
|
4908
|
+
// attribute" via its Array.isArray guard, same as an empty array.
|
|
4909
|
+
applyLeftoverFields(s.jobs[i2], newlyDirtyAll);
|
|
3932
4910
|
|
|
3933
4911
|
if (isNotifiableTerminalStatus(effectiveStatus)) {
|
|
3934
4912
|
terminalNotifySnapshot = { ...s.jobs[i2] };
|
|
@@ -3946,6 +4924,19 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3946
4924
|
// takes the treatAsPending branch above and never reaches here).
|
|
3947
4925
|
needsReviewRcaSnapshot = { ...s.jobs[i2] };
|
|
3948
4926
|
|
|
4927
|
+
// Resume-first recovery (PRD 1111): evaluated BEFORE the auto-fix
|
|
4928
|
+
// eligibility check below — a job whose verdict is
|
|
4929
|
+
// 'uncommitted_changes' with a live sessionId gets one bounded
|
|
4930
|
+
// `--resume` dispatch instead of a cold-read fix-plan
|
|
4931
|
+
// investigation. Snapshot only (no I/O inside mutate()); the
|
|
4932
|
+
// actual dispatch happens outside mutate(), below. Never sets
|
|
4933
|
+
// needsInvestigationNow — the two are mutually exclusive for the
|
|
4934
|
+
// same tick, mirroring the `else if` used outside mutate().
|
|
4935
|
+
const target = selectResumeRecoveryTarget(s.jobs[i2]);
|
|
4936
|
+
if (target) {
|
|
4937
|
+
resumeRecoveryJob = { ...s.jobs[i2] };
|
|
4938
|
+
resumeRecoveryTarget = target;
|
|
4939
|
+
} else {
|
|
3949
4940
|
// Same-tick auto-fix (feedback 2026-07-12): rather than waiting up to
|
|
3950
4941
|
// 10 min for reverifyNeedsReview()'s periodic pass, check right here
|
|
3951
4942
|
// whether this job qualifies for auto-fix (same eligibility rule
|
|
@@ -3969,6 +4960,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3969
4960
|
needsInvestigationNow = true;
|
|
3970
4961
|
investigationJobSnapshot = { ...s.jobs[i2] };
|
|
3971
4962
|
}
|
|
4963
|
+
}
|
|
3972
4964
|
}
|
|
3973
4965
|
// Auto-promote: when a fix-* PRD completes successfully, the original
|
|
3974
4966
|
// failed PRD's work is logically done. Flip its status to 'completed'
|
|
@@ -3984,6 +4976,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3984
4976
|
orig.exitCode = 0;
|
|
3985
4977
|
orig.error = null;
|
|
3986
4978
|
orig.completedBy = job.slug;
|
|
4979
|
+
delete orig.looksDone;
|
|
3987
4980
|
if (priorStatus === 'needs_review') {
|
|
3988
4981
|
delete orig.verifierVerdict;
|
|
3989
4982
|
}
|
|
@@ -3998,6 +4991,24 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3998
4991
|
}
|
|
3999
4992
|
await broadcast({ flush: true });
|
|
4000
4993
|
|
|
4994
|
+
// Per-run outcome sidecar (issue #11 list B5): turns/tokens/verdict in one
|
|
4995
|
+
// small JSON next to the log so fleet health never needs a transcript parse.
|
|
4996
|
+
launchFailure.writeOutcomeSidecar(runDir, job.slug, {
|
|
4997
|
+
runId,
|
|
4998
|
+
exitCode: res.exitCode,
|
|
4999
|
+
durationMs: res.durationMs ?? null,
|
|
5000
|
+
numTurns: res.resultStats?.numTurns ?? null,
|
|
5001
|
+
outputTokens: res.resultStats?.outputTokens ?? null,
|
|
5002
|
+
totalCostUsd: res.resultStats?.totalCostUsd ?? null,
|
|
5003
|
+
verdict: verifyResult?.verdict ?? (res.exitCode === 0 ? 'clean' : null),
|
|
5004
|
+
status: terminalNotifySnapshot?.status ?? failedJobSnapshot?.status ?? null,
|
|
5005
|
+
terminalReason: terminalNotifySnapshot?.terminalReason ?? failedJobSnapshot?.terminalReason ?? null,
|
|
5006
|
+
launchFailure: null,
|
|
5007
|
+
launchEnvApplied: launchEnv ? Object.keys(launchEnv) : [],
|
|
5008
|
+
filesChanged: Array.isArray(newlyDirtyAll) ? newlyDirtyAll.length : null,
|
|
5009
|
+
landedCommit: jobLandedCommitThisRun ?? null,
|
|
5010
|
+
});
|
|
5011
|
+
|
|
4001
5012
|
if (terminalNotifySnapshot) {
|
|
4002
5013
|
notifyOriginatingTab(terminalNotifySnapshot).catch((e) => {
|
|
4003
5014
|
console.error('[scheduler] notifyOriginatingTab error', job.slug, e);
|
|
@@ -4013,12 +5024,28 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
4013
5024
|
verdict: needsReviewRcaSnapshot.verifierVerdict,
|
|
4014
5025
|
annotations: needsReviewRcaSnapshot.verifierAnnotations,
|
|
4015
5026
|
})
|
|
4016
|
-
.then((report) =>
|
|
5027
|
+
.then(async (report) => {
|
|
5028
|
+
// Persist the classification onto the parked job row so the scheduler
|
|
5029
|
+
// can route on it (e.g. selectAutoFixTargets excluding 'archive')
|
|
5030
|
+
// without re-parsing the RCA markdown on every pass.
|
|
5031
|
+
await mutate((s) => {
|
|
5032
|
+
const j = s.jobs.find((x) => x.slug === needsReviewRcaSnapshot.slug);
|
|
5033
|
+
applyRcaClassification(j, report);
|
|
5034
|
+
}).catch(() => {});
|
|
5035
|
+
return notifyNeedsReview(needsReviewRcaSnapshot, report);
|
|
5036
|
+
})
|
|
4017
5037
|
.catch((e) => {
|
|
4018
5038
|
console.error('[scheduler] writeRcaReport error', job.slug, e);
|
|
4019
5039
|
});
|
|
4020
5040
|
}
|
|
4021
5041
|
|
|
5042
|
+
if (resumeRecoveryJob && resumeRecoveryTarget) {
|
|
5043
|
+
console.log(`[scheduler] needs_review ${job.slug} → resume-recovery (session ${resumeRecoveryTarget.sessionId}, ${resumeRecoveryTarget.dirtyPaths.length} dirty path(s))`);
|
|
5044
|
+
spawnResumeRecovery(resumeRecoveryJob, resumeRecoveryTarget).catch((e) => {
|
|
5045
|
+
console.error('[scheduler] spawnResumeRecovery error', job.slug, e);
|
|
5046
|
+
});
|
|
5047
|
+
}
|
|
5048
|
+
|
|
4022
5049
|
if (actuallyFailed && failedJobSnapshot) {
|
|
4023
5050
|
// Transient-failure detector. A 143/137 exit is ALWAYS a signal kill — the
|
|
4024
5051
|
// agent never self-exits with those — so the only question is WHO killed it.
|
|
@@ -4036,25 +5063,11 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
4036
5063
|
// the threshold and still fall through to investigation.
|
|
4037
5064
|
const ec = failedJobSnapshot.exitCode;
|
|
4038
5065
|
const retries = failedJobSnapshot.transientRetries ?? 0;
|
|
4039
|
-
// Only pay for the extra git status call when the failure is plausibly
|
|
4040
|
-
// transient — a real code failure never needs the dirty-tree check.
|
|
4041
5066
|
const maybeTransient = (ec === 143 || ec === 137) || res.networkError === true;
|
|
4042
|
-
|
|
4043
|
-
|
|
4044
|
-
|
|
4045
|
-
|
|
4046
|
-
const baseSet = new Set(guardBaseline || []);
|
|
4047
|
-
// See the commit-guard block above: worktreeLeftoverDirty was captured
|
|
4048
|
-
// (and the checkout already torn down) before this point, so it must
|
|
4049
|
-
// be folded in here too or a transiently-killed job's leftover WIP
|
|
4050
|
-
// silently disappears with its worktree.
|
|
4051
|
-
const newlyDirty = [...new Set([
|
|
4052
|
-
...(afterFailure || []).filter((p) => !baseSet.has(p)),
|
|
4053
|
-
...worktreeLeftoverDirty,
|
|
4054
|
-
])];
|
|
4055
|
-
newlyDirtyCount = newlyDirty.length;
|
|
4056
|
-
dirtySample = newlyDirty.slice(0, 3).join(', ');
|
|
4057
|
-
}
|
|
5067
|
+
// newlyDirtyAll was computed once, above, right after the try/finally —
|
|
5068
|
+
// reused here rather than re-querying git status a third time.
|
|
5069
|
+
const newlyDirtyCount = maybeTransient ? (newlyDirtyAll || []).length : 0;
|
|
5070
|
+
const dirtySample = maybeTransient ? (newlyDirtyAll || []).slice(0, 3).join(', ') : '';
|
|
4058
5071
|
const decision = classifyFailureOutcome({
|
|
4059
5072
|
exitCode: ec,
|
|
4060
5073
|
networkError: res.networkError,
|
|
@@ -4074,7 +5087,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
4074
5087
|
});
|
|
4075
5088
|
await broadcast({ flush: true });
|
|
4076
5089
|
} else if (decision.action === 'fail-dirty') {
|
|
4077
|
-
const salvageNote =
|
|
5090
|
+
const salvageNote = salvagePatch ? ` — recoverable from salvage patch ${salvagePatch}` : '';
|
|
4078
5091
|
console.log(`[scheduler] transient failure (${decision.transientKind}) for ${job.slug} left ${newlyDirtyCount} uncommitted file(s) (e.g. ${dirtySample})${salvageNote} — not auto-requeuing`);
|
|
4079
5092
|
await mutate((s) => {
|
|
4080
5093
|
const i = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
@@ -4115,11 +5128,32 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
4115
5128
|
runningSet.delete(job.slug);
|
|
4116
5129
|
// Slot release notifies subscribed pumps (chat lane) machine-wide.
|
|
4117
5130
|
sessionSlots.release(slotToken);
|
|
5131
|
+
// Release the exclusive quiet-machine lease on EVERY exit path this
|
|
5132
|
+
// finally covers (normal exit, timeout, SIGTERM, crash) — see the
|
|
5133
|
+
// acquire-site comment above. Bounded: a lease this function never
|
|
5134
|
+
// acquired is simply a no-op release.
|
|
5135
|
+
if (quietLeaseAcquired) quietMachineLease.release(job.slug);
|
|
4118
5136
|
// Each job completion is a signal to advance the queue.
|
|
4119
5137
|
tickQueue().catch(() => {});
|
|
4120
5138
|
}
|
|
4121
5139
|
}
|
|
4122
5140
|
|
|
5141
|
+
/**
|
|
5142
|
+
* Dispatch a resume-recovery attempt (PRD 1111) for a job already found
|
|
5143
|
+
* eligible by selectResumeRecoveryTarget. Thin wrapper around spawnJob —
|
|
5144
|
+
* reuses its entire slot-acquire/worktree/verify/commit-guard/finalize
|
|
5145
|
+
* machinery unchanged, so a resume run that itself parks or fails falls
|
|
5146
|
+
* through to the SAME spawnInvestigation fallback any other run would, with
|
|
5147
|
+
* zero special-casing. `job` and `resumeTarget` must be snapshots taken
|
|
5148
|
+
* BEFORE this call (this function does no eligibility re-check — spawnJob's
|
|
5149
|
+
* own dispatch mutate is what stamps resumeRecoveryAttempted, atomically
|
|
5150
|
+
* with the 'running' transition).
|
|
5151
|
+
*/
|
|
5152
|
+
async function spawnResumeRecovery(job, resumeTarget) {
|
|
5153
|
+
const { runId, dir: runDir } = pickRunDir();
|
|
5154
|
+
await spawnJob(job, runId, runDir, job.cwd || DEFAULT_PROJECT_CWD, resumeTarget);
|
|
5155
|
+
}
|
|
5156
|
+
|
|
4123
5157
|
// Serialized ticker: prevents two concurrent tickQueue() calls from racing
|
|
4124
5158
|
// on the same pending jobs. A simple promise tail suffices since pickNextBatch
|
|
4125
5159
|
// is synchronous and spawnJob is fire-and-forget.
|
|
@@ -4153,7 +5187,13 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
4153
5187
|
// cap that sessionSlots.cjs was written to replace — which silently
|
|
4154
5188
|
// ceilinged the queue at 3 while the pool the user configured said 5.
|
|
4155
5189
|
const freeSlots = sessionSlots.available();
|
|
4156
|
-
const
|
|
5190
|
+
const heldSlugs = await computeLaunchHolds(state);
|
|
5191
|
+
const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
|
|
5192
|
+
leaseHeld: quietMachineLease.isHeld(),
|
|
5193
|
+
machineInUse: sessionSlots.inUse(),
|
|
5194
|
+
now: Date.now(),
|
|
5195
|
+
heldSlugs,
|
|
5196
|
+
});
|
|
4157
5197
|
if (batch.length === 0 && freeSlots === 0) {
|
|
4158
5198
|
const snap = sessionSlots.snapshot();
|
|
4159
5199
|
const pendingCount = state.jobs.filter((j) => j.status === 'pending').length;
|
|
@@ -4328,6 +5368,30 @@ async function maybeLaunchWhenAvailable(state) {
|
|
|
4328
5368
|
|
|
4329
5369
|
// ---------- dead-process reaper ----------
|
|
4330
5370
|
|
|
5371
|
+
// Queue-health sweep cadence: hangs off reapDeadRunningJobs's own cycle
|
|
5372
|
+
// counter (it already runs once per poll tick) rather than a second timer,
|
|
5373
|
+
// so its cadence can never drift from the poll cadence or double-fire
|
|
5374
|
+
// across a backoff reset.
|
|
5375
|
+
let queueHealthSweepCycle = 0;
|
|
5376
|
+
const QUEUE_HEALTH_SWEEP_EVERY_N_CYCLES = 20;
|
|
5377
|
+
|
|
5378
|
+
/**
|
|
5379
|
+
* runQueueHealthSweep(jobs) — read-only reporting pass over the queue
|
|
5380
|
+
* snapshot reapDeadRunningJobs already read this cycle. Never transitions a
|
|
5381
|
+
* job, never archives a PRD, never spawns anything; only logs and appends
|
|
5382
|
+
* an audit event for any project with drift worth a human glance.
|
|
5383
|
+
*/
|
|
5384
|
+
function runQueueHealthSweep(jobs) {
|
|
5385
|
+
try {
|
|
5386
|
+
for (const { cwd, neverRan, looksDone, stuck } of computeQueueHealth(jobs)) {
|
|
5387
|
+
console.log(`[scheduler] queue-health ${cwd}: ${neverRan} never_ran, ${looksDone} looks-done, ${stuck} stuck`);
|
|
5388
|
+
appendAuditEvent('scheduler_queue_health', { cwd, neverRan, looksDone, stuck });
|
|
5389
|
+
}
|
|
5390
|
+
} catch (e) {
|
|
5391
|
+
console.warn('[scheduler] queue-health sweep error', e?.message);
|
|
5392
|
+
}
|
|
5393
|
+
}
|
|
5394
|
+
|
|
4331
5395
|
/**
|
|
4332
5396
|
* Scan running jobs, identify those whose claude process is provably dead OR
|
|
4333
5397
|
* whose spawn never got far enough to record a runtime.pid in the first
|
|
@@ -4362,22 +5426,79 @@ async function reapDeadRunningJobs() {
|
|
|
4362
5426
|
// Absent/empty run dir → classifyRunOutcome finds no result event →
|
|
4363
5427
|
// 'no_result' → non-success below → filed as failed, never completed.
|
|
4364
5428
|
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
4365
|
-
|
|
5429
|
+
// A pidless reap means the spawn never got far enough to record a
|
|
5430
|
+
// pid — the gate could not possibly have run, regardless of what
|
|
5431
|
+
// classifyRunOutcome makes of an absent/empty log.
|
|
5432
|
+
const gateOutcome = pidless ? 'never_ran' : mapOutcomeToGateOutcome(outcome);
|
|
5433
|
+
dead.push({ slug, pid, outcome, gateOutcome, pidless, reason });
|
|
5434
|
+
}
|
|
5435
|
+
|
|
5436
|
+
queueHealthSweepCycle += 1;
|
|
5437
|
+
if (queueHealthSweepCycle % QUEUE_HEALTH_SWEEP_EVERY_N_CYCLES === 0) {
|
|
5438
|
+
runQueueHealthSweep(state.jobs);
|
|
4366
5439
|
}
|
|
5440
|
+
|
|
4367
5441
|
if (dead.length === 0) return;
|
|
4368
5442
|
|
|
4369
|
-
await mutate((s) => {
|
|
4370
|
-
for (const { slug, pid, outcome, pidless, reason } of dead) {
|
|
5443
|
+
await mutate(async (s) => {
|
|
5444
|
+
for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
|
|
4371
5445
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
4372
5446
|
if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
|
|
4373
5447
|
const success = outcome === 'success';
|
|
4374
|
-
|
|
5448
|
+
|
|
5449
|
+
// Best-effort in-place leftover computation: a job whose owning
|
|
5450
|
+
// process vanished without spawnJob()'s own finally block ever
|
|
5451
|
+
// running (the exact case this reaper exists for) never got that
|
|
5452
|
+
// block's salvage OR leftover-attribution pass either. Only
|
|
5453
|
+
// attempted when the row carries a persisted pre-run baseline
|
|
5454
|
+
// (guardBaseline, persisted by spawnJob at dispatch — see there).
|
|
5455
|
+
// With no baseline there is no safe way to tell this job's own dirt
|
|
5456
|
+
// from a human's or a sibling's pre-existing WIP, so this skips
|
|
5457
|
+
// rather than ever dumping/attributing the whole tree.
|
|
5458
|
+
let deltaPaths = null;
|
|
5459
|
+
if (Array.isArray(s.jobs[idx].guardBaseline) && s.jobs[idx].runId) {
|
|
5460
|
+
try {
|
|
5461
|
+
const rowCwd = s.jobs[idx].cwd || s.config?.defaultCwd || DEFAULT_PROJECT_CWD;
|
|
5462
|
+
const after = await uncommittedChanges(rowCwd);
|
|
5463
|
+
if (after) {
|
|
5464
|
+
const baseSet = new Set(s.jobs[idx].guardBaseline);
|
|
5465
|
+
deltaPaths = after.filter((p) => !baseSet.has(p));
|
|
5466
|
+
if (deltaPaths.length) {
|
|
5467
|
+
const salvagePath = path.join(RUNS_DIR, s.jobs[idx].runId, `${slug}.uncommitted.patch`);
|
|
5468
|
+
const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: rowCwd, paths: deltaPaths, outFile: salvagePath });
|
|
5469
|
+
if (salvage && salvage.ok) {
|
|
5470
|
+
s.jobs[idx].salvagePatch = salvagePath;
|
|
5471
|
+
console.log(`[scheduler] reapDeadRunningJobs: salvaged ${salvage.bytes} byte(s) of uncommitted in-place diff for ${slug} to ${salvagePath}`);
|
|
5472
|
+
}
|
|
5473
|
+
}
|
|
5474
|
+
}
|
|
5475
|
+
} catch (e) {
|
|
5476
|
+
console.error(`[scheduler] reapDeadRunningJobs: in-place salvage failed for ${slug}`, e);
|
|
5477
|
+
}
|
|
5478
|
+
}
|
|
5479
|
+
const leftoverSuffix = deltaPaths && deltaPaths.length
|
|
5480
|
+
? ` — left ${deltaPaths.length} files uncommitted`
|
|
5481
|
+
: '';
|
|
5482
|
+
const transitionReason = (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
|
|
5483
|
+
|
|
4375
5484
|
transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
|
|
4376
5485
|
s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
|
|
4377
5486
|
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
4378
5487
|
s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
|
|
5488
|
+
s.jobs[idx].gateOutcome = gateOutcome;
|
|
4379
5489
|
delete s.jobs[idx].runtime;
|
|
5490
|
+
delete s.jobs[idx].guardBaseline;
|
|
5491
|
+
delete s.jobs[idx].guardHeadBefore;
|
|
5492
|
+
applyLeftoverFields(s.jobs[idx], deltaPaths);
|
|
4380
5493
|
runningSet.delete(slug);
|
|
5494
|
+
// A dead job reaped here never reached spawnJob's own finally block
|
|
5495
|
+
// (that's this reaper's whole reason to exist — see its header
|
|
5496
|
+
// comment) — so if it held the quiet-machine lease, spawnJob never
|
|
5497
|
+
// got the chance to release it. Release it here too, or a
|
|
5498
|
+
// quietMachine job whose process silently vanished (OOM, a crash
|
|
5499
|
+
// with no exit event) wedges the lease held forever and stalls
|
|
5500
|
+
// dispatch for every project until the app restarts.
|
|
5501
|
+
if (s.jobs[idx].quietMachine === true) quietMachineLease.release(slug);
|
|
4381
5502
|
if (pidless) {
|
|
4382
5503
|
console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
|
|
4383
5504
|
appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
|
|
@@ -4570,7 +5691,7 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
|
|
|
4570
5691
|
// investigation jobs correctly found "nothing to fix" but were flagged
|
|
4571
5692
|
// anyway). For non-fix-plan jobs the exemption never applies, so rescanning
|
|
4572
5693
|
// their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
|
|
4573
|
-
const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'pass_no_commit', 'pass_no_commit_already_shipped']);
|
|
5694
|
+
const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'abandoned_background_task', 'pass_no_commit', 'pass_no_commit_already_shipped']);
|
|
4574
5695
|
|
|
4575
5696
|
// Bounds fix-plan recursion: cap N permits at most N+1 fix jobs per original
|
|
4576
5697
|
// slug (depth 1 = the original job, depth 2 = its `-fix`, depth 3+ is
|
|
@@ -4707,22 +5828,48 @@ function isPlanUnqueued(job, queuedSlugs) {
|
|
|
4707
5828
|
* Bias to needs_review: a false yellow costs a human glance, a false green
|
|
4708
5829
|
* costs a silently-unfixed bug — which is exactly what happened.
|
|
4709
5830
|
*/
|
|
5831
|
+
// abandoned_background_task shares no_verdict_sentinel's exact rescan path
|
|
5832
|
+
// (same "sentinel === null && !commitEvidence" gate in runVerify, same
|
|
5833
|
+
// committedDuringRun repo-wide-not-per-job attribution problem) — the PRD 983
|
|
5834
|
+
// incident mechanism above applies identically, so it gets the same guard
|
|
5835
|
+
// rather than a carve-out that would silently reopen the same false-heal hole.
|
|
5836
|
+
const NO_ATTRIBUTABLE_COMMIT_VERDICTS = new Set(['no_verdict_sentinel', 'abandoned_background_task']);
|
|
5837
|
+
|
|
4710
5838
|
function healRefusalReason(job, verdict, committedDuringRun) {
|
|
4711
5839
|
if (!job || !verdict) return null;
|
|
4712
5840
|
if (!COMPLETED_EQUIVALENT_VERDICTS.has(verdict.verdict)) return null;
|
|
4713
|
-
if (job.verifierVerdict
|
|
5841
|
+
if (!NO_ATTRIBUTABLE_COMMIT_VERDICTS.has(job.verifierVerdict)) return null;
|
|
4714
5842
|
// A commit this job actually recorded as its own is real evidence; the
|
|
4715
5843
|
// repo-wide window scan is not.
|
|
4716
5844
|
if (job.landedCommit) return null;
|
|
4717
|
-
return
|
|
5845
|
+
return `${job.verifierVerdict} with no job-attributable commit — refusing to heal`
|
|
4718
5846
|
+ ` (committedInWindow=${committedDuringRun === true} is repo-wide, not proof this job delivered)`;
|
|
4719
5847
|
}
|
|
4720
5848
|
|
|
5849
|
+
/**
|
|
5850
|
+
* True when a `failed` job's failure is unverified-shaped — no result event
|
|
5851
|
+
* was ever recorded for its run (classifyRunOutcome === 'no_result'), so no
|
|
5852
|
+
* SCHEDULER_VERDICT sentinel could have been parsed either, OR it already
|
|
5853
|
+
* carries a RESCANNABLE_VERDICTS verifierVerdict. A row that failed with a
|
|
5854
|
+
* real result event (classifyRunOutcome === 'failed', i.e. a genuine red
|
|
5855
|
+
* gate or a real non-zero-exit error) is excluded — that failure is
|
|
5856
|
+
* evidence, not silence, and must never become a heal candidate (PRD 1102).
|
|
5857
|
+
*/
|
|
5858
|
+
function isFailedUnverifiedShaped(job) {
|
|
5859
|
+
if (!job || job.status !== 'failed') return false;
|
|
5860
|
+
if (job.verifierVerdict && RESCANNABLE_VERDICTS.has(job.verifierVerdict)) return true;
|
|
5861
|
+
const runId = job.runId || resolveRunId(job);
|
|
5862
|
+
if (!runId) return false;
|
|
5863
|
+
const logPath = path.join(RUNS_DIR, runId, `${job.slug}.log`);
|
|
5864
|
+
return classifyRunOutcome(logPath) === 'no_result';
|
|
5865
|
+
}
|
|
5866
|
+
|
|
4721
5867
|
function isRescanCandidate(job) {
|
|
4722
|
-
return
|
|
4723
|
-
|
|
4724
|
-
|
|
4725
|
-
|
|
5868
|
+
if (!job) return false;
|
|
5869
|
+
if (!(job.runId || resolveRunId(job))) return false;
|
|
5870
|
+
if (job.status === 'needs_review') return RESCANNABLE_VERDICTS.has(job.verifierVerdict);
|
|
5871
|
+
if (job.status === 'failed') return isFailedUnverifiedShaped(job);
|
|
5872
|
+
return false;
|
|
4726
5873
|
}
|
|
4727
5874
|
|
|
4728
5875
|
/**
|
|
@@ -4756,10 +5903,36 @@ function isRescanCandidate(job) {
|
|
|
4756
5903
|
* exhausted retry is excluded
|
|
4757
5904
|
* - no fix sibling on disk (fixSlugExists) or already in the queue
|
|
4758
5905
|
*/
|
|
5906
|
+
/**
|
|
5907
|
+
* Persist a writeRcaReport() result onto its job row — job.rcaFailureClass /
|
|
5908
|
+
* job.rcaRecoveryAction — so selectAutoFixTargets and future routing can read
|
|
5909
|
+
* the classification straight off the queue row instead of re-parsing the RCA
|
|
5910
|
+
* markdown. Pure mutation of the passed-in job object; no I/O. A no-op when
|
|
5911
|
+
* the job is missing, has moved off needs_review (e.g. resumed and completed
|
|
5912
|
+
* before this async write landed), or the report was never filed (disabled,
|
|
5913
|
+
* error, etc). Returns whether it applied, for callers/tests that want to
|
|
5914
|
+
* assert on it.
|
|
5915
|
+
*/
|
|
5916
|
+
function applyRcaClassification(job, report) {
|
|
5917
|
+
if (!job || job.status !== 'needs_review' || !report?.filed) return false;
|
|
5918
|
+
job.rcaFailureClass = report.failureClass;
|
|
5919
|
+
job.rcaRecoveryAction = report.recoveryAction;
|
|
5920
|
+
return true;
|
|
5921
|
+
}
|
|
5922
|
+
|
|
4759
5923
|
function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRunId }) {
|
|
4760
5924
|
const slugsInQueue = new Set(jobs.map((j) => j.slug));
|
|
4761
5925
|
return jobs.filter((job) => {
|
|
4762
5926
|
if (job.status !== 'needs_review') return false;
|
|
5927
|
+
// A stale re-run whose work already shipped (rcaReport's 'already-shipped'
|
|
5928
|
+
// class) must never buy a fix-plan PRD — there is nothing to fix, and the
|
|
5929
|
+
// correct recovery (archiving the PRD) is a human/reconcile action, not
|
|
5930
|
+
// an investigation.
|
|
5931
|
+
if (job.rcaRecoveryAction === 'archive') return false;
|
|
5932
|
+
// Resume-first recovery (PRD 1111): a job still eligible for its one
|
|
5933
|
+
// bounded `--resume` attempt must never also become a fix-plan target
|
|
5934
|
+
// in the same pass — see spawnInvestigation's own identical guard.
|
|
5935
|
+
if (selectResumeRecoveryTarget(job)) return false;
|
|
4763
5936
|
const runId = job.runId || resolveJobRunId(job);
|
|
4764
5937
|
if (!runId) return false;
|
|
4765
5938
|
if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth)) return false;
|
|
@@ -4795,12 +5968,52 @@ function isEligibleForImmediateAutoFix(job, allJobs, fixSlugExists) {
|
|
|
4795
5968
|
return targets.some((t) => t.slug === job.slug);
|
|
4796
5969
|
}
|
|
4797
5970
|
|
|
5971
|
+
/**
|
|
5972
|
+
* Widened evidence check (PRD 1102): does at least one commit land AFTER
|
|
5973
|
+
* this job's run window that touches a path the PRD itself declares? Scoped
|
|
5974
|
+
* to the PRD's own declared paths (never the whole repo) so a sibling job's
|
|
5975
|
+
* unrelated commit is not credited to this one — see healRefusalReason's own
|
|
5976
|
+
* rationale for why unscoped, repo-wide evidence is not attribution.
|
|
5977
|
+
*
|
|
5978
|
+
* Returns null (no annotation, never fabricated) when the PRD names no
|
|
5979
|
+
* paths — the caller then has only the existing, already-computed
|
|
5980
|
+
* committedInWindow signal to go on, same as before this PRD.
|
|
5981
|
+
*
|
|
5982
|
+
* @returns {Promise<{commits: string[], paths: string[], detectedAt: string} | null>}
|
|
5983
|
+
*/
|
|
5984
|
+
async function computeLooksDone(job) {
|
|
5985
|
+
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
5986
|
+
const paths = declaredPathsForPrd(prdPath);
|
|
5987
|
+
if (!paths.length) return null;
|
|
5988
|
+
await fetchAllRefs(job.cwd);
|
|
5989
|
+
const commits = await landedSinceRun(job.cwd, job.startedAt, paths);
|
|
5990
|
+
if (!commits.length) return null;
|
|
5991
|
+
return { commits, paths, detectedAt: new Date().toISOString() };
|
|
5992
|
+
}
|
|
5993
|
+
|
|
4798
5994
|
async function reverifyNeedsReview() {
|
|
4799
5995
|
const snap = await readQueue();
|
|
4800
5996
|
const candidates = snap.jobs.filter(isRescanCandidate);
|
|
4801
5997
|
const healed = [];
|
|
4802
5998
|
const leftForReview = [];
|
|
5999
|
+
const looksDoneUpdates = [];
|
|
4803
6000
|
for (const job of candidates) {
|
|
6001
|
+
if (job.status === 'failed') {
|
|
6002
|
+
// A failed row never runs the transcript-verifier rescan below — that
|
|
6003
|
+
// machinery (verifyRun/COMPLETED_EQUIVALENT_VERDICTS) exists to
|
|
6004
|
+
// auto-COMPLETE a stale needs_review row, and a failed row must never
|
|
6005
|
+
// auto-complete through this pass (see the AC's conservative-in-the-
|
|
6006
|
+
// completing-direction constraint). The only thing a failed candidate
|
|
6007
|
+
// can gain here is a looksDone annotation + a failed → needs_review
|
|
6008
|
+
// transition, for a human to confirm.
|
|
6009
|
+
const looksDone = await computeLooksDone(job);
|
|
6010
|
+
if (looksDone) {
|
|
6011
|
+
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: true });
|
|
6012
|
+
} else {
|
|
6013
|
+
leftForReview.push({ slug: job.slug, reason: 'failed, unverified-shaped run — no post-window evidence on declared paths' });
|
|
6014
|
+
}
|
|
6015
|
+
continue;
|
|
6016
|
+
}
|
|
4804
6017
|
const runDir = path.join(RUNS_DIR, job.runId || resolveRunId(job));
|
|
4805
6018
|
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
4806
6019
|
// Derive committedDuringRun from the recorded run window. The live
|
|
@@ -4827,13 +6040,45 @@ async function reverifyNeedsReview() {
|
|
|
4827
6040
|
});
|
|
4828
6041
|
} catch { leftForReview.push({ slug: job.slug, reason: 'verifyRun threw' }); continue; }
|
|
4829
6042
|
const refusal = healRefusalReason(job, v, committedDuringRun);
|
|
6043
|
+
let stillOpen = true;
|
|
4830
6044
|
if (refusal) {
|
|
4831
6045
|
leftForReview.push({ slug: job.slug, reason: refusal });
|
|
4832
6046
|
} else if (v && COMPLETED_EQUIVALENT_VERDICTS.has(v.verdict)) {
|
|
4833
6047
|
healed.push(job.slug);
|
|
6048
|
+
stillOpen = false;
|
|
4834
6049
|
} else {
|
|
4835
6050
|
leftForReview.push({ slug: job.slug, reason: v ? `${v.verdict}: ${v.reason}` : 'null verdict' });
|
|
4836
6051
|
}
|
|
6052
|
+
// Still needs_review after the existing heal pass — widen the evidence
|
|
6053
|
+
// window before giving up on it entirely (unchanged heal semantics for
|
|
6054
|
+
// rows that already qualified above; this only adds an annotation).
|
|
6055
|
+
if (stillOpen) {
|
|
6056
|
+
const looksDone = await computeLooksDone(job);
|
|
6057
|
+
if (looksDone) {
|
|
6058
|
+
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
|
|
6059
|
+
}
|
|
6060
|
+
}
|
|
6061
|
+
}
|
|
6062
|
+
if (looksDoneUpdates.length) {
|
|
6063
|
+
const bySlug = new Map(looksDoneUpdates.map((u) => [u.slug, u]));
|
|
6064
|
+
await mutate((s) => {
|
|
6065
|
+
for (const j of s.jobs) {
|
|
6066
|
+
const u = bySlug.get(j.slug);
|
|
6067
|
+
if (!u) continue;
|
|
6068
|
+
if (u.fromFailed) {
|
|
6069
|
+
transitionJob(j, 'needs_review', {
|
|
6070
|
+
reason: 'looks done — commit(s) since this run touch this PRD\'s declared paths; confirm before archiving',
|
|
6071
|
+
source: 'reverifyNeedsReview:looksDone',
|
|
6072
|
+
});
|
|
6073
|
+
}
|
|
6074
|
+
if (j.status !== 'needs_review') continue;
|
|
6075
|
+
j.looksDone = u.looksDone;
|
|
6076
|
+
const shaList = u.looksDone.commits.slice(0, 5).map((c) => c.slice(0, 7)).join(', ');
|
|
6077
|
+
j.error = `looks done — ${u.looksDone.commits.length} commit(s) since this run touch this PRD's paths (${shaList}); confirm before archiving`;
|
|
6078
|
+
}
|
|
6079
|
+
});
|
|
6080
|
+
console.log(`[scheduler] boot reverify: looksDone annotated for ${looksDoneUpdates.length} row(s): ${looksDoneUpdates.map((u) => u.slug).join(', ')}`);
|
|
6081
|
+
await broadcast();
|
|
4837
6082
|
}
|
|
4838
6083
|
if (healed.length) {
|
|
4839
6084
|
const healSet = new Set(healed);
|
|
@@ -4844,6 +6089,7 @@ async function reverifyNeedsReview() {
|
|
|
4844
6089
|
transitionJob(j, 'completed', { reason: 'boot reverify: stale needs_review healed', source: 'reverifyNeedsReview:heal' });
|
|
4845
6090
|
j.error = null;
|
|
4846
6091
|
delete j.verifierVerdict;
|
|
6092
|
+
delete j.looksDone;
|
|
4847
6093
|
healedPrds.push({ slug: j.slug, cwd: j.cwd });
|
|
4848
6094
|
}
|
|
4849
6095
|
}
|
|
@@ -4883,6 +6129,7 @@ async function reverifyNeedsReview() {
|
|
|
4883
6129
|
orig.exitCode = 0;
|
|
4884
6130
|
orig.error = null;
|
|
4885
6131
|
orig.completedBy = job.slug;
|
|
6132
|
+
delete orig.looksDone;
|
|
4886
6133
|
if (priorStatus === 'needs_review') delete orig.verifierVerdict;
|
|
4887
6134
|
promoted.push(`${orig.slug} (was ${priorStatus}, via ${job.slug})`);
|
|
4888
6135
|
promotedPrds.push({ slug: orig.slug, cwd: orig.cwd });
|
|
@@ -4949,14 +6196,40 @@ async function reverifyNeedsReview() {
|
|
|
4949
6196
|
await broadcast();
|
|
4950
6197
|
}
|
|
4951
6198
|
|
|
6199
|
+
// The annotate mutate above only runs conditionally — when it didn't fire,
|
|
6200
|
+
// afterHealForAnnotate is still the current on-disk state, so reuse it
|
|
6201
|
+
// instead of re-reading queue.json twice more back-to-back for the
|
|
6202
|
+
// resume-recovery and auto-fix passes below (neither of which mutates
|
|
6203
|
+
// synchronously: spawnResumeRecovery/spawnJob's own writes land later).
|
|
6204
|
+
const queueForResumeAndAutofix = (unresolvable.length || exhaustedAutoFix.length || planUnqueued.length)
|
|
6205
|
+
? await readQueue()
|
|
6206
|
+
: afterHealForAnnotate;
|
|
6207
|
+
|
|
6208
|
+
// Resume-first recovery (PRD 1111): before any fix-plan investigation is
|
|
6209
|
+
// authored below, offer the bounded one-attempt `--resume` dispatch to any
|
|
6210
|
+
// needs_review job this periodic pass finds still eligible — e.g. one the
|
|
6211
|
+
// same-tick check in spawnJob missed because the app restarted between
|
|
6212
|
+
// that job parking and this pass running. selectAutoFixTargets below
|
|
6213
|
+
// already excludes every job this loop dispatches, so a resumable job
|
|
6214
|
+
// never also gets a fix-plan PRD authored in the same pass.
|
|
6215
|
+
{
|
|
6216
|
+
for (const job of queueForResumeAndAutofix.jobs) {
|
|
6217
|
+
const target = selectResumeRecoveryTarget(job);
|
|
6218
|
+
if (!target) continue;
|
|
6219
|
+
console.log(`[scheduler] resume-recovery: needs_review ${job.slug} → resuming session ${target.sessionId}`);
|
|
6220
|
+
spawnResumeRecovery(job, target).catch((e) => {
|
|
6221
|
+
console.error('[scheduler] spawnResumeRecovery error', job.slug, e);
|
|
6222
|
+
});
|
|
6223
|
+
}
|
|
6224
|
+
}
|
|
6225
|
+
|
|
4952
6226
|
// Auto-fix: spawn a fix-plan investigation for each job still in
|
|
4953
6227
|
// needs_review after the heal pass (kill-switch: SM_AUTOFIX_DISABLE=1).
|
|
4954
6228
|
// spawnInvestigation early-returns once investigationsInFlight reaches
|
|
4955
6229
|
// MAX_CONCURRENT_INVESTIGATIONS (queues the rest for retry), so this loop
|
|
4956
6230
|
// cannot fan out past the cap regardless of how many targets are selected.
|
|
4957
6231
|
if (process.env.SM_AUTOFIX_DISABLE !== '1') {
|
|
4958
|
-
const
|
|
4959
|
-
const targets = selectAutoFixTargets(afterHeal.jobs, {
|
|
6232
|
+
const targets = selectAutoFixTargets(queueForResumeAndAutofix.jobs, {
|
|
4960
6233
|
fixSlugExists: (s) => candidatePrdsDirs().some((dir) => fs.existsSync(path.join(dir, `${s}.md`))),
|
|
4961
6234
|
});
|
|
4962
6235
|
for (const job of targets) {
|
|
@@ -4985,7 +6258,7 @@ async function reverifyNeedsReview() {
|
|
|
4985
6258
|
}
|
|
4986
6259
|
}
|
|
4987
6260
|
|
|
4988
|
-
return { rescanned: candidates.length, healed, leftForReview };
|
|
6261
|
+
return { rescanned: candidates.length, healed, leftForReview, looksDone: looksDoneUpdates.map((u) => u.slug) };
|
|
4989
6262
|
}
|
|
4990
6263
|
|
|
4991
6264
|
/**
|
|
@@ -5695,8 +6968,8 @@ async function init() {
|
|
|
5695
6968
|
// there" (parallelGroup/estimateMinutes/sourcePromptId/epicId/
|
|
5696
6969
|
// archivedStatus); `fields=full` restores them.
|
|
5697
6970
|
function toCompactPrdEntry(entry) {
|
|
5698
|
-
const { slug, title, cwd, mtimeMs, archived, status } = entry;
|
|
5699
|
-
return { slug, title, cwd, mtimeMs, archived, status };
|
|
6971
|
+
const { slug, title, cwd, mtimeMs, archived, status, agentType } = entry;
|
|
6972
|
+
return { slug, title, cwd, mtimeMs, archived, status, agentType };
|
|
5700
6973
|
}
|
|
5701
6974
|
|
|
5702
6975
|
/**
|
|
@@ -5743,6 +7016,7 @@ async function listPrdsInternal() {
|
|
|
5743
7016
|
estimateMinutes: parsed.estimateMinutes,
|
|
5744
7017
|
sourcePromptId: parsed.sourcePromptId,
|
|
5745
7018
|
epicId: parsed.epicId ?? null,
|
|
7019
|
+
agentType: parsed.agentType ?? null,
|
|
5746
7020
|
mtimeMs: stat.mtimeMs,
|
|
5747
7021
|
archived,
|
|
5748
7022
|
};
|
|
@@ -5952,7 +7226,7 @@ const remote = {
|
|
|
5952
7226
|
|
|
5953
7227
|
async listJobs() {
|
|
5954
7228
|
const state = await readQueue();
|
|
5955
|
-
return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd }));
|
|
7229
|
+
return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd, agentType: j.agentType ?? null }));
|
|
5956
7230
|
},
|
|
5957
7231
|
|
|
5958
7232
|
// Single queue row lookup, used by cancelJob/updatePrd's status guards and
|
|
@@ -6191,4 +7465,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
6191
7465
|
});
|
|
6192
7466
|
}
|
|
6193
7467
|
|
|
6194
|
-
module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS };
|
|
7468
|
+
module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure };
|