claude-code-session-manager 0.79.0 → 0.81.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-COtVRqBR.js → AgentLibrary-psZYVM2w.js} +1 -1
- package/dist/assets/{DataModel-CSEKw_OR.js → DataModel-BMied5pg.js} +1 -1
- package/dist/assets/{History-CHHovrAO.js → History-BFC0oaKc.js} +1 -1
- package/dist/assets/{Hooks-BZU6C3x6.js → Hooks-CTLfO9G8.js} +1 -1
- package/dist/assets/{HostBilko-CqTUoq37.js → HostBilko-D4I0Cpwn.js} +1 -1
- package/dist/assets/{Library-BtxdyTLz.js → Library-BTzS8KsS.js} +1 -1
- package/dist/assets/{ListDetail-qZc7Zm-6.js → ListDetail-CuRmT008.js} +1 -1
- package/dist/assets/{MarkdownEditor-BHe_4fJR.js → MarkdownEditor-BgQGtvWo.js} +1 -1
- package/dist/assets/{McpServers-7Z98HLNo.js → McpServers-DiVQUK57.js} +1 -1
- package/dist/assets/{Memory-CR72KoyP.js → Memory-DSu15JmL.js} +1 -1
- package/dist/assets/{Panel-pL6H3dpQ.js → Panel-CD5wxGSR.js} +1 -1
- package/dist/assets/{Permissions-CWSWjyXM.js → Permissions-C_tAE6yJ.js} +1 -1
- package/dist/assets/{Plugins-CN6lX2lt.js → Plugins-kwI7W-eK.js} +2 -2
- package/dist/assets/{ProvenanceBadge-BXSXwIsk.js → ProvenanceBadge-CQceOgsH.js} +1 -1
- package/dist/assets/{SaveBar-BlB5TGpR.js → SaveBar-BDk5e3Pp.js} +1 -1
- package/dist/assets/{Scheduler-DRciWUmR.js → Scheduler-DRnvEYzv.js} +7 -7
- package/dist/assets/{ScopeSwitcher-kFrXtjpr.js → ScopeSwitcher-CeifaOlq.js} +1 -1
- package/dist/assets/{Settings-BXuyf4lJ.js → Settings-BWQ1Utop.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-Dfacb0PE.js → SkillReferenceGraph-CW6e1SW1.js} +1 -1
- package/dist/assets/{Skills-CHqcpiyt.js → Skills-BLwFB0E4.js} +1 -1
- package/dist/assets/{SystemPrompt-fxXm0BZr.js → SystemPrompt-DC4ZTArJ.js} +1 -1
- package/dist/assets/{TagLibrary-DOz65ZTz.js → TagLibrary-pMeGfjUb.js} +1 -1
- package/dist/assets/{TiptapBody-D0bWx_9o.js → TiptapBody-BHFid2pZ.js} +1 -1
- package/dist/assets/{Toggle-C9jBwGSx.js → Toggle-D9eYoZh4.js} +1 -1
- package/dist/assets/{index-DPYa6jbM.js → index-CuyM9vAP.js} +5 -5
- package/dist/assets/{settingsSchema-BTPw1bR3.js → settingsSchema-DrxC67uZ.js} +1 -1
- package/dist/index.html +1 -1
- package/package.json +1 -1
- package/plugins/session-manager-dev/skills/builder/3-publish/SKILL.md +10 -0
- package/plugins/session-manager-dev/skills/develop/standards.md +1 -0
- package/src/main/__tests__/computeDepHistorySatisfaction.test.cjs +66 -0
- package/src/main/__tests__/prdCreate.test.cjs +133 -8
- package/src/main/__tests__/prdFrontmatterDependsOn.test.cjs +136 -0
- package/src/main/__tests__/prdUpdateDependsOn.test.cjs +160 -0
- package/src/main/__tests__/queueHistory.test.cjs +33 -0
- package/src/main/__tests__/rcaReport.test.cjs +24 -0
- package/src/main/__tests__/runVerify-blocked-by-foreign-wip.test.cjs +58 -0
- package/src/main/__tests__/runVerify-policy-denial.test.cjs +89 -0
- package/src/main/__tests__/scheduleJobTransitions.test.cjs +1 -0
- package/src/main/__tests__/scheduler-already-satisfied-on-main.test.cjs +105 -0
- package/src/main/__tests__/scheduler-autofix-outcome.test.cjs +73 -1
- package/src/main/__tests__/scheduler-autofix-select.test.cjs +17 -0
- package/src/main/__tests__/scheduler-blocked-by-foreign-wip.test.cjs +107 -0
- package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +20 -0
- package/src/main/__tests__/scheduler-finalize-dispatch-guards.test.cjs +229 -0
- package/src/main/__tests__/scheduler-leftover-quarantine.test.cjs +199 -0
- package/src/main/__tests__/scheduler-looks-done.test.cjs +141 -1
- package/src/main/__tests__/scheduler-mechanical-recovery.test.cjs +222 -0
- package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +81 -0
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +205 -2
- package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +51 -0
- package/src/main/__tests__/scheduler-resume-recovery.test.cjs +254 -0
- package/src/main/__tests__/schedulerBatchRootBlocker.test.cjs +117 -0
- package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -2
- package/src/main/ipcSchemas.cjs +15 -1
- package/src/main/lib/__tests__/branchSweep.test.cjs +164 -0
- package/src/main/lib/__tests__/fixtures/204-mercury-steam-horse.log.txt +13 -0
- package/src/main/lib/__tests__/gitWorktree.test.cjs +129 -10
- package/src/main/lib/__tests__/landedSinceRun.test.cjs +61 -1
- package/src/main/lib/__tests__/rateLimitWindow.test.cjs +88 -0
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +120 -1
- package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +59 -7
- package/src/main/lib/branchSweep.cjs +127 -0
- package/src/main/lib/depSlugResolve.cjs +72 -0
- package/src/main/lib/epicWorktreeMerge.cjs +3 -3
- package/src/main/lib/epicWorktreeMint.cjs +17 -5
- package/src/main/lib/fixPlanSlug.cjs +62 -0
- package/src/main/lib/gitWorktree.cjs +117 -12
- package/src/main/lib/landedSinceRun.cjs +41 -1
- package/src/main/lib/mcpToolCatalog.cjs +4 -1
- package/src/main/lib/prdCreate.cjs +84 -5
- package/src/main/lib/prdFrontmatter.cjs +56 -8
- package/src/main/lib/queueHistory.cjs +50 -5
- package/src/main/lib/rateLimitWindow.cjs +62 -0
- package/src/main/lib/rcaReport.cjs +18 -3
- package/src/main/lib/reaperHelpers.cjs +169 -3
- package/src/main/lib/scheduleJobTransitions.cjs +28 -5
- package/src/main/lib/schedulerBatch.cjs +181 -23
- package/src/main/runVerify.cjs +71 -3
- package/src/main/scheduler/prdParser.cjs +7 -0
- package/src/main/scheduler.cjs +1554 -71
- package/src/preload/api.d.ts +8 -0
package/src/main/scheduler.cjs
CHANGED
|
@@ -57,9 +57,14 @@ const { resolveClaudeBin, probeClaudeVersion } = require('./lib/claudeBin.cjs');
|
|
|
57
57
|
const launchFailure = require('./lib/launchFailure.cjs');
|
|
58
58
|
const { appendError } = require('./lib/opsErrorLog.cjs');
|
|
59
59
|
const { readTail } = require('./lib/fileTail.cjs');
|
|
60
|
-
const {
|
|
60
|
+
const {
|
|
61
|
+
claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs,
|
|
62
|
+
findLiveProcessForJob, logHasOutput, resolvePidlessGateOutcome, resolveCommitGuardOutcome,
|
|
63
|
+
} = require('./lib/reaperHelpers.cjs');
|
|
64
|
+
const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
|
|
61
65
|
const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
|
|
62
66
|
const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
|
|
67
|
+
const { resolveBindingRateLimitReset } = require('./lib/rateLimitWindow.cjs');
|
|
63
68
|
const { computeQueueHealth } = require('./lib/queueHealth.cjs');
|
|
64
69
|
const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
|
|
65
70
|
const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
|
|
@@ -71,9 +76,10 @@ const { enqueueExternalPrompt } = require('./chatRunner.cjs');
|
|
|
71
76
|
const { appendResponseEventIfKnown } = require('./promptSessionEvents.cjs');
|
|
72
77
|
const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs');
|
|
73
78
|
const promptSessionTranscript = require('./promptSessionTranscript.cjs');
|
|
74
|
-
const { verifyRun } = require('./runVerify.cjs');
|
|
79
|
+
const { verifyRun, parseLog, scanSentinel, scanForeignWipPathsClaim } = require('./runVerify.cjs');
|
|
75
80
|
const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
|
|
76
|
-
const {
|
|
81
|
+
const { isFixPlanSlug, classifyDiscoveredFixPlan, resolveIsFixPlan } = require('./lib/fixPlanSlug.cjs');
|
|
82
|
+
const { landedSinceRun, landedOnMainSince } = require('./lib/landedSinceRun.cjs');
|
|
77
83
|
const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
|
|
78
84
|
const logs = require('./logs.cjs');
|
|
79
85
|
const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
|
|
@@ -99,7 +105,7 @@ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
|
|
|
99
105
|
const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
|
|
100
106
|
? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
|
|
101
107
|
: JOB_OVERRUN_FLOOR_MS_DEFAULT;
|
|
102
|
-
const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
|
|
108
|
+
const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD, DEP_HISTORY_FAIL_OPEN } = require('./lib/schedulerBatch.cjs');
|
|
103
109
|
const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
|
|
104
110
|
const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
|
|
105
111
|
const queueHistory = require('./lib/queueHistory.cjs');
|
|
@@ -140,6 +146,7 @@ const jobWorktree = require('./lib/jobWorktree.cjs');
|
|
|
140
146
|
const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
|
|
141
147
|
const queueStore = require('./lib/queueStore.cjs');
|
|
142
148
|
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
149
|
+
const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
|
|
143
150
|
const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
|
|
144
151
|
const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
|
|
145
152
|
|
|
@@ -266,9 +273,10 @@ deferring it.
|
|
|
266
273
|
Your own work must still never be left uncommitted — this only changes
|
|
267
274
|
which paths get staged, never whether you commit.
|
|
268
275
|
5. VERDICT SENTINEL — as the LAST LINE of your final result text, emit exactly
|
|
269
|
-
one of these
|
|
276
|
+
one of these lines (no trailing text after it):
|
|
270
277
|
SCHEDULER_VERDICT: PASS
|
|
271
278
|
SCHEDULER_VERDICT: FAIL <one-line reason>
|
|
279
|
+
SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP
|
|
272
280
|
Print PASS only when the AC gate is green AND the commit from step 4 landed.
|
|
273
281
|
Print FAIL (and exit 1) if the AC gate was red or the commit could not land.
|
|
274
282
|
NEVER print PASS on a red AC gate — a lying PASS turns the verifier from a
|
|
@@ -276,6 +284,22 @@ deferring it.
|
|
|
276
284
|
landed commit lets the verifier override incidental transcript noise (grep
|
|
277
285
|
results containing "Error", a TDD red-test run early in the session, debug
|
|
278
286
|
Tracebacks) so those do not false-trip a needs_review downgrade.
|
|
287
|
+
Print BLOCKED_BY_FOREIGN_WIP (and exit 1) ONLY when your own AC gate failed
|
|
288
|
+
because it ran against a SIBLING job's in-flight, uncommitted file — never
|
|
289
|
+
because of your own regression — AND every failing path is one this prompt
|
|
290
|
+
already disclosed to you as foreign (see the "FOREIGN WORKING-TREE STATE"
|
|
291
|
+
section above, if present). It MUST be accompanied by a second line naming
|
|
292
|
+
every such path:
|
|
293
|
+
FOREIGN_WIP_PATHS: <path1>, <path2>, ...
|
|
294
|
+
The scheduler independently validates every listed path against the exact
|
|
295
|
+
foreign-WIP manifest it disclosed to you. Only list a path that (a) this
|
|
296
|
+
prompt already told you is foreign WIP, not yours, AND (b) is why your own
|
|
297
|
+
AC gate failed — never a path you own, and never a path that failed for
|
|
298
|
+
some other reason of your own making. Any listed path NOT in that manifest
|
|
299
|
+
downgrades this whole verdict back to FAIL, with your claim rejected. This
|
|
300
|
+
is not an escape hatch for your own broken code: claiming it for a
|
|
301
|
+
regression you introduced, or for a path you never received as foreign
|
|
302
|
+
WIP, is a lying verdict exactly like a false PASS.
|
|
279
303
|
|
|
280
304
|
A job that exits with uncommitted changes is treated as INCOMPLETE and flagged
|
|
281
305
|
for review. Do NOT add work beyond the acceptance criteria — this protocol is the
|
|
@@ -569,6 +593,52 @@ async function computeCommittedDuringRun(cwd, headBefore, headAfter, startedAt,
|
|
|
569
593
|
return module.exports.committedInWindow(cwd, startedAt, untilIso);
|
|
570
594
|
}
|
|
571
595
|
|
|
596
|
+
/**
|
|
597
|
+
* Read-only check: is a job's `sm-job/<slug>` worktree branch already fully
|
|
598
|
+
* integrated into `cwd`'s current HEAD — i.e. every commit on the branch is
|
|
599
|
+
* an ancestor of (or equal to) HEAD? Mirrors gitWorktree.cjs's own
|
|
600
|
+
* integrateBranch() no-op detection (`mergeBase === branchHead`) exactly, but
|
|
601
|
+
* deliberately never calls integrateBranch itself: this is used by
|
|
602
|
+
* reapDeadRunningJobs to PROVE a dead job's work already landed, never to
|
|
603
|
+
* perform the landing — attempting a real merge from an audit check is a
|
|
604
|
+
* mutating action with its own conflict/timing risk, and "land the still-
|
|
605
|
+
* stranded branch" is explicitly a separate, later decision (mechanical
|
|
606
|
+
* recovery, or a human), not something a reap pass should do on its own
|
|
607
|
+
* initiative. See reapDeadRunningJobs's own header comment for why the
|
|
608
|
+
* conservative failure direction here is needs_review, never a silent merge.
|
|
609
|
+
*
|
|
610
|
+
* Returns:
|
|
611
|
+
* - true — branch exists and is fully contained in HEAD (already
|
|
612
|
+
* integrated by a prior pass, or a genuine no-op branch that never
|
|
613
|
+
* diverged from its base — both cases legitimately "landed").
|
|
614
|
+
* - false — branch exists and still holds commits not in HEAD (a real
|
|
615
|
+
* stranded deliverable, e.g. the PRD 1118 shape).
|
|
616
|
+
* - null — branch does not exist at all (never a worktree run — an
|
|
617
|
+
* in-place run, or SM_JOB_WORKTREE_DISABLE), or the git calls errored.
|
|
618
|
+
* Never throws.
|
|
619
|
+
*/
|
|
620
|
+
function isBranchAlreadyIntegrated(cwd, branch) {
|
|
621
|
+
return new Promise((resolve) => {
|
|
622
|
+
if (!cwd || !branch) { resolve(null); return; }
|
|
623
|
+
execFile(
|
|
624
|
+
'git', ['-C', cwd, 'rev-parse', '--verify', branch],
|
|
625
|
+
{ timeout: 10_000, windowsHide: true },
|
|
626
|
+
(err, branchHeadOut) => {
|
|
627
|
+
if (err) { resolve(null); return; } // branch doesn't exist — not a worktree run
|
|
628
|
+
const branchHead = String(branchHeadOut || '').trim();
|
|
629
|
+
execFile(
|
|
630
|
+
'git', ['-C', cwd, 'merge-base', 'HEAD', branch],
|
|
631
|
+
{ timeout: 10_000, windowsHide: true },
|
|
632
|
+
(mbErr, mbOut) => {
|
|
633
|
+
const mergeBase = mbErr ? '' : String(mbOut || '').trim();
|
|
634
|
+
resolve(mergeBase !== '' && mergeBase === branchHead);
|
|
635
|
+
},
|
|
636
|
+
);
|
|
637
|
+
},
|
|
638
|
+
);
|
|
639
|
+
});
|
|
640
|
+
}
|
|
641
|
+
|
|
572
642
|
/**
|
|
573
643
|
* Override for a SIGTERM'd (143) run when a commit landed in its window.
|
|
574
644
|
* Exit 143 alone doesn't prove the deliverable is missing — the 776/779
|
|
@@ -590,6 +660,74 @@ function classifySigtermWithCommit(exitCode, commitFoundInWindow) {
|
|
|
590
660
|
};
|
|
591
661
|
}
|
|
592
662
|
|
|
663
|
+
/**
|
|
664
|
+
* Pure decision for spawnJob's finalize mutate (2026-09-06 incident): given
|
|
665
|
+
* the row this run is trying to finalize (or its absence) and this run's own
|
|
666
|
+
* identity, decide whether the finalize must be DROPPED instead of applied.
|
|
667
|
+
* Never mutates anything itself — callers apply `stampLandedCommit` and emit
|
|
668
|
+
* the audit event/log line.
|
|
669
|
+
*
|
|
670
|
+
* - row missing entirely (i2 === -1 upstream): always drop, nothing to
|
|
671
|
+
* stamp — there is no row left to carry the fact.
|
|
672
|
+
* - row present but not 'running': drop the STATUS change (a cancelled row,
|
|
673
|
+
* PRD 1024, must never be re-legalized to completed/needs_review by a
|
|
674
|
+
* stale exit handler) but still let a genuinely-landed commit be stamped
|
|
675
|
+
* as a fact — never a transition — when this run owns the row's current
|
|
676
|
+
* runId, or the row doesn't already carry a newer one of its own.
|
|
677
|
+
* - row present and 'running': not a drop; caller proceeds to finalize.
|
|
678
|
+
*/
|
|
679
|
+
function evaluateFinalizeDrop({ rowExists, rowStatus, rowRunId, rowLandedCommit, runId, landedCommit }) {
|
|
680
|
+
if (!rowExists) {
|
|
681
|
+
return { drop: true, reason: 'row-missing', stampLandedCommit: null };
|
|
682
|
+
}
|
|
683
|
+
if (rowStatus !== 'running') {
|
|
684
|
+
const stampLandedCommit = landedCommit && (rowRunId === runId || !rowLandedCommit) ? landedCommit : null;
|
|
685
|
+
return { drop: true, reason: 'row-not-running', stampLandedCommit };
|
|
686
|
+
}
|
|
687
|
+
return { drop: false, reason: null, stampLandedCommit: null };
|
|
688
|
+
}
|
|
689
|
+
|
|
690
|
+
/**
|
|
691
|
+
* Pure decision for spawnJob's dispatch mutate: given a PENDING row about to
|
|
692
|
+
* be dispatched and the newest terminalRunOutcome sidecar result for its
|
|
693
|
+
* slug, decide whether this dispatch should be SKIPPED because a prior run
|
|
694
|
+
* already completed this exact slug — the gap left by reconcile()'s own
|
|
695
|
+
* anti-resurrection guard, which only ever sees a slug BEFORE it first lands
|
|
696
|
+
* in s.jobs (2026-09-06 incident: an existing pending row was re-dispatched
|
|
697
|
+
* three times against already-shipped work).
|
|
698
|
+
*
|
|
699
|
+
* Compares the sidecar's finishedAt against THIS row's own last pending
|
|
700
|
+
* transition (or queuedAt, when the row has never been reset) rather than
|
|
701
|
+
* merely "a completed sidecar exists somewhere" — so a deliberate human
|
|
702
|
+
* re-queue of the same slug in a LATER episode (whose queuedAt/pending
|
|
703
|
+
* transition postdates the earlier completion) still runs.
|
|
704
|
+
*/
|
|
705
|
+
function evaluateDispatchSidecarReconcile({ rowStatus, rowRunId, statusHistory, queuedAt, outcome }) {
|
|
706
|
+
if (rowStatus !== 'pending') return { skip: false };
|
|
707
|
+
if (!outcome || outcome.status !== 'completed') return { skip: false };
|
|
708
|
+
if (outcome.runId === rowRunId) return { skip: false };
|
|
709
|
+
const lastPendingEntry = Array.isArray(statusHistory)
|
|
710
|
+
? [...statusHistory].reverse().find((h) => h.to === 'pending')
|
|
711
|
+
: null;
|
|
712
|
+
const lastPendingAt = lastPendingEntry?.at ?? queuedAt ?? null;
|
|
713
|
+
if (lastPendingAt && (!outcome.finishedAt || outcome.finishedAt < lastPendingAt)) {
|
|
714
|
+
return { skip: false };
|
|
715
|
+
}
|
|
716
|
+
return { skip: true, runId: outcome.runId, finishedAt: outcome.finishedAt };
|
|
717
|
+
}
|
|
718
|
+
|
|
719
|
+
/**
|
|
720
|
+
* Pure detector for mutate()'s write-path regression check: transitionJob
|
|
721
|
+
* only ever APPENDS to statusHistory (capped at STATUS_HISTORY_CAP, never
|
|
722
|
+
* shrunk below it once reached) — so a running->pending change whose
|
|
723
|
+
* statusHistory got SHORTER is not a real transition, it's evidence this
|
|
724
|
+
* row's prior state was lost (e.g. a finalize mutate working off a stale
|
|
725
|
+
* in-memory snapshot).
|
|
726
|
+
*/
|
|
727
|
+
function isQueueRowRegression({ statusBefore, statusAfter, historyLenBefore, historyLenAfter }) {
|
|
728
|
+
return statusBefore === 'running' && statusAfter === 'pending' && historyLenAfter < historyLenBefore;
|
|
729
|
+
}
|
|
730
|
+
|
|
593
731
|
const ROOT = path.join(os.homedir(), '.claude', 'session-manager', 'scheduled-plans');
|
|
594
732
|
const PRDS_DIR = path.join(ROOT, 'prds');
|
|
595
733
|
const RUNS_DIR = path.join(ROOT, 'runs');
|
|
@@ -1563,7 +1701,42 @@ function mutate(fn) {
|
|
|
1563
1701
|
if (state.unreadable) {
|
|
1564
1702
|
throw new Error(`queue mutation skipped: queue.json unreadable (${state.unreadable})`);
|
|
1565
1703
|
}
|
|
1704
|
+
const beforeBySlug = new Map(
|
|
1705
|
+
(state.jobs || []).map((j) => [
|
|
1706
|
+
j.slug,
|
|
1707
|
+
{ status: j.status, historyLen: Array.isArray(j.statusHistory) ? j.statusHistory.length : 0 },
|
|
1708
|
+
]),
|
|
1709
|
+
);
|
|
1566
1710
|
const ret = await fn(state);
|
|
1711
|
+
// Detection-only regression check: transitionJob only ever APPENDS to
|
|
1712
|
+
// statusHistory (capped at STATUS_HISTORY_CAP, never shrunk below it) —
|
|
1713
|
+
// so a running->pending change whose statusHistory got SHORTER is not a
|
|
1714
|
+
// real transition, it's evidence this row's prior state was lost (e.g. a
|
|
1715
|
+
// finalize mutate working off a stale in-memory snapshot). Never
|
|
1716
|
+
// repaired here — mutate()'s job is to persist, not reconcile.
|
|
1717
|
+
for (const j of state.jobs || []) {
|
|
1718
|
+
const prior = beforeBySlug.get(j.slug);
|
|
1719
|
+
if (!prior) continue;
|
|
1720
|
+
const afterLen = Array.isArray(j.statusHistory) ? j.statusHistory.length : 0;
|
|
1721
|
+
if (isQueueRowRegression({
|
|
1722
|
+
statusBefore: prior.status,
|
|
1723
|
+
statusAfter: j.status,
|
|
1724
|
+
historyLenBefore: prior.historyLen,
|
|
1725
|
+
historyLenAfter: afterLen,
|
|
1726
|
+
})) {
|
|
1727
|
+
console.error(
|
|
1728
|
+
`[scheduler] queue row regressed: slug=${j.slug} running->pending with `
|
|
1729
|
+
+ `statusHistory ${prior.historyLen} -> ${afterLen}`,
|
|
1730
|
+
);
|
|
1731
|
+
appendAuditEvent('queue_row_regressed', {
|
|
1732
|
+
slug: j.slug,
|
|
1733
|
+
from: 'running',
|
|
1734
|
+
to: 'pending',
|
|
1735
|
+
historyBefore: prior.historyLen,
|
|
1736
|
+
historyAfter: afterLen,
|
|
1737
|
+
});
|
|
1738
|
+
}
|
|
1739
|
+
}
|
|
1567
1740
|
await writeQueue(state);
|
|
1568
1741
|
return ret;
|
|
1569
1742
|
});
|
|
@@ -2072,11 +2245,24 @@ async function reconcile(state) {
|
|
|
2072
2245
|
exitCode: null,
|
|
2073
2246
|
error: null,
|
|
2074
2247
|
};
|
|
2075
|
-
//
|
|
2076
|
-
//
|
|
2077
|
-
//
|
|
2078
|
-
//
|
|
2079
|
-
|
|
2248
|
+
// Fix-plan classification (PRD 1131): a freshly-discovered PRD is a
|
|
2249
|
+
// genuine scheduler-authored fix plan only when its OWN provenance says
|
|
2250
|
+
// so — an explicit isFixPlan:true stamp (spawnInvestigation's prompt
|
|
2251
|
+
// template) or the absence of any createdVia stamp at all (legacy
|
|
2252
|
+
// fallback, matching the "no provenance = trust the name" rule the
|
|
2253
|
+
// quarantine gate below already applies) — never merely because the
|
|
2254
|
+
// slug looks like one. See lib/fixPlanSlug.cjs's header for why (PRD
|
|
2255
|
+
// 1126: a scheduler_create_prd-authored PRD whose slug happened to start
|
|
2256
|
+
// with "fix-" was wrongly stamped investigationDepth before it ever ran).
|
|
2257
|
+
// Persisted onto the queue row so every later consumer
|
|
2258
|
+
// (commitGuardVerdict, isFixPlanBeyondDepthCap, the fix-plan-completion
|
|
2259
|
+
// checks) reads this stamp instead of re-deriving it from the name.
|
|
2260
|
+
entry.isFixPlan = classifyDiscoveredFixPlan(p, slug);
|
|
2261
|
+
// Stamp investigationDepth relative to the original job it heals, so
|
|
2262
|
+
// selectAutoFixTargets/spawnInvestigation can bound the fix-of-a-fix
|
|
2263
|
+
// recursion (see MAX_INVESTIGATION_DEPTH). Non-fix-plan jobs get no
|
|
2264
|
+
// explicit field — they read as depth 1 via `?? 1`.
|
|
2265
|
+
if (entry.isFixPlan) {
|
|
2080
2266
|
const parent = healTargetForFix(slug, state.jobs);
|
|
2081
2267
|
entry.investigationDepth = parent ? (parent.investigationDepth ?? 1) + 1 : 2;
|
|
2082
2268
|
}
|
|
@@ -2088,8 +2274,9 @@ async function reconcile(state) {
|
|
|
2088
2274
|
// guard-prd-writes.cjs PreToolUse hook should have denied. Fix-plan PRDs
|
|
2089
2275
|
// are exempt: spawnInvestigation's own probe writes them directly by
|
|
2090
2276
|
// design (a trusted, scheduler-spawned internal loop, not an
|
|
2091
|
-
// agent/human authoring a PRD)
|
|
2092
|
-
//
|
|
2277
|
+
// agent/human authoring a PRD) — entry.isFixPlan (just classified above)
|
|
2278
|
+
// is the provenance-aware verdict for that exemption now, not a raw
|
|
2279
|
+
// isFixPlanSlug name check.
|
|
2093
2280
|
//
|
|
2094
2281
|
// Quarantine is loud and reversible, never a silent skip (see the
|
|
2095
2282
|
// 2026-08-01 23-PRD outage this file's header references for what a
|
|
@@ -2098,7 +2285,7 @@ async function reconcile(state) {
|
|
|
2098
2285
|
// (schedule:adopt-prd) that stamps the file via the same update-prd API
|
|
2099
2286
|
// route the MCP tool uses — reconcile()'s adopt path above promotes it
|
|
2100
2287
|
// to 'pending' on the very next pass, within one tick of being stamped.
|
|
2101
|
-
if (!p.createdVia && !
|
|
2288
|
+
if (!p.createdVia && !entry.isFixPlan) {
|
|
2102
2289
|
entry.status = 'quarantined';
|
|
2103
2290
|
// Stamped at creation (not via transitionJob, since this is a
|
|
2104
2291
|
// brand-new row minted directly at 'quarantined' rather than
|
|
@@ -2150,6 +2337,12 @@ async function reconcile(state) {
|
|
|
2150
2337
|
state.jobs = sorted;
|
|
2151
2338
|
}
|
|
2152
2339
|
|
|
2340
|
+
// Auto-requeue any job parked 'skipped' over a validated BLOCKED_BY_FOREIGN_WIP
|
|
2341
|
+
// verdict once the paths that blocked it are no longer dirty in its own cwd —
|
|
2342
|
+
// the only place this check runs, so a sibling landing its commit resumes the
|
|
2343
|
+
// blocked job with no human action, on the very next reconcile() pass.
|
|
2344
|
+
await requeueForeignWipBlockedJobs(state.jobs);
|
|
2345
|
+
|
|
2153
2346
|
// Auto-archive completed PRDs' .md files out of the live prds/ dir. Runs
|
|
2154
2347
|
// AFTER the history append above (which is awaited) so a job's queue row
|
|
2155
2348
|
// is always durably in history.jsonl before its file can be moved — a
|
|
@@ -2192,6 +2385,20 @@ function getNextResetCached() {
|
|
|
2192
2385
|
return cachedNextReset;
|
|
2193
2386
|
}
|
|
2194
2387
|
|
|
2388
|
+
/**
|
|
2389
|
+
* Pure: picks the reset to pause against for a rate-limited run (PRD 1118).
|
|
2390
|
+
* Prefers the BINDING window read off the run's own log — refreshNextReset()
|
|
2391
|
+
* only ever reports five_hour, which is the wrong clock when a
|
|
2392
|
+
* seven_day/seven_day_overage_included window is what actually 429'd
|
|
2393
|
+
* (five_hour can read 0% utilization at the very same moment). Falls back
|
|
2394
|
+
* to the billing-endpoint-derived reset only when the log yields nothing.
|
|
2395
|
+
*/
|
|
2396
|
+
function resolveRateLimitPauseReset(logPath, billingResetIso) {
|
|
2397
|
+
const logReset = resolveBindingRateLimitReset(logPath);
|
|
2398
|
+
if (logReset != null) return new Date(logReset * 1000).toISOString();
|
|
2399
|
+
return billingResetIso ?? null;
|
|
2400
|
+
}
|
|
2401
|
+
|
|
2195
2402
|
// ---------- health / poll state ----------
|
|
2196
2403
|
|
|
2197
2404
|
let bootedAt = Date.now();
|
|
@@ -2475,6 +2682,33 @@ function nextRapidRateLimitCount(prevCount, { rateLimited, durationMs }) {
|
|
|
2475
2682
|
return prevCount || 0;
|
|
2476
2683
|
}
|
|
2477
2684
|
|
|
2685
|
+
/**
|
|
2686
|
+
* Pure: decides the resumeAt actually armed for a pause. 'network' and
|
|
2687
|
+
* 'rate_limit' (PRD 1118) both get a bounded 30-minute fallback when no
|
|
2688
|
+
* explicit resumeAt is supplied — the live rate_limit failure mode is the
|
|
2689
|
+
* billing usage endpoint itself 429ing while the log yields no binding
|
|
2690
|
+
* window either, which used to leave an indefinite pause with no resume
|
|
2691
|
+
* timer at all (a queue that never comes back on its own).
|
|
2692
|
+
*/
|
|
2693
|
+
function computeEffectiveResumeAt(reason, resumeAtIso, nowMs = Date.now()) {
|
|
2694
|
+
if (resumeAtIso) return resumeAtIso;
|
|
2695
|
+
if (reason === 'network' || reason === 'rate_limit') {
|
|
2696
|
+
return new Date(nowMs + 30 * 60_000).toISOString();
|
|
2697
|
+
}
|
|
2698
|
+
return null;
|
|
2699
|
+
}
|
|
2700
|
+
|
|
2701
|
+
/**
|
|
2702
|
+
* Pure: the setTimeout delay for a resume timer, plus whether it overflows
|
|
2703
|
+
* setTimeout's signed-32-bit max (~24.8 days) and must not be armed.
|
|
2704
|
+
* Resume fires 30s after the reset to give the auth/billing endpoint time
|
|
2705
|
+
* to flip.
|
|
2706
|
+
*/
|
|
2707
|
+
function computeResumeDelay(effectiveResumeAtIso, nowMs = Date.now()) {
|
|
2708
|
+
const delayMs = Math.max(30_000, new Date(effectiveResumeAtIso).getTime() - nowMs + 30_000);
|
|
2709
|
+
return { delayMs, tooFar: delayMs > 0x7fffffff };
|
|
2710
|
+
}
|
|
2711
|
+
|
|
2478
2712
|
async function setPaused(reason, resumeAtIso, opts = {}) {
|
|
2479
2713
|
const { observedAt = null, force = false } = opts;
|
|
2480
2714
|
// Honor manual-override cooldown: if the user cleared a pause within the
|
|
@@ -2491,11 +2725,7 @@ async function setPaused(reason, resumeAtIso, opts = {}) {
|
|
|
2491
2725
|
console.log(`[scheduler] setPaused(${reason}) engaging despite manual override cooldown — triggering run started after the manual clear`);
|
|
2492
2726
|
}
|
|
2493
2727
|
|
|
2494
|
-
|
|
2495
|
-
let effectiveResumeAt = resumeAtIso;
|
|
2496
|
-
if (reason === 'network' && !resumeAtIso) {
|
|
2497
|
-
effectiveResumeAt = new Date(Date.now() + 30 * 60_000).toISOString();
|
|
2498
|
-
}
|
|
2728
|
+
const effectiveResumeAt = computeEffectiveResumeAt(reason, resumeAtIso);
|
|
2499
2729
|
|
|
2500
2730
|
await mutate((s) => {
|
|
2501
2731
|
if (s.paused && s.paused.reason === reason) {
|
|
@@ -2509,9 +2739,8 @@ async function setPaused(reason, resumeAtIso, opts = {}) {
|
|
|
2509
2739
|
if (resumeTimer) { clearTimeout(resumeTimer); resumeTimer = null; }
|
|
2510
2740
|
if (!effectiveResumeAt) return;
|
|
2511
2741
|
|
|
2512
|
-
|
|
2513
|
-
|
|
2514
|
-
if (delay > 0x7fffffff) {
|
|
2742
|
+
const { delayMs: delay, tooFar } = computeResumeDelay(effectiveResumeAt);
|
|
2743
|
+
if (tooFar) {
|
|
2515
2744
|
console.warn(`[scheduler] paused (${reason}); resumeAt too far for setTimeout (${delay}ms)`);
|
|
2516
2745
|
return;
|
|
2517
2746
|
}
|
|
@@ -2584,6 +2813,17 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
2584
2813
|
delete job.verifierVerdict;
|
|
2585
2814
|
delete job.uncommittedPaths;
|
|
2586
2815
|
delete job.resumeRecoveryAttempted;
|
|
2816
|
+
// Same one-attempt-per-episode category as resumeRecoveryAttempted above —
|
|
2817
|
+
// a re-fired row must be able to earn a fresh mechanical-recovery attempt
|
|
2818
|
+
// if it parks needs_review again (PRD 1130).
|
|
2819
|
+
delete job.mechanicalRecoveryAttempted;
|
|
2820
|
+
// Quarantine (PRD 1128) is scoped to THIS run's episode exactly like
|
|
2821
|
+
// resumeRecoveryAttempted above — a re-fired row must be able to earn a
|
|
2822
|
+
// fresh quarantine attempt if it parks needs_review again.
|
|
2823
|
+
delete job.leftoverQuarantineAttempted;
|
|
2824
|
+
delete job.quarantinedTo;
|
|
2825
|
+
delete job.quarantinedCommit;
|
|
2826
|
+
delete job.quarantinedPaths;
|
|
2587
2827
|
// Same "this run's outcome, not durable across a reset" category as the
|
|
2588
2828
|
// fields above — a stale 'archive' recoveryAction from a prior life of this
|
|
2589
2829
|
// slug must never survive a reset and silently exclude a genuinely-new
|
|
@@ -2592,6 +2832,10 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
2592
2832
|
// can otherwise linger forever when RCA is disabled or errors).
|
|
2593
2833
|
delete job.rcaFailureClass;
|
|
2594
2834
|
delete job.rcaRecoveryAction;
|
|
2835
|
+
// Same "this run's outcome, not durable across a reset" category — a
|
|
2836
|
+
// human-driven reset must genuinely start the auto-fix budget over,
|
|
2837
|
+
// including the one-time dead-fix-plan-child reopen (PRD 1129).
|
|
2838
|
+
delete job.autoFixReopened;
|
|
2595
2839
|
// Like exitCode: this run's outcome, not durable across a reset — a stale
|
|
2596
2840
|
// leak badge from a prior attempt must not linger once the job re-fires.
|
|
2597
2841
|
delete job.leakedDescendants;
|
|
@@ -2605,6 +2849,13 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
2605
2849
|
delete job.leftoverCount;
|
|
2606
2850
|
delete job.leftoverPathsTruncated;
|
|
2607
2851
|
delete job.preRunDirtyPaths;
|
|
2852
|
+
// A manual/force reset (e.g. a human clearing a 3-strikes needs_review
|
|
2853
|
+
// park) must not leave the row permanently excluded from the auto-fix
|
|
2854
|
+
// chain (selectAutoFixTargets checks job.blockedByForeignWip) — this run's
|
|
2855
|
+
// BLOCKED_BY_FOREIGN_WIP history is done, the human is taking over.
|
|
2856
|
+
delete job.blockedByForeignWip;
|
|
2857
|
+
delete job.foreignWipBlockedPaths;
|
|
2858
|
+
delete job.foreignWipBlockCount;
|
|
2608
2859
|
// Deliberately NOT deleting job.landedCommit: it must outlive a reset so a
|
|
2609
2860
|
// re-fired run of this same slug can pass it to verifyRun as
|
|
2610
2861
|
// priorLandedCommit (pass_no_commit_prior_run_verified exemption).
|
|
@@ -3071,6 +3322,44 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
|
|
|
3071
3322
|
return { action: 'retry', transientKind, retries };
|
|
3072
3323
|
}
|
|
3073
3324
|
|
|
3325
|
+
/**
|
|
3326
|
+
* Evidence gathering for the commit-guard's already-satisfied-on-main
|
|
3327
|
+
* exemption (PRD 1136): does a commit reachable from `main`, landed AFTER
|
|
3328
|
+
* this job's `queuedAt`, touch a path this PRD itself declares? Scoped to
|
|
3329
|
+
* the PRD's own declared paths via declaredPathsForPrd — the same
|
|
3330
|
+
* path-extraction computeLooksDone already uses — so an unrelated commit
|
|
3331
|
+
* elsewhere in the repo is never credited to this job. Returns `[]` (never
|
|
3332
|
+
* fabricates evidence) when the PRD names no paths or has no `queuedAt`.
|
|
3333
|
+
*
|
|
3334
|
+
* Requires EVERY declared path to have landed, not just one — declaredPathsForPrd's
|
|
3335
|
+
* regex matches any backtick-quoted path in the PRD's Implementation notes or
|
|
3336
|
+
* Acceptance criteria, including a path cited only as context (e.g.
|
|
3337
|
+
* `` `src/main/scheduler.cjs:1234` `` pointing at a call site, not a file this
|
|
3338
|
+
* PRD's own work touches). Unlike computeLooksDone's identical path-overlap
|
|
3339
|
+
* heuristic — which only ever annotates a still-needs_review row for a human
|
|
3340
|
+
* to confirm — this check drives an UNATTENDED transition straight to
|
|
3341
|
+
* 'completed', so a single incidental hot-file citation must never be
|
|
3342
|
+
* sufficient evidence on its own (code-review finding, 2026-09-07: a PRD
|
|
3343
|
+
* that cites both its real target file and one hot file purely as context
|
|
3344
|
+
* would auto-complete off any unrelated commit touching that hot file).
|
|
3345
|
+
* Requiring full coverage of the declared-path set trades recall for safety
|
|
3346
|
+
* exactly as this PRD's own constraint demands — a PRD that fails this
|
|
3347
|
+
* stricter check still falls back to the existing needs_review park, never
|
|
3348
|
+
* a false 'completed'.
|
|
3349
|
+
*
|
|
3350
|
+
* @returns {Promise<string[]>} full commit SHAs reachable from main, newest first
|
|
3351
|
+
*/
|
|
3352
|
+
async function findSatisfyingCommitOnMain(job) {
|
|
3353
|
+
if (!job?.queuedAt) return [];
|
|
3354
|
+
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
3355
|
+
const paths = declaredPathsForPrd(prdPath);
|
|
3356
|
+
if (!paths.length) return [];
|
|
3357
|
+
await fetchAllRefs(job.cwd);
|
|
3358
|
+
const perPathCommits = await Promise.all(paths.map((p) => landedOnMainSince(job.cwd, job.queuedAt, [p])));
|
|
3359
|
+
if (perPathCommits.some((commits) => commits.length === 0)) return [];
|
|
3360
|
+
return landedOnMainSince(job.cwd, job.queuedAt, paths);
|
|
3361
|
+
}
|
|
3362
|
+
|
|
3074
3363
|
/**
|
|
3075
3364
|
* Commit-guard verdict decision. Pure/no I/O so the false-positive defenses
|
|
3076
3365
|
* can be unit-tested directly rather than only through a live spawnJob run.
|
|
@@ -3241,6 +3530,73 @@ function buildForeignWipSection({ preRunDirtyPaths, carriedPaths } = {}) {
|
|
|
3241
3530
|
return '';
|
|
3242
3531
|
}
|
|
3243
3532
|
|
|
3533
|
+
// A job whose validated BLOCKED_BY_FOREIGN_WIP claim keeps recurring against
|
|
3534
|
+
// the same sibling WIP is not making progress by re-firing forever — cap the
|
|
3535
|
+
// auto-requeue cycle and hand it to a human instead, naming the paths that
|
|
3536
|
+
// never went clean. Matches TRANSIENT_RETRY_CAP's "bounded self-heal, then
|
|
3537
|
+
// escalate" shape.
|
|
3538
|
+
const FOREIGN_WIP_BLOCK_STREAK_LIMIT = 3;
|
|
3539
|
+
|
|
3540
|
+
/**
|
|
3541
|
+
* Pure: validate an executor's BLOCKED_BY_FOREIGN_WIP claim against the exact
|
|
3542
|
+
* foreign-WIP manifest THIS job was disclosed at dispatch time
|
|
3543
|
+
* (preRunDirtyPaths / carriedPaths — PRD 1105). Every claimed path must be a
|
|
3544
|
+
* member of that manifest; a path the job was never told was foreign cannot
|
|
3545
|
+
* be laundered into a block, whether that's a genuine regression the
|
|
3546
|
+
* executor is trying to dodge or an honest mistake. `job` is any object
|
|
3547
|
+
* carrying those two array fields (the live queue row at finalize time).
|
|
3548
|
+
*
|
|
3549
|
+
* Returns `ok: false` for an empty claim too — a BLOCKED_BY_FOREIGN_WIP
|
|
3550
|
+
* verdict with no FOREIGN_WIP_PATHS evidence at all is exactly as
|
|
3551
|
+
* unsubstantiated as one naming an unlisted path.
|
|
3552
|
+
*/
|
|
3553
|
+
function validateForeignWipBlockClaim(claimedPaths, job) {
|
|
3554
|
+
const manifest = new Set([
|
|
3555
|
+
...(Array.isArray(job?.preRunDirtyPaths) ? job.preRunDirtyPaths : []),
|
|
3556
|
+
...(Array.isArray(job?.carriedPaths) ? job.carriedPaths : []),
|
|
3557
|
+
]);
|
|
3558
|
+
const claimed = Array.isArray(claimedPaths) ? claimedPaths.filter(Boolean) : [];
|
|
3559
|
+
const validPaths = claimed.filter((p) => manifest.has(p));
|
|
3560
|
+
const invalidPaths = claimed.filter((p) => !manifest.has(p));
|
|
3561
|
+
return { ok: claimed.length > 0 && invalidPaths.length === 0, validPaths, invalidPaths };
|
|
3562
|
+
}
|
|
3563
|
+
|
|
3564
|
+
/**
|
|
3565
|
+
* Auto-requeue jobs parked 'skipped' with a validated BLOCKED_BY_FOREIGN_WIP
|
|
3566
|
+
* verdict once none of the paths that blocked them are still dirty in their
|
|
3567
|
+
* own `cwd` — so a sibling job landing its commit (or a human resolving their
|
|
3568
|
+
* own WIP) resumes the blocked job with no human action needed, on the very
|
|
3569
|
+
* next reconcile() pass. `getDirtyPaths` is injectable (defaults to the real
|
|
3570
|
+
* `uncommittedChanges` git-status wrapper) so this is unit-testable without a
|
|
3571
|
+
* real git repo. Mutates eligible jobs in place via transitionJob; returns
|
|
3572
|
+
* nothing.
|
|
3573
|
+
*/
|
|
3574
|
+
async function requeueForeignWipBlockedJobs(jobs, { getDirtyPaths = uncommittedChanges } = {}) {
|
|
3575
|
+
const candidates = (Array.isArray(jobs) ? jobs : []).filter(
|
|
3576
|
+
(j) => j && j.status === 'skipped' && j.blockedByForeignWip === true,
|
|
3577
|
+
);
|
|
3578
|
+
for (const job of candidates) {
|
|
3579
|
+
const blockedPaths = Array.isArray(job.foreignWipBlockedPaths) ? job.foreignWipBlockedPaths : [];
|
|
3580
|
+
if (!blockedPaths.length) continue;
|
|
3581
|
+
const dirty = await getDirtyPaths(job.cwd);
|
|
3582
|
+
if (dirty === null) continue; // non-git cwd or git error — best-effort, leave parked
|
|
3583
|
+
const dirtySet = new Set(dirty);
|
|
3584
|
+
const stillDirty = blockedPaths.filter((p) => dirtySet.has(p));
|
|
3585
|
+
if (stillDirty.length === 0) {
|
|
3586
|
+
// resetJobFields (not a bare transitionJob) so the row re-enters
|
|
3587
|
+
// 'pending' with none of the blocked run's stale exitCode/error/
|
|
3588
|
+
// verifierVerdict/finishedAt/leftoverPaths left on it for a viewer to
|
|
3589
|
+
// misread against the new 'pending' status — same reasoning as every
|
|
3590
|
+
// other terminal->pending path in this file. force:true is required
|
|
3591
|
+
// here: resetJobFields refuses skipped->pending without it.
|
|
3592
|
+
resetJobFields(job, `foreign WIP cleared (was blocked on: ${blockedPaths.join(', ')}) — auto-requeued`, {
|
|
3593
|
+
source: 'reconcile-foreign-wip-clear',
|
|
3594
|
+
force: true,
|
|
3595
|
+
});
|
|
3596
|
+
}
|
|
3597
|
+
}
|
|
3598
|
+
}
|
|
3599
|
+
|
|
3244
3600
|
/**
|
|
3245
3601
|
* Resume-first recovery (PRD 1111). A job parked in needs_review with verdict
|
|
3246
3602
|
* 'uncommitted_changes' has a live claude session (job.sessionId, minted by
|
|
@@ -3303,6 +3659,302 @@ As the LAST LINE of your final result text, emit exactly one of:
|
|
|
3303
3659
|
Print PASS only once the commit above has actually landed.`;
|
|
3304
3660
|
}
|
|
3305
3661
|
|
|
3662
|
+
/**
|
|
3663
|
+
* Mechanical recovery (PRD 1130). isFixPlanBeyondDepthCap (below) is the
|
|
3664
|
+
* ONLY gate on re-investigating a fix-plan job at investigationDepth >= 2 —
|
|
3665
|
+
* correct for open-ended "author another plan" recursion, but it also
|
|
3666
|
+
* strands a depth-capped job whose failure was fully mechanical (no
|
|
3667
|
+
* judgement required) with no other ladder rung, since resume-first recovery
|
|
3668
|
+
* (selectResumeRecoveryTarget above) is hard-gated on verdict
|
|
3669
|
+
* 'uncommitted_changes'. This rung is evaluated INDEPENDENTLY of
|
|
3670
|
+
* isFixPlanBeyondDepthCap — depth never disqualifies it, because unlike
|
|
3671
|
+
* auto-fix it authors no plan and spawns no model; it is pure git.
|
|
3672
|
+
*
|
|
3673
|
+
* The closed set of mechanically-resolvable verdicts starts at exactly
|
|
3674
|
+
* 'worktree_integration_failed': PRD 1125 already taught integrateBranch to
|
|
3675
|
+
* parse git's "would be overwritten by merge" stderr, verify the blocking
|
|
3676
|
+
* paths are byte-identical to the branch, discard the proven duplicates, and
|
|
3677
|
+
* retry the merge once. A job parked with this verdict has its `sm-job/
|
|
3678
|
+
* <slug>` branch preserved (integrateJobBranch never deletes the branch on
|
|
3679
|
+
* failure — see cleanupJobWorktree's `keepBranch: !integration.ok`), so a
|
|
3680
|
+
* plain re-call of integrateBranch against that same branch inherits PRD
|
|
3681
|
+
* 1125's auto-resolution for free — no re-implementation needed here.
|
|
3682
|
+
*
|
|
3683
|
+
* Bounded to exactly one attempt via job.mechanicalRecoveryAttempted,
|
|
3684
|
+
* stamped in the SAME mutate as the outcome (performMechanicalRecovery,
|
|
3685
|
+
* below) — never here — so this selector alone can be unit-tested exactly
|
|
3686
|
+
* like selectResumeRecoveryTarget/selectLeftoverQuarantineTarget.
|
|
3687
|
+
*
|
|
3688
|
+
* Kill-switch: SM_MECHANICAL_RECOVERY_DISABLE=1 restores today's behaviour
|
|
3689
|
+
* exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
|
|
3690
|
+
*/
|
|
3691
|
+
const MECHANICALLY_RESOLVABLE_VERDICTS = new Set(['worktree_integration_failed']);
|
|
3692
|
+
|
|
3693
|
+
function selectMechanicalRecoveryTarget(job) {
|
|
3694
|
+
if (process.env.SM_MECHANICAL_RECOVERY_DISABLE === '1') return null;
|
|
3695
|
+
if (!job || job.status !== 'needs_review') return null;
|
|
3696
|
+
if (!MECHANICALLY_RESOLVABLE_VERDICTS.has(job.verifierVerdict)) return null;
|
|
3697
|
+
if (job.mechanicalRecoveryAttempted === true) return null;
|
|
3698
|
+
const cwd = job.cwd || DEFAULT_PROJECT_CWD;
|
|
3699
|
+
return { slug: job.slug, cwd, branch: jobWorktree.branchNameFor(job.slug), carriedPaths: job.carriedPaths || [] };
|
|
3700
|
+
}
|
|
3701
|
+
|
|
3702
|
+
/**
|
|
3703
|
+
* Perform an already-selected mechanical recovery (selectMechanicalRecoveryTarget
|
|
3704
|
+
* above) — a direct re-attempt of integrateBranch against the job's preserved
|
|
3705
|
+
* branch, never a fresh `claude -p` dispatch. On success the job transitions
|
|
3706
|
+
* needs_review -> completed and its verifierVerdict is cleared; the branch,
|
|
3707
|
+
* now merged, is deleted like any other successfully-integrated job branch.
|
|
3708
|
+
* On failure (including a branch that no longer exists — already deleted or
|
|
3709
|
+
* already merged) the job stays needs_review, mechanicalRecoveryAttempted is
|
|
3710
|
+
* stamped, and the retry's own failure text is appended to `error`. Either
|
|
3711
|
+
* way mechanicalRecoveryAttempted is stamped in this SAME mutate, so a crash
|
|
3712
|
+
* between the git call returning and this mutate landing simply repeats an
|
|
3713
|
+
* idempotent git operation on the next pass rather than leaving the job
|
|
3714
|
+
* re-eligible forever.
|
|
3715
|
+
*/
|
|
3716
|
+
async function performMechanicalRecovery(job, target) {
|
|
3717
|
+
const integration = await jobWorktree.integrateJobBranch({
|
|
3718
|
+
cwd: target.cwd, branch: target.branch, slug: target.slug, carriedPaths: target.carriedPaths,
|
|
3719
|
+
});
|
|
3720
|
+
if (integration.ok) {
|
|
3721
|
+
await jobWorktree.cleanupJobWorktree({ cwd: target.cwd, dir: undefined, branch: target.branch, keepBranch: false });
|
|
3722
|
+
}
|
|
3723
|
+
let becameCompleted = false;
|
|
3724
|
+
await mutate((s) => {
|
|
3725
|
+
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
3726
|
+
if (!j) return;
|
|
3727
|
+
j.mechanicalRecoveryAttempted = true;
|
|
3728
|
+
if (integration.ok) {
|
|
3729
|
+
if (transitionJob(j, 'completed', {
|
|
3730
|
+
reason: `mechanical recovery: ${target.branch} re-integrated successfully`,
|
|
3731
|
+
source: 'scheduler:mechanicalRecovery',
|
|
3732
|
+
})) {
|
|
3733
|
+
delete j.verifierVerdict;
|
|
3734
|
+
j.exitCode = 0;
|
|
3735
|
+
j.error = null;
|
|
3736
|
+
becameCompleted = true;
|
|
3737
|
+
}
|
|
3738
|
+
} else {
|
|
3739
|
+
const pointer = `Mechanical recovery retry failed: ${integration.reason}`;
|
|
3740
|
+
j.error = j.error ? `${j.error}\n${pointer}` : pointer;
|
|
3741
|
+
}
|
|
3742
|
+
});
|
|
3743
|
+
if (integration.ok) {
|
|
3744
|
+
console.log(`[scheduler] mechanical-recovery: ${job.slug} → completed (branch ${target.branch} re-integrated)`);
|
|
3745
|
+
if (becameCompleted) await archiveCompletedPrd(job.slug, job.cwd);
|
|
3746
|
+
} else {
|
|
3747
|
+
console.error(`[scheduler] mechanical-recovery: ${job.slug} → retry failed: ${integration.reason}`);
|
|
3748
|
+
}
|
|
3749
|
+
}
|
|
3750
|
+
|
|
3751
|
+
/**
|
|
3752
|
+
* Leftover quarantine (PRD 1128). Resume-first recovery gets exactly one
|
|
3753
|
+
* `--resume` attempt (selectResumeRecoveryTarget above); when that attempt
|
|
3754
|
+
* ALSO parks needs_review with 'uncommitted_changes', the leftovers are
|
|
3755
|
+
* about to sit dirty in the SHARED tree forever — git then refuses any later
|
|
3756
|
+
* worktree merge for this cwd that would overwrite them, turning one parked
|
|
3757
|
+
* job into a project-wide stall (216-jupiter-sand-kazekage, 2026-09-06).
|
|
3758
|
+
* Pure/no I/O, mirroring selectResumeRecoveryTarget so the eligibility rule
|
|
3759
|
+
* is unit-testable directly.
|
|
3760
|
+
*
|
|
3761
|
+
* Bounded to exactly one attempt via job.leftoverQuarantineAttempted, stamped
|
|
3762
|
+
* synchronously by the caller in the SAME mutate as this decision (never
|
|
3763
|
+
* here) — see spawnJob's finalize and reverifyNeedsReview's periodic pass.
|
|
3764
|
+
*
|
|
3765
|
+
* Kill-switch: SM_LEFTOVER_QUARANTINE_DISABLE=1 restores today's behaviour
|
|
3766
|
+
* exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
|
|
3767
|
+
*/
|
|
3768
|
+
function selectLeftoverQuarantineTarget(job) {
|
|
3769
|
+
if (process.env.SM_LEFTOVER_QUARANTINE_DISABLE === '1') return null;
|
|
3770
|
+
if (!job || job.status !== 'needs_review') return null;
|
|
3771
|
+
if (job.verifierVerdict !== 'uncommitted_changes') return null;
|
|
3772
|
+
if (job.resumeRecoveryAttempted !== true) return null;
|
|
3773
|
+
if (job.leftoverQuarantineAttempted === true) return null;
|
|
3774
|
+
const uncommittedPaths = Array.isArray(job.uncommittedPaths)
|
|
3775
|
+
? job.uncommittedPaths.filter((p) => typeof p === 'string' && p.length > 0)
|
|
3776
|
+
: [];
|
|
3777
|
+
if (!uncommittedPaths.length) return null;
|
|
3778
|
+
// The single most important constraint: never touch a path that was
|
|
3779
|
+
// ALREADY dirty at this run's own dispatch time (preRunDirtyPaths) — that
|
|
3780
|
+
// is foreign WIP (a human's or a sibling's), not this job's own leftover.
|
|
3781
|
+
const preRunDirty = new Set(Array.isArray(job.preRunDirtyPaths) ? job.preRunDirtyPaths : []);
|
|
3782
|
+
const paths = uncommittedPaths.filter((p) => !preRunDirty.has(p));
|
|
3783
|
+
if (!paths.length) return null;
|
|
3784
|
+
return { slug: job.slug, cwd: job.cwd, paths };
|
|
3785
|
+
}
|
|
3786
|
+
|
|
3787
|
+
function execGitAt(cwd, args, { env, timeout = 20_000 } = {}) {
|
|
3788
|
+
return new Promise((resolve, reject) => {
|
|
3789
|
+
execFile(
|
|
3790
|
+
'git',
|
|
3791
|
+
['-C', cwd, ...args],
|
|
3792
|
+
{ timeout, windowsHide: true, encoding: 'utf8', env: env ? { ...process.env, ...env } : process.env },
|
|
3793
|
+
(err, stdout, stderr) => {
|
|
3794
|
+
if (err) {
|
|
3795
|
+
err.stderrText = stderr;
|
|
3796
|
+
reject(err);
|
|
3797
|
+
return;
|
|
3798
|
+
}
|
|
3799
|
+
resolve(stdout || '');
|
|
3800
|
+
},
|
|
3801
|
+
);
|
|
3802
|
+
});
|
|
3803
|
+
}
|
|
3804
|
+
|
|
3805
|
+
async function pathExistsInTree(cwd, treeish, p) {
|
|
3806
|
+
try {
|
|
3807
|
+
await execGitAt(cwd, ['cat-file', '-e', `${treeish}:${p}`]);
|
|
3808
|
+
return true;
|
|
3809
|
+
} catch {
|
|
3810
|
+
return false;
|
|
3811
|
+
}
|
|
3812
|
+
}
|
|
3813
|
+
|
|
3814
|
+
/**
|
|
3815
|
+
* Commit exactly `paths` (must already be dirty on disk) onto a dedicated
|
|
3816
|
+
* `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
|
|
3817
|
+
* unavailable) via a THROWAWAY `GIT_INDEX_FILE` — never touches the live
|
|
3818
|
+
* index, never moves the checked-out branch — then restores those paths to
|
|
3819
|
+
* match that baseline commit's tree, so the shared working tree returns to
|
|
3820
|
+
* its pre-run state. This is deliberately NOT `git stash` (the destructive-
|
|
3821
|
+
* git guard blocks stash on a shared tree, and a stash nobody restores
|
|
3822
|
+
* strands the work invisibly — see standards.md).
|
|
3823
|
+
*
|
|
3824
|
+
* Never throws: any git failure, or a non-git cwd, aborts the WHOLE attempt
|
|
3825
|
+
* with the tree untouched (no partial restore) — restore only ever runs
|
|
3826
|
+
* after the salvage ref/commit has safely landed, so a failure there leaves
|
|
3827
|
+
* the data recoverable from the ref even though the tree stayed dirty.
|
|
3828
|
+
* A path no longer dirty on disk (already committed, or reverted since) is
|
|
3829
|
+
* skipped, never force-restored.
|
|
3830
|
+
*/
|
|
3831
|
+
async function quarantineLeftovers({ cwd, slug, paths, headBefore }) {
|
|
3832
|
+
if (!cwd || !slug || !Array.isArray(paths) || paths.length === 0) {
|
|
3833
|
+
return { ok: false, reason: 'no cwd/slug/paths given' };
|
|
3834
|
+
}
|
|
3835
|
+
let baseline = headBefore || null;
|
|
3836
|
+
try {
|
|
3837
|
+
if (!baseline) {
|
|
3838
|
+
baseline = (await execGitAt(cwd, ['rev-parse', 'HEAD'])).trim();
|
|
3839
|
+
}
|
|
3840
|
+
if (!baseline) return { ok: false, reason: 'could not resolve a baseline commit (non-git cwd?)' };
|
|
3841
|
+
|
|
3842
|
+
const dirtyNowRaw = await execGitAt(cwd, ['status', '--porcelain', '--', ...paths]);
|
|
3843
|
+
const dirtyNow = new Set(parsePorcelain(dirtyNowRaw));
|
|
3844
|
+
const toQuarantine = paths.filter((p) => dirtyNow.has(p));
|
|
3845
|
+
const skippedPaths = paths.filter((p) => !dirtyNow.has(p));
|
|
3846
|
+
if (!toQuarantine.length) {
|
|
3847
|
+
return { ok: true, ref: null, commit: null, quarantinedPaths: [], skippedPaths };
|
|
3848
|
+
}
|
|
3849
|
+
|
|
3850
|
+
const tmpIndex = path.join(os.tmpdir(), `sm-salvage-index-${slug}-${process.pid}-${Date.now()}`);
|
|
3851
|
+
const env = { GIT_INDEX_FILE: tmpIndex };
|
|
3852
|
+
let treeSha;
|
|
3853
|
+
let commitSha;
|
|
3854
|
+
try {
|
|
3855
|
+
await execGitAt(cwd, ['read-tree', baseline], { env });
|
|
3856
|
+
for (const p of toQuarantine) {
|
|
3857
|
+
if (fs.existsSync(path.join(cwd, p))) {
|
|
3858
|
+
await execGitAt(cwd, ['add', '--', p], { env });
|
|
3859
|
+
} else {
|
|
3860
|
+
await execGitAt(cwd, ['rm', '--cached', '--ignore-unmatch', '--', p], { env });
|
|
3861
|
+
}
|
|
3862
|
+
}
|
|
3863
|
+
treeSha = (await execGitAt(cwd, ['write-tree'], { env })).trim();
|
|
3864
|
+
commitSha = (await execGitAt(cwd, ['commit-tree', treeSha, '-p', baseline, '-m', `salvage: leftover changes from ${slug}`], { env })).trim();
|
|
3865
|
+
} catch (e) {
|
|
3866
|
+
return { ok: false, reason: `git command failed while building the salvage commit: ${(e && (e.stderrText || e.message)) || e}` };
|
|
3867
|
+
} finally {
|
|
3868
|
+
await fsp.rm(tmpIndex, { force: true }).catch(() => {});
|
|
3869
|
+
}
|
|
3870
|
+
|
|
3871
|
+
const ref = `sm-salvage/${slug}`;
|
|
3872
|
+
try {
|
|
3873
|
+
await execGitAt(cwd, ['update-ref', `refs/heads/${ref}`, commitSha]);
|
|
3874
|
+
} catch (e) {
|
|
3875
|
+
return { ok: false, reason: `git command failed updating ${ref}: ${(e && (e.stderrText || e.message)) || e}` };
|
|
3876
|
+
}
|
|
3877
|
+
|
|
3878
|
+
// The salvage commit is safely landed at this point — a failure from here
|
|
3879
|
+
// on is reported with the ref/commit still attached so nothing looks lost
|
|
3880
|
+
// even if the tree itself couldn't be fully restored.
|
|
3881
|
+
try {
|
|
3882
|
+
const inBaseline = [];
|
|
3883
|
+
const notInBaseline = [];
|
|
3884
|
+
for (const p of toQuarantine) {
|
|
3885
|
+
// eslint-disable-next-line no-await-in-loop
|
|
3886
|
+
if (await pathExistsInTree(cwd, baseline, p)) inBaseline.push(p); else notInBaseline.push(p);
|
|
3887
|
+
}
|
|
3888
|
+
if (inBaseline.length) {
|
|
3889
|
+
await execGitAt(cwd, ['checkout', baseline, '--', ...inBaseline]);
|
|
3890
|
+
}
|
|
3891
|
+
if (notInBaseline.length) {
|
|
3892
|
+
await execGitAt(cwd, ['reset', '--', ...notInBaseline]).catch(() => {});
|
|
3893
|
+
for (const p of notInBaseline) {
|
|
3894
|
+
// eslint-disable-next-line no-await-in-loop
|
|
3895
|
+
await fsp.rm(path.join(cwd, p), { force: true });
|
|
3896
|
+
}
|
|
3897
|
+
}
|
|
3898
|
+
} catch (e) {
|
|
3899
|
+
return {
|
|
3900
|
+
ok: false,
|
|
3901
|
+
ref,
|
|
3902
|
+
commit: commitSha,
|
|
3903
|
+
reason: `salvage commit landed at ${ref} (${commitSha}) but restoring the working tree failed: ${(e && (e.stderrText || e.message)) || e}`,
|
|
3904
|
+
};
|
|
3905
|
+
}
|
|
3906
|
+
|
|
3907
|
+
return { ok: true, ref, commit: commitSha, quarantinedPaths: toQuarantine, skippedPaths };
|
|
3908
|
+
} catch (e) {
|
|
3909
|
+
return { ok: false, reason: `git command failed: ${(e && (e.stderrText || e.message)) || e}` };
|
|
3910
|
+
}
|
|
3911
|
+
}
|
|
3912
|
+
|
|
3913
|
+
/**
|
|
3914
|
+
* Perform an already-selected quarantine (job.leftoverQuarantineAttempted
|
|
3915
|
+
* must already be true, stamped by the caller) and persist the outcome onto
|
|
3916
|
+
* the job row: `quarantinedTo`/`quarantinedCommit`/`quarantinedPaths` on
|
|
3917
|
+
* success, plus a one-line pointer appended to `error` naming the ref so a
|
|
3918
|
+
* human can recover with a single named command
|
|
3919
|
+
* (`git show sm-salvage/<slug>`). The belt-and-braces `salvagePatch` (when
|
|
3920
|
+
* present) is referenced alongside it, never removed. On failure, only a
|
|
3921
|
+
* diagnostic is appended — the job row's dirt-describing fields are left as
|
|
3922
|
+
* they were, since the tree itself was left untouched (or, for a
|
|
3923
|
+
* restore-only failure, the salvage ref is still named in the note).
|
|
3924
|
+
*
|
|
3925
|
+
* `headBefore`, when the caller has it fresh (spawnJob's own finalize still
|
|
3926
|
+
* has the local `guardHeadBefore` in scope for the run that just parked —
|
|
3927
|
+
* the same value is deleted off the job ROW earlier in that same finalize),
|
|
3928
|
+
* is used as the salvage ref's baseline commit; otherwise (the periodic
|
|
3929
|
+
* reverifyNeedsReview pass, re-discovering an already-parked row) this falls
|
|
3930
|
+
* back to the current HEAD inside quarantineLeftovers itself.
|
|
3931
|
+
*/
|
|
3932
|
+
async function performLeftoverQuarantine(job, paths, headBefore = null) {
|
|
3933
|
+
const result = await quarantineLeftovers({
|
|
3934
|
+
cwd: job.cwd || DEFAULT_PROJECT_CWD,
|
|
3935
|
+
slug: job.slug,
|
|
3936
|
+
paths,
|
|
3937
|
+
headBefore: headBefore || job.guardHeadBefore || null,
|
|
3938
|
+
});
|
|
3939
|
+
await mutate((s) => {
|
|
3940
|
+
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
3941
|
+
if (!j) return;
|
|
3942
|
+
if (result.ok && Array.isArray(result.quarantinedPaths) && result.quarantinedPaths.length) {
|
|
3943
|
+
j.quarantinedTo = result.ref;
|
|
3944
|
+
j.quarantinedCommit = result.commit;
|
|
3945
|
+
j.quarantinedPaths = capDirtyPaths(result.quarantinedPaths);
|
|
3946
|
+
const salvageNote = j.salvagePatch ? `; salvage patch also at ${j.salvagePatch}` : '';
|
|
3947
|
+
const pointer = `Leftovers quarantined to ${result.ref} (commit ${result.commit}) — recover via \`git show ${result.ref}\`${salvageNote}`;
|
|
3948
|
+
j.error = j.error ? `${j.error}\n${pointer}` : pointer;
|
|
3949
|
+
console.log(`[scheduler] ${job.slug}: quarantined ${result.quarantinedPaths.length} leftover path(s) to ${result.ref} (${result.commit})`);
|
|
3950
|
+
} else if (!result.ok) {
|
|
3951
|
+
const pointer = `Leftover quarantine failed: ${result.reason}`;
|
|
3952
|
+
j.error = j.error ? `${j.error}\n${pointer}` : pointer;
|
|
3953
|
+
console.error(`[scheduler] ${job.slug}: leftover quarantine failed: ${result.reason}`);
|
|
3954
|
+
}
|
|
3955
|
+
});
|
|
3956
|
+
}
|
|
3957
|
+
|
|
3306
3958
|
/**
|
|
3307
3959
|
* Pure argv builder for a `claude -p` child spawn, shared so the
|
|
3308
3960
|
* resume-vs-fresh-session choice is made in exactly one place. `resume`
|
|
@@ -3782,13 +4434,9 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
3782
4434
|
// project (keyed by cwd) so jobs in different repos run concurrently up to
|
|
3783
4435
|
// the cap; within one project, sequential-group semantics are preserved.
|
|
3784
4436
|
|
|
3785
|
-
|
|
3786
|
-
|
|
3787
|
-
|
|
3788
|
-
*/
|
|
3789
|
-
function isFixPlanSlug(slug) {
|
|
3790
|
-
return /^\d+-fix-/.test(slug);
|
|
3791
|
-
}
|
|
4437
|
+
// isFixPlanSlug/classifyDiscoveredFixPlan/resolveIsFixPlan now live in
|
|
4438
|
+
// lib/fixPlanSlug.cjs (PRD 1131) — see that module's header for why slug
|
|
4439
|
+
// shape alone is no longer sufficient to classify a fix plan.
|
|
3792
4440
|
|
|
3793
4441
|
/**
|
|
3794
4442
|
* The fix-plan slug spawnInvestigation authors for a given failed job —
|
|
@@ -3827,7 +4475,25 @@ function healTargetForFix(fixSlug, jobs) {
|
|
|
3827
4475
|
* unit-tested (no spawn, no fs). Inputs are the already-resolved values that
|
|
3828
4476
|
* spawnInvestigation computes.
|
|
3829
4477
|
*/
|
|
3830
|
-
function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group }) {
|
|
4478
|
+
function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild = null }) {
|
|
4479
|
+
const deadFixChildNote = deadChild ? `
|
|
4480
|
+
|
|
4481
|
+
# This is a REOPENED investigation — your own prior fix plan died
|
|
4482
|
+
You already investigated this job once and produced a fix-plan PRD, \`${deadChild.slug}\`, which was
|
|
4483
|
+
supposed to heal it. That fix-plan job itself reached a terminal, non-completed status
|
|
4484
|
+
(\`${deadChild.status}\`) without ever fixing the original failure — so the parent job you are now
|
|
4485
|
+
investigating is stuck again with nothing left to retry it automatically. This is the ONE reopen
|
|
4486
|
+
this parent gets; do not cold-read the log and re-derive the plan that already failed.
|
|
4487
|
+
|
|
4488
|
+
Dead fix-plan child's own outcome:
|
|
4489
|
+
- Slug: ${deadChild.slug}
|
|
4490
|
+
- Status: ${deadChild.status}
|
|
4491
|
+
- Verifier verdict: ${deadChild.verifierVerdict ?? '(none recorded)'}
|
|
4492
|
+
- Error: ${deadChild.error ?? '(none recorded)'}
|
|
4493
|
+
|
|
4494
|
+
Read why THAT job died (its own run log, if any, under the runs directory) before writing a new
|
|
4495
|
+
fix-plan PRD, and make sure your new plan is genuinely different from — not a repeat of — whatever
|
|
4496
|
+
that dead child attempted.` : '';
|
|
3831
4497
|
const abandonedBackgroundTaskNote = failedJob.verifierVerdict === 'abandoned_background_task' ? `
|
|
3832
4498
|
|
|
3833
4499
|
# Known failure class: abandoned background task
|
|
@@ -3848,7 +4514,7 @@ The fix-plan PRD you write for this MUST instruct its executor to, in order:
|
|
|
3848
4514
|
1. Check for a salvage patch (named \`<slug>.uncommitted.patch\` in the run directory${failedJob.salvagePatch ? `, e.g. \`${failedJob.salvagePatch}\`` : ''}) and, if found, apply it to the working tree BEFORE inspecting \`git status\`/\`git diff\` in ${cwd} for uncommitted changes matching the original PRD's acceptance criteria.
|
|
3849
4515
|
2. If the work is present (via the applied patch or already in the tree) and satisfies the acceptance criteria, run the project's verify commands and COMMIT it — do not re-implement or re-plan the PRD from scratch.
|
|
3850
4516
|
3. Only fall back to re-implementing whatever acceptance criteria are genuinely missing after applying any salvage patch, not the whole PRD.` : '';
|
|
3851
|
-
return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${abandonedBackgroundTaskNote}
|
|
4517
|
+
return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${deadFixChildNote}${abandonedBackgroundTaskNote}
|
|
3852
4518
|
|
|
3853
4519
|
# Failed job
|
|
3854
4520
|
- Slug: ${failedJob.slug}
|
|
@@ -3893,8 +4559,13 @@ ${logTail}
|
|
|
3893
4559
|
cwd: ${cwd}
|
|
3894
4560
|
parallelGroup: ${group}
|
|
3895
4561
|
estimateMinutes: <your time estimate>
|
|
4562
|
+
isFixPlan: true
|
|
3896
4563
|
---
|
|
3897
4564
|
\`\`\`
|
|
4565
|
+
\`isFixPlan: true\` is REQUIRED — it is the scheduler's provenance signal that this PRD is a
|
|
4566
|
+
genuine auto-authored fix plan (not a human/agent PRD whose slug merely happens to start with
|
|
4567
|
+
"fix-"); omitting it means this fix plan will not get its depth-cap/zero-edit-commit-guard
|
|
4568
|
+
exemptions.
|
|
3898
4569
|
\`cwd\` must be the git repo root where the fix will actually land. If the failed job's cwd is
|
|
3899
4570
|
not that repo (e.g. a scratch dir like \`/tmp\`), set \`cwd:\` to the correct repo root instead —
|
|
3900
4571
|
the scheduler's commit guard and post-run verifier read git state from this path, and a
|
|
@@ -3951,6 +4622,11 @@ function readRunOutcomeSidecars(runDir, slug) {
|
|
|
3951
4622
|
return {
|
|
3952
4623
|
meta: readJson(path.join(runDir, `${slug}.meta.json`)),
|
|
3953
4624
|
verdicts: readJson(path.join(runDir, `${slug}.verdicts.json`)),
|
|
4625
|
+
// outcome.json (launchFailure.writeOutcomeSidecar) is the one sidecar
|
|
4626
|
+
// that carries landedCommit — reused by the dispatch-time sidecar-
|
|
4627
|
+
// reconcile guard and the pre-dispatch landedCommit backfill below,
|
|
4628
|
+
// rather than growing a second reader for the same directory.
|
|
4629
|
+
outcome: readJson(path.join(runDir, `${slug}.outcome.json`)),
|
|
3954
4630
|
};
|
|
3955
4631
|
}
|
|
3956
4632
|
|
|
@@ -3968,7 +4644,7 @@ function readRunOutcomeSidecars(runDir, slug) {
|
|
|
3968
4644
|
*/
|
|
3969
4645
|
const INVESTIGATION_LAUNCH_KEY = 'investigation';
|
|
3970
4646
|
|
|
3971
|
-
async function spawnInvestigation(failedJob, runDir) {
|
|
4647
|
+
async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {}) {
|
|
3972
4648
|
// The probe launches with the same CLI as the job it diagnoses. While
|
|
3973
4649
|
// that CLI cannot launch at all (launch circuit breaker, issue #11 list
|
|
3974
4650
|
// B1: probes e4f82da2/d374e6bf died on the same HTTP 400 as the runs
|
|
@@ -3997,7 +4673,14 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3997
4673
|
console.log(`[scheduler] skip investigation: ${failedJob.slug} is resume-recovery eligible`);
|
|
3998
4674
|
return { deferred: false };
|
|
3999
4675
|
}
|
|
4000
|
-
|
|
4676
|
+
// Mechanical recovery (PRD 1130): same first-refusal treatment — a job
|
|
4677
|
+
// eligible for a pure-git retry must never also get a cold-read fix-plan
|
|
4678
|
+
// PRD authored in the same pass.
|
|
4679
|
+
if (selectMechanicalRecoveryTarget(failedJob)) {
|
|
4680
|
+
console.log(`[scheduler] skip investigation: ${failedJob.slug} is mechanical-recovery eligible`);
|
|
4681
|
+
return { deferred: false };
|
|
4682
|
+
}
|
|
4683
|
+
if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth, failedJob.isFixPlan)) {
|
|
4001
4684
|
console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
|
|
4002
4685
|
return { deferred: false };
|
|
4003
4686
|
}
|
|
@@ -4054,7 +4737,13 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
4054
4737
|
|
|
4055
4738
|
const logTail = readTail(failedLogPath, 16 * 1024) || '(failed to read log)';
|
|
4056
4739
|
|
|
4057
|
-
|
|
4740
|
+
// A dead-fix-plan reopen (PRD 1129) targets the SAME fixPath its dead
|
|
4741
|
+
// child was originally authored at, by construction (fixSlugFor is a pure
|
|
4742
|
+
// function of the parent) — the file existing is not staleness here, it's
|
|
4743
|
+
// the whole reason a reopen was offered. Skip the guard in that one case
|
|
4744
|
+
// so the second investigation can overwrite the dead plan; every other
|
|
4745
|
+
// caller keeps the original protection against clobbering a live sibling.
|
|
4746
|
+
if (fs.existsSync(fixPath) && !deadChild) {
|
|
4058
4747
|
console.log(`[scheduler] skip investigation: fix plan already exists at ${fixPath}`);
|
|
4059
4748
|
releaseSlot();
|
|
4060
4749
|
return { deferred: false };
|
|
@@ -4080,7 +4769,7 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
4080
4769
|
console.warn(`[scheduler] investigation cwd is not a git repo (${cwd}); falling back to ${DEFAULT_PROJECT_CWD}`);
|
|
4081
4770
|
cwd = DEFAULT_PROJECT_CWD;
|
|
4082
4771
|
}
|
|
4083
|
-
const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group });
|
|
4772
|
+
const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild });
|
|
4084
4773
|
|
|
4085
4774
|
// Phase 1: open log fd for pre-spawn diagnostics.
|
|
4086
4775
|
const { fd, safeLog, closeFd } = openLog(investigationLogPath);
|
|
@@ -4209,6 +4898,22 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
4209
4898
|
mutate((s) => {
|
|
4210
4899
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
4211
4900
|
if (j) j.autoFixOutcome = 'plan';
|
|
4901
|
+
// Dead-fix-plan reopen (PRD 1129): fixSlugFor is a pure function of
|
|
4902
|
+
// the parent, so the freshly-authored plan landed at the SAME slug
|
|
4903
|
+
// as the dead child — reconcile() sees an already-known slug and
|
|
4904
|
+
// will never re-mint a pending row for it. Explicitly reset the
|
|
4905
|
+
// dead child's own row here so the overwritten plan actually gets
|
|
4906
|
+
// a chance to run, rather than sitting inert behind a permanently
|
|
4907
|
+
// terminal queue row. force:true because 'skipped' (a valid dead
|
|
4908
|
+
// status here) is otherwise reset-refused by design.
|
|
4909
|
+
if (deadChild) {
|
|
4910
|
+
const child = s.jobs.find((x) => x.slug === deadChild.slug);
|
|
4911
|
+
if (child) {
|
|
4912
|
+
resetJobFields(child, 'reset by dead-fix-plan reopen: parent investigation authored a new plan', {
|
|
4913
|
+
force: true, source: 'spawnInvestigation:dead-fix-plan-reopen',
|
|
4914
|
+
});
|
|
4915
|
+
}
|
|
4916
|
+
}
|
|
4212
4917
|
}).catch(() => {});
|
|
4213
4918
|
} else {
|
|
4214
4919
|
console.log(`[scheduler] investigation finished WITHOUT producing fix plan (slug=${failedJob.slug}, code=${exitCode})`);
|
|
@@ -4291,6 +4996,58 @@ async function computeLaunchHolds(state, { now = Date.now(), claudeVersion } = {
|
|
|
4291
4996
|
return held;
|
|
4292
4997
|
}
|
|
4293
4998
|
|
|
4999
|
+
/**
|
|
5000
|
+
* computeDepHistorySatisfaction(state) → Map<cwd, Set<string>|symbol>
|
|
5001
|
+
*
|
|
5002
|
+
* PRD 1122's once-per-tick dependsOn history/archive lookup: for every
|
|
5003
|
+
* distinct project cwd with jobs this tick, builds the set of dep slugs that
|
|
5004
|
+
* have no live queue row but are nonetheless known-satisfied — a completed
|
|
5005
|
+
* record in that project's own `state/history.jsonl` shard
|
|
5006
|
+
* (queueHistory.completedSlugsForCwd, scoped per-project so a same-named PRD
|
|
5007
|
+
* in an unrelated project can never satisfy a dep here), or a `.md` file
|
|
5008
|
+
* under any of that project's `prds-archived/` dirs (listArchivedPrdDirs —
|
|
5009
|
+
* covers both the retired flat layout and every Epic's own sibling archive).
|
|
5010
|
+
* findBlockingDep (schedulerBatch.cjs) treats a dep slug as blocking
|
|
5011
|
+
* whenever it has no live row AND is absent from this set, so a typo or a
|
|
5012
|
+
* double-prefixed slug (the exact 2026-09-06 starry-night-ships incident)
|
|
5013
|
+
* HOLDS its dependent instead of silently dispatching it.
|
|
5014
|
+
*
|
|
5015
|
+
* Fails OPEN per project, never queue-wide: a history-shard or archive-scan
|
|
5016
|
+
* read error for one cwd degrades that cwd's value to
|
|
5017
|
+
* `DEP_HISTORY_FAIL_OPEN` (findBlockingDep then treats every rowless dep in
|
|
5018
|
+
* that project as satisfied, exactly today's pre-1122 behaviour) with a
|
|
5019
|
+
* logged warning — it never throws out of this function and never blocks
|
|
5020
|
+
* every OTHER project's dispatch for one project's bad fs state.
|
|
5021
|
+
*
|
|
5022
|
+
* Computed ONCE here, before pickNextBatch runs, and threaded down as pure
|
|
5023
|
+
* data (quietOpts.satisfiedSlugsByCwd) — schedulerBatch.cjs itself does no
|
|
5024
|
+
* I/O, so this is the only fs read this gate costs per tick, not one per job
|
|
5025
|
+
* per dep.
|
|
5026
|
+
*/
|
|
5027
|
+
async function computeDepHistorySatisfaction(state) {
|
|
5028
|
+
const byCwd = new Map();
|
|
5029
|
+
const cwds = new Set((state?.jobs || []).map((j) => j.cwd || DEFAULT_PROJECT_CWD));
|
|
5030
|
+
for (const cwd of cwds) {
|
|
5031
|
+
const satisfied = new Set();
|
|
5032
|
+
try {
|
|
5033
|
+
for (const slug of await queueHistory.completedSlugsForCwd(cwd)) satisfied.add(slug);
|
|
5034
|
+
for (const dir of listArchivedPrdDirs(cwd)) {
|
|
5035
|
+
let entries;
|
|
5036
|
+
try { entries = await fsp.readdir(dir); } catch { continue; }
|
|
5037
|
+
for (const name of entries) {
|
|
5038
|
+
if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
|
|
5039
|
+
}
|
|
5040
|
+
}
|
|
5041
|
+
} catch (e) {
|
|
5042
|
+
console.warn(`[scheduler] depHistorySatisfaction: history/archive lookup failed for ${cwd} (${e?.message}) — falling back to fail-open dep resolution for this project this tick`);
|
|
5043
|
+
byCwd.set(cwd, DEP_HISTORY_FAIL_OPEN);
|
|
5044
|
+
continue;
|
|
5045
|
+
}
|
|
5046
|
+
byCwd.set(cwd, satisfied);
|
|
5047
|
+
}
|
|
5048
|
+
return byCwd;
|
|
5049
|
+
}
|
|
5050
|
+
|
|
4294
5051
|
/**
|
|
4295
5052
|
* A run that never got a turn (res.launchFailure — see executeJob's onExit)
|
|
4296
5053
|
* is routed here instead of the failed/investigation path (issue #11 lists
|
|
@@ -4468,9 +5225,64 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4468
5225
|
// started after the manual clear, not whatever startedAt this row
|
|
4469
5226
|
// carried from a prior run.
|
|
4470
5227
|
let dispatchStartedAtMs = null;
|
|
5228
|
+
let dispatchSkippedAlreadyCompleted = false;
|
|
4471
5229
|
await mutate((s) => {
|
|
4472
5230
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4473
5231
|
if (idx >= 0) {
|
|
5232
|
+
// Anti-resurrection guard for an EXISTING pending row (the gap left
|
|
5233
|
+
// by reconcile()'s own guard, which only ever sees a slug BEFORE it
|
|
5234
|
+
// first lands in s.jobs — 2026-09-06 incident: a finalize dropped
|
|
5235
|
+
// silently three times over left this same slug 'pending' and the
|
|
5236
|
+
// dispatcher re-fired it three more times against work that had
|
|
5237
|
+
// already shipped). Skipped for a resume-recovery dispatch (that
|
|
5238
|
+
// targets a specific prior session on purpose) and for anything not
|
|
5239
|
+
// currently 'pending' (e.g. a needs_review->running recovery row).
|
|
5240
|
+
if (!resumeTarget && s.jobs[idx].status === 'pending') {
|
|
5241
|
+
const outcome = latestTerminalOutcomeForSlug(job.slug, { runsDir: RUNS_DIR });
|
|
5242
|
+
const reconcileDecision = evaluateDispatchSidecarReconcile({
|
|
5243
|
+
rowStatus: s.jobs[idx].status,
|
|
5244
|
+
rowRunId: s.jobs[idx].runId ?? null,
|
|
5245
|
+
statusHistory: s.jobs[idx].statusHistory,
|
|
5246
|
+
queuedAt: s.jobs[idx].queuedAt ?? null,
|
|
5247
|
+
outcome,
|
|
5248
|
+
});
|
|
5249
|
+
if (reconcileDecision.skip) {
|
|
5250
|
+
const sidecar = readRunOutcomeSidecars(path.join(RUNS_DIR, reconcileDecision.runId), job.slug);
|
|
5251
|
+
transitionJob(s.jobs[idx], 'completed', {
|
|
5252
|
+
reason: `prior run ${reconcileDecision.runId} already completed this slug (sidecar-reconciled)`,
|
|
5253
|
+
source: 'spawnJob:dispatch-sidecar-reconcile',
|
|
5254
|
+
});
|
|
5255
|
+
s.jobs[idx].runId = reconcileDecision.runId;
|
|
5256
|
+
s.jobs[idx].finishedAt = reconcileDecision.finishedAt;
|
|
5257
|
+
s.jobs[idx].exitCode = 0;
|
|
5258
|
+
if (sidecar.outcome?.landedCommit) {
|
|
5259
|
+
s.jobs[idx].landedCommit = sidecar.outcome.landedCommit;
|
|
5260
|
+
}
|
|
5261
|
+
appendAuditEvent('job_dispatch_skipped_already_completed', {
|
|
5262
|
+
slug: job.slug,
|
|
5263
|
+
priorRunId: reconcileDecision.runId,
|
|
5264
|
+
cwd: job.cwd || defaultCwd,
|
|
5265
|
+
});
|
|
5266
|
+
console.warn(
|
|
5267
|
+
`[scheduler] ${job.slug}: dispatch skipped — prior run ${reconcileDecision.runId} already `
|
|
5268
|
+
+ 'completed this slug (sidecar-reconciled)',
|
|
5269
|
+
);
|
|
5270
|
+
dispatchSkippedAlreadyCompleted = true;
|
|
5271
|
+
return;
|
|
5272
|
+
}
|
|
5273
|
+
// Belt-and-braces (PRD fix-plan step 3): a row about to dispatch
|
|
5274
|
+
// with no landedCommit of its own, whose newest sidecar for this
|
|
5275
|
+
// slug DOES record one, gets it backfilled before spawn so
|
|
5276
|
+
// verifyRun receives a real priorLandedCommit and the
|
|
5277
|
+
// pass_no_commit_prior_run_verified exemption can fire on this
|
|
5278
|
+
// run if it turns out to be another no-op re-verification.
|
|
5279
|
+
if (!s.jobs[idx].landedCommit && outcome?.runId) {
|
|
5280
|
+
const sidecar = readRunOutcomeSidecars(path.join(RUNS_DIR, outcome.runId), job.slug);
|
|
5281
|
+
if (sidecar.outcome?.landedCommit) {
|
|
5282
|
+
s.jobs[idx].landedCommit = sidecar.outcome.landedCommit;
|
|
5283
|
+
}
|
|
5284
|
+
}
|
|
5285
|
+
}
|
|
4474
5286
|
transitionJob(s.jobs[idx], 'running', {
|
|
4475
5287
|
reason: resumeTarget ? 'dispatched for resume-recovery' : 'dispatched for execution',
|
|
4476
5288
|
source: 'spawnJob:dispatch',
|
|
@@ -4494,6 +5306,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4494
5306
|
}
|
|
4495
5307
|
});
|
|
4496
5308
|
await broadcast({ flush: true });
|
|
5309
|
+
if (dispatchSkippedAlreadyCompleted) return;
|
|
4497
5310
|
|
|
4498
5311
|
// Commit-guard baseline: snapshot the working tree BEFORE the run so the
|
|
4499
5312
|
// post-run check flags only paths THIS job left dirty, not pre-existing WIP.
|
|
@@ -4586,6 +5399,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4586
5399
|
let res;
|
|
4587
5400
|
let worktreeLeftoverDirty = [];
|
|
4588
5401
|
let worktreeIntegrationFailure = null;
|
|
5402
|
+
// Set only when integrateJobBranch's stderr-parsing auto-resolve fired
|
|
5403
|
+
// (PRD 1125) — surfaced on the job row so the Queue UI can say the merge
|
|
5404
|
+
// self-healed rather than silently looking like an ordinary merge.
|
|
5405
|
+
let mergeAutoResolved = null;
|
|
5406
|
+
let mergeAutoResolvedPaths = null;
|
|
4589
5407
|
// A job's uncommitted-work patch, whichever isolation mode produced it —
|
|
4590
5408
|
// set by EITHER branch below, never both (worktree.ok picks exactly one
|
|
4591
5409
|
// shape for the whole run). Named generically (not "worktree...") because
|
|
@@ -4633,6 +5451,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4633
5451
|
console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
|
|
4634
5452
|
} else if (integration.integrated) {
|
|
4635
5453
|
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
|
|
5454
|
+
if (integration.autoResolved) {
|
|
5455
|
+
mergeAutoResolved = integration.autoResolved;
|
|
5456
|
+
mergeAutoResolvedPaths = integration.resolvedPaths || [];
|
|
5457
|
+
console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
|
|
5458
|
+
}
|
|
4636
5459
|
}
|
|
4637
5460
|
await jobWorktree.cleanupJobWorktree({
|
|
4638
5461
|
cwd: guardCwd,
|
|
@@ -4727,7 +5550,9 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4727
5550
|
}
|
|
4728
5551
|
|
|
4729
5552
|
if (res.rateLimited) {
|
|
4730
|
-
const
|
|
5553
|
+
const logPath = path.join(runDir, `${job.slug}.log`);
|
|
5554
|
+
const billingResetIso = await refreshNextReset().catch(() => cachedNextReset);
|
|
5555
|
+
const resetIso = resolveRateLimitPauseReset(logPath, billingResetIso);
|
|
4731
5556
|
const observedAt = dispatchStartedAtMs;
|
|
4732
5557
|
const prevCount = consecutiveRapidRateLimitsBySlug.get(job.slug) || 0;
|
|
4733
5558
|
const nextCount = nextRapidRateLimitCount(prevCount, { rateLimited: true, durationMs: res.durationMs });
|
|
@@ -4868,14 +5693,27 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4868
5693
|
ranInWorktree: worktree.ok,
|
|
4869
5694
|
jobSelfCommitted,
|
|
4870
5695
|
legitimateNoOp: guardIsLegitimateNoOp,
|
|
4871
|
-
isFixPlanJob:
|
|
5696
|
+
isFixPlanJob: resolveIsFixPlan(job.slug, job.isFixPlan),
|
|
4872
5697
|
verifyResult,
|
|
4873
5698
|
salvagePatch,
|
|
4874
5699
|
});
|
|
4875
5700
|
if (guardVerdict) {
|
|
4876
|
-
|
|
4877
|
-
|
|
4878
|
-
|
|
5701
|
+
// Already-satisfied-on-main exemption (PRD 1136): resolveCommitGuardOutcome
|
|
5702
|
+
// only ever touches the clean-tree/no-commit shape ('silent_no_op') —
|
|
5703
|
+
// a job that left dirty files behind is a genuine finish-protocol
|
|
5704
|
+
// violation regardless of what already landed on main, so it is
|
|
5705
|
+
// never routed through this check (see that function's own doc).
|
|
5706
|
+
const satisfyingCommits = guardVerdict.verdict === 'silent_no_op'
|
|
5707
|
+
? await findSatisfyingCommitOnMain(job)
|
|
5708
|
+
: [];
|
|
5709
|
+
const finalVerdict = resolveCommitGuardOutcome(guardVerdict, satisfyingCommits);
|
|
5710
|
+
verifyResult = finalVerdict;
|
|
5711
|
+
if (finalVerdict.verdict === 'already_satisfied_on_main') {
|
|
5712
|
+
console.log(`[scheduler] already-satisfied-on-main: ${job.slug} → completed (${finalVerdict.satisfyingSha})`);
|
|
5713
|
+
} else {
|
|
5714
|
+
const what = newlyDirty.length > 0 ? `left ${newlyDirty.length} files uncommitted` : 'made no commit on an already-clean tree';
|
|
5715
|
+
console.log(`[scheduler] commit-guard: ${job.slug} ${what} → needs_review`);
|
|
5716
|
+
}
|
|
4879
5717
|
}
|
|
4880
5718
|
}
|
|
4881
5719
|
}
|
|
@@ -4941,13 +5779,37 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4941
5779
|
);
|
|
4942
5780
|
}
|
|
4943
5781
|
|
|
5782
|
+
// BLOCKED_BY_FOREIGN_WIP claim scan: the executor exits non-zero for this
|
|
5783
|
+
// outcome (same as FAIL — see FINISH_PROTOCOL), so it is never seen by the
|
|
5784
|
+
// exit=0-only verifyRun call above; scanned here, against THIS run's own
|
|
5785
|
+
// log, before the generic non-zero-exit -> 'failed' classification below.
|
|
5786
|
+
// Read outside mutate() (I/O); actual manifest validation happens inside
|
|
5787
|
+
// mutate(), against the LIVE row's preRunDirtyPaths/carriedPaths, so a
|
|
5788
|
+
// stale local `job` snapshot can never be the source of truth for it.
|
|
5789
|
+
let foreignWipClaimedPaths = null;
|
|
5790
|
+
if (res.exitCode !== 0 && !res.rateLimited) {
|
|
5791
|
+
try {
|
|
5792
|
+
const { resultEvent, events } = parseLog(path.join(runDir, `${job.slug}.log`));
|
|
5793
|
+
if (scanSentinel(resultEvent, events) === 'blocked_by_foreign_wip') {
|
|
5794
|
+
foreignWipClaimedPaths = scanForeignWipPathsClaim(resultEvent, events);
|
|
5795
|
+
}
|
|
5796
|
+
} catch (e) {
|
|
5797
|
+
console.warn(`[scheduler] ${job.slug}: foreign-WIP verdict scan failed, falling through to ordinary failed classification`, e?.message);
|
|
5798
|
+
}
|
|
5799
|
+
}
|
|
5800
|
+
|
|
4944
5801
|
let actuallyFailed = false;
|
|
4945
5802
|
let failedJobSnapshot = null;
|
|
4946
5803
|
let needsInvestigationNow = false;
|
|
4947
5804
|
let investigationJobSnapshot = null;
|
|
5805
|
+
let investigationDeadChildSnapshot = null;
|
|
4948
5806
|
let needsReviewRcaSnapshot = null;
|
|
4949
5807
|
let resumeRecoveryJob = null;
|
|
4950
5808
|
let resumeRecoveryTarget = null;
|
|
5809
|
+
let quarantineJob = null;
|
|
5810
|
+
let quarantinePaths = null;
|
|
5811
|
+
let mechanicalRecoveryJob = null;
|
|
5812
|
+
let mechanicalRecoveryTarget = null;
|
|
4951
5813
|
let terminalNotifySnapshot = null;
|
|
4952
5814
|
const newlyCompletedPrds = [];
|
|
4953
5815
|
await mutate((s) => {
|
|
@@ -4960,7 +5822,44 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4960
5822
|
// scheduleJobTransitions.cjs's LEGAL_TRANSITIONS) and silently
|
|
4961
5823
|
// undo the cancellation. Skip — the row already reflects its real
|
|
4962
5824
|
// terminal state.
|
|
4963
|
-
|
|
5825
|
+
const finalizeDrop = evaluateFinalizeDrop({
|
|
5826
|
+
rowExists: i2 >= 0,
|
|
5827
|
+
rowStatus: i2 >= 0 ? s.jobs[i2].status : null,
|
|
5828
|
+
rowRunId: i2 >= 0 ? (s.jobs[i2].runId ?? null) : null,
|
|
5829
|
+
rowLandedCommit: i2 >= 0 ? (s.jobs[i2].landedCommit ?? null) : null,
|
|
5830
|
+
runId,
|
|
5831
|
+
landedCommit: jobLandedCommitThisRun ?? null,
|
|
5832
|
+
});
|
|
5833
|
+
if (finalizeDrop.drop) {
|
|
5834
|
+
// Never silent (2026-09-06 incident: a bare early-return here dropped
|
|
5835
|
+
// three legitimate no-op verifications of an already-shipped PRD with
|
|
5836
|
+
// no trace at all, leaving the row stuck 'pending' so the dispatcher
|
|
5837
|
+
// re-fired it three more times). 'row-not-running' covers a job
|
|
5838
|
+
// already moved off 'running' by someone else (namely
|
|
5839
|
+
// remote.cancelJob, PRD 1024) — re-finalizing anyway could
|
|
5840
|
+
// re-legalize the row via a legal failed->completed/needs_review edge
|
|
5841
|
+
// and silently undo the cancellation, so the STATUS change is still
|
|
5842
|
+
// skipped; only a genuinely-landed commit is stamped as a fact.
|
|
5843
|
+
const logFn = finalizeDrop.reason === 'row-missing' ? console.error : console.warn;
|
|
5844
|
+
logFn(
|
|
5845
|
+
`[scheduler] finalize dropped: slug=${job.slug} runId=${runId} reason=${finalizeDrop.reason} `
|
|
5846
|
+
+ `actualStatus=${i2 >= 0 ? s.jobs[i2].status : '(row-missing)'} `
|
|
5847
|
+
+ `rowRunId=${i2 >= 0 ? (s.jobs[i2].runId ?? '(none)') : '(none)'}`,
|
|
5848
|
+
);
|
|
5849
|
+
appendAuditEvent('job_finalize_dropped', {
|
|
5850
|
+
slug: job.slug,
|
|
5851
|
+
runId,
|
|
5852
|
+
reason: finalizeDrop.reason,
|
|
5853
|
+
actualStatus: i2 >= 0 ? s.jobs[i2].status : null,
|
|
5854
|
+
rowRunId: i2 >= 0 ? (s.jobs[i2].runId ?? null) : null,
|
|
5855
|
+
landedCommit: jobLandedCommitThisRun ?? null,
|
|
5856
|
+
exitCode: res.exitCode,
|
|
5857
|
+
});
|
|
5858
|
+
if (finalizeDrop.stampLandedCommit) {
|
|
5859
|
+
s.jobs[i2].landedCommit = finalizeDrop.stampLandedCommit;
|
|
5860
|
+
}
|
|
5861
|
+
return;
|
|
5862
|
+
}
|
|
4964
5863
|
if (i2 >= 0) {
|
|
4965
5864
|
const treatAsPending = res.rateLimited || (s.paused && s.paused.reason === 'rate_limit');
|
|
4966
5865
|
if (treatAsPending) {
|
|
@@ -4972,14 +5871,66 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4972
5871
|
const sigtermOverride = res.exitCode !== 0
|
|
4973
5872
|
? classifySigtermWithCommit(res.exitCode, sigtermCommitFound)
|
|
4974
5873
|
: null;
|
|
5874
|
+
// Validated against the LIVE row's own foreign-WIP manifest — never
|
|
5875
|
+
// the stale outer `job` snapshot — so an unlisted path cannot
|
|
5876
|
+
// launder a real regression into a block (PRD: give the executor a
|
|
5877
|
+
// first-class verdict for "the gate failed on a sibling's in-flight
|
|
5878
|
+
// file", but VALIDATE the claim rather than trust it).
|
|
5879
|
+
const foreignWipValidation = (!sigtermOverride && foreignWipClaimedPaths !== null)
|
|
5880
|
+
? validateForeignWipBlockClaim(foreignWipClaimedPaths, s.jobs[i2])
|
|
5881
|
+
: null;
|
|
5882
|
+
// Consecutive-block streak: cleared by default on every outcome and
|
|
5883
|
+
// only re-established below inside the validated-block branch —
|
|
5884
|
+
// so a completed run, an ordinary failure, or any other outcome
|
|
5885
|
+
// between two blocks always resets "in a row" back to zero.
|
|
5886
|
+
const priorForeignWipBlockCount = s.jobs[i2].foreignWipBlockCount ?? 0;
|
|
5887
|
+
delete s.jobs[i2].foreignWipBlockCount;
|
|
4975
5888
|
if (sigtermOverride) {
|
|
4976
5889
|
effectiveStatus = sigtermOverride.status;
|
|
4977
5890
|
sigtermOverrideReason = sigtermOverride.reason;
|
|
5891
|
+
} else if (foreignWipValidation && foreignWipValidation.ok) {
|
|
5892
|
+
const blockCount = priorForeignWipBlockCount + 1;
|
|
5893
|
+
s.jobs[i2].foreignWipBlockCount = blockCount;
|
|
5894
|
+
s.jobs[i2].blockedByForeignWip = true;
|
|
5895
|
+
s.jobs[i2].foreignWipBlockedPaths = foreignWipValidation.validPaths;
|
|
5896
|
+
if (blockCount >= FOREIGN_WIP_BLOCK_STREAK_LIMIT) {
|
|
5897
|
+
// Blocked FOREIGN_WIP_BLOCK_STREAK_LIMIT times in a row on the
|
|
5898
|
+
// same tree: auto-requeuing again would spin forever against
|
|
5899
|
+
// paths that never go clean. Park for a human instead — never
|
|
5900
|
+
// routed through the auto-fix chain (selectAutoFixTargets
|
|
5901
|
+
// excludes any job.blockedByForeignWip row), since there is no
|
|
5902
|
+
// fix-plan to author against another job's WIP.
|
|
5903
|
+
effectiveStatus = 'needs_review';
|
|
5904
|
+
s.jobs[i2].verifierVerdict = 'blocked_by_foreign_wip_streak';
|
|
5905
|
+
sigtermOverrideReason = `blocked by foreign WIP ${blockCount} times in a row on persistently-dirty path(s): ${foreignWipValidation.validPaths.join(', ')} — auto-requeue exhausted, parked for human review`;
|
|
5906
|
+
} else {
|
|
5907
|
+
// Terminal-but-retryable, same shape as the existing 'skipped'
|
|
5908
|
+
// status (never counts as failed, never enters the auto-fix
|
|
5909
|
+
// chain, reconcile()'s requeueForeignWipBlockedJobs promotes it
|
|
5910
|
+
// straight back to 'pending' once these exact paths go clean).
|
|
5911
|
+
effectiveStatus = 'skipped';
|
|
5912
|
+
sigtermOverrideReason = `blocked by foreign WIP (attempt ${blockCount}/${FOREIGN_WIP_BLOCK_STREAK_LIMIT}): ${foreignWipValidation.validPaths.join(', ')} — will auto-requeue once these path(s) are no longer dirty`;
|
|
5913
|
+
}
|
|
5914
|
+
} else if (foreignWipValidation && !foreignWipValidation.ok) {
|
|
5915
|
+
// Claimed BLOCKED_BY_FOREIGN_WIP but named a path outside this
|
|
5916
|
+
// job's own disclosed manifest (or named none at all) — downgrade
|
|
5917
|
+
// to an ordinary failure and log the offending paths loudly so the
|
|
5918
|
+
// rejection is never silent.
|
|
5919
|
+
console.warn(`[scheduler] ${job.slug}: SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP downgraded to FAIL — claimed path(s) not in this job's foreign-WIP manifest: ${foreignWipValidation.invalidPaths.join(', ') || '(no FOREIGN_WIP_PATHS line)'}`);
|
|
5920
|
+
effectiveStatus = 'failed';
|
|
5921
|
+
sigtermOverrideReason = `SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP rejected — unlisted path(s) not in the disclosed foreign-WIP manifest: ${foreignWipValidation.invalidPaths.join(', ') || '(no FOREIGN_WIP_PATHS line)'}`;
|
|
4978
5922
|
} else if (res.exitCode !== 0) {
|
|
4979
5923
|
effectiveStatus = 'failed';
|
|
4980
5924
|
} else if (
|
|
4981
5925
|
!verifyResult
|
|
4982
5926
|
|| COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict)
|
|
5927
|
+
// Already-satisfied-on-main (PRD 1136): a second, independently-
|
|
5928
|
+
// evidenced route to 'completed' alongside COMPLETED_EQUIVALENT_
|
|
5929
|
+
// VERDICTS above — kept as its own explicit check rather than
|
|
5930
|
+
// folded into that shared Set so it can never leak into
|
|
5931
|
+
// reverifyNeedsReview's or runVerify.cjs's unrelated healing
|
|
5932
|
+
// decisions, which consult that Set for a different purpose.
|
|
5933
|
+
|| verifyResult.verdict === 'already_satisfied_on_main'
|
|
4983
5934
|
) {
|
|
4984
5935
|
effectiveStatus = 'completed';
|
|
4985
5936
|
} else if (verifyResult.downgradeTo === 'pending') {
|
|
@@ -4991,7 +5942,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4991
5942
|
effectiveStatus = 'needs_review';
|
|
4992
5943
|
}
|
|
4993
5944
|
|
|
4994
|
-
|
|
5945
|
+
// already_satisfied_on_main names the satisfying sha in its own
|
|
5946
|
+
// reason — surface that on the completed row's statusHistory
|
|
5947
|
+
// instead of the generic "run finished with exit 0" every other
|
|
5948
|
+
// completed run gets.
|
|
5949
|
+
const finalizeReason = (effectiveStatus === 'completed' && verifyResult?.verdict === 'already_satisfied_on_main')
|
|
5950
|
+
? verifyResult.reason
|
|
5951
|
+
: (sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`);
|
|
5952
|
+
transitionJob(s.jobs[i2], effectiveStatus, { reason: finalizeReason, source: 'spawnJob:finalize' });
|
|
4995
5953
|
s.jobs[i2].finishedAt = new Date().toISOString();
|
|
4996
5954
|
s.jobs[i2].exitCode = res.exitCode;
|
|
4997
5955
|
s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
|
|
@@ -5000,7 +5958,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5000
5958
|
} else {
|
|
5001
5959
|
delete s.jobs[i2].salvagePatch;
|
|
5002
5960
|
}
|
|
5003
|
-
s.jobs[i2].error = effectiveStatus === 'needs_review'
|
|
5961
|
+
s.jobs[i2].error = (effectiveStatus === 'needs_review' || s.jobs[i2].blockedByForeignWip === true)
|
|
5004
5962
|
? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
|
|
5005
5963
|
// A failed job (non-zero exit) never consults verifyResult above,
|
|
5006
5964
|
// but a worktree integration failure is still worth surfacing on
|
|
@@ -5017,9 +5975,12 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5017
5975
|
s.jobs[i2].landedCommit = jobLandedCommitThisRun;
|
|
5018
5976
|
}
|
|
5019
5977
|
// Persist the verifier's verdict string so the renderer can show it.
|
|
5978
|
+
// 'blocked_by_foreign_wip_streak' is set above from the sigterm/
|
|
5979
|
+
// exit-code path, never from verifyResult (which stays null on a
|
|
5980
|
+
// non-zero exit) — never clobber it here.
|
|
5020
5981
|
if (verifyResult?.verdict && verifyResult.verdict !== 'clean') {
|
|
5021
5982
|
s.jobs[i2].verifierVerdict = verifyResult.verdict;
|
|
5022
|
-
} else {
|
|
5983
|
+
} else if (s.jobs[i2].verifierVerdict !== 'blocked_by_foreign_wip_streak') {
|
|
5023
5984
|
delete s.jobs[i2].verifierVerdict;
|
|
5024
5985
|
}
|
|
5025
5986
|
// Closed-set outcome taxonomy (issue #11 list A2) so a queue row
|
|
@@ -5042,6 +6003,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5042
6003
|
} else {
|
|
5043
6004
|
delete s.jobs[i2].uncommittedPaths;
|
|
5044
6005
|
}
|
|
6006
|
+
// Worktree merge self-healed (PRD 1125) — every blocking path was
|
|
6007
|
+
// proven byte-identical to the branch, so the duplicate was
|
|
6008
|
+
// discarded and the merge retried once, successfully. Surfaced so
|
|
6009
|
+
// the Queue UI shows a self-heal instead of an ordinary merge.
|
|
6010
|
+
if (mergeAutoResolved) {
|
|
6011
|
+
s.jobs[i2].mergeAutoResolved = mergeAutoResolved;
|
|
6012
|
+
s.jobs[i2].mergeAutoResolvedPaths = capDirtyPaths(mergeAutoResolvedPaths);
|
|
6013
|
+
} else {
|
|
6014
|
+
delete s.jobs[i2].mergeAutoResolved;
|
|
6015
|
+
delete s.jobs[i2].mergeAutoResolvedPaths;
|
|
6016
|
+
}
|
|
5045
6017
|
// Non-blocking notes (e.g. a recovered missing-dependency probe, or a
|
|
5046
6018
|
// pattern hit demoted because a materially-checkable verdict outranked
|
|
5047
6019
|
// it) — surfaced even on completed jobs so the signal isn't lost.
|
|
@@ -5092,6 +6064,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5092
6064
|
// takes the treatAsPending branch above and never reaches here).
|
|
5093
6065
|
needsReviewRcaSnapshot = { ...s.jobs[i2] };
|
|
5094
6066
|
|
|
6067
|
+
// Mechanical recovery (PRD 1130): evaluated FIRST, ahead of both
|
|
6068
|
+
// resume-first recovery and auto-fix — a job parked with a
|
|
6069
|
+
// mechanically-resolvable verdict (see
|
|
6070
|
+
// selectMechanicalRecoveryTarget) needs no model, no plan, and no
|
|
6071
|
+
// depth-cap check, so it must never fall through to either.
|
|
6072
|
+
// Snapshot only (no I/O inside mutate()); the actual git retry
|
|
6073
|
+
// happens outside mutate(), below.
|
|
6074
|
+
const mTarget = selectMechanicalRecoveryTarget(s.jobs[i2]);
|
|
6075
|
+
if (mTarget) {
|
|
6076
|
+
mechanicalRecoveryJob = { ...s.jobs[i2] };
|
|
6077
|
+
mechanicalRecoveryTarget = mTarget;
|
|
6078
|
+
} else {
|
|
5095
6079
|
// Resume-first recovery (PRD 1111): evaluated BEFORE the auto-fix
|
|
5096
6080
|
// eligibility check below — a job whose verdict is
|
|
5097
6081
|
// 'uncommitted_changes' with a live sessionId gets one bounded
|
|
@@ -5105,6 +6089,22 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5105
6089
|
resumeRecoveryJob = { ...s.jobs[i2] };
|
|
5106
6090
|
resumeRecoveryTarget = target;
|
|
5107
6091
|
} else {
|
|
6092
|
+
// Leftover quarantine (PRD 1128): resume recovery is spent
|
|
6093
|
+
// (resumeRecoveryAttempted already true) and this run STILL parked
|
|
6094
|
+
// needs_review with uncommitted_changes — the leftovers are about
|
|
6095
|
+
// to sit dirty in the shared tree forever, poisoning every later
|
|
6096
|
+
// worktree merge for this cwd. Stamp the one-attempt marker HERE,
|
|
6097
|
+
// synchronously in the same mutate as this decision (mirrors
|
|
6098
|
+
// resumeRecoveryAttempted's own stamp-before-acting rule above),
|
|
6099
|
+
// so a concurrent reverifyNeedsReview pass can never double-fire
|
|
6100
|
+
// this. The actual git work is async and runs outside mutate(),
|
|
6101
|
+
// below (performLeftoverQuarantine).
|
|
6102
|
+
const quarantineTarget = selectLeftoverQuarantineTarget(s.jobs[i2]);
|
|
6103
|
+
if (quarantineTarget) {
|
|
6104
|
+
s.jobs[i2].leftoverQuarantineAttempted = true;
|
|
6105
|
+
quarantineJob = { ...s.jobs[i2] };
|
|
6106
|
+
quarantinePaths = quarantineTarget.paths;
|
|
6107
|
+
}
|
|
5108
6108
|
// Same-tick auto-fix (feedback 2026-07-12): rather than waiting up to
|
|
5109
6109
|
// 10 min for reverifyNeedsReview()'s periodic pass, check right here
|
|
5110
6110
|
// whether this job qualifies for auto-fix (same eligibility rule
|
|
@@ -5119,8 +6119,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5119
6119
|
isEligibleForImmediateAutoFix(s.jobs[i2], s.jobs, fixSlugExists)
|
|
5120
6120
|
) {
|
|
5121
6121
|
const isRetryAttempt = s.jobs[i2].autoFixAttempted === true;
|
|
6122
|
+
const isDeadFixPlanReopen = isFixPlanDead(s.jobs[i2], s.jobs);
|
|
6123
|
+
if (isDeadFixPlanReopen) {
|
|
6124
|
+
investigationDeadChildSnapshot = s.jobs.find((x) => x.slug === fixSlugFor(s.jobs[i2])) || null;
|
|
6125
|
+
}
|
|
5122
6126
|
s.jobs[i2].autoFixAttempted = true;
|
|
5123
6127
|
if (!s.jobs[i2].runId) s.jobs[i2].runId = runId;
|
|
6128
|
+
if (isDeadFixPlanReopen) s.jobs[i2].autoFixReopened = true;
|
|
5124
6129
|
if (isRetryAttempt) {
|
|
5125
6130
|
s.jobs[i2].autoFixRetries = (s.jobs[i2].autoFixRetries ?? 0) + 1;
|
|
5126
6131
|
delete s.jobs[i2].autoFixOutcome;
|
|
@@ -5129,13 +6134,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5129
6134
|
investigationJobSnapshot = { ...s.jobs[i2] };
|
|
5130
6135
|
}
|
|
5131
6136
|
}
|
|
6137
|
+
}
|
|
5132
6138
|
}
|
|
5133
6139
|
// Auto-promote: when a fix-* PRD completes successfully, the original
|
|
5134
6140
|
// failed PRD's work is logically done. Flip its status to 'completed'
|
|
5135
6141
|
// so the cross-group failure gate in pickNextBatch releases. Without
|
|
5136
6142
|
// this, the queue stalls indefinitely behind a stale failure even
|
|
5137
6143
|
// though the auto-recovery did its job.
|
|
5138
|
-
if (effectiveStatus === 'completed' &&
|
|
6144
|
+
if (effectiveStatus === 'completed' && resolveIsFixPlan(job.slug, job.isFixPlan)) {
|
|
5139
6145
|
const orig = healTargetForFix(job.slug, s.jobs);
|
|
5140
6146
|
if (orig) {
|
|
5141
6147
|
const priorStatus = orig.status;
|
|
@@ -5207,6 +6213,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5207
6213
|
});
|
|
5208
6214
|
}
|
|
5209
6215
|
|
|
6216
|
+
if (mechanicalRecoveryJob && mechanicalRecoveryTarget) {
|
|
6217
|
+
console.log(`[scheduler] needs_review ${job.slug} → mechanical-recovery (re-integrating ${mechanicalRecoveryTarget.branch})`);
|
|
6218
|
+
performMechanicalRecovery(mechanicalRecoveryJob, mechanicalRecoveryTarget).catch((e) => {
|
|
6219
|
+
console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
|
|
6220
|
+
});
|
|
6221
|
+
}
|
|
6222
|
+
|
|
5210
6223
|
if (resumeRecoveryJob && resumeRecoveryTarget) {
|
|
5211
6224
|
console.log(`[scheduler] needs_review ${job.slug} → resume-recovery (session ${resumeRecoveryTarget.sessionId}, ${resumeRecoveryTarget.dirtyPaths.length} dirty path(s))`);
|
|
5212
6225
|
spawnResumeRecovery(resumeRecoveryJob, resumeRecoveryTarget).catch((e) => {
|
|
@@ -5214,6 +6227,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5214
6227
|
});
|
|
5215
6228
|
}
|
|
5216
6229
|
|
|
6230
|
+
if (quarantineJob && quarantinePaths) {
|
|
6231
|
+
console.log(`[scheduler] needs_review ${job.slug} → quarantining ${quarantinePaths.length} leftover path(s) (resume recovery already spent)`);
|
|
6232
|
+
performLeftoverQuarantine(quarantineJob, quarantinePaths, guardHeadBefore).catch((e) => {
|
|
6233
|
+
console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
|
|
6234
|
+
});
|
|
6235
|
+
}
|
|
6236
|
+
|
|
5217
6237
|
if (actuallyFailed && failedJobSnapshot) {
|
|
5218
6238
|
// Transient-failure detector. A 143/137 exit is ALWAYS a signal kill — the
|
|
5219
6239
|
// agent never self-exits with those — so the only question is WHO killed it.
|
|
@@ -5286,7 +6306,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5286
6306
|
}
|
|
5287
6307
|
} else if (needsInvestigationNow && investigationJobSnapshot) {
|
|
5288
6308
|
console.log(`[scheduler] needs_review ${job.slug} → immediate auto-fix investigation (not waiting for periodic reverify)`);
|
|
5289
|
-
spawnInvestigation(investigationJobSnapshot, runDir).catch((e) => {
|
|
6309
|
+
spawnInvestigation(investigationJobSnapshot, runDir, { deadChild: investigationDeadChildSnapshot }).catch((e) => {
|
|
5290
6310
|
console.error('[scheduler] spawnInvestigation error', job.slug, e);
|
|
5291
6311
|
});
|
|
5292
6312
|
}
|
|
@@ -5356,11 +6376,13 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
5356
6376
|
// ceilinged the queue at 3 while the pool the user configured said 5.
|
|
5357
6377
|
const freeSlots = sessionSlots.available();
|
|
5358
6378
|
const heldSlugs = await computeLaunchHolds(state);
|
|
6379
|
+
const satisfiedSlugsByCwd = await computeDepHistorySatisfaction(state);
|
|
5359
6380
|
const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
|
|
5360
6381
|
leaseHeld: quietMachineLease.isHeld(),
|
|
5361
6382
|
machineInUse: sessionSlots.inUse(),
|
|
5362
6383
|
now: Date.now(),
|
|
5363
6384
|
heldSlugs,
|
|
6385
|
+
satisfiedSlugsByCwd,
|
|
5364
6386
|
});
|
|
5365
6387
|
if (batch.length === 0 && freeSlots === 0) {
|
|
5366
6388
|
const snap = sessionSlots.snapshot();
|
|
@@ -5663,6 +6685,55 @@ function runQueueHealthSweep(jobs) {
|
|
|
5663
6685
|
}
|
|
5664
6686
|
}
|
|
5665
6687
|
|
|
6688
|
+
// Cross-cycle "already attempted" memory for the branch sweep (PRD 1135) —
|
|
6689
|
+
// in-memory only, keyed `"<cwd>::<branch>"`. Resets on restart, same as
|
|
6690
|
+
// consecutiveRapidRateLimitsBySlug above; the goal is "don't retry a
|
|
6691
|
+
// permanently-conflicting branch every cycle forever" while the process
|
|
6692
|
+
// stays up, not durability across a restart.
|
|
6693
|
+
const branchSweepAttempted = new Set();
|
|
6694
|
+
|
|
6695
|
+
/**
|
|
6696
|
+
* Recover any `sm-job/*` branch left stranded (unmerged, owning row terminal
|
|
6697
|
+
* or gone) across every known project — the other half of PRD 1135's
|
|
6698
|
+
* invariant alongside the reaper's liveness check above: that check stops
|
|
6699
|
+
* NEW stranding, this sweep recovers anything stranded before it (or by a
|
|
6700
|
+
* crash the liveness check can't cover). Read-only reporting plus AT MOST one
|
|
6701
|
+
* bounded integrateBranch attempt per branch per process lifetime (see
|
|
6702
|
+
* branchSweepAttempted) — never forces anything, never deletes a branch.
|
|
6703
|
+
* Hangs off the same cadence as runQueueHealthSweep. Never throws.
|
|
6704
|
+
*/
|
|
6705
|
+
async function runBranchSweep(jobs) {
|
|
6706
|
+
if (process.env.SM_BRANCH_SWEEP_DISABLE === '1') return;
|
|
6707
|
+
try {
|
|
6708
|
+
for (const cwd of allProjectCwds()) {
|
|
6709
|
+
let sweep;
|
|
6710
|
+
try {
|
|
6711
|
+
sweep = await sweepStrandedJobBranches({ cwd, jobs, attemptedBranches: branchSweepAttempted });
|
|
6712
|
+
} catch (e) {
|
|
6713
|
+
console.warn(`[scheduler] branch sweep error for ${cwd}`, e?.message);
|
|
6714
|
+
continue;
|
|
6715
|
+
}
|
|
6716
|
+
for (const r of sweep.results) {
|
|
6717
|
+
if (r.action === 'integrated') {
|
|
6718
|
+
console.log(`[scheduler] branch-sweep: merged stranded branch ${r.branch} into ${cwd} HEAD`);
|
|
6719
|
+
appendAuditEvent('branch_sweep_integrated', { cwd, branch: r.branch, slug: r.slug });
|
|
6720
|
+
} else if (r.action === 'conflict') {
|
|
6721
|
+
console.warn(`[scheduler] branch-sweep: ${r.branch} could not be integrated: ${r.integration?.reason}`);
|
|
6722
|
+
appendAuditEvent('branch_sweep_conflict', { cwd, branch: r.branch, slug: r.slug, reason: r.integration?.reason });
|
|
6723
|
+
if (r.slug) {
|
|
6724
|
+
await mutate((s) => {
|
|
6725
|
+
const j = s.jobs.find((x) => x.slug === r.slug);
|
|
6726
|
+
if (j) j.strandedBranch = r.branch;
|
|
6727
|
+
});
|
|
6728
|
+
}
|
|
6729
|
+
}
|
|
6730
|
+
}
|
|
6731
|
+
}
|
|
6732
|
+
} catch (e) {
|
|
6733
|
+
console.warn('[scheduler] branch sweep error', e?.message);
|
|
6734
|
+
}
|
|
6735
|
+
}
|
|
6736
|
+
|
|
5666
6737
|
/**
|
|
5667
6738
|
* Scan running jobs, identify those whose claude process is provably dead OR
|
|
5668
6739
|
* whose spawn never got far enough to record a runtime.pid in the first
|
|
@@ -5680,14 +6751,35 @@ async function reapDeadRunningJobs() {
|
|
|
5680
6751
|
// status:"running" with no slug left in runningSet to trigger reconciliation.
|
|
5681
6752
|
// queue.json is the source of truth for which jobs are actually running.
|
|
5682
6753
|
const state = await readQueue();
|
|
5683
|
-
const { reapable, warnings } = selectReapableJobs(state.jobs, Date.now(), {
|
|
6754
|
+
const { reapable, warnings, recovered } = selectReapableJobs(state.jobs, Date.now(), {
|
|
5684
6755
|
pidAlive: claudePidAlive,
|
|
5685
6756
|
grace: PIDLESS_SPAWN_GRACE_MS,
|
|
6757
|
+
findLiveProcess: (j) => findLiveProcessForJob(j, {
|
|
6758
|
+
worktreeDir: jobWorktree.worktreeDirFor(j.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD, j.slug),
|
|
6759
|
+
}),
|
|
5686
6760
|
});
|
|
5687
6761
|
for (const w of warnings) {
|
|
5688
6762
|
console.warn(`[scheduler] reapDeadRunningJobs: ${w.reason} slug=${w.slug} — leaving row alone`);
|
|
5689
6763
|
}
|
|
5690
6764
|
|
|
6765
|
+
// A pidless row whose process was proven alive by the /proc liveness
|
|
6766
|
+
// scan must never be terminalized (2026-09-06 incident — see
|
|
6767
|
+
// findLiveProcessForJob's header). Re-stamp the recovered pid so future
|
|
6768
|
+
// cycles see it as an ordinary live-pid row, and stop here for it.
|
|
6769
|
+
if (recovered.length) {
|
|
6770
|
+
await mutate(async (s) => {
|
|
6771
|
+
for (const r of recovered) {
|
|
6772
|
+
const idx = s.jobs.findIndex((x) => x.slug === r.slug);
|
|
6773
|
+
if (idx < 0 || s.jobs[idx].status !== 'running') continue;
|
|
6774
|
+
s.jobs[idx].runtime = { ...(s.jobs[idx].runtime || {}), pid: r.pid };
|
|
6775
|
+
}
|
|
6776
|
+
});
|
|
6777
|
+
for (const r of recovered) {
|
|
6778
|
+
console.log(`[scheduler] reapDeadRunningJobs: pid=${r.pid} recovered by /proc liveness scan for slug=${r.slug} — row stays running`);
|
|
6779
|
+
appendAuditEvent('job_pid_recovered_by_liveness_scan', { slug: r.slug, pid: r.pid });
|
|
6780
|
+
}
|
|
6781
|
+
}
|
|
6782
|
+
|
|
5691
6783
|
const dead = [];
|
|
5692
6784
|
for (const { slug, pid, pidless, reason } of reapable) {
|
|
5693
6785
|
const j = state.jobs.find((x) => x.slug === slug);
|
|
@@ -5698,15 +6790,18 @@ async function reapDeadRunningJobs() {
|
|
|
5698
6790
|
// 'no_result' → non-success below → filed as failed, never completed.
|
|
5699
6791
|
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
5700
6792
|
// A pidless reap means the spawn never got far enough to record a
|
|
5701
|
-
// pid — the gate could not possibly have run
|
|
5702
|
-
//
|
|
5703
|
-
|
|
5704
|
-
|
|
6793
|
+
// pid — the gate could not possibly have run. But `never_ran` is only
|
|
6794
|
+
// a true claim when the run dir produced no log output at all; a log
|
|
6795
|
+
// with real content proves the job DID run (see
|
|
6796
|
+
// resolvePidlessGateOutcome's header).
|
|
6797
|
+
const gateOutcome = pidless ? resolvePidlessGateOutcome(outcome, logHasOutput(logPath)) : mapOutcomeToGateOutcome(outcome);
|
|
6798
|
+
dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath });
|
|
5705
6799
|
}
|
|
5706
6800
|
|
|
5707
6801
|
queueHealthSweepCycle += 1;
|
|
5708
6802
|
if (queueHealthSweepCycle % QUEUE_HEALTH_SWEEP_EVERY_N_CYCLES === 0) {
|
|
5709
6803
|
runQueueHealthSweep(state.jobs);
|
|
6804
|
+
await runBranchSweep(state.jobs);
|
|
5710
6805
|
}
|
|
5711
6806
|
|
|
5712
6807
|
if (dead.length === 0) return;
|
|
@@ -5718,8 +6813,9 @@ async function reapDeadRunningJobs() {
|
|
|
5718
6813
|
// the same still-active rate limit — the spin loop this PRD exists to
|
|
5719
6814
|
// stop. Done once, outside mutate(), before finalizing any row below.
|
|
5720
6815
|
if (dead.some((d) => d.outcome === 'rate_limited')) {
|
|
5721
|
-
const resetIso = await refreshNextReset().catch(() => cachedNextReset);
|
|
5722
6816
|
const triggering = dead.find((d) => d.outcome === 'rate_limited');
|
|
6817
|
+
const billingResetIso = await refreshNextReset().catch(() => cachedNextReset);
|
|
6818
|
+
const resetIso = resolveRateLimitPauseReset(triggering.logPath, billingResetIso);
|
|
5723
6819
|
const triggeringRow = triggering ? state.jobs.find((x) => x.slug === triggering.slug) : null;
|
|
5724
6820
|
const observedAtMs = triggeringRow?.startedAt ? Date.parse(triggeringRow.startedAt) : null;
|
|
5725
6821
|
// Same rapid-repeat circuit breaker spawnJob's own res.rateLimited
|
|
@@ -5739,6 +6835,64 @@ async function reapDeadRunningJobs() {
|
|
|
5739
6835
|
await setPaused('rate_limit', resetIso, { observedAt: observedAtMs, force: forceHardPause });
|
|
5740
6836
|
}
|
|
5741
6837
|
|
|
6838
|
+
// Prove integration BEFORE entering mutate() (PRD 1133): the check below
|
|
6839
|
+
// shells out to git — including a full `git fetch --all --prune` (up to
|
|
6840
|
+
// 20s) via computeCommittedDuringRun's committedInWindow, plus up to two
|
|
6841
|
+
// 2s retry sleeps when nothing landed — for every dead job that needs the
|
|
6842
|
+
// in-place fallback. mutate() serializes through ONE global mutateTail
|
|
6843
|
+
// promise chain shared by every project's dispatch/admin/cancel mutation,
|
|
6844
|
+
// same reason the rate-limit handling above already runs outside it — so
|
|
6845
|
+
// doing this git work inside the mutate() callback would stall the whole
|
|
6846
|
+
// scheduler's queue writes for the sum of these calls across every dead
|
|
6847
|
+
// job in the batch (e.g. an app-restart reap that dead-letters several
|
|
6848
|
+
// running jobs at once).
|
|
6849
|
+
const integrationResults = new Map();
|
|
6850
|
+
for (const d of dead) {
|
|
6851
|
+
if (d.outcome !== 'success') continue;
|
|
6852
|
+
const row = state.jobs.find((x) => x.slug === d.slug);
|
|
6853
|
+
if (!row) continue;
|
|
6854
|
+
const rowCwd = row.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD;
|
|
6855
|
+
if (!isGitRepoSync(rowCwd)) continue;
|
|
6856
|
+
const branch = jobWorktree.branchNameFor(d.slug);
|
|
6857
|
+
const integrated = await isBranchAlreadyIntegrated(rowCwd, branch);
|
|
6858
|
+
const headNow = await gitHead(rowCwd);
|
|
6859
|
+
const guardHeadBeforeVal = row.guardHeadBefore || null;
|
|
6860
|
+
let landedCommit = null;
|
|
6861
|
+
let notLandedInfo = null;
|
|
6862
|
+
let effectiveSuccess = true;
|
|
6863
|
+
if (integrated === null) {
|
|
6864
|
+
// No sm-job/<slug> branch — an in-place run (or worktree isolation
|
|
6865
|
+
// disabled). Prove landing via the exact same HEAD-advance evidence
|
|
6866
|
+
// the live commit-guard already uses — reused, not re-derived.
|
|
6867
|
+
const committed = await computeCommittedDuringRun(
|
|
6868
|
+
rowCwd, guardHeadBeforeVal, headNow, row.startedAt, new Date().toISOString(),
|
|
6869
|
+
);
|
|
6870
|
+
if (committed) {
|
|
6871
|
+
if (guardHeadBeforeVal && headNow && headNow !== guardHeadBeforeVal) landedCommit = headNow;
|
|
6872
|
+
} else {
|
|
6873
|
+
effectiveSuccess = false;
|
|
6874
|
+
notLandedInfo = {
|
|
6875
|
+
verdict: 'reaped_without_integration',
|
|
6876
|
+
reason: 'no commit landed during the run window — HEAD never advanced',
|
|
6877
|
+
};
|
|
6878
|
+
}
|
|
6879
|
+
} else if (integrated === true) {
|
|
6880
|
+
if (guardHeadBeforeVal && headNow && headNow !== guardHeadBeforeVal) landedCommit = headNow;
|
|
6881
|
+
} else {
|
|
6882
|
+
// Branch exists and still holds commits never merged into rowCwd's
|
|
6883
|
+
// HEAD — the exact PRD 1118 shape. Never merge it here (see
|
|
6884
|
+
// isBranchAlreadyIntegrated's header comment); park for review
|
|
6885
|
+
// instead, naming the branch so the work is easy to find and land by
|
|
6886
|
+
// hand or via mechanical recovery.
|
|
6887
|
+
effectiveSuccess = false;
|
|
6888
|
+
notLandedInfo = {
|
|
6889
|
+
verdict: 'reaped_without_integration',
|
|
6890
|
+
reason: `worktree branch ${branch} still holds unintegrated work — never merged into ${rowCwd}`,
|
|
6891
|
+
};
|
|
6892
|
+
}
|
|
6893
|
+
integrationResults.set(d.slug, { effectiveSuccess, landedCommit, notLandedInfo });
|
|
6894
|
+
}
|
|
6895
|
+
|
|
5742
6896
|
await mutate(async (s) => {
|
|
5743
6897
|
for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
|
|
5744
6898
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
@@ -5777,12 +6931,38 @@ async function reapDeadRunningJobs() {
|
|
|
5777
6931
|
console.error(`[scheduler] reapDeadRunningJobs: in-place salvage failed for ${slug}`, e);
|
|
5778
6932
|
}
|
|
5779
6933
|
}
|
|
6934
|
+
// Prove integration before this reap is allowed to say 'completed'
|
|
6935
|
+
// (PRD 1133): a reaped job's owning process vanished before
|
|
6936
|
+
// spawnJob's own post-run integration/commit-guard ever ran, so
|
|
6937
|
+
// 'outcome=success' alone (a clean result event in the log) is not
|
|
6938
|
+
// proof anything actually landed — the two live incidents this PRD
|
|
6939
|
+
// exists for (1118, starry-night-ships 224) both had exactly that
|
|
6940
|
+
// shape. Only evaluated for a genuinely successful, non-rate-limited
|
|
6941
|
+
// outcome; a non-git cwd (isGitRepoSync false) skips this entirely,
|
|
6942
|
+
// preserving today's behaviour exactly. The actual git work already
|
|
6943
|
+
// ran ABOVE, before this mutate() call, into integrationResults — see
|
|
6944
|
+
// that block's own header comment for why it must not run in here.
|
|
6945
|
+
let effectiveSuccess = success;
|
|
6946
|
+
let landedCommit = null;
|
|
6947
|
+
let notLandedInfo = null;
|
|
6948
|
+
if (success && !rateLimited) {
|
|
6949
|
+
const ir = integrationResults.get(slug);
|
|
6950
|
+
if (ir) {
|
|
6951
|
+
effectiveSuccess = ir.effectiveSuccess;
|
|
6952
|
+
landedCommit = ir.landedCommit;
|
|
6953
|
+
notLandedInfo = ir.notLandedInfo;
|
|
6954
|
+
}
|
|
6955
|
+
}
|
|
6956
|
+
|
|
5780
6957
|
const leftoverSuffix = deltaPaths && deltaPaths.length
|
|
5781
6958
|
? ` — left ${deltaPaths.length} files uncommitted`
|
|
5782
6959
|
: '';
|
|
6960
|
+
const baseReason = notLandedInfo
|
|
6961
|
+
? `reaped: ${notLandedInfo.reason}`
|
|
6962
|
+
: (pidless ? reason : `reaped: process gone (outcome=${outcome})`);
|
|
5783
6963
|
const transitionReason = rateLimited
|
|
5784
6964
|
? `reaped: rate limit detected — reset to pending, not failed (outcome=${outcome})${leftoverSuffix}`
|
|
5785
|
-
:
|
|
6965
|
+
: baseReason + leftoverSuffix;
|
|
5786
6966
|
|
|
5787
6967
|
if (rateLimited) {
|
|
5788
6968
|
// Retryable, never terminal (PRD 1117) — same resetJobFields path
|
|
@@ -5791,11 +6971,18 @@ async function reapDeadRunningJobs() {
|
|
|
5791
6971
|
// paused-for-rate-limit reset: fresh runId/startedAt/exitCode.
|
|
5792
6972
|
resetJobFields(s.jobs[idx], transitionReason, { source: 'reapDeadRunningJobs:rate-limit' });
|
|
5793
6973
|
} else {
|
|
5794
|
-
|
|
5795
|
-
s.jobs[idx]
|
|
6974
|
+
const targetStatus = effectiveSuccess ? 'completed' : (notLandedInfo ? 'needs_review' : 'failed');
|
|
6975
|
+
transitionJob(s.jobs[idx], targetStatus, { reason: transitionReason, source: 'reapDeadRunningJobs' });
|
|
6976
|
+
s.jobs[idx].exitCode = effectiveSuccess ? 0 : (s.jobs[idx].exitCode ?? 1);
|
|
5796
6977
|
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
5797
|
-
s.jobs[idx].error =
|
|
6978
|
+
s.jobs[idx].error = effectiveSuccess ? null : `${transitionReason} (outcome=${outcome})`;
|
|
5798
6979
|
s.jobs[idx].gateOutcome = gateOutcome;
|
|
6980
|
+
if (notLandedInfo) {
|
|
6981
|
+
s.jobs[idx].verifierVerdict = notLandedInfo.verdict;
|
|
6982
|
+
} else {
|
|
6983
|
+
delete s.jobs[idx].verifierVerdict;
|
|
6984
|
+
}
|
|
6985
|
+
if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
|
|
5799
6986
|
}
|
|
5800
6987
|
delete s.jobs[idx].runtime;
|
|
5801
6988
|
delete s.jobs[idx].guardBaseline;
|
|
@@ -6025,10 +7212,16 @@ const MAX_INVESTIGATION_DEPTH = 1;
|
|
|
6025
7212
|
* no recorded investigationDepth (a job already in the queue before this
|
|
6026
7213
|
* depth tracking shipped) is treated as excluded too, preserving the
|
|
6027
7214
|
* pre-existing blanket-exclusion behavior for legacy jobs — no retroactive
|
|
6028
|
-
* migration. Non-fix-plan
|
|
7215
|
+
* migration. Non-fix-plan jobs are never capped here.
|
|
7216
|
+
*
|
|
7217
|
+
* `isFixPlan` (PRD 1131) is the job's own persisted classification stamp
|
|
7218
|
+
* (see lib/fixPlanSlug.cjs's resolveIsFixPlan) — an explicit true/false wins
|
|
7219
|
+
* over the slug; only a row with the field entirely absent (persisted
|
|
7220
|
+
* before this change shipped) falls back to the legacy slug-only heuristic.
|
|
7221
|
+
* Exported for tests.
|
|
6029
7222
|
*/
|
|
6030
|
-
function isFixPlanBeyondDepthCap(slug, investigationDepth) {
|
|
6031
|
-
if (!
|
|
7223
|
+
function isFixPlanBeyondDepthCap(slug, investigationDepth, isFixPlan) {
|
|
7224
|
+
if (!resolveIsFixPlan(slug, isFixPlan)) return false;
|
|
6032
7225
|
if (investigationDepth == null) return true;
|
|
6033
7226
|
return investigationDepth >= MAX_INVESTIGATION_DEPTH + 1;
|
|
6034
7227
|
}
|
|
@@ -6081,12 +7274,18 @@ function isUnresolvableNeedsReview(job, { hasRunDir }) {
|
|
|
6081
7274
|
* ('no-plan', 'error', and unstamped/undefined) — mirrors the retry
|
|
6082
7275
|
* eligibility rule in selectAutoFixTargets so a job can never be retry-
|
|
6083
7276
|
* eligible there and simultaneously un-annotatable here.
|
|
7277
|
+
*
|
|
7278
|
+
* A parent stamped `autoFixReopened: true` (its dead fix-plan child earned
|
|
7279
|
+
* it exactly one further attempt — see isFixPlanDead) is a separate
|
|
7280
|
+
* exhaustion path: it is spent as soon as that second investigation
|
|
7281
|
+
* concludes with ANY outcome, including another 'plan' — a reopened parent
|
|
7282
|
+
* never gets a third attempt, so unlike the fresh case a 'plan' outcome does
|
|
7283
|
+
* not exempt it here.
|
|
6084
7284
|
*/
|
|
6085
7285
|
function isExhaustedAutoFix(job) {
|
|
6086
|
-
|
|
6087
|
-
|
|
6088
|
-
|
|
6089
|
-
&& (job.autoFixRetries ?? 0) >= 1;
|
|
7286
|
+
if (!job || job.status !== 'needs_review' || job.autoFixAttempted !== true) return false;
|
|
7287
|
+
if (job.autoFixReopened === true) return job.autoFixOutcome != null;
|
|
7288
|
+
return job.autoFixOutcome !== 'plan' && (job.autoFixRetries ?? 0) >= 1;
|
|
6090
7289
|
}
|
|
6091
7290
|
|
|
6092
7291
|
/**
|
|
@@ -6102,6 +7301,31 @@ function isPlanUnqueued(job, queuedSlugs) {
|
|
|
6102
7301
|
return !queuedSlugs.has(fixSlugFor(job));
|
|
6103
7302
|
}
|
|
6104
7303
|
|
|
7304
|
+
// Terminal-and-not-completed statuses a fix-plan child can die in — see
|
|
7305
|
+
// isFixPlanDead.
|
|
7306
|
+
const DEAD_FIX_CHILD_STATUSES = new Set(['needs_review', 'failed', 'quarantined', 'skipped']);
|
|
7307
|
+
|
|
7308
|
+
/**
|
|
7309
|
+
* Pure predicate: a parent stuck at outcome 'plan' whose own fix-plan child
|
|
7310
|
+
* (fixSlugFor(job)) has ITSELF died — reached a terminal non-completed
|
|
7311
|
+
* status — with nothing left in the ladder that will ever revisit either
|
|
7312
|
+
* row again (selectAutoFixTargets skips a 'plan' outcome outright, and
|
|
7313
|
+
* isPlanUnqueued only fires when the child never reached the queue at all,
|
|
7314
|
+
* which isn't true once a dead child row exists). `job.autoFixReopened`
|
|
7315
|
+
* gates this to exactly once per parent — once stamped, this always returns
|
|
7316
|
+
* false so the parent can never be reopened a second time. Exported for
|
|
7317
|
+
* tests.
|
|
7318
|
+
*/
|
|
7319
|
+
function isFixPlanDead(job, jobsInProject) {
|
|
7320
|
+
if (!job || job.status !== 'needs_review') return false;
|
|
7321
|
+
if (job.autoFixOutcome !== 'plan') return false;
|
|
7322
|
+
if (job.autoFixReopened === true) return false;
|
|
7323
|
+
const fixSlug = fixSlugFor(job);
|
|
7324
|
+
const child = (jobsInProject || []).find((j) => j.slug === fixSlug);
|
|
7325
|
+
if (!child) return false;
|
|
7326
|
+
return DEAD_FIX_CHILD_STATUSES.has(child.status);
|
|
7327
|
+
}
|
|
7328
|
+
|
|
6105
7329
|
/**
|
|
6106
7330
|
* Pure predicate: is this job eligible for the boot re-verify self-heal? Only
|
|
6107
7331
|
* needs_review jobs with a run log (own or backfilled via resolveRunId) AND a
|
|
@@ -6238,6 +7462,11 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
|
|
|
6238
7462
|
const slugsInQueue = new Set(jobs.map((j) => j.slug));
|
|
6239
7463
|
return jobs.filter((job) => {
|
|
6240
7464
|
if (job.status !== 'needs_review') return false;
|
|
7465
|
+
// A job parked here because it was blocked by a sibling's foreign WIP
|
|
7466
|
+
// three times in a row (never its own regression) has nothing for a
|
|
7467
|
+
// fix-plan investigation to diagnose — there is no code defect to
|
|
7468
|
+
// author a PRD against, only another job's still-uncommitted tree.
|
|
7469
|
+
if (job.blockedByForeignWip === true) return false;
|
|
6241
7470
|
// A stale re-run whose work already shipped (rcaReport's 'already-shipped'
|
|
6242
7471
|
// class) must never buy a fix-plan PRD — there is nothing to fix, and the
|
|
6243
7472
|
// correct recovery (archiving the PRD) is a human/reconcile action, not
|
|
@@ -6247,16 +7476,30 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
|
|
|
6247
7476
|
// bounded `--resume` attempt must never also become a fix-plan target
|
|
6248
7477
|
// in the same pass — see spawnInvestigation's own identical guard.
|
|
6249
7478
|
if (selectResumeRecoveryTarget(job)) return false;
|
|
7479
|
+
// Mechanical recovery (PRD 1130): a job eligible for a pure-git retry
|
|
7480
|
+
// must never also become a fix-plan target — it needs no plan and no
|
|
7481
|
+
// model. Defensive: today's single mechanically-resolvable verdict
|
|
7482
|
+
// (worktree_integration_failed) is already excluded below via the depth
|
|
7483
|
+
// cap, but this must hold even if that stops being true.
|
|
7484
|
+
if (selectMechanicalRecoveryTarget(job)) return false;
|
|
6250
7485
|
const runId = job.runId || resolveJobRunId(job);
|
|
6251
7486
|
if (!runId) return false;
|
|
6252
|
-
if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth)) return false;
|
|
7487
|
+
if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth, job.isFixPlan)) return false;
|
|
7488
|
+
// A dead fix-plan child (PRD 1129) earns its parent exactly one further
|
|
7489
|
+
// attempt, bypassing the normal 'plan' exclusion and the fix-slug/queue
|
|
7490
|
+
// membership checks below — those checks exist to stop a FRESH
|
|
7491
|
+
// investigation from clobbering a live sibling, but here the sibling is
|
|
7492
|
+
// dead and reusing its slug is the whole point of the reopen.
|
|
7493
|
+
const dead = isFixPlanDead(job, jobs);
|
|
6253
7494
|
if (job.autoFixAttempted) {
|
|
6254
|
-
const retryEligible =
|
|
7495
|
+
const retryEligible = dead
|
|
7496
|
+
|| job.autoFixOutcome === 'no-plan'
|
|
6255
7497
|
|| job.autoFixOutcome === 'error'
|
|
6256
7498
|
|| job.autoFixOutcome == null;
|
|
6257
7499
|
if (!retryEligible) return false;
|
|
6258
|
-
if ((job.autoFixRetries ?? 0) >= 1) return false;
|
|
7500
|
+
if (!dead && (job.autoFixRetries ?? 0) >= 1) return false;
|
|
6259
7501
|
}
|
|
7502
|
+
if (dead) return true;
|
|
6260
7503
|
const fixSlug = fixSlugFor(job);
|
|
6261
7504
|
if (fixSlugExists(fixSlug)) return false;
|
|
6262
7505
|
if (slugsInQueue.has(fixSlug)) return false;
|
|
@@ -6366,7 +7609,15 @@ async function reverifyNeedsReview() {
|
|
|
6366
7609
|
// Still needs_review after the existing heal pass — widen the evidence
|
|
6367
7610
|
// window before giving up on it entirely (unchanged heal semantics for
|
|
6368
7611
|
// rows that already qualified above; this only adds an annotation).
|
|
6369
|
-
|
|
7612
|
+
// Skipped when a fix-plan investigation was already minted for this row
|
|
7613
|
+
// (job.autoFixAttempted) — PRD 1136: 'looks done, confirm before
|
|
7614
|
+
// archiving' and 'a -fix- child is already investigating this' are two
|
|
7615
|
+
// different claims about the SAME evidence, and stamping both leaves a
|
|
7616
|
+
// human reading two contradictory signals off one row. autoFixAttempted
|
|
7617
|
+
// is stamped synchronously in spawnJob's same-tick auto-fix branch,
|
|
7618
|
+
// always before this periodic/boot pass can run against the same row, so
|
|
7619
|
+
// this check reliably catches the only order that can occur.
|
|
7620
|
+
if (stillOpen && job.autoFixAttempted !== true) {
|
|
6370
7621
|
const looksDone = await computeLooksDone(job);
|
|
6371
7622
|
if (looksDone) {
|
|
6372
7623
|
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
|
|
@@ -6435,7 +7686,7 @@ async function reverifyNeedsReview() {
|
|
|
6435
7686
|
const promotedPrds = [];
|
|
6436
7687
|
await mutate((s) => {
|
|
6437
7688
|
for (const job of s.jobs) {
|
|
6438
|
-
if (job.status !== 'completed' || !
|
|
7689
|
+
if (job.status !== 'completed' || !resolveIsFixPlan(job.slug, job.isFixPlan)) continue;
|
|
6439
7690
|
const orig = healTargetForFix(job.slug, s.jobs);
|
|
6440
7691
|
if (!orig) continue;
|
|
6441
7692
|
const priorStatus = orig.status;
|
|
@@ -6519,6 +7770,23 @@ async function reverifyNeedsReview() {
|
|
|
6519
7770
|
? await readQueue()
|
|
6520
7771
|
: afterHealForAnnotate;
|
|
6521
7772
|
|
|
7773
|
+
// Mechanical recovery (PRD 1130): evaluated first, ahead of both
|
|
7774
|
+
// resume-first recovery and auto-fix below — catches a job whose
|
|
7775
|
+
// mechanically-resolvable verdict this periodic pass finds still eligible
|
|
7776
|
+
// (e.g. one already parked before this rung shipped, or one the same-tick
|
|
7777
|
+
// check in spawnJob missed because the app restarted in between). Depth
|
|
7778
|
+
// never disqualifies it, so it runs regardless of investigationDepth.
|
|
7779
|
+
{
|
|
7780
|
+
for (const job of queueForResumeAndAutofix.jobs) {
|
|
7781
|
+
const target = selectMechanicalRecoveryTarget(job);
|
|
7782
|
+
if (!target) continue;
|
|
7783
|
+
console.log(`[scheduler] mechanical-recovery: needs_review ${job.slug} → re-integrating ${target.branch}`);
|
|
7784
|
+
performMechanicalRecovery(job, target).catch((e) => {
|
|
7785
|
+
console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
|
|
7786
|
+
});
|
|
7787
|
+
}
|
|
7788
|
+
}
|
|
7789
|
+
|
|
6522
7790
|
// Resume-first recovery (PRD 1111): before any fix-plan investigation is
|
|
6523
7791
|
// authored below, offer the bounded one-attempt `--resume` dispatch to any
|
|
6524
7792
|
// needs_review job this periodic pass finds still eligible — e.g. one the
|
|
@@ -6537,6 +7805,26 @@ async function reverifyNeedsReview() {
|
|
|
6537
7805
|
}
|
|
6538
7806
|
}
|
|
6539
7807
|
|
|
7808
|
+
// Leftover quarantine (PRD 1128), periodic pass: catches a job parked
|
|
7809
|
+
// needs_review with resume recovery already spent BEFORE this feature
|
|
7810
|
+
// shipped, or one the same-tick check in spawnJob missed because the app
|
|
7811
|
+
// restarted in between. Stamps the one-attempt marker in its own mutate
|
|
7812
|
+
// BEFORE the async git work starts (same race-closing rule as the resume
|
|
7813
|
+
// loop above and spawnJob's own dispatch stamp).
|
|
7814
|
+
{
|
|
7815
|
+
for (const job of queueForResumeAndAutofix.jobs) {
|
|
7816
|
+
const quarantineTarget = selectLeftoverQuarantineTarget(job);
|
|
7817
|
+
if (!quarantineTarget) continue;
|
|
7818
|
+
console.log(`[scheduler] leftover-quarantine: needs_review ${job.slug} → quarantining ${quarantineTarget.paths.length} leftover path(s)`);
|
|
7819
|
+
mutate((s) => {
|
|
7820
|
+
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
7821
|
+
if (j) j.leftoverQuarantineAttempted = true;
|
|
7822
|
+
}).then(() => performLeftoverQuarantine(job, quarantineTarget.paths)).catch((e) => {
|
|
7823
|
+
console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
|
|
7824
|
+
});
|
|
7825
|
+
}
|
|
7826
|
+
}
|
|
7827
|
+
|
|
6540
7828
|
// Auto-fix: spawn a fix-plan investigation for each job still in
|
|
6541
7829
|
// needs_review after the heal pass (kill-switch: SM_AUTOFIX_DISABLE=1).
|
|
6542
7830
|
// spawnInvestigation early-returns once investigationsInFlight reaches
|
|
@@ -6550,23 +7838,30 @@ async function reverifyNeedsReview() {
|
|
|
6550
7838
|
const runId = job.runId || resolveRunId(job);
|
|
6551
7839
|
const runDir = path.join(RUNS_DIR, runId);
|
|
6552
7840
|
const isRetryAttempt = job.autoFixAttempted === true;
|
|
7841
|
+
const isDeadFixPlanReopen = isFixPlanDead(job, queueForResumeAndAutofix.jobs);
|
|
7842
|
+
const deadChild = isDeadFixPlanReopen
|
|
7843
|
+
? queueForResumeAndAutofix.jobs.find((j) => j.slug === fixSlugFor(job))
|
|
7844
|
+
: null;
|
|
6553
7845
|
// Persist the attempt BEFORE spawning — a crash mid-investigation still
|
|
6554
7846
|
// counts it (mirrors orphanRetries). Safe even when the slot is busy: the
|
|
6555
7847
|
// investigation is queued and drained as slots free, so it is genuinely
|
|
6556
|
-
// attempted rather than silently dropped.
|
|
7848
|
+
// attempted rather than silently dropped. autoFixReopened is stamped in
|
|
7849
|
+
// this SAME mutate so a crash between selection and dispatch can never
|
|
7850
|
+
// leave the parent re-eligible for a second reopen (PRD 1129).
|
|
6557
7851
|
await mutate((s) => {
|
|
6558
7852
|
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
6559
7853
|
if (j) {
|
|
6560
7854
|
j.autoFixAttempted = true;
|
|
6561
7855
|
if (!j.runId && runId) j.runId = runId;
|
|
7856
|
+
if (isDeadFixPlanReopen) j.autoFixReopened = true;
|
|
6562
7857
|
if (isRetryAttempt) {
|
|
6563
7858
|
j.autoFixRetries = (j.autoFixRetries ?? 0) + 1;
|
|
6564
7859
|
delete j.autoFixOutcome;
|
|
6565
7860
|
}
|
|
6566
7861
|
}
|
|
6567
7862
|
});
|
|
6568
|
-
console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'})`);
|
|
6569
|
-
spawnInvestigation(job, runDir).catch((e) => {
|
|
7863
|
+
console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'}${isDeadFixPlanReopen ? ', dead fix-plan child reopen' : ''})`);
|
|
7864
|
+
spawnInvestigation(job, runDir, { deadChild }).catch((e) => {
|
|
6570
7865
|
console.error('[scheduler] auto-fix spawnInvestigation error', job.slug, e);
|
|
6571
7866
|
});
|
|
6572
7867
|
}
|
|
@@ -7657,6 +8952,36 @@ const remote = {
|
|
|
7657
8952
|
return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
|
|
7658
8953
|
}
|
|
7659
8954
|
|
|
8955
|
+
// Write-time FK check for a patched dependsOn (PRD 1124), reusing the
|
|
8956
|
+
// SAME resolution rule scheduler_create_prd's prdCreate.cjs applies (exact
|
|
8957
|
+
// slug, else bare-name after stripping one leading `NN-`) so update and
|
|
8958
|
+
// create can never disagree about what a dependsOn entry resolves to. An
|
|
8959
|
+
// explicit empty array CLEARS the dependency and skips validation — there
|
|
8960
|
+
// is nothing to resolve. A listPrds() read failure is skipped-with-a-
|
|
8961
|
+
// warning, matching createPrd's tolerance for an I/O hiccup.
|
|
8962
|
+
if (frontmatter && Array.isArray(frontmatter.dependsOn) && frontmatter.dependsOn.length) {
|
|
8963
|
+
let listing;
|
|
8964
|
+
try {
|
|
8965
|
+
listing = await this.listPrds({ cwd, limit: Number.MAX_SAFE_INTEGER });
|
|
8966
|
+
} catch (e) {
|
|
8967
|
+
console.warn(`[scheduler] updatePrd: dependsOn validation skipped (listPrds failed): ${e?.message ?? e}`);
|
|
8968
|
+
listing = null;
|
|
8969
|
+
}
|
|
8970
|
+
if (listing) {
|
|
8971
|
+
const candidateSlugs = (listing.prds ?? []).map((p) => p.slug);
|
|
8972
|
+
for (const dep of frontmatter.dependsOn) {
|
|
8973
|
+
if (resolveDepSlug(dep, candidateSlugs).length > 0) continue;
|
|
8974
|
+
const near = findNearMatches(dep, candidateSlugs);
|
|
8975
|
+
const suggestion = near.length ? ` Closest existing slug(s): ${near.join(', ')}.` : '';
|
|
8976
|
+
return {
|
|
8977
|
+
ok: false,
|
|
8978
|
+
error: `dependsOn entry "${dep}" does not resolve to any existing PRD in this project.${suggestion} ` +
|
|
8979
|
+
'Pass the bare name (preferred) or the exact NN-prefixed slug of an existing PRD.',
|
|
8980
|
+
};
|
|
8981
|
+
}
|
|
8982
|
+
}
|
|
8983
|
+
}
|
|
8984
|
+
|
|
7660
8985
|
let dir = null;
|
|
7661
8986
|
let filePath = null;
|
|
7662
8987
|
if (cwd) {
|
|
@@ -7787,4 +9112,162 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
7787
9112
|
});
|
|
7788
9113
|
}
|
|
7789
9114
|
|
|
7790
|
-
module.exports = {
|
|
9115
|
+
module.exports = {
|
|
9116
|
+
classifyQueueStarvation,
|
|
9117
|
+
runQueueStarvationWatchdog,
|
|
9118
|
+
QUEUE_STARVATION_MS,
|
|
9119
|
+
computeBlockedChains,
|
|
9120
|
+
stripAppOwnedChurn,
|
|
9121
|
+
findOverrunningJobs,
|
|
9122
|
+
JOB_OVERRUN_FACTOR,
|
|
9123
|
+
JOB_OVERRUN_FLOOR_MS,
|
|
9124
|
+
registerScheduleHandlers,
|
|
9125
|
+
attachWindow,
|
|
9126
|
+
init,
|
|
9127
|
+
ROOT,
|
|
9128
|
+
PRDS_DIR,
|
|
9129
|
+
healRefusalReason,
|
|
9130
|
+
writeQueue,
|
|
9131
|
+
reconcile,
|
|
9132
|
+
reconcileSourcePromptId,
|
|
9133
|
+
allocateParallelGroup,
|
|
9134
|
+
selectHistoryJobs,
|
|
9135
|
+
parsePorcelain,
|
|
9136
|
+
FINISH_PROTOCOL,
|
|
9137
|
+
IDLE_OUTPUT_KILL_MS,
|
|
9138
|
+
BASH_DEFAULT_TIMEOUT_MS,
|
|
9139
|
+
BASH_MAX_TIMEOUT_MS,
|
|
9140
|
+
remote,
|
|
9141
|
+
pickNextBatch,
|
|
9142
|
+
pickForProject,
|
|
9143
|
+
reapDeadRunningJobs,
|
|
9144
|
+
runBranchSweep,
|
|
9145
|
+
pollRecoveryClearSource,
|
|
9146
|
+
memoryLimitedBatchSize,
|
|
9147
|
+
availableForJobs,
|
|
9148
|
+
reverifyNeedsReview,
|
|
9149
|
+
isRescanCandidate,
|
|
9150
|
+
isFailedUnverifiedShaped,
|
|
9151
|
+
computeLooksDone,
|
|
9152
|
+
isPromotableOriginal,
|
|
9153
|
+
selectAutoFixTargets,
|
|
9154
|
+
applyRcaClassification,
|
|
9155
|
+
isEligibleForImmediateAutoFix,
|
|
9156
|
+
resolveRunId,
|
|
9157
|
+
isUnresolvableNeedsReview,
|
|
9158
|
+
isExhaustedAutoFix,
|
|
9159
|
+
isPlanUnqueued,
|
|
9160
|
+
isFixPlanDead,
|
|
9161
|
+
fixSlugFor,
|
|
9162
|
+
healTargetForFix,
|
|
9163
|
+
buildInvestigationPrompt,
|
|
9164
|
+
isGitRepoSync,
|
|
9165
|
+
committedInWindow,
|
|
9166
|
+
computeCommittedDuringRun,
|
|
9167
|
+
classifySigtermWithCommit,
|
|
9168
|
+
evaluateFinalizeDrop,
|
|
9169
|
+
evaluateDispatchSidecarReconcile,
|
|
9170
|
+
isQueueRowRegression,
|
|
9171
|
+
readRunOutcomeSidecars,
|
|
9172
|
+
isFixPlanSlug,
|
|
9173
|
+
classifyDiscoveredFixPlan,
|
|
9174
|
+
resolveIsFixPlan,
|
|
9175
|
+
isFixPlanBeyondDepthCap,
|
|
9176
|
+
MAX_INVESTIGATION_DEPTH,
|
|
9177
|
+
forceTickOutcome,
|
|
9178
|
+
applyPauseCleared,
|
|
9179
|
+
detectNetworkErrorInLog,
|
|
9180
|
+
detectRateLimitInLog,
|
|
9181
|
+
classifyFailureOutcome,
|
|
9182
|
+
commitGuardVerdict,
|
|
9183
|
+
findSatisfyingCommitOnMain,
|
|
9184
|
+
leftoverFieldsFrom,
|
|
9185
|
+
applyLeftoverFields,
|
|
9186
|
+
LEFTOVER_PATHS_CAP,
|
|
9187
|
+
capDirtyPaths,
|
|
9188
|
+
buildForeignWipSection,
|
|
9189
|
+
PRE_RUN_DIRTY_PATHS_CAP,
|
|
9190
|
+
FOREIGN_WIP_DELIMITER,
|
|
9191
|
+
FOREIGN_WIP_END_DELIMITER,
|
|
9192
|
+
TRANSIENT_RETRY_CAP,
|
|
9193
|
+
buildScheduleStatePayload,
|
|
9194
|
+
partitionBootOrphans,
|
|
9195
|
+
applyOrphanOutcome,
|
|
9196
|
+
BOOT_ORPHAN_KILL_GRACE_MS,
|
|
9197
|
+
registerAdminRoutes,
|
|
9198
|
+
notifyOriginatingTab,
|
|
9199
|
+
notifyNeedsReview,
|
|
9200
|
+
isNotifiableTerminalStatus,
|
|
9201
|
+
extractResultTextFromLog,
|
|
9202
|
+
candidatePrdsDirs,
|
|
9203
|
+
candidateArchivedPrdsDirs,
|
|
9204
|
+
resolveArchivedPrdStatus,
|
|
9205
|
+
prdDirForCwd,
|
|
9206
|
+
prdPathForJob,
|
|
9207
|
+
archivedPrdPathForJob,
|
|
9208
|
+
archivedTwinExists,
|
|
9209
|
+
findPrdDir,
|
|
9210
|
+
resolveVerifyPrdPath,
|
|
9211
|
+
resolveFixPlanPath,
|
|
9212
|
+
resolveNotifyPrd,
|
|
9213
|
+
runPrdMigration,
|
|
9214
|
+
consolidateAllFlatPrds,
|
|
9215
|
+
shouldSkipInvestigationForCleanRun,
|
|
9216
|
+
archiveCompletedPrd,
|
|
9217
|
+
retireCompletedSlugs,
|
|
9218
|
+
SCHEDULER_BOOTED_AT,
|
|
9219
|
+
SCHEDULER_CODE_SHA,
|
|
9220
|
+
resetJobFields,
|
|
9221
|
+
executeJob,
|
|
9222
|
+
prdArchivedSkipResult,
|
|
9223
|
+
spawnJob,
|
|
9224
|
+
listPrdsInternal,
|
|
9225
|
+
computeStallSummary,
|
|
9226
|
+
findStaleQuarantinedJobs,
|
|
9227
|
+
QUARANTINE_ESCALATE_MS,
|
|
9228
|
+
applyClearQueueVictims,
|
|
9229
|
+
PIDLESS_SPAWN_GRACE_MS,
|
|
9230
|
+
findStrandedInvestigations,
|
|
9231
|
+
INVESTIGATION_MAX_MS,
|
|
9232
|
+
stashList,
|
|
9233
|
+
parseStashLine,
|
|
9234
|
+
pathsChangedSince,
|
|
9235
|
+
restoreSpecificStash,
|
|
9236
|
+
evaluateSharedTreeGuard,
|
|
9237
|
+
checkSharedTreeGuard,
|
|
9238
|
+
uncommittedChanges,
|
|
9239
|
+
gitHead,
|
|
9240
|
+
isBranchAlreadyIntegrated,
|
|
9241
|
+
selectResumeRecoveryTarget,
|
|
9242
|
+
buildResumeRecoveryPreamble,
|
|
9243
|
+
buildClaudeSpawnArgs,
|
|
9244
|
+
spawnResumeRecovery,
|
|
9245
|
+
selectMechanicalRecoveryTarget,
|
|
9246
|
+
performMechanicalRecovery,
|
|
9247
|
+
MECHANICALLY_RESOLVABLE_VERDICTS,
|
|
9248
|
+
selectLeftoverQuarantineTarget,
|
|
9249
|
+
quarantineLeftovers,
|
|
9250
|
+
performLeftoverQuarantine,
|
|
9251
|
+
spawnInvestigation,
|
|
9252
|
+
computeLaunchHolds,
|
|
9253
|
+
computeDepHistorySatisfaction,
|
|
9254
|
+
handleLaunchFailure,
|
|
9255
|
+
applyLaunchFailure,
|
|
9256
|
+
setPaused,
|
|
9257
|
+
clearPause,
|
|
9258
|
+
tickQueue,
|
|
9259
|
+
runDueJobs,
|
|
9260
|
+
isCooldownSuppressed,
|
|
9261
|
+
nextRapidRateLimitCount,
|
|
9262
|
+
CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
|
|
9263
|
+
RAPID_RATE_LIMIT_WINDOW_MS,
|
|
9264
|
+
MANUAL_PAUSE_COOLDOWN_MS,
|
|
9265
|
+
RUNS_DIR,
|
|
9266
|
+
pickRunDir,
|
|
9267
|
+
resolveRateLimitPauseReset,
|
|
9268
|
+
computeEffectiveResumeAt,
|
|
9269
|
+
computeResumeDelay,
|
|
9270
|
+
FOREIGN_WIP_BLOCK_STREAK_LIMIT,
|
|
9271
|
+
validateForeignWipBlockClaim,
|
|
9272
|
+
requeueForeignWipBlockedJobs,
|
|
9273
|
+
};
|