claude-code-session-manager 0.79.0 → 0.80.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-COtVRqBR.js → AgentLibrary-zS3jw_1e.js} +1 -1
- package/dist/assets/{DataModel-CSEKw_OR.js → DataModel-Cy_vxTpi.js} +1 -1
- package/dist/assets/{History-CHHovrAO.js → History-C6JRuqfT.js} +1 -1
- package/dist/assets/{Hooks-BZU6C3x6.js → Hooks-BafPy9mB.js} +1 -1
- package/dist/assets/{HostBilko-CqTUoq37.js → HostBilko-BZwhQOFt.js} +1 -1
- package/dist/assets/{Library-BtxdyTLz.js → Library-C8JDDliz.js} +1 -1
- package/dist/assets/{ListDetail-qZc7Zm-6.js → ListDetail-CqiOdwLc.js} +1 -1
- package/dist/assets/{MarkdownEditor-BHe_4fJR.js → MarkdownEditor-CyLyP67L.js} +1 -1
- package/dist/assets/{McpServers-7Z98HLNo.js → McpServers-BzMv-_84.js} +1 -1
- package/dist/assets/{Memory-CR72KoyP.js → Memory-DSBYQdJR.js} +1 -1
- package/dist/assets/{Panel-pL6H3dpQ.js → Panel-CLUhkNNA.js} +1 -1
- package/dist/assets/{Permissions-CWSWjyXM.js → Permissions-BfC2-HN4.js} +1 -1
- package/dist/assets/{Plugins-CN6lX2lt.js → Plugins-BKi40jT5.js} +2 -2
- package/dist/assets/{ProvenanceBadge-BXSXwIsk.js → ProvenanceBadge-BzFw4KhD.js} +1 -1
- package/dist/assets/{SaveBar-BlB5TGpR.js → SaveBar-avk2p9jv.js} +1 -1
- package/dist/assets/{Scheduler-DRciWUmR.js → Scheduler-Bf_6MdJo.js} +7 -7
- package/dist/assets/{ScopeSwitcher-kFrXtjpr.js → ScopeSwitcher-C-RwYUVZ.js} +1 -1
- package/dist/assets/{Settings-BXuyf4lJ.js → Settings-Djd8OoBA.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-Dfacb0PE.js → SkillReferenceGraph-DuogY6s7.js} +1 -1
- package/dist/assets/{Skills-CHqcpiyt.js → Skills-D_qAqxZ_.js} +1 -1
- package/dist/assets/{SystemPrompt-fxXm0BZr.js → SystemPrompt-DbHFLQV3.js} +1 -1
- package/dist/assets/{TagLibrary-DOz65ZTz.js → TagLibrary-C2y91BT0.js} +1 -1
- package/dist/assets/{TiptapBody-D0bWx_9o.js → TiptapBody-D9iz4xQx.js} +1 -1
- package/dist/assets/{Toggle-C9jBwGSx.js → Toggle-BGnFL2E5.js} +1 -1
- package/dist/assets/{index-DPYa6jbM.js → index-_2ARyFDj.js} +4 -4
- package/dist/assets/{settingsSchema-BTPw1bR3.js → settingsSchema-JK15eJU8.js} +1 -1
- package/dist/index.html +1 -1
- package/package.json +1 -1
- package/plugins/session-manager-dev/skills/builder/3-publish/SKILL.md +10 -0
- package/scripts/project-pages-logic/dist/logic.cjs +12 -12
- package/scripts/render-project-pages/dist/renderer.cjs +22 -22
- package/src/main/__tests__/computeDepHistorySatisfaction.test.cjs +66 -0
- package/src/main/__tests__/prdCreate.test.cjs +133 -8
- package/src/main/__tests__/prdFrontmatterDependsOn.test.cjs +136 -0
- package/src/main/__tests__/prdUpdateDependsOn.test.cjs +160 -0
- package/src/main/__tests__/queueHistory.test.cjs +33 -0
- package/src/main/__tests__/scheduleJobTransitions.test.cjs +1 -0
- package/src/main/__tests__/scheduler-autofix-outcome.test.cjs +73 -1
- package/src/main/__tests__/scheduler-autofix-select.test.cjs +17 -0
- package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +20 -0
- package/src/main/__tests__/scheduler-leftover-quarantine.test.cjs +199 -0
- package/src/main/__tests__/scheduler-mechanical-recovery.test.cjs +222 -0
- package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +51 -0
- package/src/main/__tests__/scheduler-resume-recovery.test.cjs +254 -0
- package/src/main/__tests__/schedulerBatchRootBlocker.test.cjs +117 -0
- package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -2
- package/src/main/ipcSchemas.cjs +15 -1
- package/src/main/lib/__tests__/gitWorktree.test.cjs +129 -10
- package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +59 -7
- package/src/main/lib/depSlugResolve.cjs +72 -0
- package/src/main/lib/epicWorktreeMerge.cjs +3 -3
- package/src/main/lib/epicWorktreeMint.cjs +17 -5
- package/src/main/lib/fixPlanSlug.cjs +62 -0
- package/src/main/lib/gitWorktree.cjs +97 -12
- package/src/main/lib/mcpToolCatalog.cjs +4 -1
- package/src/main/lib/prdCreate.cjs +84 -5
- package/src/main/lib/prdFrontmatter.cjs +56 -8
- package/src/main/lib/queueHistory.cjs +50 -5
- package/src/main/lib/scheduleJobTransitions.cjs +12 -2
- package/src/main/lib/schedulerBatch.cjs +181 -23
- package/src/main/scheduler/prdParser.cjs +7 -0
- package/src/main/scheduler.cjs +813 -40
- package/src/preload/api.d.ts +8 -0
package/src/main/scheduler.cjs
CHANGED
|
@@ -73,6 +73,7 @@ const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs')
|
|
|
73
73
|
const promptSessionTranscript = require('./promptSessionTranscript.cjs');
|
|
74
74
|
const { verifyRun } = require('./runVerify.cjs');
|
|
75
75
|
const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
|
|
76
|
+
const { isFixPlanSlug, classifyDiscoveredFixPlan, resolveIsFixPlan } = require('./lib/fixPlanSlug.cjs');
|
|
76
77
|
const { landedSinceRun } = require('./lib/landedSinceRun.cjs');
|
|
77
78
|
const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
|
|
78
79
|
const logs = require('./logs.cjs');
|
|
@@ -99,7 +100,7 @@ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
|
|
|
99
100
|
const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
|
|
100
101
|
? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
|
|
101
102
|
: JOB_OVERRUN_FLOOR_MS_DEFAULT;
|
|
102
|
-
const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
|
|
103
|
+
const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD, DEP_HISTORY_FAIL_OPEN } = require('./lib/schedulerBatch.cjs');
|
|
103
104
|
const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
|
|
104
105
|
const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
|
|
105
106
|
const queueHistory = require('./lib/queueHistory.cjs');
|
|
@@ -140,6 +141,7 @@ const jobWorktree = require('./lib/jobWorktree.cjs');
|
|
|
140
141
|
const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
|
|
141
142
|
const queueStore = require('./lib/queueStore.cjs');
|
|
142
143
|
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
144
|
+
const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
|
|
143
145
|
const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
|
|
144
146
|
const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
|
|
145
147
|
|
|
@@ -2072,11 +2074,24 @@ async function reconcile(state) {
|
|
|
2072
2074
|
exitCode: null,
|
|
2073
2075
|
error: null,
|
|
2074
2076
|
};
|
|
2075
|
-
//
|
|
2076
|
-
//
|
|
2077
|
-
//
|
|
2078
|
-
//
|
|
2079
|
-
|
|
2077
|
+
// Fix-plan classification (PRD 1131): a freshly-discovered PRD is a
|
|
2078
|
+
// genuine scheduler-authored fix plan only when its OWN provenance says
|
|
2079
|
+
// so — an explicit isFixPlan:true stamp (spawnInvestigation's prompt
|
|
2080
|
+
// template) or the absence of any createdVia stamp at all (legacy
|
|
2081
|
+
// fallback, matching the "no provenance = trust the name" rule the
|
|
2082
|
+
// quarantine gate below already applies) — never merely because the
|
|
2083
|
+
// slug looks like one. See lib/fixPlanSlug.cjs's header for why (PRD
|
|
2084
|
+
// 1126: a scheduler_create_prd-authored PRD whose slug happened to start
|
|
2085
|
+
// with "fix-" was wrongly stamped investigationDepth before it ever ran).
|
|
2086
|
+
// Persisted onto the queue row so every later consumer
|
|
2087
|
+
// (commitGuardVerdict, isFixPlanBeyondDepthCap, the fix-plan-completion
|
|
2088
|
+
// checks) reads this stamp instead of re-deriving it from the name.
|
|
2089
|
+
entry.isFixPlan = classifyDiscoveredFixPlan(p, slug);
|
|
2090
|
+
// Stamp investigationDepth relative to the original job it heals, so
|
|
2091
|
+
// selectAutoFixTargets/spawnInvestigation can bound the fix-of-a-fix
|
|
2092
|
+
// recursion (see MAX_INVESTIGATION_DEPTH). Non-fix-plan jobs get no
|
|
2093
|
+
// explicit field — they read as depth 1 via `?? 1`.
|
|
2094
|
+
if (entry.isFixPlan) {
|
|
2080
2095
|
const parent = healTargetForFix(slug, state.jobs);
|
|
2081
2096
|
entry.investigationDepth = parent ? (parent.investigationDepth ?? 1) + 1 : 2;
|
|
2082
2097
|
}
|
|
@@ -2088,8 +2103,9 @@ async function reconcile(state) {
|
|
|
2088
2103
|
// guard-prd-writes.cjs PreToolUse hook should have denied. Fix-plan PRDs
|
|
2089
2104
|
// are exempt: spawnInvestigation's own probe writes them directly by
|
|
2090
2105
|
// design (a trusted, scheduler-spawned internal loop, not an
|
|
2091
|
-
// agent/human authoring a PRD)
|
|
2092
|
-
//
|
|
2106
|
+
// agent/human authoring a PRD) — entry.isFixPlan (just classified above)
|
|
2107
|
+
// is the provenance-aware verdict for that exemption now, not a raw
|
|
2108
|
+
// isFixPlanSlug name check.
|
|
2093
2109
|
//
|
|
2094
2110
|
// Quarantine is loud and reversible, never a silent skip (see the
|
|
2095
2111
|
// 2026-08-01 23-PRD outage this file's header references for what a
|
|
@@ -2098,7 +2114,7 @@ async function reconcile(state) {
|
|
|
2098
2114
|
// (schedule:adopt-prd) that stamps the file via the same update-prd API
|
|
2099
2115
|
// route the MCP tool uses — reconcile()'s adopt path above promotes it
|
|
2100
2116
|
// to 'pending' on the very next pass, within one tick of being stamped.
|
|
2101
|
-
if (!p.createdVia && !
|
|
2117
|
+
if (!p.createdVia && !entry.isFixPlan) {
|
|
2102
2118
|
entry.status = 'quarantined';
|
|
2103
2119
|
// Stamped at creation (not via transitionJob, since this is a
|
|
2104
2120
|
// brand-new row minted directly at 'quarantined' rather than
|
|
@@ -2584,6 +2600,17 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
2584
2600
|
delete job.verifierVerdict;
|
|
2585
2601
|
delete job.uncommittedPaths;
|
|
2586
2602
|
delete job.resumeRecoveryAttempted;
|
|
2603
|
+
// Same one-attempt-per-episode category as resumeRecoveryAttempted above —
|
|
2604
|
+
// a re-fired row must be able to earn a fresh mechanical-recovery attempt
|
|
2605
|
+
// if it parks needs_review again (PRD 1130).
|
|
2606
|
+
delete job.mechanicalRecoveryAttempted;
|
|
2607
|
+
// Quarantine (PRD 1128) is scoped to THIS run's episode exactly like
|
|
2608
|
+
// resumeRecoveryAttempted above — a re-fired row must be able to earn a
|
|
2609
|
+
// fresh quarantine attempt if it parks needs_review again.
|
|
2610
|
+
delete job.leftoverQuarantineAttempted;
|
|
2611
|
+
delete job.quarantinedTo;
|
|
2612
|
+
delete job.quarantinedCommit;
|
|
2613
|
+
delete job.quarantinedPaths;
|
|
2587
2614
|
// Same "this run's outcome, not durable across a reset" category as the
|
|
2588
2615
|
// fields above — a stale 'archive' recoveryAction from a prior life of this
|
|
2589
2616
|
// slug must never survive a reset and silently exclude a genuinely-new
|
|
@@ -2592,6 +2619,10 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
2592
2619
|
// can otherwise linger forever when RCA is disabled or errors).
|
|
2593
2620
|
delete job.rcaFailureClass;
|
|
2594
2621
|
delete job.rcaRecoveryAction;
|
|
2622
|
+
// Same "this run's outcome, not durable across a reset" category — a
|
|
2623
|
+
// human-driven reset must genuinely start the auto-fix budget over,
|
|
2624
|
+
// including the one-time dead-fix-plan-child reopen (PRD 1129).
|
|
2625
|
+
delete job.autoFixReopened;
|
|
2595
2626
|
// Like exitCode: this run's outcome, not durable across a reset — a stale
|
|
2596
2627
|
// leak badge from a prior attempt must not linger once the job re-fires.
|
|
2597
2628
|
delete job.leakedDescendants;
|
|
@@ -3303,6 +3334,302 @@ As the LAST LINE of your final result text, emit exactly one of:
|
|
|
3303
3334
|
Print PASS only once the commit above has actually landed.`;
|
|
3304
3335
|
}
|
|
3305
3336
|
|
|
3337
|
+
/**
|
|
3338
|
+
* Mechanical recovery (PRD 1130). isFixPlanBeyondDepthCap (below) is the
|
|
3339
|
+
* ONLY gate on re-investigating a fix-plan job at investigationDepth >= 2 —
|
|
3340
|
+
* correct for open-ended "author another plan" recursion, but it also
|
|
3341
|
+
* strands a depth-capped job whose failure was fully mechanical (no
|
|
3342
|
+
* judgement required) with no other ladder rung, since resume-first recovery
|
|
3343
|
+
* (selectResumeRecoveryTarget above) is hard-gated on verdict
|
|
3344
|
+
* 'uncommitted_changes'. This rung is evaluated INDEPENDENTLY of
|
|
3345
|
+
* isFixPlanBeyondDepthCap — depth never disqualifies it, because unlike
|
|
3346
|
+
* auto-fix it authors no plan and spawns no model; it is pure git.
|
|
3347
|
+
*
|
|
3348
|
+
* The closed set of mechanically-resolvable verdicts starts at exactly
|
|
3349
|
+
* 'worktree_integration_failed': PRD 1125 already taught integrateBranch to
|
|
3350
|
+
* parse git's "would be overwritten by merge" stderr, verify the blocking
|
|
3351
|
+
* paths are byte-identical to the branch, discard the proven duplicates, and
|
|
3352
|
+
* retry the merge once. A job parked with this verdict has its `sm-job/
|
|
3353
|
+
* <slug>` branch preserved (integrateJobBranch never deletes the branch on
|
|
3354
|
+
* failure — see cleanupJobWorktree's `keepBranch: !integration.ok`), so a
|
|
3355
|
+
* plain re-call of integrateBranch against that same branch inherits PRD
|
|
3356
|
+
* 1125's auto-resolution for free — no re-implementation needed here.
|
|
3357
|
+
*
|
|
3358
|
+
* Bounded to exactly one attempt via job.mechanicalRecoveryAttempted,
|
|
3359
|
+
* stamped in the SAME mutate as the outcome (performMechanicalRecovery,
|
|
3360
|
+
* below) — never here — so this selector alone can be unit-tested exactly
|
|
3361
|
+
* like selectResumeRecoveryTarget/selectLeftoverQuarantineTarget.
|
|
3362
|
+
*
|
|
3363
|
+
* Kill-switch: SM_MECHANICAL_RECOVERY_DISABLE=1 restores today's behaviour
|
|
3364
|
+
* exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
|
|
3365
|
+
*/
|
|
3366
|
+
const MECHANICALLY_RESOLVABLE_VERDICTS = new Set(['worktree_integration_failed']);
|
|
3367
|
+
|
|
3368
|
+
function selectMechanicalRecoveryTarget(job) {
|
|
3369
|
+
if (process.env.SM_MECHANICAL_RECOVERY_DISABLE === '1') return null;
|
|
3370
|
+
if (!job || job.status !== 'needs_review') return null;
|
|
3371
|
+
if (!MECHANICALLY_RESOLVABLE_VERDICTS.has(job.verifierVerdict)) return null;
|
|
3372
|
+
if (job.mechanicalRecoveryAttempted === true) return null;
|
|
3373
|
+
const cwd = job.cwd || DEFAULT_PROJECT_CWD;
|
|
3374
|
+
return { slug: job.slug, cwd, branch: jobWorktree.branchNameFor(job.slug), carriedPaths: job.carriedPaths || [] };
|
|
3375
|
+
}
|
|
3376
|
+
|
|
3377
|
+
/**
|
|
3378
|
+
* Perform an already-selected mechanical recovery (selectMechanicalRecoveryTarget
|
|
3379
|
+
* above) — a direct re-attempt of integrateBranch against the job's preserved
|
|
3380
|
+
* branch, never a fresh `claude -p` dispatch. On success the job transitions
|
|
3381
|
+
* needs_review -> completed and its verifierVerdict is cleared; the branch,
|
|
3382
|
+
* now merged, is deleted like any other successfully-integrated job branch.
|
|
3383
|
+
* On failure (including a branch that no longer exists — already deleted or
|
|
3384
|
+
* already merged) the job stays needs_review, mechanicalRecoveryAttempted is
|
|
3385
|
+
* stamped, and the retry's own failure text is appended to `error`. Either
|
|
3386
|
+
* way mechanicalRecoveryAttempted is stamped in this SAME mutate, so a crash
|
|
3387
|
+
* between the git call returning and this mutate landing simply repeats an
|
|
3388
|
+
* idempotent git operation on the next pass rather than leaving the job
|
|
3389
|
+
* re-eligible forever.
|
|
3390
|
+
*/
|
|
3391
|
+
async function performMechanicalRecovery(job, target) {
|
|
3392
|
+
const integration = await jobWorktree.integrateJobBranch({
|
|
3393
|
+
cwd: target.cwd, branch: target.branch, slug: target.slug, carriedPaths: target.carriedPaths,
|
|
3394
|
+
});
|
|
3395
|
+
if (integration.ok) {
|
|
3396
|
+
await jobWorktree.cleanupJobWorktree({ cwd: target.cwd, dir: undefined, branch: target.branch, keepBranch: false });
|
|
3397
|
+
}
|
|
3398
|
+
let becameCompleted = false;
|
|
3399
|
+
await mutate((s) => {
|
|
3400
|
+
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
3401
|
+
if (!j) return;
|
|
3402
|
+
j.mechanicalRecoveryAttempted = true;
|
|
3403
|
+
if (integration.ok) {
|
|
3404
|
+
if (transitionJob(j, 'completed', {
|
|
3405
|
+
reason: `mechanical recovery: ${target.branch} re-integrated successfully`,
|
|
3406
|
+
source: 'scheduler:mechanicalRecovery',
|
|
3407
|
+
})) {
|
|
3408
|
+
delete j.verifierVerdict;
|
|
3409
|
+
j.exitCode = 0;
|
|
3410
|
+
j.error = null;
|
|
3411
|
+
becameCompleted = true;
|
|
3412
|
+
}
|
|
3413
|
+
} else {
|
|
3414
|
+
const pointer = `Mechanical recovery retry failed: ${integration.reason}`;
|
|
3415
|
+
j.error = j.error ? `${j.error}\n${pointer}` : pointer;
|
|
3416
|
+
}
|
|
3417
|
+
});
|
|
3418
|
+
if (integration.ok) {
|
|
3419
|
+
console.log(`[scheduler] mechanical-recovery: ${job.slug} → completed (branch ${target.branch} re-integrated)`);
|
|
3420
|
+
if (becameCompleted) await archiveCompletedPrd(job.slug, job.cwd);
|
|
3421
|
+
} else {
|
|
3422
|
+
console.error(`[scheduler] mechanical-recovery: ${job.slug} → retry failed: ${integration.reason}`);
|
|
3423
|
+
}
|
|
3424
|
+
}
|
|
3425
|
+
|
|
3426
|
+
/**
|
|
3427
|
+
* Leftover quarantine (PRD 1128). Resume-first recovery gets exactly one
|
|
3428
|
+
* `--resume` attempt (selectResumeRecoveryTarget above); when that attempt
|
|
3429
|
+
* ALSO parks needs_review with 'uncommitted_changes', the leftovers are
|
|
3430
|
+
* about to sit dirty in the SHARED tree forever — git then refuses any later
|
|
3431
|
+
* worktree merge for this cwd that would overwrite them, turning one parked
|
|
3432
|
+
* job into a project-wide stall (216-jupiter-sand-kazekage, 2026-09-06).
|
|
3433
|
+
* Pure/no I/O, mirroring selectResumeRecoveryTarget so the eligibility rule
|
|
3434
|
+
* is unit-testable directly.
|
|
3435
|
+
*
|
|
3436
|
+
* Bounded to exactly one attempt via job.leftoverQuarantineAttempted, stamped
|
|
3437
|
+
* synchronously by the caller in the SAME mutate as this decision (never
|
|
3438
|
+
* here) — see spawnJob's finalize and reverifyNeedsReview's periodic pass.
|
|
3439
|
+
*
|
|
3440
|
+
* Kill-switch: SM_LEFTOVER_QUARANTINE_DISABLE=1 restores today's behaviour
|
|
3441
|
+
* exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
|
|
3442
|
+
*/
|
|
3443
|
+
function selectLeftoverQuarantineTarget(job) {
|
|
3444
|
+
if (process.env.SM_LEFTOVER_QUARANTINE_DISABLE === '1') return null;
|
|
3445
|
+
if (!job || job.status !== 'needs_review') return null;
|
|
3446
|
+
if (job.verifierVerdict !== 'uncommitted_changes') return null;
|
|
3447
|
+
if (job.resumeRecoveryAttempted !== true) return null;
|
|
3448
|
+
if (job.leftoverQuarantineAttempted === true) return null;
|
|
3449
|
+
const uncommittedPaths = Array.isArray(job.uncommittedPaths)
|
|
3450
|
+
? job.uncommittedPaths.filter((p) => typeof p === 'string' && p.length > 0)
|
|
3451
|
+
: [];
|
|
3452
|
+
if (!uncommittedPaths.length) return null;
|
|
3453
|
+
// The single most important constraint: never touch a path that was
|
|
3454
|
+
// ALREADY dirty at this run's own dispatch time (preRunDirtyPaths) — that
|
|
3455
|
+
// is foreign WIP (a human's or a sibling's), not this job's own leftover.
|
|
3456
|
+
const preRunDirty = new Set(Array.isArray(job.preRunDirtyPaths) ? job.preRunDirtyPaths : []);
|
|
3457
|
+
const paths = uncommittedPaths.filter((p) => !preRunDirty.has(p));
|
|
3458
|
+
if (!paths.length) return null;
|
|
3459
|
+
return { slug: job.slug, cwd: job.cwd, paths };
|
|
3460
|
+
}
|
|
3461
|
+
|
|
3462
|
+
function execGitAt(cwd, args, { env, timeout = 20_000 } = {}) {
|
|
3463
|
+
return new Promise((resolve, reject) => {
|
|
3464
|
+
execFile(
|
|
3465
|
+
'git',
|
|
3466
|
+
['-C', cwd, ...args],
|
|
3467
|
+
{ timeout, windowsHide: true, encoding: 'utf8', env: env ? { ...process.env, ...env } : process.env },
|
|
3468
|
+
(err, stdout, stderr) => {
|
|
3469
|
+
if (err) {
|
|
3470
|
+
err.stderrText = stderr;
|
|
3471
|
+
reject(err);
|
|
3472
|
+
return;
|
|
3473
|
+
}
|
|
3474
|
+
resolve(stdout || '');
|
|
3475
|
+
},
|
|
3476
|
+
);
|
|
3477
|
+
});
|
|
3478
|
+
}
|
|
3479
|
+
|
|
3480
|
+
async function pathExistsInTree(cwd, treeish, p) {
|
|
3481
|
+
try {
|
|
3482
|
+
await execGitAt(cwd, ['cat-file', '-e', `${treeish}:${p}`]);
|
|
3483
|
+
return true;
|
|
3484
|
+
} catch {
|
|
3485
|
+
return false;
|
|
3486
|
+
}
|
|
3487
|
+
}
|
|
3488
|
+
|
|
3489
|
+
/**
|
|
3490
|
+
* Commit exactly `paths` (must already be dirty on disk) onto a dedicated
|
|
3491
|
+
* `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
|
|
3492
|
+
* unavailable) via a THROWAWAY `GIT_INDEX_FILE` — never touches the live
|
|
3493
|
+
* index, never moves the checked-out branch — then restores those paths to
|
|
3494
|
+
* match that baseline commit's tree, so the shared working tree returns to
|
|
3495
|
+
* its pre-run state. This is deliberately NOT `git stash` (the destructive-
|
|
3496
|
+
* git guard blocks stash on a shared tree, and a stash nobody restores
|
|
3497
|
+
* strands the work invisibly — see standards.md).
|
|
3498
|
+
*
|
|
3499
|
+
* Never throws: any git failure, or a non-git cwd, aborts the WHOLE attempt
|
|
3500
|
+
* with the tree untouched (no partial restore) — restore only ever runs
|
|
3501
|
+
* after the salvage ref/commit has safely landed, so a failure there leaves
|
|
3502
|
+
* the data recoverable from the ref even though the tree stayed dirty.
|
|
3503
|
+
* A path no longer dirty on disk (already committed, or reverted since) is
|
|
3504
|
+
* skipped, never force-restored.
|
|
3505
|
+
*/
|
|
3506
|
+
async function quarantineLeftovers({ cwd, slug, paths, headBefore }) {
|
|
3507
|
+
if (!cwd || !slug || !Array.isArray(paths) || paths.length === 0) {
|
|
3508
|
+
return { ok: false, reason: 'no cwd/slug/paths given' };
|
|
3509
|
+
}
|
|
3510
|
+
let baseline = headBefore || null;
|
|
3511
|
+
try {
|
|
3512
|
+
if (!baseline) {
|
|
3513
|
+
baseline = (await execGitAt(cwd, ['rev-parse', 'HEAD'])).trim();
|
|
3514
|
+
}
|
|
3515
|
+
if (!baseline) return { ok: false, reason: 'could not resolve a baseline commit (non-git cwd?)' };
|
|
3516
|
+
|
|
3517
|
+
const dirtyNowRaw = await execGitAt(cwd, ['status', '--porcelain', '--', ...paths]);
|
|
3518
|
+
const dirtyNow = new Set(parsePorcelain(dirtyNowRaw));
|
|
3519
|
+
const toQuarantine = paths.filter((p) => dirtyNow.has(p));
|
|
3520
|
+
const skippedPaths = paths.filter((p) => !dirtyNow.has(p));
|
|
3521
|
+
if (!toQuarantine.length) {
|
|
3522
|
+
return { ok: true, ref: null, commit: null, quarantinedPaths: [], skippedPaths };
|
|
3523
|
+
}
|
|
3524
|
+
|
|
3525
|
+
const tmpIndex = path.join(os.tmpdir(), `sm-salvage-index-${slug}-${process.pid}-${Date.now()}`);
|
|
3526
|
+
const env = { GIT_INDEX_FILE: tmpIndex };
|
|
3527
|
+
let treeSha;
|
|
3528
|
+
let commitSha;
|
|
3529
|
+
try {
|
|
3530
|
+
await execGitAt(cwd, ['read-tree', baseline], { env });
|
|
3531
|
+
for (const p of toQuarantine) {
|
|
3532
|
+
if (fs.existsSync(path.join(cwd, p))) {
|
|
3533
|
+
await execGitAt(cwd, ['add', '--', p], { env });
|
|
3534
|
+
} else {
|
|
3535
|
+
await execGitAt(cwd, ['rm', '--cached', '--ignore-unmatch', '--', p], { env });
|
|
3536
|
+
}
|
|
3537
|
+
}
|
|
3538
|
+
treeSha = (await execGitAt(cwd, ['write-tree'], { env })).trim();
|
|
3539
|
+
commitSha = (await execGitAt(cwd, ['commit-tree', treeSha, '-p', baseline, '-m', `salvage: leftover changes from ${slug}`], { env })).trim();
|
|
3540
|
+
} catch (e) {
|
|
3541
|
+
return { ok: false, reason: `git command failed while building the salvage commit: ${(e && (e.stderrText || e.message)) || e}` };
|
|
3542
|
+
} finally {
|
|
3543
|
+
await fsp.rm(tmpIndex, { force: true }).catch(() => {});
|
|
3544
|
+
}
|
|
3545
|
+
|
|
3546
|
+
const ref = `sm-salvage/${slug}`;
|
|
3547
|
+
try {
|
|
3548
|
+
await execGitAt(cwd, ['update-ref', `refs/heads/${ref}`, commitSha]);
|
|
3549
|
+
} catch (e) {
|
|
3550
|
+
return { ok: false, reason: `git command failed updating ${ref}: ${(e && (e.stderrText || e.message)) || e}` };
|
|
3551
|
+
}
|
|
3552
|
+
|
|
3553
|
+
// The salvage commit is safely landed at this point — a failure from here
|
|
3554
|
+
// on is reported with the ref/commit still attached so nothing looks lost
|
|
3555
|
+
// even if the tree itself couldn't be fully restored.
|
|
3556
|
+
try {
|
|
3557
|
+
const inBaseline = [];
|
|
3558
|
+
const notInBaseline = [];
|
|
3559
|
+
for (const p of toQuarantine) {
|
|
3560
|
+
// eslint-disable-next-line no-await-in-loop
|
|
3561
|
+
if (await pathExistsInTree(cwd, baseline, p)) inBaseline.push(p); else notInBaseline.push(p);
|
|
3562
|
+
}
|
|
3563
|
+
if (inBaseline.length) {
|
|
3564
|
+
await execGitAt(cwd, ['checkout', baseline, '--', ...inBaseline]);
|
|
3565
|
+
}
|
|
3566
|
+
if (notInBaseline.length) {
|
|
3567
|
+
await execGitAt(cwd, ['reset', '--', ...notInBaseline]).catch(() => {});
|
|
3568
|
+
for (const p of notInBaseline) {
|
|
3569
|
+
// eslint-disable-next-line no-await-in-loop
|
|
3570
|
+
await fsp.rm(path.join(cwd, p), { force: true });
|
|
3571
|
+
}
|
|
3572
|
+
}
|
|
3573
|
+
} catch (e) {
|
|
3574
|
+
return {
|
|
3575
|
+
ok: false,
|
|
3576
|
+
ref,
|
|
3577
|
+
commit: commitSha,
|
|
3578
|
+
reason: `salvage commit landed at ${ref} (${commitSha}) but restoring the working tree failed: ${(e && (e.stderrText || e.message)) || e}`,
|
|
3579
|
+
};
|
|
3580
|
+
}
|
|
3581
|
+
|
|
3582
|
+
return { ok: true, ref, commit: commitSha, quarantinedPaths: toQuarantine, skippedPaths };
|
|
3583
|
+
} catch (e) {
|
|
3584
|
+
return { ok: false, reason: `git command failed: ${(e && (e.stderrText || e.message)) || e}` };
|
|
3585
|
+
}
|
|
3586
|
+
}
|
|
3587
|
+
|
|
3588
|
+
/**
|
|
3589
|
+
* Perform an already-selected quarantine (job.leftoverQuarantineAttempted
|
|
3590
|
+
* must already be true, stamped by the caller) and persist the outcome onto
|
|
3591
|
+
* the job row: `quarantinedTo`/`quarantinedCommit`/`quarantinedPaths` on
|
|
3592
|
+
* success, plus a one-line pointer appended to `error` naming the ref so a
|
|
3593
|
+
* human can recover with a single named command
|
|
3594
|
+
* (`git show sm-salvage/<slug>`). The belt-and-braces `salvagePatch` (when
|
|
3595
|
+
* present) is referenced alongside it, never removed. On failure, only a
|
|
3596
|
+
* diagnostic is appended — the job row's dirt-describing fields are left as
|
|
3597
|
+
* they were, since the tree itself was left untouched (or, for a
|
|
3598
|
+
* restore-only failure, the salvage ref is still named in the note).
|
|
3599
|
+
*
|
|
3600
|
+
* `headBefore`, when the caller has it fresh (spawnJob's own finalize still
|
|
3601
|
+
* has the local `guardHeadBefore` in scope for the run that just parked —
|
|
3602
|
+
* the same value is deleted off the job ROW earlier in that same finalize),
|
|
3603
|
+
* is used as the salvage ref's baseline commit; otherwise (the periodic
|
|
3604
|
+
* reverifyNeedsReview pass, re-discovering an already-parked row) this falls
|
|
3605
|
+
* back to the current HEAD inside quarantineLeftovers itself.
|
|
3606
|
+
*/
|
|
3607
|
+
async function performLeftoverQuarantine(job, paths, headBefore = null) {
|
|
3608
|
+
const result = await quarantineLeftovers({
|
|
3609
|
+
cwd: job.cwd || DEFAULT_PROJECT_CWD,
|
|
3610
|
+
slug: job.slug,
|
|
3611
|
+
paths,
|
|
3612
|
+
headBefore: headBefore || job.guardHeadBefore || null,
|
|
3613
|
+
});
|
|
3614
|
+
await mutate((s) => {
|
|
3615
|
+
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
3616
|
+
if (!j) return;
|
|
3617
|
+
if (result.ok && Array.isArray(result.quarantinedPaths) && result.quarantinedPaths.length) {
|
|
3618
|
+
j.quarantinedTo = result.ref;
|
|
3619
|
+
j.quarantinedCommit = result.commit;
|
|
3620
|
+
j.quarantinedPaths = capDirtyPaths(result.quarantinedPaths);
|
|
3621
|
+
const salvageNote = j.salvagePatch ? `; salvage patch also at ${j.salvagePatch}` : '';
|
|
3622
|
+
const pointer = `Leftovers quarantined to ${result.ref} (commit ${result.commit}) — recover via \`git show ${result.ref}\`${salvageNote}`;
|
|
3623
|
+
j.error = j.error ? `${j.error}\n${pointer}` : pointer;
|
|
3624
|
+
console.log(`[scheduler] ${job.slug}: quarantined ${result.quarantinedPaths.length} leftover path(s) to ${result.ref} (${result.commit})`);
|
|
3625
|
+
} else if (!result.ok) {
|
|
3626
|
+
const pointer = `Leftover quarantine failed: ${result.reason}`;
|
|
3627
|
+
j.error = j.error ? `${j.error}\n${pointer}` : pointer;
|
|
3628
|
+
console.error(`[scheduler] ${job.slug}: leftover quarantine failed: ${result.reason}`);
|
|
3629
|
+
}
|
|
3630
|
+
});
|
|
3631
|
+
}
|
|
3632
|
+
|
|
3306
3633
|
/**
|
|
3307
3634
|
* Pure argv builder for a `claude -p` child spawn, shared so the
|
|
3308
3635
|
* resume-vs-fresh-session choice is made in exactly one place. `resume`
|
|
@@ -3782,13 +4109,9 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
3782
4109
|
// project (keyed by cwd) so jobs in different repos run concurrently up to
|
|
3783
4110
|
// the cap; within one project, sequential-group semantics are preserved.
|
|
3784
4111
|
|
|
3785
|
-
|
|
3786
|
-
|
|
3787
|
-
|
|
3788
|
-
*/
|
|
3789
|
-
function isFixPlanSlug(slug) {
|
|
3790
|
-
return /^\d+-fix-/.test(slug);
|
|
3791
|
-
}
|
|
4112
|
+
// isFixPlanSlug/classifyDiscoveredFixPlan/resolveIsFixPlan now live in
|
|
4113
|
+
// lib/fixPlanSlug.cjs (PRD 1131) — see that module's header for why slug
|
|
4114
|
+
// shape alone is no longer sufficient to classify a fix plan.
|
|
3792
4115
|
|
|
3793
4116
|
/**
|
|
3794
4117
|
* The fix-plan slug spawnInvestigation authors for a given failed job —
|
|
@@ -3827,7 +4150,25 @@ function healTargetForFix(fixSlug, jobs) {
|
|
|
3827
4150
|
* unit-tested (no spawn, no fs). Inputs are the already-resolved values that
|
|
3828
4151
|
* spawnInvestigation computes.
|
|
3829
4152
|
*/
|
|
3830
|
-
function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group }) {
|
|
4153
|
+
function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild = null }) {
|
|
4154
|
+
const deadFixChildNote = deadChild ? `
|
|
4155
|
+
|
|
4156
|
+
# This is a REOPENED investigation — your own prior fix plan died
|
|
4157
|
+
You already investigated this job once and produced a fix-plan PRD, \`${deadChild.slug}\`, which was
|
|
4158
|
+
supposed to heal it. That fix-plan job itself reached a terminal, non-completed status
|
|
4159
|
+
(\`${deadChild.status}\`) without ever fixing the original failure — so the parent job you are now
|
|
4160
|
+
investigating is stuck again with nothing left to retry it automatically. This is the ONE reopen
|
|
4161
|
+
this parent gets; do not cold-read the log and re-derive the plan that already failed.
|
|
4162
|
+
|
|
4163
|
+
Dead fix-plan child's own outcome:
|
|
4164
|
+
- Slug: ${deadChild.slug}
|
|
4165
|
+
- Status: ${deadChild.status}
|
|
4166
|
+
- Verifier verdict: ${deadChild.verifierVerdict ?? '(none recorded)'}
|
|
4167
|
+
- Error: ${deadChild.error ?? '(none recorded)'}
|
|
4168
|
+
|
|
4169
|
+
Read why THAT job died (its own run log, if any, under the runs directory) before writing a new
|
|
4170
|
+
fix-plan PRD, and make sure your new plan is genuinely different from — not a repeat of — whatever
|
|
4171
|
+
that dead child attempted.` : '';
|
|
3831
4172
|
const abandonedBackgroundTaskNote = failedJob.verifierVerdict === 'abandoned_background_task' ? `
|
|
3832
4173
|
|
|
3833
4174
|
# Known failure class: abandoned background task
|
|
@@ -3848,7 +4189,7 @@ The fix-plan PRD you write for this MUST instruct its executor to, in order:
|
|
|
3848
4189
|
1. Check for a salvage patch (named \`<slug>.uncommitted.patch\` in the run directory${failedJob.salvagePatch ? `, e.g. \`${failedJob.salvagePatch}\`` : ''}) and, if found, apply it to the working tree BEFORE inspecting \`git status\`/\`git diff\` in ${cwd} for uncommitted changes matching the original PRD's acceptance criteria.
|
|
3849
4190
|
2. If the work is present (via the applied patch or already in the tree) and satisfies the acceptance criteria, run the project's verify commands and COMMIT it — do not re-implement or re-plan the PRD from scratch.
|
|
3850
4191
|
3. Only fall back to re-implementing whatever acceptance criteria are genuinely missing after applying any salvage patch, not the whole PRD.` : '';
|
|
3851
|
-
return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${abandonedBackgroundTaskNote}
|
|
4192
|
+
return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${deadFixChildNote}${abandonedBackgroundTaskNote}
|
|
3852
4193
|
|
|
3853
4194
|
# Failed job
|
|
3854
4195
|
- Slug: ${failedJob.slug}
|
|
@@ -3893,8 +4234,13 @@ ${logTail}
|
|
|
3893
4234
|
cwd: ${cwd}
|
|
3894
4235
|
parallelGroup: ${group}
|
|
3895
4236
|
estimateMinutes: <your time estimate>
|
|
4237
|
+
isFixPlan: true
|
|
3896
4238
|
---
|
|
3897
4239
|
\`\`\`
|
|
4240
|
+
\`isFixPlan: true\` is REQUIRED — it is the scheduler's provenance signal that this PRD is a
|
|
4241
|
+
genuine auto-authored fix plan (not a human/agent PRD whose slug merely happens to start with
|
|
4242
|
+
"fix-"); omitting it means this fix plan will not get its depth-cap/zero-edit-commit-guard
|
|
4243
|
+
exemptions.
|
|
3898
4244
|
\`cwd\` must be the git repo root where the fix will actually land. If the failed job's cwd is
|
|
3899
4245
|
not that repo (e.g. a scratch dir like \`/tmp\`), set \`cwd:\` to the correct repo root instead —
|
|
3900
4246
|
the scheduler's commit guard and post-run verifier read git state from this path, and a
|
|
@@ -3968,7 +4314,7 @@ function readRunOutcomeSidecars(runDir, slug) {
|
|
|
3968
4314
|
*/
|
|
3969
4315
|
const INVESTIGATION_LAUNCH_KEY = 'investigation';
|
|
3970
4316
|
|
|
3971
|
-
async function spawnInvestigation(failedJob, runDir) {
|
|
4317
|
+
async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {}) {
|
|
3972
4318
|
// The probe launches with the same CLI as the job it diagnoses. While
|
|
3973
4319
|
// that CLI cannot launch at all (launch circuit breaker, issue #11 list
|
|
3974
4320
|
// B1: probes e4f82da2/d374e6bf died on the same HTTP 400 as the runs
|
|
@@ -3997,7 +4343,14 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3997
4343
|
console.log(`[scheduler] skip investigation: ${failedJob.slug} is resume-recovery eligible`);
|
|
3998
4344
|
return { deferred: false };
|
|
3999
4345
|
}
|
|
4000
|
-
|
|
4346
|
+
// Mechanical recovery (PRD 1130): same first-refusal treatment — a job
|
|
4347
|
+
// eligible for a pure-git retry must never also get a cold-read fix-plan
|
|
4348
|
+
// PRD authored in the same pass.
|
|
4349
|
+
if (selectMechanicalRecoveryTarget(failedJob)) {
|
|
4350
|
+
console.log(`[scheduler] skip investigation: ${failedJob.slug} is mechanical-recovery eligible`);
|
|
4351
|
+
return { deferred: false };
|
|
4352
|
+
}
|
|
4353
|
+
if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth, failedJob.isFixPlan)) {
|
|
4001
4354
|
console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
|
|
4002
4355
|
return { deferred: false };
|
|
4003
4356
|
}
|
|
@@ -4054,7 +4407,13 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
4054
4407
|
|
|
4055
4408
|
const logTail = readTail(failedLogPath, 16 * 1024) || '(failed to read log)';
|
|
4056
4409
|
|
|
4057
|
-
|
|
4410
|
+
// A dead-fix-plan reopen (PRD 1129) targets the SAME fixPath its dead
|
|
4411
|
+
// child was originally authored at, by construction (fixSlugFor is a pure
|
|
4412
|
+
// function of the parent) — the file existing is not staleness here, it's
|
|
4413
|
+
// the whole reason a reopen was offered. Skip the guard in that one case
|
|
4414
|
+
// so the second investigation can overwrite the dead plan; every other
|
|
4415
|
+
// caller keeps the original protection against clobbering a live sibling.
|
|
4416
|
+
if (fs.existsSync(fixPath) && !deadChild) {
|
|
4058
4417
|
console.log(`[scheduler] skip investigation: fix plan already exists at ${fixPath}`);
|
|
4059
4418
|
releaseSlot();
|
|
4060
4419
|
return { deferred: false };
|
|
@@ -4080,7 +4439,7 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
4080
4439
|
console.warn(`[scheduler] investigation cwd is not a git repo (${cwd}); falling back to ${DEFAULT_PROJECT_CWD}`);
|
|
4081
4440
|
cwd = DEFAULT_PROJECT_CWD;
|
|
4082
4441
|
}
|
|
4083
|
-
const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group });
|
|
4442
|
+
const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild });
|
|
4084
4443
|
|
|
4085
4444
|
// Phase 1: open log fd for pre-spawn diagnostics.
|
|
4086
4445
|
const { fd, safeLog, closeFd } = openLog(investigationLogPath);
|
|
@@ -4209,6 +4568,22 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
4209
4568
|
mutate((s) => {
|
|
4210
4569
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
4211
4570
|
if (j) j.autoFixOutcome = 'plan';
|
|
4571
|
+
// Dead-fix-plan reopen (PRD 1129): fixSlugFor is a pure function of
|
|
4572
|
+
// the parent, so the freshly-authored plan landed at the SAME slug
|
|
4573
|
+
// as the dead child — reconcile() sees an already-known slug and
|
|
4574
|
+
// will never re-mint a pending row for it. Explicitly reset the
|
|
4575
|
+
// dead child's own row here so the overwritten plan actually gets
|
|
4576
|
+
// a chance to run, rather than sitting inert behind a permanently
|
|
4577
|
+
// terminal queue row. force:true because 'skipped' (a valid dead
|
|
4578
|
+
// status here) is otherwise reset-refused by design.
|
|
4579
|
+
if (deadChild) {
|
|
4580
|
+
const child = s.jobs.find((x) => x.slug === deadChild.slug);
|
|
4581
|
+
if (child) {
|
|
4582
|
+
resetJobFields(child, 'reset by dead-fix-plan reopen: parent investigation authored a new plan', {
|
|
4583
|
+
force: true, source: 'spawnInvestigation:dead-fix-plan-reopen',
|
|
4584
|
+
});
|
|
4585
|
+
}
|
|
4586
|
+
}
|
|
4212
4587
|
}).catch(() => {});
|
|
4213
4588
|
} else {
|
|
4214
4589
|
console.log(`[scheduler] investigation finished WITHOUT producing fix plan (slug=${failedJob.slug}, code=${exitCode})`);
|
|
@@ -4291,6 +4666,58 @@ async function computeLaunchHolds(state, { now = Date.now(), claudeVersion } = {
|
|
|
4291
4666
|
return held;
|
|
4292
4667
|
}
|
|
4293
4668
|
|
|
4669
|
+
/**
|
|
4670
|
+
* computeDepHistorySatisfaction(state) → Map<cwd, Set<string>|symbol>
|
|
4671
|
+
*
|
|
4672
|
+
* PRD 1122's once-per-tick dependsOn history/archive lookup: for every
|
|
4673
|
+
* distinct project cwd with jobs this tick, builds the set of dep slugs that
|
|
4674
|
+
* have no live queue row but are nonetheless known-satisfied — a completed
|
|
4675
|
+
* record in that project's own `state/history.jsonl` shard
|
|
4676
|
+
* (queueHistory.completedSlugsForCwd, scoped per-project so a same-named PRD
|
|
4677
|
+
* in an unrelated project can never satisfy a dep here), or a `.md` file
|
|
4678
|
+
* under any of that project's `prds-archived/` dirs (listArchivedPrdDirs —
|
|
4679
|
+
* covers both the retired flat layout and every Epic's own sibling archive).
|
|
4680
|
+
* findBlockingDep (schedulerBatch.cjs) treats a dep slug as blocking
|
|
4681
|
+
* whenever it has no live row AND is absent from this set, so a typo or a
|
|
4682
|
+
* double-prefixed slug (the exact 2026-09-06 starry-night-ships incident)
|
|
4683
|
+
* HOLDS its dependent instead of silently dispatching it.
|
|
4684
|
+
*
|
|
4685
|
+
* Fails OPEN per project, never queue-wide: a history-shard or archive-scan
|
|
4686
|
+
* read error for one cwd degrades that cwd's value to
|
|
4687
|
+
* `DEP_HISTORY_FAIL_OPEN` (findBlockingDep then treats every rowless dep in
|
|
4688
|
+
* that project as satisfied, exactly today's pre-1122 behaviour) with a
|
|
4689
|
+
* logged warning — it never throws out of this function and never blocks
|
|
4690
|
+
* every OTHER project's dispatch for one project's bad fs state.
|
|
4691
|
+
*
|
|
4692
|
+
* Computed ONCE here, before pickNextBatch runs, and threaded down as pure
|
|
4693
|
+
* data (quietOpts.satisfiedSlugsByCwd) — schedulerBatch.cjs itself does no
|
|
4694
|
+
* I/O, so this is the only fs read this gate costs per tick, not one per job
|
|
4695
|
+
* per dep.
|
|
4696
|
+
*/
|
|
4697
|
+
async function computeDepHistorySatisfaction(state) {
|
|
4698
|
+
const byCwd = new Map();
|
|
4699
|
+
const cwds = new Set((state?.jobs || []).map((j) => j.cwd || DEFAULT_PROJECT_CWD));
|
|
4700
|
+
for (const cwd of cwds) {
|
|
4701
|
+
const satisfied = new Set();
|
|
4702
|
+
try {
|
|
4703
|
+
for (const slug of await queueHistory.completedSlugsForCwd(cwd)) satisfied.add(slug);
|
|
4704
|
+
for (const dir of listArchivedPrdDirs(cwd)) {
|
|
4705
|
+
let entries;
|
|
4706
|
+
try { entries = await fsp.readdir(dir); } catch { continue; }
|
|
4707
|
+
for (const name of entries) {
|
|
4708
|
+
if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
|
|
4709
|
+
}
|
|
4710
|
+
}
|
|
4711
|
+
} catch (e) {
|
|
4712
|
+
console.warn(`[scheduler] depHistorySatisfaction: history/archive lookup failed for ${cwd} (${e?.message}) — falling back to fail-open dep resolution for this project this tick`);
|
|
4713
|
+
byCwd.set(cwd, DEP_HISTORY_FAIL_OPEN);
|
|
4714
|
+
continue;
|
|
4715
|
+
}
|
|
4716
|
+
byCwd.set(cwd, satisfied);
|
|
4717
|
+
}
|
|
4718
|
+
return byCwd;
|
|
4719
|
+
}
|
|
4720
|
+
|
|
4294
4721
|
/**
|
|
4295
4722
|
* A run that never got a turn (res.launchFailure — see executeJob's onExit)
|
|
4296
4723
|
* is routed here instead of the failed/investigation path (issue #11 lists
|
|
@@ -4586,6 +5013,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4586
5013
|
let res;
|
|
4587
5014
|
let worktreeLeftoverDirty = [];
|
|
4588
5015
|
let worktreeIntegrationFailure = null;
|
|
5016
|
+
// Set only when integrateJobBranch's stderr-parsing auto-resolve fired
|
|
5017
|
+
// (PRD 1125) — surfaced on the job row so the Queue UI can say the merge
|
|
5018
|
+
// self-healed rather than silently looking like an ordinary merge.
|
|
5019
|
+
let mergeAutoResolved = null;
|
|
5020
|
+
let mergeAutoResolvedPaths = null;
|
|
4589
5021
|
// A job's uncommitted-work patch, whichever isolation mode produced it —
|
|
4590
5022
|
// set by EITHER branch below, never both (worktree.ok picks exactly one
|
|
4591
5023
|
// shape for the whole run). Named generically (not "worktree...") because
|
|
@@ -4633,6 +5065,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4633
5065
|
console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
|
|
4634
5066
|
} else if (integration.integrated) {
|
|
4635
5067
|
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
|
|
5068
|
+
if (integration.autoResolved) {
|
|
5069
|
+
mergeAutoResolved = integration.autoResolved;
|
|
5070
|
+
mergeAutoResolvedPaths = integration.resolvedPaths || [];
|
|
5071
|
+
console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
|
|
5072
|
+
}
|
|
4636
5073
|
}
|
|
4637
5074
|
await jobWorktree.cleanupJobWorktree({
|
|
4638
5075
|
cwd: guardCwd,
|
|
@@ -4868,7 +5305,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4868
5305
|
ranInWorktree: worktree.ok,
|
|
4869
5306
|
jobSelfCommitted,
|
|
4870
5307
|
legitimateNoOp: guardIsLegitimateNoOp,
|
|
4871
|
-
isFixPlanJob:
|
|
5308
|
+
isFixPlanJob: resolveIsFixPlan(job.slug, job.isFixPlan),
|
|
4872
5309
|
verifyResult,
|
|
4873
5310
|
salvagePatch,
|
|
4874
5311
|
});
|
|
@@ -4945,9 +5382,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4945
5382
|
let failedJobSnapshot = null;
|
|
4946
5383
|
let needsInvestigationNow = false;
|
|
4947
5384
|
let investigationJobSnapshot = null;
|
|
5385
|
+
let investigationDeadChildSnapshot = null;
|
|
4948
5386
|
let needsReviewRcaSnapshot = null;
|
|
4949
5387
|
let resumeRecoveryJob = null;
|
|
4950
5388
|
let resumeRecoveryTarget = null;
|
|
5389
|
+
let quarantineJob = null;
|
|
5390
|
+
let quarantinePaths = null;
|
|
5391
|
+
let mechanicalRecoveryJob = null;
|
|
5392
|
+
let mechanicalRecoveryTarget = null;
|
|
4951
5393
|
let terminalNotifySnapshot = null;
|
|
4952
5394
|
const newlyCompletedPrds = [];
|
|
4953
5395
|
await mutate((s) => {
|
|
@@ -5042,6 +5484,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5042
5484
|
} else {
|
|
5043
5485
|
delete s.jobs[i2].uncommittedPaths;
|
|
5044
5486
|
}
|
|
5487
|
+
// Worktree merge self-healed (PRD 1125) — every blocking path was
|
|
5488
|
+
// proven byte-identical to the branch, so the duplicate was
|
|
5489
|
+
// discarded and the merge retried once, successfully. Surfaced so
|
|
5490
|
+
// the Queue UI shows a self-heal instead of an ordinary merge.
|
|
5491
|
+
if (mergeAutoResolved) {
|
|
5492
|
+
s.jobs[i2].mergeAutoResolved = mergeAutoResolved;
|
|
5493
|
+
s.jobs[i2].mergeAutoResolvedPaths = capDirtyPaths(mergeAutoResolvedPaths);
|
|
5494
|
+
} else {
|
|
5495
|
+
delete s.jobs[i2].mergeAutoResolved;
|
|
5496
|
+
delete s.jobs[i2].mergeAutoResolvedPaths;
|
|
5497
|
+
}
|
|
5045
5498
|
// Non-blocking notes (e.g. a recovered missing-dependency probe, or a
|
|
5046
5499
|
// pattern hit demoted because a materially-checkable verdict outranked
|
|
5047
5500
|
// it) — surfaced even on completed jobs so the signal isn't lost.
|
|
@@ -5092,6 +5545,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5092
5545
|
// takes the treatAsPending branch above and never reaches here).
|
|
5093
5546
|
needsReviewRcaSnapshot = { ...s.jobs[i2] };
|
|
5094
5547
|
|
|
5548
|
+
// Mechanical recovery (PRD 1130): evaluated FIRST, ahead of both
|
|
5549
|
+
// resume-first recovery and auto-fix — a job parked with a
|
|
5550
|
+
// mechanically-resolvable verdict (see
|
|
5551
|
+
// selectMechanicalRecoveryTarget) needs no model, no plan, and no
|
|
5552
|
+
// depth-cap check, so it must never fall through to either.
|
|
5553
|
+
// Snapshot only (no I/O inside mutate()); the actual git retry
|
|
5554
|
+
// happens outside mutate(), below.
|
|
5555
|
+
const mTarget = selectMechanicalRecoveryTarget(s.jobs[i2]);
|
|
5556
|
+
if (mTarget) {
|
|
5557
|
+
mechanicalRecoveryJob = { ...s.jobs[i2] };
|
|
5558
|
+
mechanicalRecoveryTarget = mTarget;
|
|
5559
|
+
} else {
|
|
5095
5560
|
// Resume-first recovery (PRD 1111): evaluated BEFORE the auto-fix
|
|
5096
5561
|
// eligibility check below — a job whose verdict is
|
|
5097
5562
|
// 'uncommitted_changes' with a live sessionId gets one bounded
|
|
@@ -5105,6 +5570,22 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5105
5570
|
resumeRecoveryJob = { ...s.jobs[i2] };
|
|
5106
5571
|
resumeRecoveryTarget = target;
|
|
5107
5572
|
} else {
|
|
5573
|
+
// Leftover quarantine (PRD 1128): resume recovery is spent
|
|
5574
|
+
// (resumeRecoveryAttempted already true) and this run STILL parked
|
|
5575
|
+
// needs_review with uncommitted_changes — the leftovers are about
|
|
5576
|
+
// to sit dirty in the shared tree forever, poisoning every later
|
|
5577
|
+
// worktree merge for this cwd. Stamp the one-attempt marker HERE,
|
|
5578
|
+
// synchronously in the same mutate as this decision (mirrors
|
|
5579
|
+
// resumeRecoveryAttempted's own stamp-before-acting rule above),
|
|
5580
|
+
// so a concurrent reverifyNeedsReview pass can never double-fire
|
|
5581
|
+
// this. The actual git work is async and runs outside mutate(),
|
|
5582
|
+
// below (performLeftoverQuarantine).
|
|
5583
|
+
const quarantineTarget = selectLeftoverQuarantineTarget(s.jobs[i2]);
|
|
5584
|
+
if (quarantineTarget) {
|
|
5585
|
+
s.jobs[i2].leftoverQuarantineAttempted = true;
|
|
5586
|
+
quarantineJob = { ...s.jobs[i2] };
|
|
5587
|
+
quarantinePaths = quarantineTarget.paths;
|
|
5588
|
+
}
|
|
5108
5589
|
// Same-tick auto-fix (feedback 2026-07-12): rather than waiting up to
|
|
5109
5590
|
// 10 min for reverifyNeedsReview()'s periodic pass, check right here
|
|
5110
5591
|
// whether this job qualifies for auto-fix (same eligibility rule
|
|
@@ -5119,8 +5600,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5119
5600
|
isEligibleForImmediateAutoFix(s.jobs[i2], s.jobs, fixSlugExists)
|
|
5120
5601
|
) {
|
|
5121
5602
|
const isRetryAttempt = s.jobs[i2].autoFixAttempted === true;
|
|
5603
|
+
const isDeadFixPlanReopen = isFixPlanDead(s.jobs[i2], s.jobs);
|
|
5604
|
+
if (isDeadFixPlanReopen) {
|
|
5605
|
+
investigationDeadChildSnapshot = s.jobs.find((x) => x.slug === fixSlugFor(s.jobs[i2])) || null;
|
|
5606
|
+
}
|
|
5122
5607
|
s.jobs[i2].autoFixAttempted = true;
|
|
5123
5608
|
if (!s.jobs[i2].runId) s.jobs[i2].runId = runId;
|
|
5609
|
+
if (isDeadFixPlanReopen) s.jobs[i2].autoFixReopened = true;
|
|
5124
5610
|
if (isRetryAttempt) {
|
|
5125
5611
|
s.jobs[i2].autoFixRetries = (s.jobs[i2].autoFixRetries ?? 0) + 1;
|
|
5126
5612
|
delete s.jobs[i2].autoFixOutcome;
|
|
@@ -5129,13 +5615,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5129
5615
|
investigationJobSnapshot = { ...s.jobs[i2] };
|
|
5130
5616
|
}
|
|
5131
5617
|
}
|
|
5618
|
+
}
|
|
5132
5619
|
}
|
|
5133
5620
|
// Auto-promote: when a fix-* PRD completes successfully, the original
|
|
5134
5621
|
// failed PRD's work is logically done. Flip its status to 'completed'
|
|
5135
5622
|
// so the cross-group failure gate in pickNextBatch releases. Without
|
|
5136
5623
|
// this, the queue stalls indefinitely behind a stale failure even
|
|
5137
5624
|
// though the auto-recovery did its job.
|
|
5138
|
-
if (effectiveStatus === 'completed' &&
|
|
5625
|
+
if (effectiveStatus === 'completed' && resolveIsFixPlan(job.slug, job.isFixPlan)) {
|
|
5139
5626
|
const orig = healTargetForFix(job.slug, s.jobs);
|
|
5140
5627
|
if (orig) {
|
|
5141
5628
|
const priorStatus = orig.status;
|
|
@@ -5207,6 +5694,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5207
5694
|
});
|
|
5208
5695
|
}
|
|
5209
5696
|
|
|
5697
|
+
if (mechanicalRecoveryJob && mechanicalRecoveryTarget) {
|
|
5698
|
+
console.log(`[scheduler] needs_review ${job.slug} → mechanical-recovery (re-integrating ${mechanicalRecoveryTarget.branch})`);
|
|
5699
|
+
performMechanicalRecovery(mechanicalRecoveryJob, mechanicalRecoveryTarget).catch((e) => {
|
|
5700
|
+
console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
|
|
5701
|
+
});
|
|
5702
|
+
}
|
|
5703
|
+
|
|
5210
5704
|
if (resumeRecoveryJob && resumeRecoveryTarget) {
|
|
5211
5705
|
console.log(`[scheduler] needs_review ${job.slug} → resume-recovery (session ${resumeRecoveryTarget.sessionId}, ${resumeRecoveryTarget.dirtyPaths.length} dirty path(s))`);
|
|
5212
5706
|
spawnResumeRecovery(resumeRecoveryJob, resumeRecoveryTarget).catch((e) => {
|
|
@@ -5214,6 +5708,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5214
5708
|
});
|
|
5215
5709
|
}
|
|
5216
5710
|
|
|
5711
|
+
if (quarantineJob && quarantinePaths) {
|
|
5712
|
+
console.log(`[scheduler] needs_review ${job.slug} → quarantining ${quarantinePaths.length} leftover path(s) (resume recovery already spent)`);
|
|
5713
|
+
performLeftoverQuarantine(quarantineJob, quarantinePaths, guardHeadBefore).catch((e) => {
|
|
5714
|
+
console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
|
|
5715
|
+
});
|
|
5716
|
+
}
|
|
5717
|
+
|
|
5217
5718
|
if (actuallyFailed && failedJobSnapshot) {
|
|
5218
5719
|
// Transient-failure detector. A 143/137 exit is ALWAYS a signal kill — the
|
|
5219
5720
|
// agent never self-exits with those — so the only question is WHO killed it.
|
|
@@ -5286,7 +5787,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5286
5787
|
}
|
|
5287
5788
|
} else if (needsInvestigationNow && investigationJobSnapshot) {
|
|
5288
5789
|
console.log(`[scheduler] needs_review ${job.slug} → immediate auto-fix investigation (not waiting for periodic reverify)`);
|
|
5289
|
-
spawnInvestigation(investigationJobSnapshot, runDir).catch((e) => {
|
|
5790
|
+
spawnInvestigation(investigationJobSnapshot, runDir, { deadChild: investigationDeadChildSnapshot }).catch((e) => {
|
|
5290
5791
|
console.error('[scheduler] spawnInvestigation error', job.slug, e);
|
|
5291
5792
|
});
|
|
5292
5793
|
}
|
|
@@ -5356,11 +5857,13 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
5356
5857
|
// ceilinged the queue at 3 while the pool the user configured said 5.
|
|
5357
5858
|
const freeSlots = sessionSlots.available();
|
|
5358
5859
|
const heldSlugs = await computeLaunchHolds(state);
|
|
5860
|
+
const satisfiedSlugsByCwd = await computeDepHistorySatisfaction(state);
|
|
5359
5861
|
const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
|
|
5360
5862
|
leaseHeld: quietMachineLease.isHeld(),
|
|
5361
5863
|
machineInUse: sessionSlots.inUse(),
|
|
5362
5864
|
now: Date.now(),
|
|
5363
5865
|
heldSlugs,
|
|
5866
|
+
satisfiedSlugsByCwd,
|
|
5364
5867
|
});
|
|
5365
5868
|
if (batch.length === 0 && freeSlots === 0) {
|
|
5366
5869
|
const snap = sessionSlots.snapshot();
|
|
@@ -6025,10 +6528,16 @@ const MAX_INVESTIGATION_DEPTH = 1;
|
|
|
6025
6528
|
* no recorded investigationDepth (a job already in the queue before this
|
|
6026
6529
|
* depth tracking shipped) is treated as excluded too, preserving the
|
|
6027
6530
|
* pre-existing blanket-exclusion behavior for legacy jobs — no retroactive
|
|
6028
|
-
* migration. Non-fix-plan
|
|
6531
|
+
* migration. Non-fix-plan jobs are never capped here.
|
|
6532
|
+
*
|
|
6533
|
+
* `isFixPlan` (PRD 1131) is the job's own persisted classification stamp
|
|
6534
|
+
* (see lib/fixPlanSlug.cjs's resolveIsFixPlan) — an explicit true/false wins
|
|
6535
|
+
* over the slug; only a row with the field entirely absent (persisted
|
|
6536
|
+
* before this change shipped) falls back to the legacy slug-only heuristic.
|
|
6537
|
+
* Exported for tests.
|
|
6029
6538
|
*/
|
|
6030
|
-
function isFixPlanBeyondDepthCap(slug, investigationDepth) {
|
|
6031
|
-
if (!
|
|
6539
|
+
function isFixPlanBeyondDepthCap(slug, investigationDepth, isFixPlan) {
|
|
6540
|
+
if (!resolveIsFixPlan(slug, isFixPlan)) return false;
|
|
6032
6541
|
if (investigationDepth == null) return true;
|
|
6033
6542
|
return investigationDepth >= MAX_INVESTIGATION_DEPTH + 1;
|
|
6034
6543
|
}
|
|
@@ -6081,12 +6590,18 @@ function isUnresolvableNeedsReview(job, { hasRunDir }) {
|
|
|
6081
6590
|
* ('no-plan', 'error', and unstamped/undefined) — mirrors the retry
|
|
6082
6591
|
* eligibility rule in selectAutoFixTargets so a job can never be retry-
|
|
6083
6592
|
* eligible there and simultaneously un-annotatable here.
|
|
6593
|
+
*
|
|
6594
|
+
* A parent stamped `autoFixReopened: true` (its dead fix-plan child earned
|
|
6595
|
+
* it exactly one further attempt — see isFixPlanDead) is a separate
|
|
6596
|
+
* exhaustion path: it is spent as soon as that second investigation
|
|
6597
|
+
* concludes with ANY outcome, including another 'plan' — a reopened parent
|
|
6598
|
+
* never gets a third attempt, so unlike the fresh case a 'plan' outcome does
|
|
6599
|
+
* not exempt it here.
|
|
6084
6600
|
*/
|
|
6085
6601
|
function isExhaustedAutoFix(job) {
|
|
6086
|
-
|
|
6087
|
-
|
|
6088
|
-
|
|
6089
|
-
&& (job.autoFixRetries ?? 0) >= 1;
|
|
6602
|
+
if (!job || job.status !== 'needs_review' || job.autoFixAttempted !== true) return false;
|
|
6603
|
+
if (job.autoFixReopened === true) return job.autoFixOutcome != null;
|
|
6604
|
+
return job.autoFixOutcome !== 'plan' && (job.autoFixRetries ?? 0) >= 1;
|
|
6090
6605
|
}
|
|
6091
6606
|
|
|
6092
6607
|
/**
|
|
@@ -6102,6 +6617,31 @@ function isPlanUnqueued(job, queuedSlugs) {
|
|
|
6102
6617
|
return !queuedSlugs.has(fixSlugFor(job));
|
|
6103
6618
|
}
|
|
6104
6619
|
|
|
6620
|
+
// Terminal-and-not-completed statuses a fix-plan child can die in — see
|
|
6621
|
+
// isFixPlanDead.
|
|
6622
|
+
const DEAD_FIX_CHILD_STATUSES = new Set(['needs_review', 'failed', 'quarantined', 'skipped']);
|
|
6623
|
+
|
|
6624
|
+
/**
|
|
6625
|
+
* Pure predicate: a parent stuck at outcome 'plan' whose own fix-plan child
|
|
6626
|
+
* (fixSlugFor(job)) has ITSELF died — reached a terminal non-completed
|
|
6627
|
+
* status — with nothing left in the ladder that will ever revisit either
|
|
6628
|
+
* row again (selectAutoFixTargets skips a 'plan' outcome outright, and
|
|
6629
|
+
* isPlanUnqueued only fires when the child never reached the queue at all,
|
|
6630
|
+
* which isn't true once a dead child row exists). `job.autoFixReopened`
|
|
6631
|
+
* gates this to exactly once per parent — once stamped, this always returns
|
|
6632
|
+
* false so the parent can never be reopened a second time. Exported for
|
|
6633
|
+
* tests.
|
|
6634
|
+
*/
|
|
6635
|
+
function isFixPlanDead(job, jobsInProject) {
|
|
6636
|
+
if (!job || job.status !== 'needs_review') return false;
|
|
6637
|
+
if (job.autoFixOutcome !== 'plan') return false;
|
|
6638
|
+
if (job.autoFixReopened === true) return false;
|
|
6639
|
+
const fixSlug = fixSlugFor(job);
|
|
6640
|
+
const child = (jobsInProject || []).find((j) => j.slug === fixSlug);
|
|
6641
|
+
if (!child) return false;
|
|
6642
|
+
return DEAD_FIX_CHILD_STATUSES.has(child.status);
|
|
6643
|
+
}
|
|
6644
|
+
|
|
6105
6645
|
/**
|
|
6106
6646
|
* Pure predicate: is this job eligible for the boot re-verify self-heal? Only
|
|
6107
6647
|
* needs_review jobs with a run log (own or backfilled via resolveRunId) AND a
|
|
@@ -6247,16 +6787,30 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
|
|
|
6247
6787
|
// bounded `--resume` attempt must never also become a fix-plan target
|
|
6248
6788
|
// in the same pass — see spawnInvestigation's own identical guard.
|
|
6249
6789
|
if (selectResumeRecoveryTarget(job)) return false;
|
|
6790
|
+
// Mechanical recovery (PRD 1130): a job eligible for a pure-git retry
|
|
6791
|
+
// must never also become a fix-plan target — it needs no plan and no
|
|
6792
|
+
// model. Defensive: today's single mechanically-resolvable verdict
|
|
6793
|
+
// (worktree_integration_failed) is already excluded below via the depth
|
|
6794
|
+
// cap, but this must hold even if that stops being true.
|
|
6795
|
+
if (selectMechanicalRecoveryTarget(job)) return false;
|
|
6250
6796
|
const runId = job.runId || resolveJobRunId(job);
|
|
6251
6797
|
if (!runId) return false;
|
|
6252
|
-
if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth)) return false;
|
|
6798
|
+
if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth, job.isFixPlan)) return false;
|
|
6799
|
+
// A dead fix-plan child (PRD 1129) earns its parent exactly one further
|
|
6800
|
+
// attempt, bypassing the normal 'plan' exclusion and the fix-slug/queue
|
|
6801
|
+
// membership checks below — those checks exist to stop a FRESH
|
|
6802
|
+
// investigation from clobbering a live sibling, but here the sibling is
|
|
6803
|
+
// dead and reusing its slug is the whole point of the reopen.
|
|
6804
|
+
const dead = isFixPlanDead(job, jobs);
|
|
6253
6805
|
if (job.autoFixAttempted) {
|
|
6254
|
-
const retryEligible =
|
|
6806
|
+
const retryEligible = dead
|
|
6807
|
+
|| job.autoFixOutcome === 'no-plan'
|
|
6255
6808
|
|| job.autoFixOutcome === 'error'
|
|
6256
6809
|
|| job.autoFixOutcome == null;
|
|
6257
6810
|
if (!retryEligible) return false;
|
|
6258
|
-
if ((job.autoFixRetries ?? 0) >= 1) return false;
|
|
6811
|
+
if (!dead && (job.autoFixRetries ?? 0) >= 1) return false;
|
|
6259
6812
|
}
|
|
6813
|
+
if (dead) return true;
|
|
6260
6814
|
const fixSlug = fixSlugFor(job);
|
|
6261
6815
|
if (fixSlugExists(fixSlug)) return false;
|
|
6262
6816
|
if (slugsInQueue.has(fixSlug)) return false;
|
|
@@ -6435,7 +6989,7 @@ async function reverifyNeedsReview() {
|
|
|
6435
6989
|
const promotedPrds = [];
|
|
6436
6990
|
await mutate((s) => {
|
|
6437
6991
|
for (const job of s.jobs) {
|
|
6438
|
-
if (job.status !== 'completed' || !
|
|
6992
|
+
if (job.status !== 'completed' || !resolveIsFixPlan(job.slug, job.isFixPlan)) continue;
|
|
6439
6993
|
const orig = healTargetForFix(job.slug, s.jobs);
|
|
6440
6994
|
if (!orig) continue;
|
|
6441
6995
|
const priorStatus = orig.status;
|
|
@@ -6519,6 +7073,23 @@ async function reverifyNeedsReview() {
|
|
|
6519
7073
|
? await readQueue()
|
|
6520
7074
|
: afterHealForAnnotate;
|
|
6521
7075
|
|
|
7076
|
+
// Mechanical recovery (PRD 1130): evaluated first, ahead of both
|
|
7077
|
+
// resume-first recovery and auto-fix below — catches a job whose
|
|
7078
|
+
// mechanically-resolvable verdict this periodic pass finds still eligible
|
|
7079
|
+
// (e.g. one already parked before this rung shipped, or one the same-tick
|
|
7080
|
+
// check in spawnJob missed because the app restarted in between). Depth
|
|
7081
|
+
// never disqualifies it, so it runs regardless of investigationDepth.
|
|
7082
|
+
{
|
|
7083
|
+
for (const job of queueForResumeAndAutofix.jobs) {
|
|
7084
|
+
const target = selectMechanicalRecoveryTarget(job);
|
|
7085
|
+
if (!target) continue;
|
|
7086
|
+
console.log(`[scheduler] mechanical-recovery: needs_review ${job.slug} → re-integrating ${target.branch}`);
|
|
7087
|
+
performMechanicalRecovery(job, target).catch((e) => {
|
|
7088
|
+
console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
|
|
7089
|
+
});
|
|
7090
|
+
}
|
|
7091
|
+
}
|
|
7092
|
+
|
|
6522
7093
|
// Resume-first recovery (PRD 1111): before any fix-plan investigation is
|
|
6523
7094
|
// authored below, offer the bounded one-attempt `--resume` dispatch to any
|
|
6524
7095
|
// needs_review job this periodic pass finds still eligible — e.g. one the
|
|
@@ -6537,6 +7108,26 @@ async function reverifyNeedsReview() {
|
|
|
6537
7108
|
}
|
|
6538
7109
|
}
|
|
6539
7110
|
|
|
7111
|
+
// Leftover quarantine (PRD 1128), periodic pass: catches a job parked
|
|
7112
|
+
// needs_review with resume recovery already spent BEFORE this feature
|
|
7113
|
+
// shipped, or one the same-tick check in spawnJob missed because the app
|
|
7114
|
+
// restarted in between. Stamps the one-attempt marker in its own mutate
|
|
7115
|
+
// BEFORE the async git work starts (same race-closing rule as the resume
|
|
7116
|
+
// loop above and spawnJob's own dispatch stamp).
|
|
7117
|
+
{
|
|
7118
|
+
for (const job of queueForResumeAndAutofix.jobs) {
|
|
7119
|
+
const quarantineTarget = selectLeftoverQuarantineTarget(job);
|
|
7120
|
+
if (!quarantineTarget) continue;
|
|
7121
|
+
console.log(`[scheduler] leftover-quarantine: needs_review ${job.slug} → quarantining ${quarantineTarget.paths.length} leftover path(s)`);
|
|
7122
|
+
mutate((s) => {
|
|
7123
|
+
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
7124
|
+
if (j) j.leftoverQuarantineAttempted = true;
|
|
7125
|
+
}).then(() => performLeftoverQuarantine(job, quarantineTarget.paths)).catch((e) => {
|
|
7126
|
+
console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
|
|
7127
|
+
});
|
|
7128
|
+
}
|
|
7129
|
+
}
|
|
7130
|
+
|
|
6540
7131
|
// Auto-fix: spawn a fix-plan investigation for each job still in
|
|
6541
7132
|
// needs_review after the heal pass (kill-switch: SM_AUTOFIX_DISABLE=1).
|
|
6542
7133
|
// spawnInvestigation early-returns once investigationsInFlight reaches
|
|
@@ -6550,23 +7141,30 @@ async function reverifyNeedsReview() {
|
|
|
6550
7141
|
const runId = job.runId || resolveRunId(job);
|
|
6551
7142
|
const runDir = path.join(RUNS_DIR, runId);
|
|
6552
7143
|
const isRetryAttempt = job.autoFixAttempted === true;
|
|
7144
|
+
const isDeadFixPlanReopen = isFixPlanDead(job, queueForResumeAndAutofix.jobs);
|
|
7145
|
+
const deadChild = isDeadFixPlanReopen
|
|
7146
|
+
? queueForResumeAndAutofix.jobs.find((j) => j.slug === fixSlugFor(job))
|
|
7147
|
+
: null;
|
|
6553
7148
|
// Persist the attempt BEFORE spawning — a crash mid-investigation still
|
|
6554
7149
|
// counts it (mirrors orphanRetries). Safe even when the slot is busy: the
|
|
6555
7150
|
// investigation is queued and drained as slots free, so it is genuinely
|
|
6556
|
-
// attempted rather than silently dropped.
|
|
7151
|
+
// attempted rather than silently dropped. autoFixReopened is stamped in
|
|
7152
|
+
// this SAME mutate so a crash between selection and dispatch can never
|
|
7153
|
+
// leave the parent re-eligible for a second reopen (PRD 1129).
|
|
6557
7154
|
await mutate((s) => {
|
|
6558
7155
|
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
6559
7156
|
if (j) {
|
|
6560
7157
|
j.autoFixAttempted = true;
|
|
6561
7158
|
if (!j.runId && runId) j.runId = runId;
|
|
7159
|
+
if (isDeadFixPlanReopen) j.autoFixReopened = true;
|
|
6562
7160
|
if (isRetryAttempt) {
|
|
6563
7161
|
j.autoFixRetries = (j.autoFixRetries ?? 0) + 1;
|
|
6564
7162
|
delete j.autoFixOutcome;
|
|
6565
7163
|
}
|
|
6566
7164
|
}
|
|
6567
7165
|
});
|
|
6568
|
-
console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'})`);
|
|
6569
|
-
spawnInvestigation(job, runDir).catch((e) => {
|
|
7166
|
+
console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'}${isDeadFixPlanReopen ? ', dead fix-plan child reopen' : ''})`);
|
|
7167
|
+
spawnInvestigation(job, runDir, { deadChild }).catch((e) => {
|
|
6570
7168
|
console.error('[scheduler] auto-fix spawnInvestigation error', job.slug, e);
|
|
6571
7169
|
});
|
|
6572
7170
|
}
|
|
@@ -7657,6 +8255,36 @@ const remote = {
|
|
|
7657
8255
|
return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
|
|
7658
8256
|
}
|
|
7659
8257
|
|
|
8258
|
+
// Write-time FK check for a patched dependsOn (PRD 1124), reusing the
|
|
8259
|
+
// SAME resolution rule scheduler_create_prd's prdCreate.cjs applies (exact
|
|
8260
|
+
// slug, else bare-name after stripping one leading `NN-`) so update and
|
|
8261
|
+
// create can never disagree about what a dependsOn entry resolves to. An
|
|
8262
|
+
// explicit empty array CLEARS the dependency and skips validation — there
|
|
8263
|
+
// is nothing to resolve. A listPrds() read failure is skipped-with-a-
|
|
8264
|
+
// warning, matching createPrd's tolerance for an I/O hiccup.
|
|
8265
|
+
if (frontmatter && Array.isArray(frontmatter.dependsOn) && frontmatter.dependsOn.length) {
|
|
8266
|
+
let listing;
|
|
8267
|
+
try {
|
|
8268
|
+
listing = await this.listPrds({ cwd, limit: Number.MAX_SAFE_INTEGER });
|
|
8269
|
+
} catch (e) {
|
|
8270
|
+
console.warn(`[scheduler] updatePrd: dependsOn validation skipped (listPrds failed): ${e?.message ?? e}`);
|
|
8271
|
+
listing = null;
|
|
8272
|
+
}
|
|
8273
|
+
if (listing) {
|
|
8274
|
+
const candidateSlugs = (listing.prds ?? []).map((p) => p.slug);
|
|
8275
|
+
for (const dep of frontmatter.dependsOn) {
|
|
8276
|
+
if (resolveDepSlug(dep, candidateSlugs).length > 0) continue;
|
|
8277
|
+
const near = findNearMatches(dep, candidateSlugs);
|
|
8278
|
+
const suggestion = near.length ? ` Closest existing slug(s): ${near.join(', ')}.` : '';
|
|
8279
|
+
return {
|
|
8280
|
+
ok: false,
|
|
8281
|
+
error: `dependsOn entry "${dep}" does not resolve to any existing PRD in this project.${suggestion} ` +
|
|
8282
|
+
'Pass the bare name (preferred) or the exact NN-prefixed slug of an existing PRD.',
|
|
8283
|
+
};
|
|
8284
|
+
}
|
|
8285
|
+
}
|
|
8286
|
+
}
|
|
8287
|
+
|
|
7660
8288
|
let dir = null;
|
|
7661
8289
|
let filePath = null;
|
|
7662
8290
|
if (cwd) {
|
|
@@ -7787,4 +8415,149 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
7787
8415
|
});
|
|
7788
8416
|
}
|
|
7789
8417
|
|
|
7790
|
-
module.exports = {
|
|
8418
|
+
module.exports = {
|
|
8419
|
+
classifyQueueStarvation,
|
|
8420
|
+
runQueueStarvationWatchdog,
|
|
8421
|
+
QUEUE_STARVATION_MS,
|
|
8422
|
+
computeBlockedChains,
|
|
8423
|
+
stripAppOwnedChurn,
|
|
8424
|
+
findOverrunningJobs,
|
|
8425
|
+
JOB_OVERRUN_FACTOR,
|
|
8426
|
+
JOB_OVERRUN_FLOOR_MS,
|
|
8427
|
+
registerScheduleHandlers,
|
|
8428
|
+
attachWindow,
|
|
8429
|
+
init,
|
|
8430
|
+
ROOT,
|
|
8431
|
+
PRDS_DIR,
|
|
8432
|
+
healRefusalReason,
|
|
8433
|
+
writeQueue,
|
|
8434
|
+
reconcile,
|
|
8435
|
+
reconcileSourcePromptId,
|
|
8436
|
+
allocateParallelGroup,
|
|
8437
|
+
selectHistoryJobs,
|
|
8438
|
+
parsePorcelain,
|
|
8439
|
+
FINISH_PROTOCOL,
|
|
8440
|
+
IDLE_OUTPUT_KILL_MS,
|
|
8441
|
+
BASH_DEFAULT_TIMEOUT_MS,
|
|
8442
|
+
BASH_MAX_TIMEOUT_MS,
|
|
8443
|
+
remote,
|
|
8444
|
+
pickNextBatch,
|
|
8445
|
+
pickForProject,
|
|
8446
|
+
reapDeadRunningJobs,
|
|
8447
|
+
pollRecoveryClearSource,
|
|
8448
|
+
memoryLimitedBatchSize,
|
|
8449
|
+
availableForJobs,
|
|
8450
|
+
reverifyNeedsReview,
|
|
8451
|
+
isRescanCandidate,
|
|
8452
|
+
isFailedUnverifiedShaped,
|
|
8453
|
+
computeLooksDone,
|
|
8454
|
+
isPromotableOriginal,
|
|
8455
|
+
selectAutoFixTargets,
|
|
8456
|
+
applyRcaClassification,
|
|
8457
|
+
isEligibleForImmediateAutoFix,
|
|
8458
|
+
resolveRunId,
|
|
8459
|
+
isUnresolvableNeedsReview,
|
|
8460
|
+
isExhaustedAutoFix,
|
|
8461
|
+
isPlanUnqueued,
|
|
8462
|
+
isFixPlanDead,
|
|
8463
|
+
fixSlugFor,
|
|
8464
|
+
healTargetForFix,
|
|
8465
|
+
buildInvestigationPrompt,
|
|
8466
|
+
isGitRepoSync,
|
|
8467
|
+
committedInWindow,
|
|
8468
|
+
computeCommittedDuringRun,
|
|
8469
|
+
classifySigtermWithCommit,
|
|
8470
|
+
isFixPlanSlug,
|
|
8471
|
+
classifyDiscoveredFixPlan,
|
|
8472
|
+
resolveIsFixPlan,
|
|
8473
|
+
isFixPlanBeyondDepthCap,
|
|
8474
|
+
MAX_INVESTIGATION_DEPTH,
|
|
8475
|
+
forceTickOutcome,
|
|
8476
|
+
applyPauseCleared,
|
|
8477
|
+
detectNetworkErrorInLog,
|
|
8478
|
+
detectRateLimitInLog,
|
|
8479
|
+
classifyFailureOutcome,
|
|
8480
|
+
commitGuardVerdict,
|
|
8481
|
+
leftoverFieldsFrom,
|
|
8482
|
+
applyLeftoverFields,
|
|
8483
|
+
LEFTOVER_PATHS_CAP,
|
|
8484
|
+
capDirtyPaths,
|
|
8485
|
+
buildForeignWipSection,
|
|
8486
|
+
PRE_RUN_DIRTY_PATHS_CAP,
|
|
8487
|
+
FOREIGN_WIP_DELIMITER,
|
|
8488
|
+
FOREIGN_WIP_END_DELIMITER,
|
|
8489
|
+
TRANSIENT_RETRY_CAP,
|
|
8490
|
+
buildScheduleStatePayload,
|
|
8491
|
+
partitionBootOrphans,
|
|
8492
|
+
applyOrphanOutcome,
|
|
8493
|
+
BOOT_ORPHAN_KILL_GRACE_MS,
|
|
8494
|
+
registerAdminRoutes,
|
|
8495
|
+
notifyOriginatingTab,
|
|
8496
|
+
notifyNeedsReview,
|
|
8497
|
+
isNotifiableTerminalStatus,
|
|
8498
|
+
extractResultTextFromLog,
|
|
8499
|
+
candidatePrdsDirs,
|
|
8500
|
+
candidateArchivedPrdsDirs,
|
|
8501
|
+
resolveArchivedPrdStatus,
|
|
8502
|
+
prdDirForCwd,
|
|
8503
|
+
prdPathForJob,
|
|
8504
|
+
archivedPrdPathForJob,
|
|
8505
|
+
archivedTwinExists,
|
|
8506
|
+
findPrdDir,
|
|
8507
|
+
resolveVerifyPrdPath,
|
|
8508
|
+
resolveFixPlanPath,
|
|
8509
|
+
resolveNotifyPrd,
|
|
8510
|
+
runPrdMigration,
|
|
8511
|
+
consolidateAllFlatPrds,
|
|
8512
|
+
shouldSkipInvestigationForCleanRun,
|
|
8513
|
+
archiveCompletedPrd,
|
|
8514
|
+
retireCompletedSlugs,
|
|
8515
|
+
SCHEDULER_BOOTED_AT,
|
|
8516
|
+
SCHEDULER_CODE_SHA,
|
|
8517
|
+
resetJobFields,
|
|
8518
|
+
executeJob,
|
|
8519
|
+
prdArchivedSkipResult,
|
|
8520
|
+
spawnJob,
|
|
8521
|
+
listPrdsInternal,
|
|
8522
|
+
computeStallSummary,
|
|
8523
|
+
findStaleQuarantinedJobs,
|
|
8524
|
+
QUARANTINE_ESCALATE_MS,
|
|
8525
|
+
applyClearQueueVictims,
|
|
8526
|
+
PIDLESS_SPAWN_GRACE_MS,
|
|
8527
|
+
findStrandedInvestigations,
|
|
8528
|
+
INVESTIGATION_MAX_MS,
|
|
8529
|
+
stashList,
|
|
8530
|
+
parseStashLine,
|
|
8531
|
+
pathsChangedSince,
|
|
8532
|
+
restoreSpecificStash,
|
|
8533
|
+
evaluateSharedTreeGuard,
|
|
8534
|
+
checkSharedTreeGuard,
|
|
8535
|
+
uncommittedChanges,
|
|
8536
|
+
gitHead,
|
|
8537
|
+
selectResumeRecoveryTarget,
|
|
8538
|
+
buildResumeRecoveryPreamble,
|
|
8539
|
+
buildClaudeSpawnArgs,
|
|
8540
|
+
spawnResumeRecovery,
|
|
8541
|
+
selectMechanicalRecoveryTarget,
|
|
8542
|
+
performMechanicalRecovery,
|
|
8543
|
+
MECHANICALLY_RESOLVABLE_VERDICTS,
|
|
8544
|
+
selectLeftoverQuarantineTarget,
|
|
8545
|
+
quarantineLeftovers,
|
|
8546
|
+
performLeftoverQuarantine,
|
|
8547
|
+
spawnInvestigation,
|
|
8548
|
+
computeLaunchHolds,
|
|
8549
|
+
computeDepHistorySatisfaction,
|
|
8550
|
+
handleLaunchFailure,
|
|
8551
|
+
applyLaunchFailure,
|
|
8552
|
+
setPaused,
|
|
8553
|
+
clearPause,
|
|
8554
|
+
tickQueue,
|
|
8555
|
+
runDueJobs,
|
|
8556
|
+
isCooldownSuppressed,
|
|
8557
|
+
nextRapidRateLimitCount,
|
|
8558
|
+
CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
|
|
8559
|
+
RAPID_RATE_LIMIT_WINDOW_MS,
|
|
8560
|
+
MANUAL_PAUSE_COOLDOWN_MS,
|
|
8561
|
+
RUNS_DIR,
|
|
8562
|
+
pickRunDir,
|
|
8563
|
+
};
|