claude-code-session-manager 0.79.0 → 0.80.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/dist/assets/{AgentLibrary-COtVRqBR.js → AgentLibrary-zS3jw_1e.js} +1 -1
  2. package/dist/assets/{DataModel-CSEKw_OR.js → DataModel-Cy_vxTpi.js} +1 -1
  3. package/dist/assets/{History-CHHovrAO.js → History-C6JRuqfT.js} +1 -1
  4. package/dist/assets/{Hooks-BZU6C3x6.js → Hooks-BafPy9mB.js} +1 -1
  5. package/dist/assets/{HostBilko-CqTUoq37.js → HostBilko-BZwhQOFt.js} +1 -1
  6. package/dist/assets/{Library-BtxdyTLz.js → Library-C8JDDliz.js} +1 -1
  7. package/dist/assets/{ListDetail-qZc7Zm-6.js → ListDetail-CqiOdwLc.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-BHe_4fJR.js → MarkdownEditor-CyLyP67L.js} +1 -1
  9. package/dist/assets/{McpServers-7Z98HLNo.js → McpServers-BzMv-_84.js} +1 -1
  10. package/dist/assets/{Memory-CR72KoyP.js → Memory-DSBYQdJR.js} +1 -1
  11. package/dist/assets/{Panel-pL6H3dpQ.js → Panel-CLUhkNNA.js} +1 -1
  12. package/dist/assets/{Permissions-CWSWjyXM.js → Permissions-BfC2-HN4.js} +1 -1
  13. package/dist/assets/{Plugins-CN6lX2lt.js → Plugins-BKi40jT5.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-BXSXwIsk.js → ProvenanceBadge-BzFw4KhD.js} +1 -1
  15. package/dist/assets/{SaveBar-BlB5TGpR.js → SaveBar-avk2p9jv.js} +1 -1
  16. package/dist/assets/{Scheduler-DRciWUmR.js → Scheduler-Bf_6MdJo.js} +7 -7
  17. package/dist/assets/{ScopeSwitcher-kFrXtjpr.js → ScopeSwitcher-C-RwYUVZ.js} +1 -1
  18. package/dist/assets/{Settings-BXuyf4lJ.js → Settings-Djd8OoBA.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-Dfacb0PE.js → SkillReferenceGraph-DuogY6s7.js} +1 -1
  20. package/dist/assets/{Skills-CHqcpiyt.js → Skills-D_qAqxZ_.js} +1 -1
  21. package/dist/assets/{SystemPrompt-fxXm0BZr.js → SystemPrompt-DbHFLQV3.js} +1 -1
  22. package/dist/assets/{TagLibrary-DOz65ZTz.js → TagLibrary-C2y91BT0.js} +1 -1
  23. package/dist/assets/{TiptapBody-D0bWx_9o.js → TiptapBody-D9iz4xQx.js} +1 -1
  24. package/dist/assets/{Toggle-C9jBwGSx.js → Toggle-BGnFL2E5.js} +1 -1
  25. package/dist/assets/{index-DPYa6jbM.js → index-_2ARyFDj.js} +4 -4
  26. package/dist/assets/{settingsSchema-BTPw1bR3.js → settingsSchema-JK15eJU8.js} +1 -1
  27. package/dist/index.html +1 -1
  28. package/package.json +1 -1
  29. package/plugins/session-manager-dev/skills/builder/3-publish/SKILL.md +10 -0
  30. package/scripts/project-pages-logic/dist/logic.cjs +12 -12
  31. package/scripts/render-project-pages/dist/renderer.cjs +22 -22
  32. package/src/main/__tests__/computeDepHistorySatisfaction.test.cjs +66 -0
  33. package/src/main/__tests__/prdCreate.test.cjs +133 -8
  34. package/src/main/__tests__/prdFrontmatterDependsOn.test.cjs +136 -0
  35. package/src/main/__tests__/prdUpdateDependsOn.test.cjs +160 -0
  36. package/src/main/__tests__/queueHistory.test.cjs +33 -0
  37. package/src/main/__tests__/scheduleJobTransitions.test.cjs +1 -0
  38. package/src/main/__tests__/scheduler-autofix-outcome.test.cjs +73 -1
  39. package/src/main/__tests__/scheduler-autofix-select.test.cjs +17 -0
  40. package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +20 -0
  41. package/src/main/__tests__/scheduler-leftover-quarantine.test.cjs +199 -0
  42. package/src/main/__tests__/scheduler-mechanical-recovery.test.cjs +222 -0
  43. package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +51 -0
  44. package/src/main/__tests__/scheduler-resume-recovery.test.cjs +254 -0
  45. package/src/main/__tests__/schedulerBatchRootBlocker.test.cjs +117 -0
  46. package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -2
  47. package/src/main/ipcSchemas.cjs +15 -1
  48. package/src/main/lib/__tests__/gitWorktree.test.cjs +129 -10
  49. package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +59 -7
  50. package/src/main/lib/depSlugResolve.cjs +72 -0
  51. package/src/main/lib/epicWorktreeMerge.cjs +3 -3
  52. package/src/main/lib/epicWorktreeMint.cjs +17 -5
  53. package/src/main/lib/fixPlanSlug.cjs +62 -0
  54. package/src/main/lib/gitWorktree.cjs +97 -12
  55. package/src/main/lib/mcpToolCatalog.cjs +4 -1
  56. package/src/main/lib/prdCreate.cjs +84 -5
  57. package/src/main/lib/prdFrontmatter.cjs +56 -8
  58. package/src/main/lib/queueHistory.cjs +50 -5
  59. package/src/main/lib/scheduleJobTransitions.cjs +12 -2
  60. package/src/main/lib/schedulerBatch.cjs +181 -23
  61. package/src/main/scheduler/prdParser.cjs +7 -0
  62. package/src/main/scheduler.cjs +813 -40
  63. package/src/preload/api.d.ts +8 -0
@@ -73,6 +73,7 @@ const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs')
73
73
  const promptSessionTranscript = require('./promptSessionTranscript.cjs');
74
74
  const { verifyRun } = require('./runVerify.cjs');
75
75
  const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
76
+ const { isFixPlanSlug, classifyDiscoveredFixPlan, resolveIsFixPlan } = require('./lib/fixPlanSlug.cjs');
76
77
  const { landedSinceRun } = require('./lib/landedSinceRun.cjs');
77
78
  const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
78
79
  const logs = require('./logs.cjs');
@@ -99,7 +100,7 @@ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
99
100
  const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
100
101
  ? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
101
102
  : JOB_OVERRUN_FLOOR_MS_DEFAULT;
102
- const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
103
+ const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD, DEP_HISTORY_FAIL_OPEN } = require('./lib/schedulerBatch.cjs');
103
104
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
104
105
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
105
106
  const queueHistory = require('./lib/queueHistory.cjs');
@@ -140,6 +141,7 @@ const jobWorktree = require('./lib/jobWorktree.cjs');
140
141
  const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
141
142
  const queueStore = require('./lib/queueStore.cjs');
142
143
  const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
144
+ const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
143
145
  const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
144
146
  const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
145
147
 
@@ -2072,11 +2074,24 @@ async function reconcile(state) {
2072
2074
  exitCode: null,
2073
2075
  error: null,
2074
2076
  };
2075
- // Newly-discovered fix-plan PRD: stamp its investigationDepth relative to
2076
- // the original job it heals, so selectAutoFixTargets/spawnInvestigation
2077
- // can bound the fix-of-a-fix recursion (see MAX_INVESTIGATION_DEPTH).
2078
- // Non-fix-plan jobs get no explicit field — they read as depth 1 via `?? 1`.
2079
- if (isFixPlanSlug(slug)) {
2077
+ // Fix-plan classification (PRD 1131): a freshly-discovered PRD is a
2078
+ // genuine scheduler-authored fix plan only when its OWN provenance says
2079
+ // so — an explicit isFixPlan:true stamp (spawnInvestigation's prompt
2080
+ // template) or the absence of any createdVia stamp at all (legacy
2081
+ // fallback, matching the "no provenance = trust the name" rule the
2082
+ // quarantine gate below already applies) — never merely because the
2083
+ // slug looks like one. See lib/fixPlanSlug.cjs's header for why (PRD
2084
+ // 1126: a scheduler_create_prd-authored PRD whose slug happened to start
2085
+ // with "fix-" was wrongly stamped investigationDepth before it ever ran).
2086
+ // Persisted onto the queue row so every later consumer
2087
+ // (commitGuardVerdict, isFixPlanBeyondDepthCap, the fix-plan-completion
2088
+ // checks) reads this stamp instead of re-deriving it from the name.
2089
+ entry.isFixPlan = classifyDiscoveredFixPlan(p, slug);
2090
+ // Stamp investigationDepth relative to the original job it heals, so
2091
+ // selectAutoFixTargets/spawnInvestigation can bound the fix-of-a-fix
2092
+ // recursion (see MAX_INVESTIGATION_DEPTH). Non-fix-plan jobs get no
2093
+ // explicit field — they read as depth 1 via `?? 1`.
2094
+ if (entry.isFixPlan) {
2080
2095
  const parent = healTargetForFix(slug, state.jobs);
2081
2096
  entry.investigationDepth = parent ? (parent.investigationDepth ?? 1) + 1 : 2;
2082
2097
  }
@@ -2088,8 +2103,9 @@ async function reconcile(state) {
2088
2103
  // guard-prd-writes.cjs PreToolUse hook should have denied. Fix-plan PRDs
2089
2104
  // are exempt: spawnInvestigation's own probe writes them directly by
2090
2105
  // design (a trusted, scheduler-spawned internal loop, not an
2091
- // agent/human authoring a PRD), matching the isFixPlanSlug convention
2092
- // used everywhere else this distinction matters.
2106
+ // agent/human authoring a PRD) — entry.isFixPlan (just classified above)
2107
+ // is the provenance-aware verdict for that exemption now, not a raw
2108
+ // isFixPlanSlug name check.
2093
2109
  //
2094
2110
  // Quarantine is loud and reversible, never a silent skip (see the
2095
2111
  // 2026-08-01 23-PRD outage this file's header references for what a
@@ -2098,7 +2114,7 @@ async function reconcile(state) {
2098
2114
  // (schedule:adopt-prd) that stamps the file via the same update-prd API
2099
2115
  // route the MCP tool uses — reconcile()'s adopt path above promotes it
2100
2116
  // to 'pending' on the very next pass, within one tick of being stamped.
2101
- if (!p.createdVia && !isFixPlanSlug(slug)) {
2117
+ if (!p.createdVia && !entry.isFixPlan) {
2102
2118
  entry.status = 'quarantined';
2103
2119
  // Stamped at creation (not via transitionJob, since this is a
2104
2120
  // brand-new row minted directly at 'quarantined' rather than
@@ -2584,6 +2600,17 @@ function resetJobFields(job, errorMsg, opts = {}) {
2584
2600
  delete job.verifierVerdict;
2585
2601
  delete job.uncommittedPaths;
2586
2602
  delete job.resumeRecoveryAttempted;
2603
+ // Same one-attempt-per-episode category as resumeRecoveryAttempted above —
2604
+ // a re-fired row must be able to earn a fresh mechanical-recovery attempt
2605
+ // if it parks needs_review again (PRD 1130).
2606
+ delete job.mechanicalRecoveryAttempted;
2607
+ // Quarantine (PRD 1128) is scoped to THIS run's episode exactly like
2608
+ // resumeRecoveryAttempted above — a re-fired row must be able to earn a
2609
+ // fresh quarantine attempt if it parks needs_review again.
2610
+ delete job.leftoverQuarantineAttempted;
2611
+ delete job.quarantinedTo;
2612
+ delete job.quarantinedCommit;
2613
+ delete job.quarantinedPaths;
2587
2614
  // Same "this run's outcome, not durable across a reset" category as the
2588
2615
  // fields above — a stale 'archive' recoveryAction from a prior life of this
2589
2616
  // slug must never survive a reset and silently exclude a genuinely-new
@@ -2592,6 +2619,10 @@ function resetJobFields(job, errorMsg, opts = {}) {
2592
2619
  // can otherwise linger forever when RCA is disabled or errors).
2593
2620
  delete job.rcaFailureClass;
2594
2621
  delete job.rcaRecoveryAction;
2622
+ // Same "this run's outcome, not durable across a reset" category — a
2623
+ // human-driven reset must genuinely start the auto-fix budget over,
2624
+ // including the one-time dead-fix-plan-child reopen (PRD 1129).
2625
+ delete job.autoFixReopened;
2595
2626
  // Like exitCode: this run's outcome, not durable across a reset — a stale
2596
2627
  // leak badge from a prior attempt must not linger once the job re-fires.
2597
2628
  delete job.leakedDescendants;
@@ -3303,6 +3334,302 @@ As the LAST LINE of your final result text, emit exactly one of:
3303
3334
  Print PASS only once the commit above has actually landed.`;
3304
3335
  }
3305
3336
 
3337
+ /**
3338
+ * Mechanical recovery (PRD 1130). isFixPlanBeyondDepthCap (below) is the
3339
+ * ONLY gate on re-investigating a fix-plan job at investigationDepth >= 2 —
3340
+ * correct for open-ended "author another plan" recursion, but it also
3341
+ * strands a depth-capped job whose failure was fully mechanical (no
3342
+ * judgement required) with no other ladder rung, since resume-first recovery
3343
+ * (selectResumeRecoveryTarget above) is hard-gated on verdict
3344
+ * 'uncommitted_changes'. This rung is evaluated INDEPENDENTLY of
3345
+ * isFixPlanBeyondDepthCap — depth never disqualifies it, because unlike
3346
+ * auto-fix it authors no plan and spawns no model; it is pure git.
3347
+ *
3348
+ * The closed set of mechanically-resolvable verdicts starts at exactly
3349
+ * 'worktree_integration_failed': PRD 1125 already taught integrateBranch to
3350
+ * parse git's "would be overwritten by merge" stderr, verify the blocking
3351
+ * paths are byte-identical to the branch, discard the proven duplicates, and
3352
+ * retry the merge once. A job parked with this verdict has its `sm-job/
3353
+ * <slug>` branch preserved (integrateJobBranch never deletes the branch on
3354
+ * failure — see cleanupJobWorktree's `keepBranch: !integration.ok`), so a
3355
+ * plain re-call of integrateBranch against that same branch inherits PRD
3356
+ * 1125's auto-resolution for free — no re-implementation needed here.
3357
+ *
3358
+ * Bounded to exactly one attempt via job.mechanicalRecoveryAttempted,
3359
+ * stamped in the SAME mutate as the outcome (performMechanicalRecovery,
3360
+ * below) — never here — so this selector alone can be unit-tested exactly
3361
+ * like selectResumeRecoveryTarget/selectLeftoverQuarantineTarget.
3362
+ *
3363
+ * Kill-switch: SM_MECHANICAL_RECOVERY_DISABLE=1 restores today's behaviour
3364
+ * exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
3365
+ */
3366
+ const MECHANICALLY_RESOLVABLE_VERDICTS = new Set(['worktree_integration_failed']);
3367
+
3368
+ function selectMechanicalRecoveryTarget(job) {
3369
+ if (process.env.SM_MECHANICAL_RECOVERY_DISABLE === '1') return null;
3370
+ if (!job || job.status !== 'needs_review') return null;
3371
+ if (!MECHANICALLY_RESOLVABLE_VERDICTS.has(job.verifierVerdict)) return null;
3372
+ if (job.mechanicalRecoveryAttempted === true) return null;
3373
+ const cwd = job.cwd || DEFAULT_PROJECT_CWD;
3374
+ return { slug: job.slug, cwd, branch: jobWorktree.branchNameFor(job.slug), carriedPaths: job.carriedPaths || [] };
3375
+ }
3376
+
3377
+ /**
3378
+ * Perform an already-selected mechanical recovery (selectMechanicalRecoveryTarget
3379
+ * above) — a direct re-attempt of integrateBranch against the job's preserved
3380
+ * branch, never a fresh `claude -p` dispatch. On success the job transitions
3381
+ * needs_review -> completed and its verifierVerdict is cleared; the branch,
3382
+ * now merged, is deleted like any other successfully-integrated job branch.
3383
+ * On failure (including a branch that no longer exists — already deleted or
3384
+ * already merged) the job stays needs_review, mechanicalRecoveryAttempted is
3385
+ * stamped, and the retry's own failure text is appended to `error`. Either
3386
+ * way mechanicalRecoveryAttempted is stamped in this SAME mutate, so a crash
3387
+ * between the git call returning and this mutate landing simply repeats an
3388
+ * idempotent git operation on the next pass rather than leaving the job
3389
+ * re-eligible forever.
3390
+ */
3391
+ async function performMechanicalRecovery(job, target) {
3392
+ const integration = await jobWorktree.integrateJobBranch({
3393
+ cwd: target.cwd, branch: target.branch, slug: target.slug, carriedPaths: target.carriedPaths,
3394
+ });
3395
+ if (integration.ok) {
3396
+ await jobWorktree.cleanupJobWorktree({ cwd: target.cwd, dir: undefined, branch: target.branch, keepBranch: false });
3397
+ }
3398
+ let becameCompleted = false;
3399
+ await mutate((s) => {
3400
+ const j = s.jobs.find((x) => x.slug === job.slug);
3401
+ if (!j) return;
3402
+ j.mechanicalRecoveryAttempted = true;
3403
+ if (integration.ok) {
3404
+ if (transitionJob(j, 'completed', {
3405
+ reason: `mechanical recovery: ${target.branch} re-integrated successfully`,
3406
+ source: 'scheduler:mechanicalRecovery',
3407
+ })) {
3408
+ delete j.verifierVerdict;
3409
+ j.exitCode = 0;
3410
+ j.error = null;
3411
+ becameCompleted = true;
3412
+ }
3413
+ } else {
3414
+ const pointer = `Mechanical recovery retry failed: ${integration.reason}`;
3415
+ j.error = j.error ? `${j.error}\n${pointer}` : pointer;
3416
+ }
3417
+ });
3418
+ if (integration.ok) {
3419
+ console.log(`[scheduler] mechanical-recovery: ${job.slug} → completed (branch ${target.branch} re-integrated)`);
3420
+ if (becameCompleted) await archiveCompletedPrd(job.slug, job.cwd);
3421
+ } else {
3422
+ console.error(`[scheduler] mechanical-recovery: ${job.slug} → retry failed: ${integration.reason}`);
3423
+ }
3424
+ }
3425
+
3426
+ /**
3427
+ * Leftover quarantine (PRD 1128). Resume-first recovery gets exactly one
3428
+ * `--resume` attempt (selectResumeRecoveryTarget above); when that attempt
3429
+ * ALSO parks needs_review with 'uncommitted_changes', the leftovers are
3430
+ * about to sit dirty in the SHARED tree forever — git then refuses any later
3431
+ * worktree merge for this cwd that would overwrite them, turning one parked
3432
+ * job into a project-wide stall (216-jupiter-sand-kazekage, 2026-09-06).
3433
+ * Pure/no I/O, mirroring selectResumeRecoveryTarget so the eligibility rule
3434
+ * is unit-testable directly.
3435
+ *
3436
+ * Bounded to exactly one attempt via job.leftoverQuarantineAttempted, stamped
3437
+ * synchronously by the caller in the SAME mutate as this decision (never
3438
+ * here) — see spawnJob's finalize and reverifyNeedsReview's periodic pass.
3439
+ *
3440
+ * Kill-switch: SM_LEFTOVER_QUARANTINE_DISABLE=1 restores today's behaviour
3441
+ * exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
3442
+ */
3443
+ function selectLeftoverQuarantineTarget(job) {
3444
+ if (process.env.SM_LEFTOVER_QUARANTINE_DISABLE === '1') return null;
3445
+ if (!job || job.status !== 'needs_review') return null;
3446
+ if (job.verifierVerdict !== 'uncommitted_changes') return null;
3447
+ if (job.resumeRecoveryAttempted !== true) return null;
3448
+ if (job.leftoverQuarantineAttempted === true) return null;
3449
+ const uncommittedPaths = Array.isArray(job.uncommittedPaths)
3450
+ ? job.uncommittedPaths.filter((p) => typeof p === 'string' && p.length > 0)
3451
+ : [];
3452
+ if (!uncommittedPaths.length) return null;
3453
+ // The single most important constraint: never touch a path that was
3454
+ // ALREADY dirty at this run's own dispatch time (preRunDirtyPaths) — that
3455
+ // is foreign WIP (a human's or a sibling's), not this job's own leftover.
3456
+ const preRunDirty = new Set(Array.isArray(job.preRunDirtyPaths) ? job.preRunDirtyPaths : []);
3457
+ const paths = uncommittedPaths.filter((p) => !preRunDirty.has(p));
3458
+ if (!paths.length) return null;
3459
+ return { slug: job.slug, cwd: job.cwd, paths };
3460
+ }
3461
+
3462
+ function execGitAt(cwd, args, { env, timeout = 20_000 } = {}) {
3463
+ return new Promise((resolve, reject) => {
3464
+ execFile(
3465
+ 'git',
3466
+ ['-C', cwd, ...args],
3467
+ { timeout, windowsHide: true, encoding: 'utf8', env: env ? { ...process.env, ...env } : process.env },
3468
+ (err, stdout, stderr) => {
3469
+ if (err) {
3470
+ err.stderrText = stderr;
3471
+ reject(err);
3472
+ return;
3473
+ }
3474
+ resolve(stdout || '');
3475
+ },
3476
+ );
3477
+ });
3478
+ }
3479
+
3480
+ async function pathExistsInTree(cwd, treeish, p) {
3481
+ try {
3482
+ await execGitAt(cwd, ['cat-file', '-e', `${treeish}:${p}`]);
3483
+ return true;
3484
+ } catch {
3485
+ return false;
3486
+ }
3487
+ }
3488
+
3489
+ /**
3490
+ * Commit exactly `paths` (must already be dirty on disk) onto a dedicated
3491
+ * `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
3492
+ * unavailable) via a THROWAWAY `GIT_INDEX_FILE` — never touches the live
3493
+ * index, never moves the checked-out branch — then restores those paths to
3494
+ * match that baseline commit's tree, so the shared working tree returns to
3495
+ * its pre-run state. This is deliberately NOT `git stash` (the destructive-
3496
+ * git guard blocks stash on a shared tree, and a stash nobody restores
3497
+ * strands the work invisibly — see standards.md).
3498
+ *
3499
+ * Never throws: any git failure, or a non-git cwd, aborts the WHOLE attempt
3500
+ * with the tree untouched (no partial restore) — restore only ever runs
3501
+ * after the salvage ref/commit has safely landed, so a failure there leaves
3502
+ * the data recoverable from the ref even though the tree stayed dirty.
3503
+ * A path no longer dirty on disk (already committed, or reverted since) is
3504
+ * skipped, never force-restored.
3505
+ */
3506
+ async function quarantineLeftovers({ cwd, slug, paths, headBefore }) {
3507
+ if (!cwd || !slug || !Array.isArray(paths) || paths.length === 0) {
3508
+ return { ok: false, reason: 'no cwd/slug/paths given' };
3509
+ }
3510
+ let baseline = headBefore || null;
3511
+ try {
3512
+ if (!baseline) {
3513
+ baseline = (await execGitAt(cwd, ['rev-parse', 'HEAD'])).trim();
3514
+ }
3515
+ if (!baseline) return { ok: false, reason: 'could not resolve a baseline commit (non-git cwd?)' };
3516
+
3517
+ const dirtyNowRaw = await execGitAt(cwd, ['status', '--porcelain', '--', ...paths]);
3518
+ const dirtyNow = new Set(parsePorcelain(dirtyNowRaw));
3519
+ const toQuarantine = paths.filter((p) => dirtyNow.has(p));
3520
+ const skippedPaths = paths.filter((p) => !dirtyNow.has(p));
3521
+ if (!toQuarantine.length) {
3522
+ return { ok: true, ref: null, commit: null, quarantinedPaths: [], skippedPaths };
3523
+ }
3524
+
3525
+ const tmpIndex = path.join(os.tmpdir(), `sm-salvage-index-${slug}-${process.pid}-${Date.now()}`);
3526
+ const env = { GIT_INDEX_FILE: tmpIndex };
3527
+ let treeSha;
3528
+ let commitSha;
3529
+ try {
3530
+ await execGitAt(cwd, ['read-tree', baseline], { env });
3531
+ for (const p of toQuarantine) {
3532
+ if (fs.existsSync(path.join(cwd, p))) {
3533
+ await execGitAt(cwd, ['add', '--', p], { env });
3534
+ } else {
3535
+ await execGitAt(cwd, ['rm', '--cached', '--ignore-unmatch', '--', p], { env });
3536
+ }
3537
+ }
3538
+ treeSha = (await execGitAt(cwd, ['write-tree'], { env })).trim();
3539
+ commitSha = (await execGitAt(cwd, ['commit-tree', treeSha, '-p', baseline, '-m', `salvage: leftover changes from ${slug}`], { env })).trim();
3540
+ } catch (e) {
3541
+ return { ok: false, reason: `git command failed while building the salvage commit: ${(e && (e.stderrText || e.message)) || e}` };
3542
+ } finally {
3543
+ await fsp.rm(tmpIndex, { force: true }).catch(() => {});
3544
+ }
3545
+
3546
+ const ref = `sm-salvage/${slug}`;
3547
+ try {
3548
+ await execGitAt(cwd, ['update-ref', `refs/heads/${ref}`, commitSha]);
3549
+ } catch (e) {
3550
+ return { ok: false, reason: `git command failed updating ${ref}: ${(e && (e.stderrText || e.message)) || e}` };
3551
+ }
3552
+
3553
+ // The salvage commit is safely landed at this point — a failure from here
3554
+ // on is reported with the ref/commit still attached so nothing looks lost
3555
+ // even if the tree itself couldn't be fully restored.
3556
+ try {
3557
+ const inBaseline = [];
3558
+ const notInBaseline = [];
3559
+ for (const p of toQuarantine) {
3560
+ // eslint-disable-next-line no-await-in-loop
3561
+ if (await pathExistsInTree(cwd, baseline, p)) inBaseline.push(p); else notInBaseline.push(p);
3562
+ }
3563
+ if (inBaseline.length) {
3564
+ await execGitAt(cwd, ['checkout', baseline, '--', ...inBaseline]);
3565
+ }
3566
+ if (notInBaseline.length) {
3567
+ await execGitAt(cwd, ['reset', '--', ...notInBaseline]).catch(() => {});
3568
+ for (const p of notInBaseline) {
3569
+ // eslint-disable-next-line no-await-in-loop
3570
+ await fsp.rm(path.join(cwd, p), { force: true });
3571
+ }
3572
+ }
3573
+ } catch (e) {
3574
+ return {
3575
+ ok: false,
3576
+ ref,
3577
+ commit: commitSha,
3578
+ reason: `salvage commit landed at ${ref} (${commitSha}) but restoring the working tree failed: ${(e && (e.stderrText || e.message)) || e}`,
3579
+ };
3580
+ }
3581
+
3582
+ return { ok: true, ref, commit: commitSha, quarantinedPaths: toQuarantine, skippedPaths };
3583
+ } catch (e) {
3584
+ return { ok: false, reason: `git command failed: ${(e && (e.stderrText || e.message)) || e}` };
3585
+ }
3586
+ }
3587
+
3588
+ /**
3589
+ * Perform an already-selected quarantine (job.leftoverQuarantineAttempted
3590
+ * must already be true, stamped by the caller) and persist the outcome onto
3591
+ * the job row: `quarantinedTo`/`quarantinedCommit`/`quarantinedPaths` on
3592
+ * success, plus a one-line pointer appended to `error` naming the ref so a
3593
+ * human can recover with a single named command
3594
+ * (`git show sm-salvage/<slug>`). The belt-and-braces `salvagePatch` (when
3595
+ * present) is referenced alongside it, never removed. On failure, only a
3596
+ * diagnostic is appended — the job row's dirt-describing fields are left as
3597
+ * they were, since the tree itself was left untouched (or, for a
3598
+ * restore-only failure, the salvage ref is still named in the note).
3599
+ *
3600
+ * `headBefore`, when the caller has it fresh (spawnJob's own finalize still
3601
+ * has the local `guardHeadBefore` in scope for the run that just parked —
3602
+ * the same value is deleted off the job ROW earlier in that same finalize),
3603
+ * is used as the salvage ref's baseline commit; otherwise (the periodic
3604
+ * reverifyNeedsReview pass, re-discovering an already-parked row) this falls
3605
+ * back to the current HEAD inside quarantineLeftovers itself.
3606
+ */
3607
+ async function performLeftoverQuarantine(job, paths, headBefore = null) {
3608
+ const result = await quarantineLeftovers({
3609
+ cwd: job.cwd || DEFAULT_PROJECT_CWD,
3610
+ slug: job.slug,
3611
+ paths,
3612
+ headBefore: headBefore || job.guardHeadBefore || null,
3613
+ });
3614
+ await mutate((s) => {
3615
+ const j = s.jobs.find((x) => x.slug === job.slug);
3616
+ if (!j) return;
3617
+ if (result.ok && Array.isArray(result.quarantinedPaths) && result.quarantinedPaths.length) {
3618
+ j.quarantinedTo = result.ref;
3619
+ j.quarantinedCommit = result.commit;
3620
+ j.quarantinedPaths = capDirtyPaths(result.quarantinedPaths);
3621
+ const salvageNote = j.salvagePatch ? `; salvage patch also at ${j.salvagePatch}` : '';
3622
+ const pointer = `Leftovers quarantined to ${result.ref} (commit ${result.commit}) — recover via \`git show ${result.ref}\`${salvageNote}`;
3623
+ j.error = j.error ? `${j.error}\n${pointer}` : pointer;
3624
+ console.log(`[scheduler] ${job.slug}: quarantined ${result.quarantinedPaths.length} leftover path(s) to ${result.ref} (${result.commit})`);
3625
+ } else if (!result.ok) {
3626
+ const pointer = `Leftover quarantine failed: ${result.reason}`;
3627
+ j.error = j.error ? `${j.error}\n${pointer}` : pointer;
3628
+ console.error(`[scheduler] ${job.slug}: leftover quarantine failed: ${result.reason}`);
3629
+ }
3630
+ });
3631
+ }
3632
+
3306
3633
  /**
3307
3634
  * Pure argv builder for a `claude -p` child spawn, shared so the
3308
3635
  * resume-vs-fresh-session choice is made in exactly one place. `resume`
@@ -3782,13 +4109,9 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
3782
4109
  // project (keyed by cwd) so jobs in different repos run concurrently up to
3783
4110
  // the cap; within one project, sequential-group semantics are preserved.
3784
4111
 
3785
- /**
3786
- * Recognize fix-plan slugs (NN-fix-...) so we don't recurse on a fix-plan that
3787
- * itself failed. The pattern matches the slug we generate in spawnInvestigation.
3788
- */
3789
- function isFixPlanSlug(slug) {
3790
- return /^\d+-fix-/.test(slug);
3791
- }
4112
+ // isFixPlanSlug/classifyDiscoveredFixPlan/resolveIsFixPlan now live in
4113
+ // lib/fixPlanSlug.cjs (PRD 1131) — see that module's header for why slug
4114
+ // shape alone is no longer sufficient to classify a fix plan.
3792
4115
 
3793
4116
  /**
3794
4117
  * The fix-plan slug spawnInvestigation authors for a given failed job —
@@ -3827,7 +4150,25 @@ function healTargetForFix(fixSlug, jobs) {
3827
4150
  * unit-tested (no spawn, no fs). Inputs are the already-resolved values that
3828
4151
  * spawnInvestigation computes.
3829
4152
  */
3830
- function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group }) {
4153
+ function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild = null }) {
4154
+ const deadFixChildNote = deadChild ? `
4155
+
4156
+ # This is a REOPENED investigation — your own prior fix plan died
4157
+ You already investigated this job once and produced a fix-plan PRD, \`${deadChild.slug}\`, which was
4158
+ supposed to heal it. That fix-plan job itself reached a terminal, non-completed status
4159
+ (\`${deadChild.status}\`) without ever fixing the original failure — so the parent job you are now
4160
+ investigating is stuck again with nothing left to retry it automatically. This is the ONE reopen
4161
+ this parent gets; do not cold-read the log and re-derive the plan that already failed.
4162
+
4163
+ Dead fix-plan child's own outcome:
4164
+ - Slug: ${deadChild.slug}
4165
+ - Status: ${deadChild.status}
4166
+ - Verifier verdict: ${deadChild.verifierVerdict ?? '(none recorded)'}
4167
+ - Error: ${deadChild.error ?? '(none recorded)'}
4168
+
4169
+ Read why THAT job died (its own run log, if any, under the runs directory) before writing a new
4170
+ fix-plan PRD, and make sure your new plan is genuinely different from — not a repeat of — whatever
4171
+ that dead child attempted.` : '';
3831
4172
  const abandonedBackgroundTaskNote = failedJob.verifierVerdict === 'abandoned_background_task' ? `
3832
4173
 
3833
4174
  # Known failure class: abandoned background task
@@ -3848,7 +4189,7 @@ The fix-plan PRD you write for this MUST instruct its executor to, in order:
3848
4189
  1. Check for a salvage patch (named \`<slug>.uncommitted.patch\` in the run directory${failedJob.salvagePatch ? `, e.g. \`${failedJob.salvagePatch}\`` : ''}) and, if found, apply it to the working tree BEFORE inspecting \`git status\`/\`git diff\` in ${cwd} for uncommitted changes matching the original PRD's acceptance criteria.
3849
4190
  2. If the work is present (via the applied patch or already in the tree) and satisfies the acceptance criteria, run the project's verify commands and COMMIT it — do not re-implement or re-plan the PRD from scratch.
3850
4191
  3. Only fall back to re-implementing whatever acceptance criteria are genuinely missing after applying any salvage patch, not the whole PRD.` : '';
3851
- return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${abandonedBackgroundTaskNote}
4192
+ return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${deadFixChildNote}${abandonedBackgroundTaskNote}
3852
4193
 
3853
4194
  # Failed job
3854
4195
  - Slug: ${failedJob.slug}
@@ -3893,8 +4234,13 @@ ${logTail}
3893
4234
  cwd: ${cwd}
3894
4235
  parallelGroup: ${group}
3895
4236
  estimateMinutes: <your time estimate>
4237
+ isFixPlan: true
3896
4238
  ---
3897
4239
  \`\`\`
4240
+ \`isFixPlan: true\` is REQUIRED — it is the scheduler's provenance signal that this PRD is a
4241
+ genuine auto-authored fix plan (not a human/agent PRD whose slug merely happens to start with
4242
+ "fix-"); omitting it means this fix plan will not get its depth-cap/zero-edit-commit-guard
4243
+ exemptions.
3898
4244
  \`cwd\` must be the git repo root where the fix will actually land. If the failed job's cwd is
3899
4245
  not that repo (e.g. a scratch dir like \`/tmp\`), set \`cwd:\` to the correct repo root instead —
3900
4246
  the scheduler's commit guard and post-run verifier read git state from this path, and a
@@ -3968,7 +4314,7 @@ function readRunOutcomeSidecars(runDir, slug) {
3968
4314
  */
3969
4315
  const INVESTIGATION_LAUNCH_KEY = 'investigation';
3970
4316
 
3971
- async function spawnInvestigation(failedJob, runDir) {
4317
+ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {}) {
3972
4318
  // The probe launches with the same CLI as the job it diagnoses. While
3973
4319
  // that CLI cannot launch at all (launch circuit breaker, issue #11 list
3974
4320
  // B1: probes e4f82da2/d374e6bf died on the same HTTP 400 as the runs
@@ -3997,7 +4343,14 @@ async function spawnInvestigation(failedJob, runDir) {
3997
4343
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is resume-recovery eligible`);
3998
4344
  return { deferred: false };
3999
4345
  }
4000
- if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth)) {
4346
+ // Mechanical recovery (PRD 1130): same first-refusal treatment — a job
4347
+ // eligible for a pure-git retry must never also get a cold-read fix-plan
4348
+ // PRD authored in the same pass.
4349
+ if (selectMechanicalRecoveryTarget(failedJob)) {
4350
+ console.log(`[scheduler] skip investigation: ${failedJob.slug} is mechanical-recovery eligible`);
4351
+ return { deferred: false };
4352
+ }
4353
+ if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth, failedJob.isFixPlan)) {
4001
4354
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
4002
4355
  return { deferred: false };
4003
4356
  }
@@ -4054,7 +4407,13 @@ async function spawnInvestigation(failedJob, runDir) {
4054
4407
 
4055
4408
  const logTail = readTail(failedLogPath, 16 * 1024) || '(failed to read log)';
4056
4409
 
4057
- if (fs.existsSync(fixPath)) {
4410
+ // A dead-fix-plan reopen (PRD 1129) targets the SAME fixPath its dead
4411
+ // child was originally authored at, by construction (fixSlugFor is a pure
4412
+ // function of the parent) — the file existing is not staleness here, it's
4413
+ // the whole reason a reopen was offered. Skip the guard in that one case
4414
+ // so the second investigation can overwrite the dead plan; every other
4415
+ // caller keeps the original protection against clobbering a live sibling.
4416
+ if (fs.existsSync(fixPath) && !deadChild) {
4058
4417
  console.log(`[scheduler] skip investigation: fix plan already exists at ${fixPath}`);
4059
4418
  releaseSlot();
4060
4419
  return { deferred: false };
@@ -4080,7 +4439,7 @@ async function spawnInvestigation(failedJob, runDir) {
4080
4439
  console.warn(`[scheduler] investigation cwd is not a git repo (${cwd}); falling back to ${DEFAULT_PROJECT_CWD}`);
4081
4440
  cwd = DEFAULT_PROJECT_CWD;
4082
4441
  }
4083
- const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group });
4442
+ const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild });
4084
4443
 
4085
4444
  // Phase 1: open log fd for pre-spawn diagnostics.
4086
4445
  const { fd, safeLog, closeFd } = openLog(investigationLogPath);
@@ -4209,6 +4568,22 @@ async function spawnInvestigation(failedJob, runDir) {
4209
4568
  mutate((s) => {
4210
4569
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
4211
4570
  if (j) j.autoFixOutcome = 'plan';
4571
+ // Dead-fix-plan reopen (PRD 1129): fixSlugFor is a pure function of
4572
+ // the parent, so the freshly-authored plan landed at the SAME slug
4573
+ // as the dead child — reconcile() sees an already-known slug and
4574
+ // will never re-mint a pending row for it. Explicitly reset the
4575
+ // dead child's own row here so the overwritten plan actually gets
4576
+ // a chance to run, rather than sitting inert behind a permanently
4577
+ // terminal queue row. force:true because 'skipped' (a valid dead
4578
+ // status here) is otherwise reset-refused by design.
4579
+ if (deadChild) {
4580
+ const child = s.jobs.find((x) => x.slug === deadChild.slug);
4581
+ if (child) {
4582
+ resetJobFields(child, 'reset by dead-fix-plan reopen: parent investigation authored a new plan', {
4583
+ force: true, source: 'spawnInvestigation:dead-fix-plan-reopen',
4584
+ });
4585
+ }
4586
+ }
4212
4587
  }).catch(() => {});
4213
4588
  } else {
4214
4589
  console.log(`[scheduler] investigation finished WITHOUT producing fix plan (slug=${failedJob.slug}, code=${exitCode})`);
@@ -4291,6 +4666,58 @@ async function computeLaunchHolds(state, { now = Date.now(), claudeVersion } = {
4291
4666
  return held;
4292
4667
  }
4293
4668
 
4669
+ /**
4670
+ * computeDepHistorySatisfaction(state) → Map<cwd, Set<string>|symbol>
4671
+ *
4672
+ * PRD 1122's once-per-tick dependsOn history/archive lookup: for every
4673
+ * distinct project cwd with jobs this tick, builds the set of dep slugs that
4674
+ * have no live queue row but are nonetheless known-satisfied — a completed
4675
+ * record in that project's own `state/history.jsonl` shard
4676
+ * (queueHistory.completedSlugsForCwd, scoped per-project so a same-named PRD
4677
+ * in an unrelated project can never satisfy a dep here), or a `.md` file
4678
+ * under any of that project's `prds-archived/` dirs (listArchivedPrdDirs —
4679
+ * covers both the retired flat layout and every Epic's own sibling archive).
4680
+ * findBlockingDep (schedulerBatch.cjs) treats a dep slug as blocking
4681
+ * whenever it has no live row AND is absent from this set, so a typo or a
4682
+ * double-prefixed slug (the exact 2026-09-06 starry-night-ships incident)
4683
+ * HOLDS its dependent instead of silently dispatching it.
4684
+ *
4685
+ * Fails OPEN per project, never queue-wide: a history-shard or archive-scan
4686
+ * read error for one cwd degrades that cwd's value to
4687
+ * `DEP_HISTORY_FAIL_OPEN` (findBlockingDep then treats every rowless dep in
4688
+ * that project as satisfied, exactly today's pre-1122 behaviour) with a
4689
+ * logged warning — it never throws out of this function and never blocks
4690
+ * every OTHER project's dispatch for one project's bad fs state.
4691
+ *
4692
+ * Computed ONCE here, before pickNextBatch runs, and threaded down as pure
4693
+ * data (quietOpts.satisfiedSlugsByCwd) — schedulerBatch.cjs itself does no
4694
+ * I/O, so this is the only fs read this gate costs per tick, not one per job
4695
+ * per dep.
4696
+ */
4697
+ async function computeDepHistorySatisfaction(state) {
4698
+ const byCwd = new Map();
4699
+ const cwds = new Set((state?.jobs || []).map((j) => j.cwd || DEFAULT_PROJECT_CWD));
4700
+ for (const cwd of cwds) {
4701
+ const satisfied = new Set();
4702
+ try {
4703
+ for (const slug of await queueHistory.completedSlugsForCwd(cwd)) satisfied.add(slug);
4704
+ for (const dir of listArchivedPrdDirs(cwd)) {
4705
+ let entries;
4706
+ try { entries = await fsp.readdir(dir); } catch { continue; }
4707
+ for (const name of entries) {
4708
+ if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
4709
+ }
4710
+ }
4711
+ } catch (e) {
4712
+ console.warn(`[scheduler] depHistorySatisfaction: history/archive lookup failed for ${cwd} (${e?.message}) — falling back to fail-open dep resolution for this project this tick`);
4713
+ byCwd.set(cwd, DEP_HISTORY_FAIL_OPEN);
4714
+ continue;
4715
+ }
4716
+ byCwd.set(cwd, satisfied);
4717
+ }
4718
+ return byCwd;
4719
+ }
4720
+
4294
4721
  /**
4295
4722
  * A run that never got a turn (res.launchFailure — see executeJob's onExit)
4296
4723
  * is routed here instead of the failed/investigation path (issue #11 lists
@@ -4586,6 +5013,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4586
5013
  let res;
4587
5014
  let worktreeLeftoverDirty = [];
4588
5015
  let worktreeIntegrationFailure = null;
5016
+ // Set only when integrateJobBranch's stderr-parsing auto-resolve fired
5017
+ // (PRD 1125) — surfaced on the job row so the Queue UI can say the merge
5018
+ // self-healed rather than silently looking like an ordinary merge.
5019
+ let mergeAutoResolved = null;
5020
+ let mergeAutoResolvedPaths = null;
4589
5021
  // A job's uncommitted-work patch, whichever isolation mode produced it —
4590
5022
  // set by EITHER branch below, never both (worktree.ok picks exactly one
4591
5023
  // shape for the whole run). Named generically (not "worktree...") because
@@ -4633,6 +5065,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4633
5065
  console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
4634
5066
  } else if (integration.integrated) {
4635
5067
  console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
5068
+ if (integration.autoResolved) {
5069
+ mergeAutoResolved = integration.autoResolved;
5070
+ mergeAutoResolvedPaths = integration.resolvedPaths || [];
5071
+ console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
5072
+ }
4636
5073
  }
4637
5074
  await jobWorktree.cleanupJobWorktree({
4638
5075
  cwd: guardCwd,
@@ -4868,7 +5305,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4868
5305
  ranInWorktree: worktree.ok,
4869
5306
  jobSelfCommitted,
4870
5307
  legitimateNoOp: guardIsLegitimateNoOp,
4871
- isFixPlanJob: isFixPlanSlug(job.slug),
5308
+ isFixPlanJob: resolveIsFixPlan(job.slug, job.isFixPlan),
4872
5309
  verifyResult,
4873
5310
  salvagePatch,
4874
5311
  });
@@ -4945,9 +5382,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4945
5382
  let failedJobSnapshot = null;
4946
5383
  let needsInvestigationNow = false;
4947
5384
  let investigationJobSnapshot = null;
5385
+ let investigationDeadChildSnapshot = null;
4948
5386
  let needsReviewRcaSnapshot = null;
4949
5387
  let resumeRecoveryJob = null;
4950
5388
  let resumeRecoveryTarget = null;
5389
+ let quarantineJob = null;
5390
+ let quarantinePaths = null;
5391
+ let mechanicalRecoveryJob = null;
5392
+ let mechanicalRecoveryTarget = null;
4951
5393
  let terminalNotifySnapshot = null;
4952
5394
  const newlyCompletedPrds = [];
4953
5395
  await mutate((s) => {
@@ -5042,6 +5484,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5042
5484
  } else {
5043
5485
  delete s.jobs[i2].uncommittedPaths;
5044
5486
  }
5487
+ // Worktree merge self-healed (PRD 1125) — every blocking path was
5488
+ // proven byte-identical to the branch, so the duplicate was
5489
+ // discarded and the merge retried once, successfully. Surfaced so
5490
+ // the Queue UI shows a self-heal instead of an ordinary merge.
5491
+ if (mergeAutoResolved) {
5492
+ s.jobs[i2].mergeAutoResolved = mergeAutoResolved;
5493
+ s.jobs[i2].mergeAutoResolvedPaths = capDirtyPaths(mergeAutoResolvedPaths);
5494
+ } else {
5495
+ delete s.jobs[i2].mergeAutoResolved;
5496
+ delete s.jobs[i2].mergeAutoResolvedPaths;
5497
+ }
5045
5498
  // Non-blocking notes (e.g. a recovered missing-dependency probe, or a
5046
5499
  // pattern hit demoted because a materially-checkable verdict outranked
5047
5500
  // it) — surfaced even on completed jobs so the signal isn't lost.
@@ -5092,6 +5545,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5092
5545
  // takes the treatAsPending branch above and never reaches here).
5093
5546
  needsReviewRcaSnapshot = { ...s.jobs[i2] };
5094
5547
 
5548
+ // Mechanical recovery (PRD 1130): evaluated FIRST, ahead of both
5549
+ // resume-first recovery and auto-fix — a job parked with a
5550
+ // mechanically-resolvable verdict (see
5551
+ // selectMechanicalRecoveryTarget) needs no model, no plan, and no
5552
+ // depth-cap check, so it must never fall through to either.
5553
+ // Snapshot only (no I/O inside mutate()); the actual git retry
5554
+ // happens outside mutate(), below.
5555
+ const mTarget = selectMechanicalRecoveryTarget(s.jobs[i2]);
5556
+ if (mTarget) {
5557
+ mechanicalRecoveryJob = { ...s.jobs[i2] };
5558
+ mechanicalRecoveryTarget = mTarget;
5559
+ } else {
5095
5560
  // Resume-first recovery (PRD 1111): evaluated BEFORE the auto-fix
5096
5561
  // eligibility check below — a job whose verdict is
5097
5562
  // 'uncommitted_changes' with a live sessionId gets one bounded
@@ -5105,6 +5570,22 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5105
5570
  resumeRecoveryJob = { ...s.jobs[i2] };
5106
5571
  resumeRecoveryTarget = target;
5107
5572
  } else {
5573
+ // Leftover quarantine (PRD 1128): resume recovery is spent
5574
+ // (resumeRecoveryAttempted already true) and this run STILL parked
5575
+ // needs_review with uncommitted_changes — the leftovers are about
5576
+ // to sit dirty in the shared tree forever, poisoning every later
5577
+ // worktree merge for this cwd. Stamp the one-attempt marker HERE,
5578
+ // synchronously in the same mutate as this decision (mirrors
5579
+ // resumeRecoveryAttempted's own stamp-before-acting rule above),
5580
+ // so a concurrent reverifyNeedsReview pass can never double-fire
5581
+ // this. The actual git work is async and runs outside mutate(),
5582
+ // below (performLeftoverQuarantine).
5583
+ const quarantineTarget = selectLeftoverQuarantineTarget(s.jobs[i2]);
5584
+ if (quarantineTarget) {
5585
+ s.jobs[i2].leftoverQuarantineAttempted = true;
5586
+ quarantineJob = { ...s.jobs[i2] };
5587
+ quarantinePaths = quarantineTarget.paths;
5588
+ }
5108
5589
  // Same-tick auto-fix (feedback 2026-07-12): rather than waiting up to
5109
5590
  // 10 min for reverifyNeedsReview()'s periodic pass, check right here
5110
5591
  // whether this job qualifies for auto-fix (same eligibility rule
@@ -5119,8 +5600,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5119
5600
  isEligibleForImmediateAutoFix(s.jobs[i2], s.jobs, fixSlugExists)
5120
5601
  ) {
5121
5602
  const isRetryAttempt = s.jobs[i2].autoFixAttempted === true;
5603
+ const isDeadFixPlanReopen = isFixPlanDead(s.jobs[i2], s.jobs);
5604
+ if (isDeadFixPlanReopen) {
5605
+ investigationDeadChildSnapshot = s.jobs.find((x) => x.slug === fixSlugFor(s.jobs[i2])) || null;
5606
+ }
5122
5607
  s.jobs[i2].autoFixAttempted = true;
5123
5608
  if (!s.jobs[i2].runId) s.jobs[i2].runId = runId;
5609
+ if (isDeadFixPlanReopen) s.jobs[i2].autoFixReopened = true;
5124
5610
  if (isRetryAttempt) {
5125
5611
  s.jobs[i2].autoFixRetries = (s.jobs[i2].autoFixRetries ?? 0) + 1;
5126
5612
  delete s.jobs[i2].autoFixOutcome;
@@ -5129,13 +5615,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5129
5615
  investigationJobSnapshot = { ...s.jobs[i2] };
5130
5616
  }
5131
5617
  }
5618
+ }
5132
5619
  }
5133
5620
  // Auto-promote: when a fix-* PRD completes successfully, the original
5134
5621
  // failed PRD's work is logically done. Flip its status to 'completed'
5135
5622
  // so the cross-group failure gate in pickNextBatch releases. Without
5136
5623
  // this, the queue stalls indefinitely behind a stale failure even
5137
5624
  // though the auto-recovery did its job.
5138
- if (effectiveStatus === 'completed' && isFixPlanSlug(job.slug)) {
5625
+ if (effectiveStatus === 'completed' && resolveIsFixPlan(job.slug, job.isFixPlan)) {
5139
5626
  const orig = healTargetForFix(job.slug, s.jobs);
5140
5627
  if (orig) {
5141
5628
  const priorStatus = orig.status;
@@ -5207,6 +5694,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5207
5694
  });
5208
5695
  }
5209
5696
 
5697
+ if (mechanicalRecoveryJob && mechanicalRecoveryTarget) {
5698
+ console.log(`[scheduler] needs_review ${job.slug} → mechanical-recovery (re-integrating ${mechanicalRecoveryTarget.branch})`);
5699
+ performMechanicalRecovery(mechanicalRecoveryJob, mechanicalRecoveryTarget).catch((e) => {
5700
+ console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
5701
+ });
5702
+ }
5703
+
5210
5704
  if (resumeRecoveryJob && resumeRecoveryTarget) {
5211
5705
  console.log(`[scheduler] needs_review ${job.slug} → resume-recovery (session ${resumeRecoveryTarget.sessionId}, ${resumeRecoveryTarget.dirtyPaths.length} dirty path(s))`);
5212
5706
  spawnResumeRecovery(resumeRecoveryJob, resumeRecoveryTarget).catch((e) => {
@@ -5214,6 +5708,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5214
5708
  });
5215
5709
  }
5216
5710
 
5711
+ if (quarantineJob && quarantinePaths) {
5712
+ console.log(`[scheduler] needs_review ${job.slug} → quarantining ${quarantinePaths.length} leftover path(s) (resume recovery already spent)`);
5713
+ performLeftoverQuarantine(quarantineJob, quarantinePaths, guardHeadBefore).catch((e) => {
5714
+ console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
5715
+ });
5716
+ }
5717
+
5217
5718
  if (actuallyFailed && failedJobSnapshot) {
5218
5719
  // Transient-failure detector. A 143/137 exit is ALWAYS a signal kill — the
5219
5720
  // agent never self-exits with those — so the only question is WHO killed it.
@@ -5286,7 +5787,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5286
5787
  }
5287
5788
  } else if (needsInvestigationNow && investigationJobSnapshot) {
5288
5789
  console.log(`[scheduler] needs_review ${job.slug} → immediate auto-fix investigation (not waiting for periodic reverify)`);
5289
- spawnInvestigation(investigationJobSnapshot, runDir).catch((e) => {
5790
+ spawnInvestigation(investigationJobSnapshot, runDir, { deadChild: investigationDeadChildSnapshot }).catch((e) => {
5290
5791
  console.error('[scheduler] spawnInvestigation error', job.slug, e);
5291
5792
  });
5292
5793
  }
@@ -5356,11 +5857,13 @@ function tickQueue({ bypassLoadGate = false } = {}) {
5356
5857
  // ceilinged the queue at 3 while the pool the user configured said 5.
5357
5858
  const freeSlots = sessionSlots.available();
5358
5859
  const heldSlugs = await computeLaunchHolds(state);
5860
+ const satisfiedSlugsByCwd = await computeDepHistorySatisfaction(state);
5359
5861
  const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
5360
5862
  leaseHeld: quietMachineLease.isHeld(),
5361
5863
  machineInUse: sessionSlots.inUse(),
5362
5864
  now: Date.now(),
5363
5865
  heldSlugs,
5866
+ satisfiedSlugsByCwd,
5364
5867
  });
5365
5868
  if (batch.length === 0 && freeSlots === 0) {
5366
5869
  const snap = sessionSlots.snapshot();
@@ -6025,10 +6528,16 @@ const MAX_INVESTIGATION_DEPTH = 1;
6025
6528
  * no recorded investigationDepth (a job already in the queue before this
6026
6529
  * depth tracking shipped) is treated as excluded too, preserving the
6027
6530
  * pre-existing blanket-exclusion behavior for legacy jobs — no retroactive
6028
- * migration. Non-fix-plan slugs are never capped here. Exported for tests.
6531
+ * migration. Non-fix-plan jobs are never capped here.
6532
+ *
6533
+ * `isFixPlan` (PRD 1131) is the job's own persisted classification stamp
6534
+ * (see lib/fixPlanSlug.cjs's resolveIsFixPlan) — an explicit true/false wins
6535
+ * over the slug; only a row with the field entirely absent (persisted
6536
+ * before this change shipped) falls back to the legacy slug-only heuristic.
6537
+ * Exported for tests.
6029
6538
  */
6030
- function isFixPlanBeyondDepthCap(slug, investigationDepth) {
6031
- if (!isFixPlanSlug(slug)) return false;
6539
+ function isFixPlanBeyondDepthCap(slug, investigationDepth, isFixPlan) {
6540
+ if (!resolveIsFixPlan(slug, isFixPlan)) return false;
6032
6541
  if (investigationDepth == null) return true;
6033
6542
  return investigationDepth >= MAX_INVESTIGATION_DEPTH + 1;
6034
6543
  }
@@ -6081,12 +6590,18 @@ function isUnresolvableNeedsReview(job, { hasRunDir }) {
6081
6590
  * ('no-plan', 'error', and unstamped/undefined) — mirrors the retry
6082
6591
  * eligibility rule in selectAutoFixTargets so a job can never be retry-
6083
6592
  * eligible there and simultaneously un-annotatable here.
6593
+ *
6594
+ * A parent stamped `autoFixReopened: true` (its dead fix-plan child earned
6595
+ * it exactly one further attempt — see isFixPlanDead) is a separate
6596
+ * exhaustion path: it is spent as soon as that second investigation
6597
+ * concludes with ANY outcome, including another 'plan' — a reopened parent
6598
+ * never gets a third attempt, so unlike the fresh case a 'plan' outcome does
6599
+ * not exempt it here.
6084
6600
  */
6085
6601
  function isExhaustedAutoFix(job) {
6086
- return !!job && job.status === 'needs_review'
6087
- && job.autoFixAttempted === true
6088
- && job.autoFixOutcome !== 'plan'
6089
- && (job.autoFixRetries ?? 0) >= 1;
6602
+ if (!job || job.status !== 'needs_review' || job.autoFixAttempted !== true) return false;
6603
+ if (job.autoFixReopened === true) return job.autoFixOutcome != null;
6604
+ return job.autoFixOutcome !== 'plan' && (job.autoFixRetries ?? 0) >= 1;
6090
6605
  }
6091
6606
 
6092
6607
  /**
@@ -6102,6 +6617,31 @@ function isPlanUnqueued(job, queuedSlugs) {
6102
6617
  return !queuedSlugs.has(fixSlugFor(job));
6103
6618
  }
6104
6619
 
6620
+ // Terminal-and-not-completed statuses a fix-plan child can die in — see
6621
+ // isFixPlanDead.
6622
+ const DEAD_FIX_CHILD_STATUSES = new Set(['needs_review', 'failed', 'quarantined', 'skipped']);
6623
+
6624
+ /**
6625
+ * Pure predicate: a parent stuck at outcome 'plan' whose own fix-plan child
6626
+ * (fixSlugFor(job)) has ITSELF died — reached a terminal non-completed
6627
+ * status — with nothing left in the ladder that will ever revisit either
6628
+ * row again (selectAutoFixTargets skips a 'plan' outcome outright, and
6629
+ * isPlanUnqueued only fires when the child never reached the queue at all,
6630
+ * which isn't true once a dead child row exists). `job.autoFixReopened`
6631
+ * gates this to exactly once per parent — once stamped, this always returns
6632
+ * false so the parent can never be reopened a second time. Exported for
6633
+ * tests.
6634
+ */
6635
+ function isFixPlanDead(job, jobsInProject) {
6636
+ if (!job || job.status !== 'needs_review') return false;
6637
+ if (job.autoFixOutcome !== 'plan') return false;
6638
+ if (job.autoFixReopened === true) return false;
6639
+ const fixSlug = fixSlugFor(job);
6640
+ const child = (jobsInProject || []).find((j) => j.slug === fixSlug);
6641
+ if (!child) return false;
6642
+ return DEAD_FIX_CHILD_STATUSES.has(child.status);
6643
+ }
6644
+
6105
6645
  /**
6106
6646
  * Pure predicate: is this job eligible for the boot re-verify self-heal? Only
6107
6647
  * needs_review jobs with a run log (own or backfilled via resolveRunId) AND a
@@ -6247,16 +6787,30 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
6247
6787
  // bounded `--resume` attempt must never also become a fix-plan target
6248
6788
  // in the same pass — see spawnInvestigation's own identical guard.
6249
6789
  if (selectResumeRecoveryTarget(job)) return false;
6790
+ // Mechanical recovery (PRD 1130): a job eligible for a pure-git retry
6791
+ // must never also become a fix-plan target — it needs no plan and no
6792
+ // model. Defensive: today's single mechanically-resolvable verdict
6793
+ // (worktree_integration_failed) is already excluded below via the depth
6794
+ // cap, but this must hold even if that stops being true.
6795
+ if (selectMechanicalRecoveryTarget(job)) return false;
6250
6796
  const runId = job.runId || resolveJobRunId(job);
6251
6797
  if (!runId) return false;
6252
- if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth)) return false;
6798
+ if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth, job.isFixPlan)) return false;
6799
+ // A dead fix-plan child (PRD 1129) earns its parent exactly one further
6800
+ // attempt, bypassing the normal 'plan' exclusion and the fix-slug/queue
6801
+ // membership checks below — those checks exist to stop a FRESH
6802
+ // investigation from clobbering a live sibling, but here the sibling is
6803
+ // dead and reusing its slug is the whole point of the reopen.
6804
+ const dead = isFixPlanDead(job, jobs);
6253
6805
  if (job.autoFixAttempted) {
6254
- const retryEligible = job.autoFixOutcome === 'no-plan'
6806
+ const retryEligible = dead
6807
+ || job.autoFixOutcome === 'no-plan'
6255
6808
  || job.autoFixOutcome === 'error'
6256
6809
  || job.autoFixOutcome == null;
6257
6810
  if (!retryEligible) return false;
6258
- if ((job.autoFixRetries ?? 0) >= 1) return false;
6811
+ if (!dead && (job.autoFixRetries ?? 0) >= 1) return false;
6259
6812
  }
6813
+ if (dead) return true;
6260
6814
  const fixSlug = fixSlugFor(job);
6261
6815
  if (fixSlugExists(fixSlug)) return false;
6262
6816
  if (slugsInQueue.has(fixSlug)) return false;
@@ -6435,7 +6989,7 @@ async function reverifyNeedsReview() {
6435
6989
  const promotedPrds = [];
6436
6990
  await mutate((s) => {
6437
6991
  for (const job of s.jobs) {
6438
- if (job.status !== 'completed' || !isFixPlanSlug(job.slug)) continue;
6992
+ if (job.status !== 'completed' || !resolveIsFixPlan(job.slug, job.isFixPlan)) continue;
6439
6993
  const orig = healTargetForFix(job.slug, s.jobs);
6440
6994
  if (!orig) continue;
6441
6995
  const priorStatus = orig.status;
@@ -6519,6 +7073,23 @@ async function reverifyNeedsReview() {
6519
7073
  ? await readQueue()
6520
7074
  : afterHealForAnnotate;
6521
7075
 
7076
+ // Mechanical recovery (PRD 1130): evaluated first, ahead of both
7077
+ // resume-first recovery and auto-fix below — catches a job whose
7078
+ // mechanically-resolvable verdict this periodic pass finds still eligible
7079
+ // (e.g. one already parked before this rung shipped, or one the same-tick
7080
+ // check in spawnJob missed because the app restarted in between). Depth
7081
+ // never disqualifies it, so it runs regardless of investigationDepth.
7082
+ {
7083
+ for (const job of queueForResumeAndAutofix.jobs) {
7084
+ const target = selectMechanicalRecoveryTarget(job);
7085
+ if (!target) continue;
7086
+ console.log(`[scheduler] mechanical-recovery: needs_review ${job.slug} → re-integrating ${target.branch}`);
7087
+ performMechanicalRecovery(job, target).catch((e) => {
7088
+ console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
7089
+ });
7090
+ }
7091
+ }
7092
+
6522
7093
  // Resume-first recovery (PRD 1111): before any fix-plan investigation is
6523
7094
  // authored below, offer the bounded one-attempt `--resume` dispatch to any
6524
7095
  // needs_review job this periodic pass finds still eligible — e.g. one the
@@ -6537,6 +7108,26 @@ async function reverifyNeedsReview() {
6537
7108
  }
6538
7109
  }
6539
7110
 
7111
+ // Leftover quarantine (PRD 1128), periodic pass: catches a job parked
7112
+ // needs_review with resume recovery already spent BEFORE this feature
7113
+ // shipped, or one the same-tick check in spawnJob missed because the app
7114
+ // restarted in between. Stamps the one-attempt marker in its own mutate
7115
+ // BEFORE the async git work starts (same race-closing rule as the resume
7116
+ // loop above and spawnJob's own dispatch stamp).
7117
+ {
7118
+ for (const job of queueForResumeAndAutofix.jobs) {
7119
+ const quarantineTarget = selectLeftoverQuarantineTarget(job);
7120
+ if (!quarantineTarget) continue;
7121
+ console.log(`[scheduler] leftover-quarantine: needs_review ${job.slug} → quarantining ${quarantineTarget.paths.length} leftover path(s)`);
7122
+ mutate((s) => {
7123
+ const j = s.jobs.find((x) => x.slug === job.slug);
7124
+ if (j) j.leftoverQuarantineAttempted = true;
7125
+ }).then(() => performLeftoverQuarantine(job, quarantineTarget.paths)).catch((e) => {
7126
+ console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
7127
+ });
7128
+ }
7129
+ }
7130
+
6540
7131
  // Auto-fix: spawn a fix-plan investigation for each job still in
6541
7132
  // needs_review after the heal pass (kill-switch: SM_AUTOFIX_DISABLE=1).
6542
7133
  // spawnInvestigation early-returns once investigationsInFlight reaches
@@ -6550,23 +7141,30 @@ async function reverifyNeedsReview() {
6550
7141
  const runId = job.runId || resolveRunId(job);
6551
7142
  const runDir = path.join(RUNS_DIR, runId);
6552
7143
  const isRetryAttempt = job.autoFixAttempted === true;
7144
+ const isDeadFixPlanReopen = isFixPlanDead(job, queueForResumeAndAutofix.jobs);
7145
+ const deadChild = isDeadFixPlanReopen
7146
+ ? queueForResumeAndAutofix.jobs.find((j) => j.slug === fixSlugFor(job))
7147
+ : null;
6553
7148
  // Persist the attempt BEFORE spawning — a crash mid-investigation still
6554
7149
  // counts it (mirrors orphanRetries). Safe even when the slot is busy: the
6555
7150
  // investigation is queued and drained as slots free, so it is genuinely
6556
- // attempted rather than silently dropped.
7151
+ // attempted rather than silently dropped. autoFixReopened is stamped in
7152
+ // this SAME mutate so a crash between selection and dispatch can never
7153
+ // leave the parent re-eligible for a second reopen (PRD 1129).
6557
7154
  await mutate((s) => {
6558
7155
  const j = s.jobs.find((x) => x.slug === job.slug);
6559
7156
  if (j) {
6560
7157
  j.autoFixAttempted = true;
6561
7158
  if (!j.runId && runId) j.runId = runId;
7159
+ if (isDeadFixPlanReopen) j.autoFixReopened = true;
6562
7160
  if (isRetryAttempt) {
6563
7161
  j.autoFixRetries = (j.autoFixRetries ?? 0) + 1;
6564
7162
  delete j.autoFixOutcome;
6565
7163
  }
6566
7164
  }
6567
7165
  });
6568
- console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'})`);
6569
- spawnInvestigation(job, runDir).catch((e) => {
7166
+ console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'}${isDeadFixPlanReopen ? ', dead fix-plan child reopen' : ''})`);
7167
+ spawnInvestigation(job, runDir, { deadChild }).catch((e) => {
6570
7168
  console.error('[scheduler] auto-fix spawnInvestigation error', job.slug, e);
6571
7169
  });
6572
7170
  }
@@ -7657,6 +8255,36 @@ const remote = {
7657
8255
  return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
7658
8256
  }
7659
8257
 
8258
+ // Write-time FK check for a patched dependsOn (PRD 1124), reusing the
8259
+ // SAME resolution rule scheduler_create_prd's prdCreate.cjs applies (exact
8260
+ // slug, else bare-name after stripping one leading `NN-`) so update and
8261
+ // create can never disagree about what a dependsOn entry resolves to. An
8262
+ // explicit empty array CLEARS the dependency and skips validation — there
8263
+ // is nothing to resolve. A listPrds() read failure is skipped-with-a-
8264
+ // warning, matching createPrd's tolerance for an I/O hiccup.
8265
+ if (frontmatter && Array.isArray(frontmatter.dependsOn) && frontmatter.dependsOn.length) {
8266
+ let listing;
8267
+ try {
8268
+ listing = await this.listPrds({ cwd, limit: Number.MAX_SAFE_INTEGER });
8269
+ } catch (e) {
8270
+ console.warn(`[scheduler] updatePrd: dependsOn validation skipped (listPrds failed): ${e?.message ?? e}`);
8271
+ listing = null;
8272
+ }
8273
+ if (listing) {
8274
+ const candidateSlugs = (listing.prds ?? []).map((p) => p.slug);
8275
+ for (const dep of frontmatter.dependsOn) {
8276
+ if (resolveDepSlug(dep, candidateSlugs).length > 0) continue;
8277
+ const near = findNearMatches(dep, candidateSlugs);
8278
+ const suggestion = near.length ? ` Closest existing slug(s): ${near.join(', ')}.` : '';
8279
+ return {
8280
+ ok: false,
8281
+ error: `dependsOn entry "${dep}" does not resolve to any existing PRD in this project.${suggestion} ` +
8282
+ 'Pass the bare name (preferred) or the exact NN-prefixed slug of an existing PRD.',
8283
+ };
8284
+ }
8285
+ }
8286
+ }
8287
+
7660
8288
  let dir = null;
7661
8289
  let filePath = null;
7662
8290
  if (cwd) {
@@ -7787,4 +8415,149 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
7787
8415
  });
7788
8416
  }
7789
8417
 
7790
- module.exports = { classifyQueueStarvation, runQueueStarvationWatchdog, QUEUE_STARVATION_MS, computeBlockedChains, stripAppOwnedChurn, findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure, setPaused, clearPause, tickQueue, runDueJobs, isCooldownSuppressed, nextRapidRateLimitCount, CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD, RAPID_RATE_LIMIT_WINDOW_MS, MANUAL_PAUSE_COOLDOWN_MS, RUNS_DIR, pickRunDir };
8418
+ module.exports = {
8419
+ classifyQueueStarvation,
8420
+ runQueueStarvationWatchdog,
8421
+ QUEUE_STARVATION_MS,
8422
+ computeBlockedChains,
8423
+ stripAppOwnedChurn,
8424
+ findOverrunningJobs,
8425
+ JOB_OVERRUN_FACTOR,
8426
+ JOB_OVERRUN_FLOOR_MS,
8427
+ registerScheduleHandlers,
8428
+ attachWindow,
8429
+ init,
8430
+ ROOT,
8431
+ PRDS_DIR,
8432
+ healRefusalReason,
8433
+ writeQueue,
8434
+ reconcile,
8435
+ reconcileSourcePromptId,
8436
+ allocateParallelGroup,
8437
+ selectHistoryJobs,
8438
+ parsePorcelain,
8439
+ FINISH_PROTOCOL,
8440
+ IDLE_OUTPUT_KILL_MS,
8441
+ BASH_DEFAULT_TIMEOUT_MS,
8442
+ BASH_MAX_TIMEOUT_MS,
8443
+ remote,
8444
+ pickNextBatch,
8445
+ pickForProject,
8446
+ reapDeadRunningJobs,
8447
+ pollRecoveryClearSource,
8448
+ memoryLimitedBatchSize,
8449
+ availableForJobs,
8450
+ reverifyNeedsReview,
8451
+ isRescanCandidate,
8452
+ isFailedUnverifiedShaped,
8453
+ computeLooksDone,
8454
+ isPromotableOriginal,
8455
+ selectAutoFixTargets,
8456
+ applyRcaClassification,
8457
+ isEligibleForImmediateAutoFix,
8458
+ resolveRunId,
8459
+ isUnresolvableNeedsReview,
8460
+ isExhaustedAutoFix,
8461
+ isPlanUnqueued,
8462
+ isFixPlanDead,
8463
+ fixSlugFor,
8464
+ healTargetForFix,
8465
+ buildInvestigationPrompt,
8466
+ isGitRepoSync,
8467
+ committedInWindow,
8468
+ computeCommittedDuringRun,
8469
+ classifySigtermWithCommit,
8470
+ isFixPlanSlug,
8471
+ classifyDiscoveredFixPlan,
8472
+ resolveIsFixPlan,
8473
+ isFixPlanBeyondDepthCap,
8474
+ MAX_INVESTIGATION_DEPTH,
8475
+ forceTickOutcome,
8476
+ applyPauseCleared,
8477
+ detectNetworkErrorInLog,
8478
+ detectRateLimitInLog,
8479
+ classifyFailureOutcome,
8480
+ commitGuardVerdict,
8481
+ leftoverFieldsFrom,
8482
+ applyLeftoverFields,
8483
+ LEFTOVER_PATHS_CAP,
8484
+ capDirtyPaths,
8485
+ buildForeignWipSection,
8486
+ PRE_RUN_DIRTY_PATHS_CAP,
8487
+ FOREIGN_WIP_DELIMITER,
8488
+ FOREIGN_WIP_END_DELIMITER,
8489
+ TRANSIENT_RETRY_CAP,
8490
+ buildScheduleStatePayload,
8491
+ partitionBootOrphans,
8492
+ applyOrphanOutcome,
8493
+ BOOT_ORPHAN_KILL_GRACE_MS,
8494
+ registerAdminRoutes,
8495
+ notifyOriginatingTab,
8496
+ notifyNeedsReview,
8497
+ isNotifiableTerminalStatus,
8498
+ extractResultTextFromLog,
8499
+ candidatePrdsDirs,
8500
+ candidateArchivedPrdsDirs,
8501
+ resolveArchivedPrdStatus,
8502
+ prdDirForCwd,
8503
+ prdPathForJob,
8504
+ archivedPrdPathForJob,
8505
+ archivedTwinExists,
8506
+ findPrdDir,
8507
+ resolveVerifyPrdPath,
8508
+ resolveFixPlanPath,
8509
+ resolveNotifyPrd,
8510
+ runPrdMigration,
8511
+ consolidateAllFlatPrds,
8512
+ shouldSkipInvestigationForCleanRun,
8513
+ archiveCompletedPrd,
8514
+ retireCompletedSlugs,
8515
+ SCHEDULER_BOOTED_AT,
8516
+ SCHEDULER_CODE_SHA,
8517
+ resetJobFields,
8518
+ executeJob,
8519
+ prdArchivedSkipResult,
8520
+ spawnJob,
8521
+ listPrdsInternal,
8522
+ computeStallSummary,
8523
+ findStaleQuarantinedJobs,
8524
+ QUARANTINE_ESCALATE_MS,
8525
+ applyClearQueueVictims,
8526
+ PIDLESS_SPAWN_GRACE_MS,
8527
+ findStrandedInvestigations,
8528
+ INVESTIGATION_MAX_MS,
8529
+ stashList,
8530
+ parseStashLine,
8531
+ pathsChangedSince,
8532
+ restoreSpecificStash,
8533
+ evaluateSharedTreeGuard,
8534
+ checkSharedTreeGuard,
8535
+ uncommittedChanges,
8536
+ gitHead,
8537
+ selectResumeRecoveryTarget,
8538
+ buildResumeRecoveryPreamble,
8539
+ buildClaudeSpawnArgs,
8540
+ spawnResumeRecovery,
8541
+ selectMechanicalRecoveryTarget,
8542
+ performMechanicalRecovery,
8543
+ MECHANICALLY_RESOLVABLE_VERDICTS,
8544
+ selectLeftoverQuarantineTarget,
8545
+ quarantineLeftovers,
8546
+ performLeftoverQuarantine,
8547
+ spawnInvestigation,
8548
+ computeLaunchHolds,
8549
+ computeDepHistorySatisfaction,
8550
+ handleLaunchFailure,
8551
+ applyLaunchFailure,
8552
+ setPaused,
8553
+ clearPause,
8554
+ tickQueue,
8555
+ runDueJobs,
8556
+ isCooldownSuppressed,
8557
+ nextRapidRateLimitCount,
8558
+ CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
8559
+ RAPID_RATE_LIMIT_WINDOW_MS,
8560
+ MANUAL_PAUSE_COOLDOWN_MS,
8561
+ RUNS_DIR,
8562
+ pickRunDir,
8563
+ };