claude-code-session-manager 0.78.0 → 0.80.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/dist/assets/{AgentLibrary-13pfo8uY.js → AgentLibrary-zS3jw_1e.js} +1 -1
  2. package/dist/assets/{DataModel-SUyQbFlg.js → DataModel-Cy_vxTpi.js} +1 -1
  3. package/dist/assets/{History-2GJMS703.js → History-C6JRuqfT.js} +1 -1
  4. package/dist/assets/{Hooks-DM2nS3RT.js → Hooks-BafPy9mB.js} +1 -1
  5. package/dist/assets/{HostBilko-BLeC-lpp.js → HostBilko-BZwhQOFt.js} +1 -1
  6. package/dist/assets/{Library-BaRkU9m0.js → Library-C8JDDliz.js} +1 -1
  7. package/dist/assets/{ListDetail-D5scjSKq.js → ListDetail-CqiOdwLc.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-B1lgAo9T.js → MarkdownEditor-CyLyP67L.js} +1 -1
  9. package/dist/assets/{McpServers-BzyThQSM.js → McpServers-BzMv-_84.js} +1 -1
  10. package/dist/assets/{Memory-7UdaOTtl.js → Memory-DSBYQdJR.js} +1 -1
  11. package/dist/assets/{Panel-JbTMaOPq.js → Panel-CLUhkNNA.js} +1 -1
  12. package/dist/assets/{Permissions-UBam0bJG.js → Permissions-BfC2-HN4.js} +1 -1
  13. package/dist/assets/{Plugins-B3gUDkeb.js → Plugins-BKi40jT5.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-CeHOub7m.js → ProvenanceBadge-BzFw4KhD.js} +1 -1
  15. package/dist/assets/{SaveBar-BcvQEq6h.js → SaveBar-avk2p9jv.js} +1 -1
  16. package/dist/assets/{Scheduler-Dc5qiP24.js → Scheduler-Bf_6MdJo.js} +7 -7
  17. package/dist/assets/{ScopeSwitcher-BvGQmw4Y.js → ScopeSwitcher-C-RwYUVZ.js} +1 -1
  18. package/dist/assets/{Settings-C2dEFb-v.js → Settings-Djd8OoBA.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-CIwlosBc.js → SkillReferenceGraph-DuogY6s7.js} +1 -1
  20. package/dist/assets/{Skills-C0GzzVrQ.js → Skills-D_qAqxZ_.js} +1 -1
  21. package/dist/assets/{SystemPrompt-mtGPK8zo.js → SystemPrompt-DbHFLQV3.js} +1 -1
  22. package/dist/assets/{TagLibrary-DX54-mpd.js → TagLibrary-C2y91BT0.js} +1 -1
  23. package/dist/assets/{TiptapBody-yADC2RWE.js → TiptapBody-D9iz4xQx.js} +1 -1
  24. package/dist/assets/{Toggle-CRxaCYLI.js → Toggle-BGnFL2E5.js} +1 -1
  25. package/dist/assets/{index-D6ymGESc.js → index-_2ARyFDj.js} +4 -4
  26. package/dist/assets/{settingsSchema-TtMvT5Sx.js → settingsSchema-JK15eJU8.js} +1 -1
  27. package/dist/index.html +1 -1
  28. package/package.json +1 -1
  29. package/plugins/session-manager-dev/skills/builder/3-publish/SKILL.md +10 -0
  30. package/scripts/project-pages-logic/dist/logic.cjs +12 -12
  31. package/scripts/render-project-pages/dist/renderer.cjs +22 -22
  32. package/src/main/__tests__/computeDepHistorySatisfaction.test.cjs +66 -0
  33. package/src/main/__tests__/prdCreate.test.cjs +133 -8
  34. package/src/main/__tests__/prdFrontmatterDependsOn.test.cjs +136 -0
  35. package/src/main/__tests__/prdUpdateDependsOn.test.cjs +160 -0
  36. package/src/main/__tests__/queueHistory.test.cjs +33 -0
  37. package/src/main/__tests__/runLogRetention.test.cjs +59 -0
  38. package/src/main/__tests__/scheduleJobTransitions.test.cjs +1 -0
  39. package/src/main/__tests__/scheduler-autofix-outcome.test.cjs +73 -1
  40. package/src/main/__tests__/scheduler-autofix-select.test.cjs +17 -0
  41. package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +20 -0
  42. package/src/main/__tests__/scheduler-leftover-quarantine.test.cjs +199 -0
  43. package/src/main/__tests__/scheduler-mechanical-recovery.test.cjs +222 -0
  44. package/src/main/__tests__/scheduler-never-stop.test.cjs +157 -0
  45. package/src/main/__tests__/scheduler-no-orphan-run-dir.test.cjs +81 -0
  46. package/src/main/__tests__/scheduler-rate-limit-cooldown-freshness.test.cjs +123 -0
  47. package/src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs +158 -0
  48. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +30 -0
  49. package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +51 -0
  50. package/src/main/__tests__/scheduler-resume-recovery.test.cjs +254 -0
  51. package/src/main/__tests__/schedulerBatchRootBlocker.test.cjs +117 -0
  52. package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -2
  53. package/src/main/ipcSchemas.cjs +15 -1
  54. package/src/main/lib/__tests__/gitWorktree.test.cjs +129 -10
  55. package/src/main/lib/__tests__/reaperHelpers.test.cjs +90 -1
  56. package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +59 -7
  57. package/src/main/lib/depSlugResolve.cjs +72 -0
  58. package/src/main/lib/epicWorktreeMerge.cjs +3 -3
  59. package/src/main/lib/epicWorktreeMint.cjs +17 -5
  60. package/src/main/lib/fixPlanSlug.cjs +62 -0
  61. package/src/main/lib/gitWorktree.cjs +97 -12
  62. package/src/main/lib/jobDirtFilter.cjs +54 -0
  63. package/src/main/lib/mcpToolCatalog.cjs +4 -1
  64. package/src/main/lib/prdCreate.cjs +84 -5
  65. package/src/main/lib/prdFrontmatter.cjs +56 -8
  66. package/src/main/lib/queueHistory.cjs +50 -5
  67. package/src/main/lib/rateLimitDetect.cjs +35 -0
  68. package/src/main/lib/reaperHelpers.cjs +31 -6
  69. package/src/main/lib/runLogRetention.cjs +82 -4
  70. package/src/main/lib/scheduleJobTransitions.cjs +12 -2
  71. package/src/main/lib/schedulerBatch.cjs +181 -23
  72. package/src/main/scheduler/prdParser.cjs +7 -0
  73. package/src/main/scheduler.cjs +1165 -70
  74. package/src/preload/api.d.ts +8 -0
@@ -58,6 +58,8 @@ const launchFailure = require('./lib/launchFailure.cjs');
58
58
  const { appendError } = require('./lib/opsErrorLog.cjs');
59
59
  const { readTail } = require('./lib/fileTail.cjs');
60
60
  const { claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
61
+ const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
62
+ const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
61
63
  const { computeQueueHealth } = require('./lib/queueHealth.cjs');
62
64
  const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
63
65
  const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
@@ -71,6 +73,7 @@ const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs')
71
73
  const promptSessionTranscript = require('./promptSessionTranscript.cjs');
72
74
  const { verifyRun } = require('./runVerify.cjs');
73
75
  const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
76
+ const { isFixPlanSlug, classifyDiscoveredFixPlan, resolveIsFixPlan } = require('./lib/fixPlanSlug.cjs');
74
77
  const { landedSinceRun } = require('./lib/landedSinceRun.cjs');
75
78
  const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
76
79
  const logs = require('./logs.cjs');
@@ -97,7 +100,7 @@ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
97
100
  const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
98
101
  ? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
99
102
  : JOB_OVERRUN_FLOOR_MS_DEFAULT;
100
- const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
103
+ const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD, DEP_HISTORY_FAIL_OPEN } = require('./lib/schedulerBatch.cjs');
101
104
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
102
105
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
103
106
  const queueHistory = require('./lib/queueHistory.cjs');
@@ -138,6 +141,7 @@ const jobWorktree = require('./lib/jobWorktree.cjs');
138
141
  const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
139
142
  const queueStore = require('./lib/queueStore.cjs');
140
143
  const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
144
+ const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
141
145
  const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
142
146
  const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
143
147
 
@@ -1252,6 +1256,89 @@ function computeStallSummary(state) {
1252
1256
  return { stalled, total, running, pending, byProject };
1253
1257
  }
1254
1258
 
1259
+ /**
1260
+ * computeBlockedChains(jobs) → [{ cwd, blockedBy, blocked }]
1261
+ *
1262
+ * Pure, no IO. The gap computeStallSummary above cannot see.
1263
+ *
1264
+ * `stalled` is defined as `running === 0 && pending === 0`, which encodes an
1265
+ * assumption that a PENDING row is healthy in-progress work. It is not: a
1266
+ * pending row whose `dependsOn` chain terminates in a TERMINAL non-completed
1267
+ * status (`failed`/`skipped`) can never be dispatched by pickForProject, and
1268
+ * never will be, but it still counts toward `pending` and so reads as a
1269
+ * healthy queue to every monitor in the app.
1270
+ *
1271
+ * On 2026-09-05 starry-night-ships held 42 such rows behind one `failed`
1272
+ * job for three hours. Machine-wide `stalled` was false (42 pending),
1273
+ * per-project `stalled` was false (42 pending), the queue-health sweep
1274
+ * doesn't count `failed` at all, and the supervisor only probes `running` —
1275
+ * so nothing anywhere reported a problem while nothing could ever run.
1276
+ *
1277
+ * Reported per project as { blockedBy: [terminal slugs], blocked: count }.
1278
+ * Transitive by construction: a row blocked by a row that is itself blocked
1279
+ * resolves through the same walk.
1280
+ */
1281
+ function computeBlockedChains(jobs) {
1282
+ const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
1283
+ const byCwd = new Map();
1284
+ for (const j of rows) {
1285
+ const key = j.cwd || '(unknown)';
1286
+ if (!byCwd.has(key)) byCwd.set(key, []);
1287
+ byCwd.get(key).push(j);
1288
+ }
1289
+
1290
+ const out = [];
1291
+ for (const [cwd, projectJobs] of byCwd) {
1292
+ // Reuse the picker's OWN dep resolution so this can never disagree with
1293
+ // what the scheduler will actually dispatch (bare-name fallback included).
1294
+ const rowBySlug = new Map(projectJobs.map((j) => [j.slug, j]));
1295
+ const rowsByBareSlug = new Map();
1296
+ for (const j of projectJobs) {
1297
+ const bare = String(j.slug ?? '').replace(/^\d+-/, '');
1298
+ if (!rowsByBareSlug.has(bare)) rowsByBareSlug.set(bare, []);
1299
+ rowsByBareSlug.get(bare).push(j);
1300
+ }
1301
+ const rowsForDep = (slug) => {
1302
+ const exact = rowBySlug.get(slug);
1303
+ if (exact) return [exact];
1304
+ return rowsByBareSlug.get(String(slug ?? '').replace(/^\d+-/, '')) ?? [];
1305
+ };
1306
+
1307
+ // Memoised walk: does this row's dep closure hit a terminally-stuck row?
1308
+ const TERMINAL_STUCK = new Set(['failed', 'skipped']);
1309
+ const verdicts = new Map(); // slug -> Set of terminal blocker slugs
1310
+ const visiting = new Set();
1311
+ const blockersFor = (job) => {
1312
+ if (!job) return new Set();
1313
+ if (verdicts.has(job.slug)) return verdicts.get(job.slug);
1314
+ if (visiting.has(job.slug)) return new Set(); // dependsOn cycle — not our problem here
1315
+ visiting.add(job.slug);
1316
+ const found = new Set();
1317
+ for (const depSlug of job.dependsOn ?? []) {
1318
+ for (const dep of rowsForDep(depSlug)) {
1319
+ if (TERMINAL_STUCK.has(dep.status)) found.add(dep.slug);
1320
+ else if (dep.status !== 'completed') for (const b of blockersFor(dep)) found.add(b);
1321
+ }
1322
+ }
1323
+ visiting.delete(job.slug);
1324
+ verdicts.set(job.slug, found);
1325
+ return found;
1326
+ };
1327
+
1328
+ const blockedBy = new Set();
1329
+ let blocked = 0;
1330
+ for (const j of projectJobs) {
1331
+ if (j.status !== 'pending') continue;
1332
+ const bs = blockersFor(j);
1333
+ if (bs.size === 0) continue;
1334
+ blocked += 1;
1335
+ for (const b of bs) blockedBy.add(b);
1336
+ }
1337
+ if (blocked > 0) out.push({ cwd, blockedBy: [...blockedBy].sort(), blocked });
1338
+ }
1339
+ return out;
1340
+ }
1341
+
1255
1342
  /**
1256
1343
  * findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
1257
1344
  *
@@ -1987,11 +2074,24 @@ async function reconcile(state) {
1987
2074
  exitCode: null,
1988
2075
  error: null,
1989
2076
  };
1990
- // Newly-discovered fix-plan PRD: stamp its investigationDepth relative to
1991
- // the original job it heals, so selectAutoFixTargets/spawnInvestigation
1992
- // can bound the fix-of-a-fix recursion (see MAX_INVESTIGATION_DEPTH).
1993
- // Non-fix-plan jobs get no explicit field — they read as depth 1 via `?? 1`.
1994
- if (isFixPlanSlug(slug)) {
2077
+ // Fix-plan classification (PRD 1131): a freshly-discovered PRD is a
2078
+ // genuine scheduler-authored fix plan only when its OWN provenance says
2079
+ // so — an explicit isFixPlan:true stamp (spawnInvestigation's prompt
2080
+ // template) or the absence of any createdVia stamp at all (legacy
2081
+ // fallback, matching the "no provenance = trust the name" rule the
2082
+ // quarantine gate below already applies) — never merely because the
2083
+ // slug looks like one. See lib/fixPlanSlug.cjs's header for why (PRD
2084
+ // 1126: a scheduler_create_prd-authored PRD whose slug happened to start
2085
+ // with "fix-" was wrongly stamped investigationDepth before it ever ran).
2086
+ // Persisted onto the queue row so every later consumer
2087
+ // (commitGuardVerdict, isFixPlanBeyondDepthCap, the fix-plan-completion
2088
+ // checks) reads this stamp instead of re-deriving it from the name.
2089
+ entry.isFixPlan = classifyDiscoveredFixPlan(p, slug);
2090
+ // Stamp investigationDepth relative to the original job it heals, so
2091
+ // selectAutoFixTargets/spawnInvestigation can bound the fix-of-a-fix
2092
+ // recursion (see MAX_INVESTIGATION_DEPTH). Non-fix-plan jobs get no
2093
+ // explicit field — they read as depth 1 via `?? 1`.
2094
+ if (entry.isFixPlan) {
1995
2095
  const parent = healTargetForFix(slug, state.jobs);
1996
2096
  entry.investigationDepth = parent ? (parent.investigationDepth ?? 1) + 1 : 2;
1997
2097
  }
@@ -2003,8 +2103,9 @@ async function reconcile(state) {
2003
2103
  // guard-prd-writes.cjs PreToolUse hook should have denied. Fix-plan PRDs
2004
2104
  // are exempt: spawnInvestigation's own probe writes them directly by
2005
2105
  // design (a trusted, scheduler-spawned internal loop, not an
2006
- // agent/human authoring a PRD), matching the isFixPlanSlug convention
2007
- // used everywhere else this distinction matters.
2106
+ // agent/human authoring a PRD) — entry.isFixPlan (just classified above)
2107
+ // is the provenance-aware verdict for that exemption now, not a raw
2108
+ // isFixPlanSlug name check.
2008
2109
  //
2009
2110
  // Quarantine is loud and reversible, never a silent skip (see the
2010
2111
  // 2026-08-01 23-PRD outage this file's header references for what a
@@ -2013,7 +2114,7 @@ async function reconcile(state) {
2013
2114
  // (schedule:adopt-prd) that stamps the file via the same update-prd API
2014
2115
  // route the MCP tool uses — reconcile()'s adopt path above promotes it
2015
2116
  // to 'pending' on the very next pass, within one tick of being stamped.
2016
- if (!p.createdVia && !isFixPlanSlug(slug)) {
2117
+ if (!p.createdVia && !entry.isFixPlan) {
2017
2118
  entry.status = 'quarantined';
2018
2119
  // Stamped at creation (not via transitionJob, since this is a
2019
2120
  // brand-new row minted directly at 'quarantined' rather than
@@ -2119,6 +2220,9 @@ let firstFailureAt = null;
2119
2220
  let firstNon429FailureAt = null; // tracks only transient/config failures; 429s don't count toward network-pause threshold
2120
2221
  let lastFailureKind = null; // 'transient' | 'meter_rate_limited' | 'auth' | null
2121
2222
  let pauseClearedManuallyAt = null;
2223
+ // PRD 1119: consecutive-rapid-rate-limit hard-pause tracking, keyed per slug.
2224
+ // See isCooldownSuppressed/nextRapidRateLimitCount below for the pure rules.
2225
+ const consecutiveRapidRateLimitsBySlug = new Map();
2122
2226
 
2123
2227
  // ---------- timer ----------
2124
2228
 
@@ -2337,13 +2441,71 @@ async function rescheduleTimer() {
2337
2441
 
2338
2442
  // ---------- pause / resume ----------
2339
2443
 
2340
- async function setPaused(reason, resumeAtIso) {
2444
+ const MANUAL_PAUSE_COOLDOWN_MS = 300_000;
2445
+ // PRD 1119: after this many consecutive rate-limited dispatches of the SAME
2446
+ // slug that EACH also finished in under RAPID_RATE_LIMIT_WINDOW_MS, the rate
2447
+ // limit is not a stale/flaky auto-detection any more — it's real and
2448
+ // persistent for this job. Engage a hard pause the manual-clear cooldown
2449
+ // cannot suppress at all. This exists because the freshness check alone
2450
+ // (isCooldownSuppressed) is not sufficient: if the computed resumeAt is
2451
+ // itself wrong or stale (e.g. a failed usage-API fetch), the resume timer
2452
+ // can keep re-clearing the pause every ~30s, and every SUBSEQUENT dispatch
2453
+ // is genuinely "fresh" (it started after that re-clear) — so freshness alone
2454
+ // would let the spin continue indefinitely within the same 5-minute cooldown
2455
+ // window. The rapid-repeat count is an independent circuit breaker of last
2456
+ // resort for exactly that case.
2457
+ const CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD = 3;
2458
+ const RAPID_RATE_LIMIT_WINDOW_MS = 30_000;
2459
+
2460
+ /**
2461
+ * Pure: should setPaused()'s manual-override cooldown suppress WRITING this
2462
+ * pause? `force` (the rapid-repeat hard pause) always answers no — that path
2463
+ * exists precisely to bypass the cooldown. Otherwise, suppress only while
2464
+ * inside the cooldown window AND the triggering observation is stale, i.e.
2465
+ * it was NOT produced by a run that started after the human's manual clear.
2466
+ * A run that started after the clear is fresh evidence the human's fix (if
2467
+ * any) did not hold, and must be allowed to re-engage the pause regardless
2468
+ * of the cooldown — the cooldown's job is to ignore STALE auto-detections,
2469
+ * never to ignore new evidence.
2470
+ */
2471
+ function isCooldownSuppressed({ pauseClearedManuallyAt: clearedAt, now, observedAt, force }) {
2472
+ if (force) return false;
2473
+ if (!clearedAt) return false;
2474
+ if (now - clearedAt >= MANUAL_PAUSE_COOLDOWN_MS) return false;
2475
+ const isFresh = typeof observedAt === 'number' && observedAt > clearedAt;
2476
+ return !isFresh;
2477
+ }
2478
+
2479
+ /**
2480
+ * Pure: the next consecutive-rapid-rate-limit count for a slug, given its
2481
+ * previous count and this run's outcome. Increments only on a rate-limited
2482
+ * run that ALSO ran under RAPID_RATE_LIMIT_WINDOW_MS (a genuine "dispatch,
2483
+ * 429, die" cycle — not a job that ran for a while before hitting the
2484
+ * limit). Resets to 0 on any non-rate-limited outcome. A rate-limited-but-
2485
+ * slow run leaves the count unchanged: still a rate limit, just not the
2486
+ * rapid-spin shape this cap exists to catch.
2487
+ */
2488
+ function nextRapidRateLimitCount(prevCount, { rateLimited, durationMs }) {
2489
+ if (!rateLimited) return 0;
2490
+ if (durationMs < RAPID_RATE_LIMIT_WINDOW_MS) return (prevCount || 0) + 1;
2491
+ return prevCount || 0;
2492
+ }
2493
+
2494
+ async function setPaused(reason, resumeAtIso, opts = {}) {
2495
+ const { observedAt = null, force = false } = opts;
2341
2496
  // Honor manual-override cooldown: if the user cleared a pause within the
2342
- // last 5 minutes, suppress auto-pause re-engagement on the same condition.
2343
- if (pauseClearedManuallyAt && Date.now() - pauseClearedManuallyAt < 300_000) {
2497
+ // last 5 minutes, suppress auto-pause re-engagement UNLESS this pause is
2498
+ // backed by a fresh observation (a run that started after the clear) or is
2499
+ // forced (the rapid-repeat hard pause, which the cooldown cannot suppress).
2500
+ if (isCooldownSuppressed({ pauseClearedManuallyAt, now: Date.now(), observedAt, force })) {
2344
2501
  console.log(`[scheduler] setPaused(${reason}) suppressed by manual override cooldown`);
2345
2502
  return;
2346
2503
  }
2504
+ if (force) {
2505
+ console.log(`[scheduler] setPaused(${reason}) forced past manual override cooldown — rapid-repeat rate-limit cap engaged`);
2506
+ } else if (pauseClearedManuallyAt && Date.now() - pauseClearedManuallyAt < MANUAL_PAUSE_COOLDOWN_MS) {
2507
+ console.log(`[scheduler] setPaused(${reason}) engaging despite manual override cooldown — triggering run started after the manual clear`);
2508
+ }
2347
2509
 
2348
2510
  // For 'network' with no explicit resumeAt, auto-resume after 30 minutes.
2349
2511
  let effectiveResumeAt = resumeAtIso;
@@ -2438,6 +2600,17 @@ function resetJobFields(job, errorMsg, opts = {}) {
2438
2600
  delete job.verifierVerdict;
2439
2601
  delete job.uncommittedPaths;
2440
2602
  delete job.resumeRecoveryAttempted;
2603
+ // Same one-attempt-per-episode category as resumeRecoveryAttempted above —
2604
+ // a re-fired row must be able to earn a fresh mechanical-recovery attempt
2605
+ // if it parks needs_review again (PRD 1130).
2606
+ delete job.mechanicalRecoveryAttempted;
2607
+ // Quarantine (PRD 1128) is scoped to THIS run's episode exactly like
2608
+ // resumeRecoveryAttempted above — a re-fired row must be able to earn a
2609
+ // fresh quarantine attempt if it parks needs_review again.
2610
+ delete job.leftoverQuarantineAttempted;
2611
+ delete job.quarantinedTo;
2612
+ delete job.quarantinedCommit;
2613
+ delete job.quarantinedPaths;
2441
2614
  // Same "this run's outcome, not durable across a reset" category as the
2442
2615
  // fields above — a stale 'archive' recoveryAction from a prior life of this
2443
2616
  // slug must never survive a reset and silently exclude a genuinely-new
@@ -2446,6 +2619,10 @@ function resetJobFields(job, errorMsg, opts = {}) {
2446
2619
  // can otherwise linger forever when RCA is disabled or errors).
2447
2620
  delete job.rcaFailureClass;
2448
2621
  delete job.rcaRecoveryAction;
2622
+ // Same "this run's outcome, not durable across a reset" category — a
2623
+ // human-driven reset must genuinely start the auto-fix budget over,
2624
+ // including the one-time dead-fix-plan-child reopen (PRD 1129).
2625
+ delete job.autoFixReopened;
2449
2626
  // Like exitCode: this run's outcome, not durable across a reset — a stale
2450
2627
  // leak badge from a prior attempt must not linger once the job re-fires.
2451
2628
  delete job.leakedDescendants;
@@ -2862,21 +3039,6 @@ async function notifyNeedsReview(job, report, {
2862
3039
  }
2863
3040
  }
2864
3041
 
2865
- /** Scan the tail of a job's log for the canonical rate-limit signal. We look
2866
- * at the last 16 KB — final result event always lands at the end.
2867
- * Uses readTail() so no raw fd lifecycle is needed here. */
2868
- function detectRateLimitInLog(logPath) {
2869
- try {
2870
- const text = readTail(logPath, 16384);
2871
- if (!text) return false;
2872
- return /"rateLimitType":"five_hour"/.test(text)
2873
- || /"api_error_status":429/.test(text)
2874
- || /You'?ve hit your limit/.test(text);
2875
- } catch {
2876
- return false;
2877
- }
2878
- }
2879
-
2880
3042
  /** Scan the tail of a job's log for a network-outage signal: the structured
2881
3043
  * `terminal_reason":"api_error"` field alongside a network-class error
2882
3044
  * string. This is NOT a real code defect — spawning an auto-fix
@@ -3172,6 +3334,302 @@ As the LAST LINE of your final result text, emit exactly one of:
3172
3334
  Print PASS only once the commit above has actually landed.`;
3173
3335
  }
3174
3336
 
3337
+ /**
3338
+ * Mechanical recovery (PRD 1130). isFixPlanBeyondDepthCap (below) is the
3339
+ * ONLY gate on re-investigating a fix-plan job at investigationDepth >= 2 —
3340
+ * correct for open-ended "author another plan" recursion, but it also
3341
+ * strands a depth-capped job whose failure was fully mechanical (no
3342
+ * judgement required) with no other ladder rung, since resume-first recovery
3343
+ * (selectResumeRecoveryTarget above) is hard-gated on verdict
3344
+ * 'uncommitted_changes'. This rung is evaluated INDEPENDENTLY of
3345
+ * isFixPlanBeyondDepthCap — depth never disqualifies it, because unlike
3346
+ * auto-fix it authors no plan and spawns no model; it is pure git.
3347
+ *
3348
+ * The closed set of mechanically-resolvable verdicts starts at exactly
3349
+ * 'worktree_integration_failed': PRD 1125 already taught integrateBranch to
3350
+ * parse git's "would be overwritten by merge" stderr, verify the blocking
3351
+ * paths are byte-identical to the branch, discard the proven duplicates, and
3352
+ * retry the merge once. A job parked with this verdict has its `sm-job/
3353
+ * <slug>` branch preserved (integrateJobBranch never deletes the branch on
3354
+ * failure — see cleanupJobWorktree's `keepBranch: !integration.ok`), so a
3355
+ * plain re-call of integrateBranch against that same branch inherits PRD
3356
+ * 1125's auto-resolution for free — no re-implementation needed here.
3357
+ *
3358
+ * Bounded to exactly one attempt via job.mechanicalRecoveryAttempted,
3359
+ * stamped in the SAME mutate as the outcome (performMechanicalRecovery,
3360
+ * below) — never here — so this selector alone can be unit-tested exactly
3361
+ * like selectResumeRecoveryTarget/selectLeftoverQuarantineTarget.
3362
+ *
3363
+ * Kill-switch: SM_MECHANICAL_RECOVERY_DISABLE=1 restores today's behaviour
3364
+ * exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
3365
+ */
3366
+ const MECHANICALLY_RESOLVABLE_VERDICTS = new Set(['worktree_integration_failed']);
3367
+
3368
+ function selectMechanicalRecoveryTarget(job) {
3369
+ if (process.env.SM_MECHANICAL_RECOVERY_DISABLE === '1') return null;
3370
+ if (!job || job.status !== 'needs_review') return null;
3371
+ if (!MECHANICALLY_RESOLVABLE_VERDICTS.has(job.verifierVerdict)) return null;
3372
+ if (job.mechanicalRecoveryAttempted === true) return null;
3373
+ const cwd = job.cwd || DEFAULT_PROJECT_CWD;
3374
+ return { slug: job.slug, cwd, branch: jobWorktree.branchNameFor(job.slug), carriedPaths: job.carriedPaths || [] };
3375
+ }
3376
+
3377
+ /**
3378
+ * Perform an already-selected mechanical recovery (selectMechanicalRecoveryTarget
3379
+ * above) — a direct re-attempt of integrateBranch against the job's preserved
3380
+ * branch, never a fresh `claude -p` dispatch. On success the job transitions
3381
+ * needs_review -> completed and its verifierVerdict is cleared; the branch,
3382
+ * now merged, is deleted like any other successfully-integrated job branch.
3383
+ * On failure (including a branch that no longer exists — already deleted or
3384
+ * already merged) the job stays needs_review, mechanicalRecoveryAttempted is
3385
+ * stamped, and the retry's own failure text is appended to `error`. Either
3386
+ * way mechanicalRecoveryAttempted is stamped in this SAME mutate, so a crash
3387
+ * between the git call returning and this mutate landing simply repeats an
3388
+ * idempotent git operation on the next pass rather than leaving the job
3389
+ * re-eligible forever.
3390
+ */
3391
+ async function performMechanicalRecovery(job, target) {
3392
+ const integration = await jobWorktree.integrateJobBranch({
3393
+ cwd: target.cwd, branch: target.branch, slug: target.slug, carriedPaths: target.carriedPaths,
3394
+ });
3395
+ if (integration.ok) {
3396
+ await jobWorktree.cleanupJobWorktree({ cwd: target.cwd, dir: undefined, branch: target.branch, keepBranch: false });
3397
+ }
3398
+ let becameCompleted = false;
3399
+ await mutate((s) => {
3400
+ const j = s.jobs.find((x) => x.slug === job.slug);
3401
+ if (!j) return;
3402
+ j.mechanicalRecoveryAttempted = true;
3403
+ if (integration.ok) {
3404
+ if (transitionJob(j, 'completed', {
3405
+ reason: `mechanical recovery: ${target.branch} re-integrated successfully`,
3406
+ source: 'scheduler:mechanicalRecovery',
3407
+ })) {
3408
+ delete j.verifierVerdict;
3409
+ j.exitCode = 0;
3410
+ j.error = null;
3411
+ becameCompleted = true;
3412
+ }
3413
+ } else {
3414
+ const pointer = `Mechanical recovery retry failed: ${integration.reason}`;
3415
+ j.error = j.error ? `${j.error}\n${pointer}` : pointer;
3416
+ }
3417
+ });
3418
+ if (integration.ok) {
3419
+ console.log(`[scheduler] mechanical-recovery: ${job.slug} → completed (branch ${target.branch} re-integrated)`);
3420
+ if (becameCompleted) await archiveCompletedPrd(job.slug, job.cwd);
3421
+ } else {
3422
+ console.error(`[scheduler] mechanical-recovery: ${job.slug} → retry failed: ${integration.reason}`);
3423
+ }
3424
+ }
3425
+
3426
+ /**
3427
+ * Leftover quarantine (PRD 1128). Resume-first recovery gets exactly one
3428
+ * `--resume` attempt (selectResumeRecoveryTarget above); when that attempt
3429
+ * ALSO parks needs_review with 'uncommitted_changes', the leftovers are
3430
+ * about to sit dirty in the SHARED tree forever — git then refuses any later
3431
+ * worktree merge for this cwd that would overwrite them, turning one parked
3432
+ * job into a project-wide stall (216-jupiter-sand-kazekage, 2026-09-06).
3433
+ * Pure/no I/O, mirroring selectResumeRecoveryTarget so the eligibility rule
3434
+ * is unit-testable directly.
3435
+ *
3436
+ * Bounded to exactly one attempt via job.leftoverQuarantineAttempted, stamped
3437
+ * synchronously by the caller in the SAME mutate as this decision (never
3438
+ * here) — see spawnJob's finalize and reverifyNeedsReview's periodic pass.
3439
+ *
3440
+ * Kill-switch: SM_LEFTOVER_QUARANTINE_DISABLE=1 restores today's behaviour
3441
+ * exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
3442
+ */
3443
+ function selectLeftoverQuarantineTarget(job) {
3444
+ if (process.env.SM_LEFTOVER_QUARANTINE_DISABLE === '1') return null;
3445
+ if (!job || job.status !== 'needs_review') return null;
3446
+ if (job.verifierVerdict !== 'uncommitted_changes') return null;
3447
+ if (job.resumeRecoveryAttempted !== true) return null;
3448
+ if (job.leftoverQuarantineAttempted === true) return null;
3449
+ const uncommittedPaths = Array.isArray(job.uncommittedPaths)
3450
+ ? job.uncommittedPaths.filter((p) => typeof p === 'string' && p.length > 0)
3451
+ : [];
3452
+ if (!uncommittedPaths.length) return null;
3453
+ // The single most important constraint: never touch a path that was
3454
+ // ALREADY dirty at this run's own dispatch time (preRunDirtyPaths) — that
3455
+ // is foreign WIP (a human's or a sibling's), not this job's own leftover.
3456
+ const preRunDirty = new Set(Array.isArray(job.preRunDirtyPaths) ? job.preRunDirtyPaths : []);
3457
+ const paths = uncommittedPaths.filter((p) => !preRunDirty.has(p));
3458
+ if (!paths.length) return null;
3459
+ return { slug: job.slug, cwd: job.cwd, paths };
3460
+ }
3461
+
3462
+ function execGitAt(cwd, args, { env, timeout = 20_000 } = {}) {
3463
+ return new Promise((resolve, reject) => {
3464
+ execFile(
3465
+ 'git',
3466
+ ['-C', cwd, ...args],
3467
+ { timeout, windowsHide: true, encoding: 'utf8', env: env ? { ...process.env, ...env } : process.env },
3468
+ (err, stdout, stderr) => {
3469
+ if (err) {
3470
+ err.stderrText = stderr;
3471
+ reject(err);
3472
+ return;
3473
+ }
3474
+ resolve(stdout || '');
3475
+ },
3476
+ );
3477
+ });
3478
+ }
3479
+
3480
+ async function pathExistsInTree(cwd, treeish, p) {
3481
+ try {
3482
+ await execGitAt(cwd, ['cat-file', '-e', `${treeish}:${p}`]);
3483
+ return true;
3484
+ } catch {
3485
+ return false;
3486
+ }
3487
+ }
3488
+
3489
+ /**
3490
+ * Commit exactly `paths` (must already be dirty on disk) onto a dedicated
3491
+ * `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
3492
+ * unavailable) via a THROWAWAY `GIT_INDEX_FILE` — never touches the live
3493
+ * index, never moves the checked-out branch — then restores those paths to
3494
+ * match that baseline commit's tree, so the shared working tree returns to
3495
+ * its pre-run state. This is deliberately NOT `git stash` (the destructive-
3496
+ * git guard blocks stash on a shared tree, and a stash nobody restores
3497
+ * strands the work invisibly — see standards.md).
3498
+ *
3499
+ * Never throws: any git failure, or a non-git cwd, aborts the WHOLE attempt
3500
+ * with the tree untouched (no partial restore) — restore only ever runs
3501
+ * after the salvage ref/commit has safely landed, so a failure there leaves
3502
+ * the data recoverable from the ref even though the tree stayed dirty.
3503
+ * A path no longer dirty on disk (already committed, or reverted since) is
3504
+ * skipped, never force-restored.
3505
+ */
3506
+ async function quarantineLeftovers({ cwd, slug, paths, headBefore }) {
3507
+ if (!cwd || !slug || !Array.isArray(paths) || paths.length === 0) {
3508
+ return { ok: false, reason: 'no cwd/slug/paths given' };
3509
+ }
3510
+ let baseline = headBefore || null;
3511
+ try {
3512
+ if (!baseline) {
3513
+ baseline = (await execGitAt(cwd, ['rev-parse', 'HEAD'])).trim();
3514
+ }
3515
+ if (!baseline) return { ok: false, reason: 'could not resolve a baseline commit (non-git cwd?)' };
3516
+
3517
+ const dirtyNowRaw = await execGitAt(cwd, ['status', '--porcelain', '--', ...paths]);
3518
+ const dirtyNow = new Set(parsePorcelain(dirtyNowRaw));
3519
+ const toQuarantine = paths.filter((p) => dirtyNow.has(p));
3520
+ const skippedPaths = paths.filter((p) => !dirtyNow.has(p));
3521
+ if (!toQuarantine.length) {
3522
+ return { ok: true, ref: null, commit: null, quarantinedPaths: [], skippedPaths };
3523
+ }
3524
+
3525
+ const tmpIndex = path.join(os.tmpdir(), `sm-salvage-index-${slug}-${process.pid}-${Date.now()}`);
3526
+ const env = { GIT_INDEX_FILE: tmpIndex };
3527
+ let treeSha;
3528
+ let commitSha;
3529
+ try {
3530
+ await execGitAt(cwd, ['read-tree', baseline], { env });
3531
+ for (const p of toQuarantine) {
3532
+ if (fs.existsSync(path.join(cwd, p))) {
3533
+ await execGitAt(cwd, ['add', '--', p], { env });
3534
+ } else {
3535
+ await execGitAt(cwd, ['rm', '--cached', '--ignore-unmatch', '--', p], { env });
3536
+ }
3537
+ }
3538
+ treeSha = (await execGitAt(cwd, ['write-tree'], { env })).trim();
3539
+ commitSha = (await execGitAt(cwd, ['commit-tree', treeSha, '-p', baseline, '-m', `salvage: leftover changes from ${slug}`], { env })).trim();
3540
+ } catch (e) {
3541
+ return { ok: false, reason: `git command failed while building the salvage commit: ${(e && (e.stderrText || e.message)) || e}` };
3542
+ } finally {
3543
+ await fsp.rm(tmpIndex, { force: true }).catch(() => {});
3544
+ }
3545
+
3546
+ const ref = `sm-salvage/${slug}`;
3547
+ try {
3548
+ await execGitAt(cwd, ['update-ref', `refs/heads/${ref}`, commitSha]);
3549
+ } catch (e) {
3550
+ return { ok: false, reason: `git command failed updating ${ref}: ${(e && (e.stderrText || e.message)) || e}` };
3551
+ }
3552
+
3553
+ // The salvage commit is safely landed at this point — a failure from here
3554
+ // on is reported with the ref/commit still attached so nothing looks lost
3555
+ // even if the tree itself couldn't be fully restored.
3556
+ try {
3557
+ const inBaseline = [];
3558
+ const notInBaseline = [];
3559
+ for (const p of toQuarantine) {
3560
+ // eslint-disable-next-line no-await-in-loop
3561
+ if (await pathExistsInTree(cwd, baseline, p)) inBaseline.push(p); else notInBaseline.push(p);
3562
+ }
3563
+ if (inBaseline.length) {
3564
+ await execGitAt(cwd, ['checkout', baseline, '--', ...inBaseline]);
3565
+ }
3566
+ if (notInBaseline.length) {
3567
+ await execGitAt(cwd, ['reset', '--', ...notInBaseline]).catch(() => {});
3568
+ for (const p of notInBaseline) {
3569
+ // eslint-disable-next-line no-await-in-loop
3570
+ await fsp.rm(path.join(cwd, p), { force: true });
3571
+ }
3572
+ }
3573
+ } catch (e) {
3574
+ return {
3575
+ ok: false,
3576
+ ref,
3577
+ commit: commitSha,
3578
+ reason: `salvage commit landed at ${ref} (${commitSha}) but restoring the working tree failed: ${(e && (e.stderrText || e.message)) || e}`,
3579
+ };
3580
+ }
3581
+
3582
+ return { ok: true, ref, commit: commitSha, quarantinedPaths: toQuarantine, skippedPaths };
3583
+ } catch (e) {
3584
+ return { ok: false, reason: `git command failed: ${(e && (e.stderrText || e.message)) || e}` };
3585
+ }
3586
+ }
3587
+
3588
+ /**
3589
+ * Perform an already-selected quarantine (job.leftoverQuarantineAttempted
3590
+ * must already be true, stamped by the caller) and persist the outcome onto
3591
+ * the job row: `quarantinedTo`/`quarantinedCommit`/`quarantinedPaths` on
3592
+ * success, plus a one-line pointer appended to `error` naming the ref so a
3593
+ * human can recover with a single named command
3594
+ * (`git show sm-salvage/<slug>`). The belt-and-braces `salvagePatch` (when
3595
+ * present) is referenced alongside it, never removed. On failure, only a
3596
+ * diagnostic is appended — the job row's dirt-describing fields are left as
3597
+ * they were, since the tree itself was left untouched (or, for a
3598
+ * restore-only failure, the salvage ref is still named in the note).
3599
+ *
3600
+ * `headBefore`, when the caller has it fresh (spawnJob's own finalize still
3601
+ * has the local `guardHeadBefore` in scope for the run that just parked —
3602
+ * the same value is deleted off the job ROW earlier in that same finalize),
3603
+ * is used as the salvage ref's baseline commit; otherwise (the periodic
3604
+ * reverifyNeedsReview pass, re-discovering an already-parked row) this falls
3605
+ * back to the current HEAD inside quarantineLeftovers itself.
3606
+ */
3607
+ async function performLeftoverQuarantine(job, paths, headBefore = null) {
3608
+ const result = await quarantineLeftovers({
3609
+ cwd: job.cwd || DEFAULT_PROJECT_CWD,
3610
+ slug: job.slug,
3611
+ paths,
3612
+ headBefore: headBefore || job.guardHeadBefore || null,
3613
+ });
3614
+ await mutate((s) => {
3615
+ const j = s.jobs.find((x) => x.slug === job.slug);
3616
+ if (!j) return;
3617
+ if (result.ok && Array.isArray(result.quarantinedPaths) && result.quarantinedPaths.length) {
3618
+ j.quarantinedTo = result.ref;
3619
+ j.quarantinedCommit = result.commit;
3620
+ j.quarantinedPaths = capDirtyPaths(result.quarantinedPaths);
3621
+ const salvageNote = j.salvagePatch ? `; salvage patch also at ${j.salvagePatch}` : '';
3622
+ const pointer = `Leftovers quarantined to ${result.ref} (commit ${result.commit}) — recover via \`git show ${result.ref}\`${salvageNote}`;
3623
+ j.error = j.error ? `${j.error}\n${pointer}` : pointer;
3624
+ console.log(`[scheduler] ${job.slug}: quarantined ${result.quarantinedPaths.length} leftover path(s) to ${result.ref} (${result.commit})`);
3625
+ } else if (!result.ok) {
3626
+ const pointer = `Leftover quarantine failed: ${result.reason}`;
3627
+ j.error = j.error ? `${j.error}\n${pointer}` : pointer;
3628
+ console.error(`[scheduler] ${job.slug}: leftover quarantine failed: ${result.reason}`);
3629
+ }
3630
+ });
3631
+ }
3632
+
3175
3633
  /**
3176
3634
  * Pure argv builder for a `claude -p` child spawn, shared so the
3177
3635
  * resume-vs-fresh-session choice is made in exactly one place. `resume`
@@ -3197,10 +3655,17 @@ function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }
3197
3655
 
3198
3656
  // ---------- execution ----------
3199
3657
 
3658
+ // Allocates the runId/dir pair at dispatch time WITHOUT creating the
3659
+ // directory — a dispatch that aborts inside spawnJob before executeJob's
3660
+ // openLog() call (slot-acquire miss, worktree-cap deferral, launch-gate
3661
+ // block, ...) must leave no trace on disk. The directory is materialised
3662
+ // lazily, the first time something actually needs to write into it (see
3663
+ // openLog's mkdirSync in executeJob below). Because tickQueue hands ONE
3664
+ // shared batch dir to every spawnJob in the batch, several jobs may race to
3665
+ // create it — `recursive: true` makes that race safe.
3200
3666
  function pickRunDir() {
3201
3667
  const ts = new Date().toISOString().replace(/[:.]/g, '-');
3202
3668
  const dir = path.join(RUNS_DIR, ts);
3203
- fs.mkdirSync(dir, { recursive: true });
3204
3669
  return { runId: ts, dir };
3205
3670
  }
3206
3671
 
@@ -3228,6 +3693,12 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
3228
3693
  // fresh one via `--session-id` is the entire point of the recovery.
3229
3694
  const sessionId = resumeTarget ? resumeTarget.sessionId : randomUUID();
3230
3695
 
3696
+ // Materialise the (possibly shared-batch) run dir lazily, right before the
3697
+ // first write into it — see pickRunDir's comment for why this is deferred
3698
+ // this far. recursive:true makes it safe if a sibling job in the same
3699
+ // batch dir already created it.
3700
+ fs.mkdirSync(runDir, { recursive: true });
3701
+
3231
3702
  // Phase 1: open log fd so we can emit pre-spawn diagnostics (early-exit
3232
3703
  // error paths) before the child is created. withChildAndLog takes ownership
3233
3704
  // of fd/safeLog/closeFd from the point it is called.
@@ -3638,13 +4109,9 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
3638
4109
  // project (keyed by cwd) so jobs in different repos run concurrently up to
3639
4110
  // the cap; within one project, sequential-group semantics are preserved.
3640
4111
 
3641
- /**
3642
- * Recognize fix-plan slugs (NN-fix-...) so we don't recurse on a fix-plan that
3643
- * itself failed. The pattern matches the slug we generate in spawnInvestigation.
3644
- */
3645
- function isFixPlanSlug(slug) {
3646
- return /^\d+-fix-/.test(slug);
3647
- }
4112
+ // isFixPlanSlug/classifyDiscoveredFixPlan/resolveIsFixPlan now live in
4113
+ // lib/fixPlanSlug.cjs (PRD 1131) — see that module's header for why slug
4114
+ // shape alone is no longer sufficient to classify a fix plan.
3648
4115
 
3649
4116
  /**
3650
4117
  * The fix-plan slug spawnInvestigation authors for a given failed job —
@@ -3683,7 +4150,25 @@ function healTargetForFix(fixSlug, jobs) {
3683
4150
  * unit-tested (no spawn, no fs). Inputs are the already-resolved values that
3684
4151
  * spawnInvestigation computes.
3685
4152
  */
3686
- function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group }) {
4153
+ function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild = null }) {
4154
+ const deadFixChildNote = deadChild ? `
4155
+
4156
+ # This is a REOPENED investigation — your own prior fix plan died
4157
+ You already investigated this job once and produced a fix-plan PRD, \`${deadChild.slug}\`, which was
4158
+ supposed to heal it. That fix-plan job itself reached a terminal, non-completed status
4159
+ (\`${deadChild.status}\`) without ever fixing the original failure — so the parent job you are now
4160
+ investigating is stuck again with nothing left to retry it automatically. This is the ONE reopen
4161
+ this parent gets; do not cold-read the log and re-derive the plan that already failed.
4162
+
4163
+ Dead fix-plan child's own outcome:
4164
+ - Slug: ${deadChild.slug}
4165
+ - Status: ${deadChild.status}
4166
+ - Verifier verdict: ${deadChild.verifierVerdict ?? '(none recorded)'}
4167
+ - Error: ${deadChild.error ?? '(none recorded)'}
4168
+
4169
+ Read why THAT job died (its own run log, if any, under the runs directory) before writing a new
4170
+ fix-plan PRD, and make sure your new plan is genuinely different from — not a repeat of — whatever
4171
+ that dead child attempted.` : '';
3687
4172
  const abandonedBackgroundTaskNote = failedJob.verifierVerdict === 'abandoned_background_task' ? `
3688
4173
 
3689
4174
  # Known failure class: abandoned background task
@@ -3704,7 +4189,7 @@ The fix-plan PRD you write for this MUST instruct its executor to, in order:
3704
4189
  1. Check for a salvage patch (named \`<slug>.uncommitted.patch\` in the run directory${failedJob.salvagePatch ? `, e.g. \`${failedJob.salvagePatch}\`` : ''}) and, if found, apply it to the working tree BEFORE inspecting \`git status\`/\`git diff\` in ${cwd} for uncommitted changes matching the original PRD's acceptance criteria.
3705
4190
  2. If the work is present (via the applied patch or already in the tree) and satisfies the acceptance criteria, run the project's verify commands and COMMIT it — do not re-implement or re-plan the PRD from scratch.
3706
4191
  3. Only fall back to re-implementing whatever acceptance criteria are genuinely missing after applying any salvage patch, not the whole PRD.` : '';
3707
- return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${abandonedBackgroundTaskNote}
4192
+ return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${deadFixChildNote}${abandonedBackgroundTaskNote}
3708
4193
 
3709
4194
  # Failed job
3710
4195
  - Slug: ${failedJob.slug}
@@ -3749,8 +4234,13 @@ ${logTail}
3749
4234
  cwd: ${cwd}
3750
4235
  parallelGroup: ${group}
3751
4236
  estimateMinutes: <your time estimate>
4237
+ isFixPlan: true
3752
4238
  ---
3753
4239
  \`\`\`
4240
+ \`isFixPlan: true\` is REQUIRED — it is the scheduler's provenance signal that this PRD is a
4241
+ genuine auto-authored fix plan (not a human/agent PRD whose slug merely happens to start with
4242
+ "fix-"); omitting it means this fix plan will not get its depth-cap/zero-edit-commit-guard
4243
+ exemptions.
3754
4244
  \`cwd\` must be the git repo root where the fix will actually land. If the failed job's cwd is
3755
4245
  not that repo (e.g. a scratch dir like \`/tmp\`), set \`cwd:\` to the correct repo root instead —
3756
4246
  the scheduler's commit guard and post-run verifier read git state from this path, and a
@@ -3824,7 +4314,7 @@ function readRunOutcomeSidecars(runDir, slug) {
3824
4314
  */
3825
4315
  const INVESTIGATION_LAUNCH_KEY = 'investigation';
3826
4316
 
3827
- async function spawnInvestigation(failedJob, runDir) {
4317
+ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {}) {
3828
4318
  // The probe launches with the same CLI as the job it diagnoses. While
3829
4319
  // that CLI cannot launch at all (launch circuit breaker, issue #11 list
3830
4320
  // B1: probes e4f82da2/d374e6bf died on the same HTTP 400 as the runs
@@ -3853,7 +4343,14 @@ async function spawnInvestigation(failedJob, runDir) {
3853
4343
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is resume-recovery eligible`);
3854
4344
  return { deferred: false };
3855
4345
  }
3856
- if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth)) {
4346
+ // Mechanical recovery (PRD 1130): same first-refusal treatment — a job
4347
+ // eligible for a pure-git retry must never also get a cold-read fix-plan
4348
+ // PRD authored in the same pass.
4349
+ if (selectMechanicalRecoveryTarget(failedJob)) {
4350
+ console.log(`[scheduler] skip investigation: ${failedJob.slug} is mechanical-recovery eligible`);
4351
+ return { deferred: false };
4352
+ }
4353
+ if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth, failedJob.isFixPlan)) {
3857
4354
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
3858
4355
  return { deferred: false };
3859
4356
  }
@@ -3910,7 +4407,13 @@ async function spawnInvestigation(failedJob, runDir) {
3910
4407
 
3911
4408
  const logTail = readTail(failedLogPath, 16 * 1024) || '(failed to read log)';
3912
4409
 
3913
- if (fs.existsSync(fixPath)) {
4410
+ // A dead-fix-plan reopen (PRD 1129) targets the SAME fixPath its dead
4411
+ // child was originally authored at, by construction (fixSlugFor is a pure
4412
+ // function of the parent) — the file existing is not staleness here, it's
4413
+ // the whole reason a reopen was offered. Skip the guard in that one case
4414
+ // so the second investigation can overwrite the dead plan; every other
4415
+ // caller keeps the original protection against clobbering a live sibling.
4416
+ if (fs.existsSync(fixPath) && !deadChild) {
3914
4417
  console.log(`[scheduler] skip investigation: fix plan already exists at ${fixPath}`);
3915
4418
  releaseSlot();
3916
4419
  return { deferred: false };
@@ -3936,7 +4439,7 @@ async function spawnInvestigation(failedJob, runDir) {
3936
4439
  console.warn(`[scheduler] investigation cwd is not a git repo (${cwd}); falling back to ${DEFAULT_PROJECT_CWD}`);
3937
4440
  cwd = DEFAULT_PROJECT_CWD;
3938
4441
  }
3939
- const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group });
4442
+ const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild });
3940
4443
 
3941
4444
  // Phase 1: open log fd for pre-spawn diagnostics.
3942
4445
  const { fd, safeLog, closeFd } = openLog(investigationLogPath);
@@ -4065,6 +4568,22 @@ async function spawnInvestigation(failedJob, runDir) {
4065
4568
  mutate((s) => {
4066
4569
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
4067
4570
  if (j) j.autoFixOutcome = 'plan';
4571
+ // Dead-fix-plan reopen (PRD 1129): fixSlugFor is a pure function of
4572
+ // the parent, so the freshly-authored plan landed at the SAME slug
4573
+ // as the dead child — reconcile() sees an already-known slug and
4574
+ // will never re-mint a pending row for it. Explicitly reset the
4575
+ // dead child's own row here so the overwritten plan actually gets
4576
+ // a chance to run, rather than sitting inert behind a permanently
4577
+ // terminal queue row. force:true because 'skipped' (a valid dead
4578
+ // status here) is otherwise reset-refused by design.
4579
+ if (deadChild) {
4580
+ const child = s.jobs.find((x) => x.slug === deadChild.slug);
4581
+ if (child) {
4582
+ resetJobFields(child, 'reset by dead-fix-plan reopen: parent investigation authored a new plan', {
4583
+ force: true, source: 'spawnInvestigation:dead-fix-plan-reopen',
4584
+ });
4585
+ }
4586
+ }
4068
4587
  }).catch(() => {});
4069
4588
  } else {
4070
4589
  console.log(`[scheduler] investigation finished WITHOUT producing fix plan (slug=${failedJob.slug}, code=${exitCode})`);
@@ -4147,6 +4666,58 @@ async function computeLaunchHolds(state, { now = Date.now(), claudeVersion } = {
4147
4666
  return held;
4148
4667
  }
4149
4668
 
4669
+ /**
4670
+ * computeDepHistorySatisfaction(state) → Map<cwd, Set<string>|symbol>
4671
+ *
4672
+ * PRD 1122's once-per-tick dependsOn history/archive lookup: for every
4673
+ * distinct project cwd with jobs this tick, builds the set of dep slugs that
4674
+ * have no live queue row but are nonetheless known-satisfied — a completed
4675
+ * record in that project's own `state/history.jsonl` shard
4676
+ * (queueHistory.completedSlugsForCwd, scoped per-project so a same-named PRD
4677
+ * in an unrelated project can never satisfy a dep here), or a `.md` file
4678
+ * under any of that project's `prds-archived/` dirs (listArchivedPrdDirs —
4679
+ * covers both the retired flat layout and every Epic's own sibling archive).
4680
+ * findBlockingDep (schedulerBatch.cjs) treats a dep slug as blocking
4681
+ * whenever it has no live row AND is absent from this set, so a typo or a
4682
+ * double-prefixed slug (the exact 2026-09-06 starry-night-ships incident)
4683
+ * HOLDS its dependent instead of silently dispatching it.
4684
+ *
4685
+ * Fails OPEN per project, never queue-wide: a history-shard or archive-scan
4686
+ * read error for one cwd degrades that cwd's value to
4687
+ * `DEP_HISTORY_FAIL_OPEN` (findBlockingDep then treats every rowless dep in
4688
+ * that project as satisfied, exactly today's pre-1122 behaviour) with a
4689
+ * logged warning — it never throws out of this function and never blocks
4690
+ * every OTHER project's dispatch for one project's bad fs state.
4691
+ *
4692
+ * Computed ONCE here, before pickNextBatch runs, and threaded down as pure
4693
+ * data (quietOpts.satisfiedSlugsByCwd) — schedulerBatch.cjs itself does no
4694
+ * I/O, so this is the only fs read this gate costs per tick, not one per job
4695
+ * per dep.
4696
+ */
4697
+ async function computeDepHistorySatisfaction(state) {
4698
+ const byCwd = new Map();
4699
+ const cwds = new Set((state?.jobs || []).map((j) => j.cwd || DEFAULT_PROJECT_CWD));
4700
+ for (const cwd of cwds) {
4701
+ const satisfied = new Set();
4702
+ try {
4703
+ for (const slug of await queueHistory.completedSlugsForCwd(cwd)) satisfied.add(slug);
4704
+ for (const dir of listArchivedPrdDirs(cwd)) {
4705
+ let entries;
4706
+ try { entries = await fsp.readdir(dir); } catch { continue; }
4707
+ for (const name of entries) {
4708
+ if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
4709
+ }
4710
+ }
4711
+ } catch (e) {
4712
+ console.warn(`[scheduler] depHistorySatisfaction: history/archive lookup failed for ${cwd} (${e?.message}) — falling back to fail-open dep resolution for this project this tick`);
4713
+ byCwd.set(cwd, DEP_HISTORY_FAIL_OPEN);
4714
+ continue;
4715
+ }
4716
+ byCwd.set(cwd, satisfied);
4717
+ }
4718
+ return byCwd;
4719
+ }
4720
+
4150
4721
  /**
4151
4722
  * A run that never got a turn (res.launchFailure — see executeJob's onExit)
4152
4723
  * is routed here instead of the failed/investigation path (issue #11 lists
@@ -4317,6 +4888,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4317
4888
  console.log(`[scheduler] ${job.slug}: dispatching as launch probe for '${launchKey}'${launchEnv ? ` with mitigation ${JSON.stringify(launchEnv)}` : ''}`);
4318
4889
  }
4319
4890
 
4891
+ // Captured here (not read back off `job`, a pre-dispatch snapshot that
4892
+ // mutate()'s fresh-from-disk read never touches) so the rate-limited
4893
+ // branch below has this run's OWN start time — the freshness check
4894
+ // (isCooldownSuppressed) needs to know whether this specific dispatch
4895
+ // started after the manual clear, not whatever startedAt this row
4896
+ // carried from a prior run.
4897
+ let dispatchStartedAtMs = null;
4320
4898
  await mutate((s) => {
4321
4899
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
4322
4900
  if (idx >= 0) {
@@ -4327,6 +4905,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4327
4905
  delete s.jobs[idx].heldReason;
4328
4906
  s.jobs[idx].runId = runId;
4329
4907
  s.jobs[idx].startedAt = new Date().toISOString();
4908
+ dispatchStartedAtMs = Date.parse(s.jobs[idx].startedAt);
4330
4909
  if (job.quietMachine === true) {
4331
4910
  s.jobs[idx].quietMachine = true;
4332
4911
  s.jobs[idx].quietLeaseDegraded = job.quietLeaseDegraded === true;
@@ -4434,6 +5013,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4434
5013
  let res;
4435
5014
  let worktreeLeftoverDirty = [];
4436
5015
  let worktreeIntegrationFailure = null;
5016
+ // Set only when integrateJobBranch's stderr-parsing auto-resolve fired
5017
+ // (PRD 1125) — surfaced on the job row so the Queue UI can say the merge
5018
+ // self-healed rather than silently looking like an ordinary merge.
5019
+ let mergeAutoResolved = null;
5020
+ let mergeAutoResolvedPaths = null;
4437
5021
  // A job's uncommitted-work patch, whichever isolation mode produced it —
4438
5022
  // set by EITHER branch below, never both (worktree.ok picks exactly one
4439
5023
  // shape for the whole run). Named generically (not "worktree...") because
@@ -4481,6 +5065,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4481
5065
  console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
4482
5066
  } else if (integration.integrated) {
4483
5067
  console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
5068
+ if (integration.autoResolved) {
5069
+ mergeAutoResolved = integration.autoResolved;
5070
+ mergeAutoResolvedPaths = integration.resolvedPaths || [];
5071
+ console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
5072
+ }
4484
5073
  }
4485
5074
  await jobWorktree.cleanupJobWorktree({
4486
5075
  cwd: guardCwd,
@@ -4531,12 +5120,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4531
5120
  // (non-git cwd / git errored) — NEVER treated as "left nothing", exactly
4532
5121
  // like every other best-effort git-state check in this function.
4533
5122
  const afterGuardCwd = await uncommittedChanges(guardCwd);
5123
+ // stripAppOwnedChurn: the app writes session-manager-operations/ (queue.json,
5124
+ // history.jsonl, active-index.json, transcripts) DURING this job's own guard
5125
+ // window, so those land in the delta and get blamed on the job. A job can
5126
+ // never be responsible for them — see jobDirtFilter.cjs. Applied here, at
5127
+ // the single place the delta is computed, so the commit guard, the
5128
+ // transient-retry dirty check and leftoverPaths all agree.
4534
5129
  const newlyDirtyAll = afterGuardCwd === null
4535
5130
  ? null
4536
- : [...new Set([
5131
+ : stripAppOwnedChurn([...new Set([
4537
5132
  ...afterGuardCwd.filter((p) => !new Set(guardBaseline || []).has(p)),
4538
5133
  ...worktreeLeftoverDirty,
4539
- ])];
5134
+ ])]);
4540
5135
 
4541
5136
  if (res.launchFailure) {
4542
5137
  await handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion: claudeVersionNow });
@@ -4570,7 +5165,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4570
5165
 
4571
5166
  if (res.rateLimited) {
4572
5167
  const resetIso = await refreshNextReset().catch(() => cachedNextReset);
4573
- await setPaused('rate_limit', resetIso);
5168
+ const observedAt = dispatchStartedAtMs;
5169
+ const prevCount = consecutiveRapidRateLimitsBySlug.get(job.slug) || 0;
5170
+ const nextCount = nextRapidRateLimitCount(prevCount, { rateLimited: true, durationMs: res.durationMs });
5171
+ consecutiveRapidRateLimitsBySlug.set(job.slug, nextCount);
5172
+ const forceHardPause = nextCount >= CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD;
5173
+ if (forceHardPause) {
5174
+ console.log(`[scheduler] ${job.slug}: ${nextCount} consecutive rate-limited dispatches under ${RAPID_RATE_LIMIT_WINDOW_MS / 1000}s each — engaging hard pause`);
5175
+ }
5176
+ await setPaused('rate_limit', resetIso, { observedAt, force: forceHardPause });
5177
+ } else {
5178
+ consecutiveRapidRateLimitsBySlug.delete(job.slug);
4574
5179
  }
4575
5180
 
4576
5181
  // Stale queue entry: the PRD was archived (already shipped) or is gone
@@ -4700,7 +5305,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4700
5305
  ranInWorktree: worktree.ok,
4701
5306
  jobSelfCommitted,
4702
5307
  legitimateNoOp: guardIsLegitimateNoOp,
4703
- isFixPlanJob: isFixPlanSlug(job.slug),
5308
+ isFixPlanJob: resolveIsFixPlan(job.slug, job.isFixPlan),
4704
5309
  verifyResult,
4705
5310
  salvagePatch,
4706
5311
  });
@@ -4777,9 +5382,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4777
5382
  let failedJobSnapshot = null;
4778
5383
  let needsInvestigationNow = false;
4779
5384
  let investigationJobSnapshot = null;
5385
+ let investigationDeadChildSnapshot = null;
4780
5386
  let needsReviewRcaSnapshot = null;
4781
5387
  let resumeRecoveryJob = null;
4782
5388
  let resumeRecoveryTarget = null;
5389
+ let quarantineJob = null;
5390
+ let quarantinePaths = null;
5391
+ let mechanicalRecoveryJob = null;
5392
+ let mechanicalRecoveryTarget = null;
4783
5393
  let terminalNotifySnapshot = null;
4784
5394
  const newlyCompletedPrds = [];
4785
5395
  await mutate((s) => {
@@ -4874,6 +5484,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4874
5484
  } else {
4875
5485
  delete s.jobs[i2].uncommittedPaths;
4876
5486
  }
5487
+ // Worktree merge self-healed (PRD 1125) — every blocking path was
5488
+ // proven byte-identical to the branch, so the duplicate was
5489
+ // discarded and the merge retried once, successfully. Surfaced so
5490
+ // the Queue UI shows a self-heal instead of an ordinary merge.
5491
+ if (mergeAutoResolved) {
5492
+ s.jobs[i2].mergeAutoResolved = mergeAutoResolved;
5493
+ s.jobs[i2].mergeAutoResolvedPaths = capDirtyPaths(mergeAutoResolvedPaths);
5494
+ } else {
5495
+ delete s.jobs[i2].mergeAutoResolved;
5496
+ delete s.jobs[i2].mergeAutoResolvedPaths;
5497
+ }
4877
5498
  // Non-blocking notes (e.g. a recovered missing-dependency probe, or a
4878
5499
  // pattern hit demoted because a materially-checkable verdict outranked
4879
5500
  // it) — surfaced even on completed jobs so the signal isn't lost.
@@ -4924,6 +5545,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4924
5545
  // takes the treatAsPending branch above and never reaches here).
4925
5546
  needsReviewRcaSnapshot = { ...s.jobs[i2] };
4926
5547
 
5548
+ // Mechanical recovery (PRD 1130): evaluated FIRST, ahead of both
5549
+ // resume-first recovery and auto-fix — a job parked with a
5550
+ // mechanically-resolvable verdict (see
5551
+ // selectMechanicalRecoveryTarget) needs no model, no plan, and no
5552
+ // depth-cap check, so it must never fall through to either.
5553
+ // Snapshot only (no I/O inside mutate()); the actual git retry
5554
+ // happens outside mutate(), below.
5555
+ const mTarget = selectMechanicalRecoveryTarget(s.jobs[i2]);
5556
+ if (mTarget) {
5557
+ mechanicalRecoveryJob = { ...s.jobs[i2] };
5558
+ mechanicalRecoveryTarget = mTarget;
5559
+ } else {
4927
5560
  // Resume-first recovery (PRD 1111): evaluated BEFORE the auto-fix
4928
5561
  // eligibility check below — a job whose verdict is
4929
5562
  // 'uncommitted_changes' with a live sessionId gets one bounded
@@ -4937,6 +5570,22 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4937
5570
  resumeRecoveryJob = { ...s.jobs[i2] };
4938
5571
  resumeRecoveryTarget = target;
4939
5572
  } else {
5573
+ // Leftover quarantine (PRD 1128): resume recovery is spent
5574
+ // (resumeRecoveryAttempted already true) and this run STILL parked
5575
+ // needs_review with uncommitted_changes — the leftovers are about
5576
+ // to sit dirty in the shared tree forever, poisoning every later
5577
+ // worktree merge for this cwd. Stamp the one-attempt marker HERE,
5578
+ // synchronously in the same mutate as this decision (mirrors
5579
+ // resumeRecoveryAttempted's own stamp-before-acting rule above),
5580
+ // so a concurrent reverifyNeedsReview pass can never double-fire
5581
+ // this. The actual git work is async and runs outside mutate(),
5582
+ // below (performLeftoverQuarantine).
5583
+ const quarantineTarget = selectLeftoverQuarantineTarget(s.jobs[i2]);
5584
+ if (quarantineTarget) {
5585
+ s.jobs[i2].leftoverQuarantineAttempted = true;
5586
+ quarantineJob = { ...s.jobs[i2] };
5587
+ quarantinePaths = quarantineTarget.paths;
5588
+ }
4940
5589
  // Same-tick auto-fix (feedback 2026-07-12): rather than waiting up to
4941
5590
  // 10 min for reverifyNeedsReview()'s periodic pass, check right here
4942
5591
  // whether this job qualifies for auto-fix (same eligibility rule
@@ -4951,8 +5600,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4951
5600
  isEligibleForImmediateAutoFix(s.jobs[i2], s.jobs, fixSlugExists)
4952
5601
  ) {
4953
5602
  const isRetryAttempt = s.jobs[i2].autoFixAttempted === true;
5603
+ const isDeadFixPlanReopen = isFixPlanDead(s.jobs[i2], s.jobs);
5604
+ if (isDeadFixPlanReopen) {
5605
+ investigationDeadChildSnapshot = s.jobs.find((x) => x.slug === fixSlugFor(s.jobs[i2])) || null;
5606
+ }
4954
5607
  s.jobs[i2].autoFixAttempted = true;
4955
5608
  if (!s.jobs[i2].runId) s.jobs[i2].runId = runId;
5609
+ if (isDeadFixPlanReopen) s.jobs[i2].autoFixReopened = true;
4956
5610
  if (isRetryAttempt) {
4957
5611
  s.jobs[i2].autoFixRetries = (s.jobs[i2].autoFixRetries ?? 0) + 1;
4958
5612
  delete s.jobs[i2].autoFixOutcome;
@@ -4961,13 +5615,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
4961
5615
  investigationJobSnapshot = { ...s.jobs[i2] };
4962
5616
  }
4963
5617
  }
5618
+ }
4964
5619
  }
4965
5620
  // Auto-promote: when a fix-* PRD completes successfully, the original
4966
5621
  // failed PRD's work is logically done. Flip its status to 'completed'
4967
5622
  // so the cross-group failure gate in pickNextBatch releases. Without
4968
5623
  // this, the queue stalls indefinitely behind a stale failure even
4969
5624
  // though the auto-recovery did its job.
4970
- if (effectiveStatus === 'completed' && isFixPlanSlug(job.slug)) {
5625
+ if (effectiveStatus === 'completed' && resolveIsFixPlan(job.slug, job.isFixPlan)) {
4971
5626
  const orig = healTargetForFix(job.slug, s.jobs);
4972
5627
  if (orig) {
4973
5628
  const priorStatus = orig.status;
@@ -5039,6 +5694,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5039
5694
  });
5040
5695
  }
5041
5696
 
5697
+ if (mechanicalRecoveryJob && mechanicalRecoveryTarget) {
5698
+ console.log(`[scheduler] needs_review ${job.slug} → mechanical-recovery (re-integrating ${mechanicalRecoveryTarget.branch})`);
5699
+ performMechanicalRecovery(mechanicalRecoveryJob, mechanicalRecoveryTarget).catch((e) => {
5700
+ console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
5701
+ });
5702
+ }
5703
+
5042
5704
  if (resumeRecoveryJob && resumeRecoveryTarget) {
5043
5705
  console.log(`[scheduler] needs_review ${job.slug} → resume-recovery (session ${resumeRecoveryTarget.sessionId}, ${resumeRecoveryTarget.dirtyPaths.length} dirty path(s))`);
5044
5706
  spawnResumeRecovery(resumeRecoveryJob, resumeRecoveryTarget).catch((e) => {
@@ -5046,6 +5708,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5046
5708
  });
5047
5709
  }
5048
5710
 
5711
+ if (quarantineJob && quarantinePaths) {
5712
+ console.log(`[scheduler] needs_review ${job.slug} → quarantining ${quarantinePaths.length} leftover path(s) (resume recovery already spent)`);
5713
+ performLeftoverQuarantine(quarantineJob, quarantinePaths, guardHeadBefore).catch((e) => {
5714
+ console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
5715
+ });
5716
+ }
5717
+
5049
5718
  if (actuallyFailed && failedJobSnapshot) {
5050
5719
  // Transient-failure detector. A 143/137 exit is ALWAYS a signal kill — the
5051
5720
  // agent never self-exits with those — so the only question is WHO killed it.
@@ -5118,7 +5787,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
5118
5787
  }
5119
5788
  } else if (needsInvestigationNow && investigationJobSnapshot) {
5120
5789
  console.log(`[scheduler] needs_review ${job.slug} → immediate auto-fix investigation (not waiting for periodic reverify)`);
5121
- spawnInvestigation(investigationJobSnapshot, runDir).catch((e) => {
5790
+ spawnInvestigation(investigationJobSnapshot, runDir, { deadChild: investigationDeadChildSnapshot }).catch((e) => {
5122
5791
  console.error('[scheduler] spawnInvestigation error', job.slug, e);
5123
5792
  });
5124
5793
  }
@@ -5188,11 +5857,13 @@ function tickQueue({ bypassLoadGate = false } = {}) {
5188
5857
  // ceilinged the queue at 3 while the pool the user configured said 5.
5189
5858
  const freeSlots = sessionSlots.available();
5190
5859
  const heldSlugs = await computeLaunchHolds(state);
5860
+ const satisfiedSlugsByCwd = await computeDepHistorySatisfaction(state);
5191
5861
  const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
5192
5862
  leaseHeld: quietMachineLease.isHeld(),
5193
5863
  machineInUse: sessionSlots.inUse(),
5194
5864
  now: Date.now(),
5195
5865
  heldSlugs,
5866
+ satisfiedSlugsByCwd,
5196
5867
  });
5197
5868
  if (batch.length === 0 && freeSlots === 0) {
5198
5869
  const snap = sessionSlots.snapshot();
@@ -5366,6 +6037,109 @@ async function maybeLaunchWhenAvailable(state) {
5366
6037
  tickQueue().catch((e) => console.error('[scheduler] tickQueue error', e));
5367
6038
  }
5368
6039
 
6040
+ /** How long a queue may hold ready work with nothing running before the
6041
+ * starvation watchdog forces a tick. Deliberately longer than the poll
6042
+ * loop's own cadence + backoff, so this only ever fires when the normal
6043
+ * path has genuinely stopped driving the queue — it is a safety net, not a
6044
+ * second scheduler. */
6045
+ const QUEUE_STARVATION_MS = 10 * 60_000;
6046
+
6047
+ /**
6048
+ * classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs })
6049
+ * → null | { kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }
6050
+ *
6051
+ * Pure, no IO. Answers the one question the user's invariant reduces to:
6052
+ * "there are PRDs in a queue — is anything actually going to run them?"
6053
+ *
6054
+ * Every stall this codebase has seen was a DIFFERENT cause with the SAME
6055
+ * shape: ready rows, nothing running, nobody ticking. A rate-limited exit
6056
+ * stamped terminal `failed` (2026-09-05, 42 rows); a spin loop past the
6057
+ * manual-clear cooldown; a worktree merge-back that left the project on a
6058
+ * job branch; a job parked `needs_review` with no fix plan; app churn
6059
+ * counted as unfinished work. Guarding each cause individually will always
6060
+ * lag the next one, so this guards the SHAPE instead.
6061
+ *
6062
+ * Two outcomes, deliberately distinguished — they need opposite responses:
6063
+ * 'starved' — at least one pending row is dispatchable RIGHT NOW and
6064
+ * nothing is running. Whatever should have ticked, didn't.
6065
+ * Forcing a tick is safe and fixes it.
6066
+ * 'blocked' — every pending row is behind a terminal/parked dependency.
6067
+ * A tick cannot help; this needs a human (or a heal pass) to
6068
+ * resolve the blocker, and must be reported as such rather
6069
+ * than silently re-ticking forever.
6070
+ *
6071
+ * Returns null when the queue is healthy (work running, nothing pending,
6072
+ * paused on purpose, or simply not idle long enough yet).
6073
+ */
6074
+ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
6075
+ if (paused) return null; // paused is a DECISION, not a stall
6076
+ if (runningCount > 0) return null; // work is flowing
6077
+ const rows = Array.isArray(jobs) ? jobs : [];
6078
+ const pending = rows.filter((j) => j && j.status === 'pending');
6079
+ if (pending.length === 0) return null; // nothing to run — not a stall
6080
+
6081
+ const idleMs = Number.isFinite(lastRunAtMs) ? now - lastRunAtMs : Infinity;
6082
+ if (idleMs < thresholdMs) return null; // give the normal path its chance first
6083
+
6084
+ // Which pending rows could actually dispatch? Anything NOT named by a
6085
+ // blocked chain. computeBlockedChains already walks dependsOn with the
6086
+ // picker's own resolution, so the two can never disagree.
6087
+ const blockedChains = computeBlockedChains(rows);
6088
+ const blockedTotal = blockedChains.reduce((n, c) => n + c.blocked, 0);
6089
+ const dispatchable = pending.length - blockedTotal;
6090
+
6091
+ return {
6092
+ kind: dispatchable > 0 ? 'starved' : 'blocked',
6093
+ pending: pending.length,
6094
+ dispatchable,
6095
+ blockedChains,
6096
+ idleMs,
6097
+ };
6098
+ }
6099
+
6100
+ /**
6101
+ * The watchdog half: acts on classifyQueueStarvation. Called from the
6102
+ * heartbeat, which already runs on its own timer independent of the billing
6103
+ * poll loop — so a wedged or never-succeeding poll (the /api/oauth/usage
6104
+ * endpoint was itself 429ing all of 2026-09-05) can no longer leave a queue
6105
+ * with ready work idle indefinitely.
6106
+ */
6107
+ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
6108
+ const verdict = classifyQueueStarvation({
6109
+ jobs: state?.jobs,
6110
+ paused: state?.paused,
6111
+ runningCount: runningSet.size,
6112
+ lastRunAtMs: Date.parse(state?.lastRunAt ?? ''),
6113
+ now,
6114
+ thresholdMs,
6115
+ });
6116
+ if (!verdict) return null;
6117
+
6118
+ const mins = Math.round(verdict.idleMs / 60_000);
6119
+ if (verdict.kind === 'blocked') {
6120
+ console.warn(
6121
+ `[scheduler] QUEUE BLOCKED: ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
6122
+ + `terminal or parked dependency, so ticking cannot help. Blockers: `
6123
+ + verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
6124
+ );
6125
+ appendAuditEvent('queue_blocked_stall', { pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
6126
+ return verdict;
6127
+ }
6128
+
6129
+ console.warn(
6130
+ `[scheduler] QUEUE STARVED: ${verdict.dispatchable} dispatchable job(s) of ${verdict.pending} pending, 0 running, `
6131
+ + `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
6132
+ );
6133
+ appendAuditEvent('queue_starvation_forced_tick', { pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
6134
+ // A never-populated utilization reading is itself one of the ways the
6135
+ // when-available path silently never fires (maybeLaunchWhenAvailable
6136
+ // returns early on null). Treat unknown as safe here, exactly as the
6137
+ // billing meter's own 429 fallback already does.
6138
+ if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
6139
+ await tickQueue({ bypassLoadGate: false }).catch((e) => console.error('[scheduler] starvation tick error', e));
6140
+ return verdict;
6141
+ }
6142
+
5369
6143
  // ---------- dead-process reaper ----------
5370
6144
 
5371
6145
  // Queue-health sweep cadence: hangs off reapDeadRunningJobs's own cycle
@@ -5440,11 +6214,41 @@ async function reapDeadRunningJobs() {
5440
6214
 
5441
6215
  if (dead.length === 0) return;
5442
6216
 
6217
+ // A rate-limited death is retryable, not terminal — mirror spawnJob's own
6218
+ // live-process handling (PRD 1117) exactly: engage the SAME setPaused
6219
+ // pause here too. Skipping this would reset the row to 'pending' but
6220
+ // leave dispatch unpaused, so the next tick immediately re-fires it into
6221
+ // the same still-active rate limit — the spin loop this PRD exists to
6222
+ // stop. Done once, outside mutate(), before finalizing any row below.
6223
+ if (dead.some((d) => d.outcome === 'rate_limited')) {
6224
+ const resetIso = await refreshNextReset().catch(() => cachedNextReset);
6225
+ const triggering = dead.find((d) => d.outcome === 'rate_limited');
6226
+ const triggeringRow = triggering ? state.jobs.find((x) => x.slug === triggering.slug) : null;
6227
+ const observedAtMs = triggeringRow?.startedAt ? Date.parse(triggeringRow.startedAt) : null;
6228
+ // Same rapid-repeat circuit breaker spawnJob's own res.rateLimited
6229
+ // branch drives (see consecutiveRapidRateLimitsBySlug above) — a
6230
+ // process that gets rate-limited and then dies without spawnJob's own
6231
+ // branch ever running is reconciled HERE instead, and must feed the
6232
+ // same counter or a stale/wrong resumeAt could keep re-clearing this
6233
+ // path's "fresh" pause every reap cycle with no hard cap ever engaging.
6234
+ const durationMs = Number.isFinite(observedAtMs) ? Date.now() - observedAtMs : Infinity;
6235
+ const prevCount = consecutiveRapidRateLimitsBySlug.get(triggering.slug) || 0;
6236
+ const nextCount = nextRapidRateLimitCount(prevCount, { rateLimited: true, durationMs });
6237
+ consecutiveRapidRateLimitsBySlug.set(triggering.slug, nextCount);
6238
+ const forceHardPause = nextCount >= CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD;
6239
+ if (forceHardPause) {
6240
+ console.log(`[scheduler] ${triggering.slug}: ${nextCount} consecutive rate-limited dispatches under ${RAPID_RATE_LIMIT_WINDOW_MS / 1000}s each (reaped) — engaging hard pause`);
6241
+ }
6242
+ await setPaused('rate_limit', resetIso, { observedAt: observedAtMs, force: forceHardPause });
6243
+ }
6244
+
5443
6245
  await mutate(async (s) => {
5444
6246
  for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
5445
6247
  const idx = s.jobs.findIndex((x) => x.slug === slug);
5446
6248
  if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
6249
+ const rateLimited = outcome === 'rate_limited';
5447
6250
  const success = outcome === 'success';
6251
+ if (!rateLimited) consecutiveRapidRateLimitsBySlug.delete(slug);
5448
6252
 
5449
6253
  // Best-effort in-place leftover computation: a job whose owning
5450
6254
  // process vanished without spawnJob()'s own finally block ever
@@ -5479,13 +6283,23 @@ async function reapDeadRunningJobs() {
5479
6283
  const leftoverSuffix = deltaPaths && deltaPaths.length
5480
6284
  ? ` — left ${deltaPaths.length} files uncommitted`
5481
6285
  : '';
5482
- const transitionReason = (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
5483
-
5484
- transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
5485
- s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
5486
- s.jobs[idx].finishedAt = new Date().toISOString();
5487
- s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
5488
- s.jobs[idx].gateOutcome = gateOutcome;
6286
+ const transitionReason = rateLimited
6287
+ ? `reaped: rate limit detected — reset to pending, not failed (outcome=${outcome})${leftoverSuffix}`
6288
+ : (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
6289
+
6290
+ if (rateLimited) {
6291
+ // Retryable, never terminal (PRD 1117) — same resetJobFields path
6292
+ // spawnJob's own rateLimited branch uses (see ~4797's
6293
+ // treatAsPending), so the row comes back exactly like any other
6294
+ // paused-for-rate-limit reset: fresh runId/startedAt/exitCode.
6295
+ resetJobFields(s.jobs[idx], transitionReason, { source: 'reapDeadRunningJobs:rate-limit' });
6296
+ } else {
6297
+ transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
6298
+ s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
6299
+ s.jobs[idx].finishedAt = new Date().toISOString();
6300
+ s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
6301
+ s.jobs[idx].gateOutcome = gateOutcome;
6302
+ }
5489
6303
  delete s.jobs[idx].runtime;
5490
6304
  delete s.jobs[idx].guardBaseline;
5491
6305
  delete s.jobs[idx].guardHeadBefore;
@@ -5499,7 +6313,10 @@ async function reapDeadRunningJobs() {
5499
6313
  // with no exit event) wedges the lease held forever and stalls
5500
6314
  // dispatch for every project until the app restarts.
5501
6315
  if (s.jobs[idx].quietMachine === true) quietMachineLease.release(slug);
5502
- if (pidless) {
6316
+ if (rateLimited) {
6317
+ console.log(`[scheduler] reaped rate-limited job slug=${slug} — reset to pending, pause engaged`);
6318
+ appendAuditEvent('job_reaped_rate_limited', { slug, cwd: s.jobs[idx].cwd ?? null });
6319
+ } else if (pidless) {
5503
6320
  console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
5504
6321
  appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
5505
6322
  } else {
@@ -5711,10 +6528,16 @@ const MAX_INVESTIGATION_DEPTH = 1;
5711
6528
  * no recorded investigationDepth (a job already in the queue before this
5712
6529
  * depth tracking shipped) is treated as excluded too, preserving the
5713
6530
  * pre-existing blanket-exclusion behavior for legacy jobs — no retroactive
5714
- * migration. Non-fix-plan slugs are never capped here. Exported for tests.
6531
+ * migration. Non-fix-plan jobs are never capped here.
6532
+ *
6533
+ * `isFixPlan` (PRD 1131) is the job's own persisted classification stamp
6534
+ * (see lib/fixPlanSlug.cjs's resolveIsFixPlan) — an explicit true/false wins
6535
+ * over the slug; only a row with the field entirely absent (persisted
6536
+ * before this change shipped) falls back to the legacy slug-only heuristic.
6537
+ * Exported for tests.
5715
6538
  */
5716
- function isFixPlanBeyondDepthCap(slug, investigationDepth) {
5717
- if (!isFixPlanSlug(slug)) return false;
6539
+ function isFixPlanBeyondDepthCap(slug, investigationDepth, isFixPlan) {
6540
+ if (!resolveIsFixPlan(slug, isFixPlan)) return false;
5718
6541
  if (investigationDepth == null) return true;
5719
6542
  return investigationDepth >= MAX_INVESTIGATION_DEPTH + 1;
5720
6543
  }
@@ -5767,12 +6590,18 @@ function isUnresolvableNeedsReview(job, { hasRunDir }) {
5767
6590
  * ('no-plan', 'error', and unstamped/undefined) — mirrors the retry
5768
6591
  * eligibility rule in selectAutoFixTargets so a job can never be retry-
5769
6592
  * eligible there and simultaneously un-annotatable here.
6593
+ *
6594
+ * A parent stamped `autoFixReopened: true` (its dead fix-plan child earned
6595
+ * it exactly one further attempt — see isFixPlanDead) is a separate
6596
+ * exhaustion path: it is spent as soon as that second investigation
6597
+ * concludes with ANY outcome, including another 'plan' — a reopened parent
6598
+ * never gets a third attempt, so unlike the fresh case a 'plan' outcome does
6599
+ * not exempt it here.
5770
6600
  */
5771
6601
  function isExhaustedAutoFix(job) {
5772
- return !!job && job.status === 'needs_review'
5773
- && job.autoFixAttempted === true
5774
- && job.autoFixOutcome !== 'plan'
5775
- && (job.autoFixRetries ?? 0) >= 1;
6602
+ if (!job || job.status !== 'needs_review' || job.autoFixAttempted !== true) return false;
6603
+ if (job.autoFixReopened === true) return job.autoFixOutcome != null;
6604
+ return job.autoFixOutcome !== 'plan' && (job.autoFixRetries ?? 0) >= 1;
5776
6605
  }
5777
6606
 
5778
6607
  /**
@@ -5788,6 +6617,31 @@ function isPlanUnqueued(job, queuedSlugs) {
5788
6617
  return !queuedSlugs.has(fixSlugFor(job));
5789
6618
  }
5790
6619
 
6620
+ // Terminal-and-not-completed statuses a fix-plan child can die in — see
6621
+ // isFixPlanDead.
6622
+ const DEAD_FIX_CHILD_STATUSES = new Set(['needs_review', 'failed', 'quarantined', 'skipped']);
6623
+
6624
+ /**
6625
+ * Pure predicate: a parent stuck at outcome 'plan' whose own fix-plan child
6626
+ * (fixSlugFor(job)) has ITSELF died — reached a terminal non-completed
6627
+ * status — with nothing left in the ladder that will ever revisit either
6628
+ * row again (selectAutoFixTargets skips a 'plan' outcome outright, and
6629
+ * isPlanUnqueued only fires when the child never reached the queue at all,
6630
+ * which isn't true once a dead child row exists). `job.autoFixReopened`
6631
+ * gates this to exactly once per parent — once stamped, this always returns
6632
+ * false so the parent can never be reopened a second time. Exported for
6633
+ * tests.
6634
+ */
6635
+ function isFixPlanDead(job, jobsInProject) {
6636
+ if (!job || job.status !== 'needs_review') return false;
6637
+ if (job.autoFixOutcome !== 'plan') return false;
6638
+ if (job.autoFixReopened === true) return false;
6639
+ const fixSlug = fixSlugFor(job);
6640
+ const child = (jobsInProject || []).find((j) => j.slug === fixSlug);
6641
+ if (!child) return false;
6642
+ return DEAD_FIX_CHILD_STATUSES.has(child.status);
6643
+ }
6644
+
5791
6645
  /**
5792
6646
  * Pure predicate: is this job eligible for the boot re-verify self-heal? Only
5793
6647
  * needs_review jobs with a run log (own or backfilled via resolveRunId) AND a
@@ -5933,16 +6787,30 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
5933
6787
  // bounded `--resume` attempt must never also become a fix-plan target
5934
6788
  // in the same pass — see spawnInvestigation's own identical guard.
5935
6789
  if (selectResumeRecoveryTarget(job)) return false;
6790
+ // Mechanical recovery (PRD 1130): a job eligible for a pure-git retry
6791
+ // must never also become a fix-plan target — it needs no plan and no
6792
+ // model. Defensive: today's single mechanically-resolvable verdict
6793
+ // (worktree_integration_failed) is already excluded below via the depth
6794
+ // cap, but this must hold even if that stops being true.
6795
+ if (selectMechanicalRecoveryTarget(job)) return false;
5936
6796
  const runId = job.runId || resolveJobRunId(job);
5937
6797
  if (!runId) return false;
5938
- if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth)) return false;
6798
+ if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth, job.isFixPlan)) return false;
6799
+ // A dead fix-plan child (PRD 1129) earns its parent exactly one further
6800
+ // attempt, bypassing the normal 'plan' exclusion and the fix-slug/queue
6801
+ // membership checks below — those checks exist to stop a FRESH
6802
+ // investigation from clobbering a live sibling, but here the sibling is
6803
+ // dead and reusing its slug is the whole point of the reopen.
6804
+ const dead = isFixPlanDead(job, jobs);
5939
6805
  if (job.autoFixAttempted) {
5940
- const retryEligible = job.autoFixOutcome === 'no-plan'
6806
+ const retryEligible = dead
6807
+ || job.autoFixOutcome === 'no-plan'
5941
6808
  || job.autoFixOutcome === 'error'
5942
6809
  || job.autoFixOutcome == null;
5943
6810
  if (!retryEligible) return false;
5944
- if ((job.autoFixRetries ?? 0) >= 1) return false;
6811
+ if (!dead && (job.autoFixRetries ?? 0) >= 1) return false;
5945
6812
  }
6813
+ if (dead) return true;
5946
6814
  const fixSlug = fixSlugFor(job);
5947
6815
  if (fixSlugExists(fixSlug)) return false;
5948
6816
  if (slugsInQueue.has(fixSlug)) return false;
@@ -6121,7 +6989,7 @@ async function reverifyNeedsReview() {
6121
6989
  const promotedPrds = [];
6122
6990
  await mutate((s) => {
6123
6991
  for (const job of s.jobs) {
6124
- if (job.status !== 'completed' || !isFixPlanSlug(job.slug)) continue;
6992
+ if (job.status !== 'completed' || !resolveIsFixPlan(job.slug, job.isFixPlan)) continue;
6125
6993
  const orig = healTargetForFix(job.slug, s.jobs);
6126
6994
  if (!orig) continue;
6127
6995
  const priorStatus = orig.status;
@@ -6205,6 +7073,23 @@ async function reverifyNeedsReview() {
6205
7073
  ? await readQueue()
6206
7074
  : afterHealForAnnotate;
6207
7075
 
7076
+ // Mechanical recovery (PRD 1130): evaluated first, ahead of both
7077
+ // resume-first recovery and auto-fix below — catches a job whose
7078
+ // mechanically-resolvable verdict this periodic pass finds still eligible
7079
+ // (e.g. one already parked before this rung shipped, or one the same-tick
7080
+ // check in spawnJob missed because the app restarted in between). Depth
7081
+ // never disqualifies it, so it runs regardless of investigationDepth.
7082
+ {
7083
+ for (const job of queueForResumeAndAutofix.jobs) {
7084
+ const target = selectMechanicalRecoveryTarget(job);
7085
+ if (!target) continue;
7086
+ console.log(`[scheduler] mechanical-recovery: needs_review ${job.slug} → re-integrating ${target.branch}`);
7087
+ performMechanicalRecovery(job, target).catch((e) => {
7088
+ console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
7089
+ });
7090
+ }
7091
+ }
7092
+
6208
7093
  // Resume-first recovery (PRD 1111): before any fix-plan investigation is
6209
7094
  // authored below, offer the bounded one-attempt `--resume` dispatch to any
6210
7095
  // needs_review job this periodic pass finds still eligible — e.g. one the
@@ -6223,6 +7108,26 @@ async function reverifyNeedsReview() {
6223
7108
  }
6224
7109
  }
6225
7110
 
7111
+ // Leftover quarantine (PRD 1128), periodic pass: catches a job parked
7112
+ // needs_review with resume recovery already spent BEFORE this feature
7113
+ // shipped, or one the same-tick check in spawnJob missed because the app
7114
+ // restarted in between. Stamps the one-attempt marker in its own mutate
7115
+ // BEFORE the async git work starts (same race-closing rule as the resume
7116
+ // loop above and spawnJob's own dispatch stamp).
7117
+ {
7118
+ for (const job of queueForResumeAndAutofix.jobs) {
7119
+ const quarantineTarget = selectLeftoverQuarantineTarget(job);
7120
+ if (!quarantineTarget) continue;
7121
+ console.log(`[scheduler] leftover-quarantine: needs_review ${job.slug} → quarantining ${quarantineTarget.paths.length} leftover path(s)`);
7122
+ mutate((s) => {
7123
+ const j = s.jobs.find((x) => x.slug === job.slug);
7124
+ if (j) j.leftoverQuarantineAttempted = true;
7125
+ }).then(() => performLeftoverQuarantine(job, quarantineTarget.paths)).catch((e) => {
7126
+ console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
7127
+ });
7128
+ }
7129
+ }
7130
+
6226
7131
  // Auto-fix: spawn a fix-plan investigation for each job still in
6227
7132
  // needs_review after the heal pass (kill-switch: SM_AUTOFIX_DISABLE=1).
6228
7133
  // spawnInvestigation early-returns once investigationsInFlight reaches
@@ -6236,23 +7141,30 @@ async function reverifyNeedsReview() {
6236
7141
  const runId = job.runId || resolveRunId(job);
6237
7142
  const runDir = path.join(RUNS_DIR, runId);
6238
7143
  const isRetryAttempt = job.autoFixAttempted === true;
7144
+ const isDeadFixPlanReopen = isFixPlanDead(job, queueForResumeAndAutofix.jobs);
7145
+ const deadChild = isDeadFixPlanReopen
7146
+ ? queueForResumeAndAutofix.jobs.find((j) => j.slug === fixSlugFor(job))
7147
+ : null;
6239
7148
  // Persist the attempt BEFORE spawning — a crash mid-investigation still
6240
7149
  // counts it (mirrors orphanRetries). Safe even when the slot is busy: the
6241
7150
  // investigation is queued and drained as slots free, so it is genuinely
6242
- // attempted rather than silently dropped.
7151
+ // attempted rather than silently dropped. autoFixReopened is stamped in
7152
+ // this SAME mutate so a crash between selection and dispatch can never
7153
+ // leave the parent re-eligible for a second reopen (PRD 1129).
6243
7154
  await mutate((s) => {
6244
7155
  const j = s.jobs.find((x) => x.slug === job.slug);
6245
7156
  if (j) {
6246
7157
  j.autoFixAttempted = true;
6247
7158
  if (!j.runId && runId) j.runId = runId;
7159
+ if (isDeadFixPlanReopen) j.autoFixReopened = true;
6248
7160
  if (isRetryAttempt) {
6249
7161
  j.autoFixRetries = (j.autoFixRetries ?? 0) + 1;
6250
7162
  delete j.autoFixOutcome;
6251
7163
  }
6252
7164
  }
6253
7165
  });
6254
- console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'})`);
6255
- spawnInvestigation(job, runDir).catch((e) => {
7166
+ console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'}${isDeadFixPlanReopen ? ', dead fix-plan child reopen' : ''})`);
7167
+ spawnInvestigation(job, runDir, { deadChild }).catch((e) => {
6256
7168
  console.error('[scheduler] auto-fix spawnInvestigation error', job.slug, e);
6257
7169
  });
6258
7170
  }
@@ -6865,6 +7777,14 @@ async function init() {
6865
7777
  if (heartbeatInterval) clearInterval(heartbeatInterval);
6866
7778
  heartbeatInterval = setInterval(() => {
6867
7779
  const s = readQueueSync();
7780
+ // NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
7781
+ // running, something must drive it. This is the only driver that does
7782
+ // not depend on the billing poll loop, a pause timer, or a completing
7783
+ // job to schedule the next tick — every one of which has failed at
7784
+ // least once. See classifyQueueStarvation.
7785
+ if (!s.unreadable) {
7786
+ runQueueStarvationWatchdog(s).catch((e) => console.error('[scheduler] starvation watchdog error', e));
7787
+ }
6868
7788
  // Initialise from the real status union (scheduleJobSchema.cjs) rather
6869
7789
  // than a hand-maintained subset — the old `{ pending, running, completed,
6870
7790
  // failed }` literal silently minted a NEW key for any other value
@@ -7335,6 +8255,36 @@ const remote = {
7335
8255
  return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
7336
8256
  }
7337
8257
 
8258
+ // Write-time FK check for a patched dependsOn (PRD 1124), reusing the
8259
+ // SAME resolution rule scheduler_create_prd's prdCreate.cjs applies (exact
8260
+ // slug, else bare-name after stripping one leading `NN-`) so update and
8261
+ // create can never disagree about what a dependsOn entry resolves to. An
8262
+ // explicit empty array CLEARS the dependency and skips validation — there
8263
+ // is nothing to resolve. A listPrds() read failure is skipped-with-a-
8264
+ // warning, matching createPrd's tolerance for an I/O hiccup.
8265
+ if (frontmatter && Array.isArray(frontmatter.dependsOn) && frontmatter.dependsOn.length) {
8266
+ let listing;
8267
+ try {
8268
+ listing = await this.listPrds({ cwd, limit: Number.MAX_SAFE_INTEGER });
8269
+ } catch (e) {
8270
+ console.warn(`[scheduler] updatePrd: dependsOn validation skipped (listPrds failed): ${e?.message ?? e}`);
8271
+ listing = null;
8272
+ }
8273
+ if (listing) {
8274
+ const candidateSlugs = (listing.prds ?? []).map((p) => p.slug);
8275
+ for (const dep of frontmatter.dependsOn) {
8276
+ if (resolveDepSlug(dep, candidateSlugs).length > 0) continue;
8277
+ const near = findNearMatches(dep, candidateSlugs);
8278
+ const suggestion = near.length ? ` Closest existing slug(s): ${near.join(', ')}.` : '';
8279
+ return {
8280
+ ok: false,
8281
+ error: `dependsOn entry "${dep}" does not resolve to any existing PRD in this project.${suggestion} ` +
8282
+ 'Pass the bare name (preferred) or the exact NN-prefixed slug of an existing PRD.',
8283
+ };
8284
+ }
8285
+ }
8286
+ }
8287
+
7338
8288
  let dir = null;
7339
8289
  let filePath = null;
7340
8290
  if (cwd) {
@@ -7465,4 +8415,149 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
7465
8415
  });
7466
8416
  }
7467
8417
 
7468
- module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure };
8418
+ module.exports = {
8419
+ classifyQueueStarvation,
8420
+ runQueueStarvationWatchdog,
8421
+ QUEUE_STARVATION_MS,
8422
+ computeBlockedChains,
8423
+ stripAppOwnedChurn,
8424
+ findOverrunningJobs,
8425
+ JOB_OVERRUN_FACTOR,
8426
+ JOB_OVERRUN_FLOOR_MS,
8427
+ registerScheduleHandlers,
8428
+ attachWindow,
8429
+ init,
8430
+ ROOT,
8431
+ PRDS_DIR,
8432
+ healRefusalReason,
8433
+ writeQueue,
8434
+ reconcile,
8435
+ reconcileSourcePromptId,
8436
+ allocateParallelGroup,
8437
+ selectHistoryJobs,
8438
+ parsePorcelain,
8439
+ FINISH_PROTOCOL,
8440
+ IDLE_OUTPUT_KILL_MS,
8441
+ BASH_DEFAULT_TIMEOUT_MS,
8442
+ BASH_MAX_TIMEOUT_MS,
8443
+ remote,
8444
+ pickNextBatch,
8445
+ pickForProject,
8446
+ reapDeadRunningJobs,
8447
+ pollRecoveryClearSource,
8448
+ memoryLimitedBatchSize,
8449
+ availableForJobs,
8450
+ reverifyNeedsReview,
8451
+ isRescanCandidate,
8452
+ isFailedUnverifiedShaped,
8453
+ computeLooksDone,
8454
+ isPromotableOriginal,
8455
+ selectAutoFixTargets,
8456
+ applyRcaClassification,
8457
+ isEligibleForImmediateAutoFix,
8458
+ resolveRunId,
8459
+ isUnresolvableNeedsReview,
8460
+ isExhaustedAutoFix,
8461
+ isPlanUnqueued,
8462
+ isFixPlanDead,
8463
+ fixSlugFor,
8464
+ healTargetForFix,
8465
+ buildInvestigationPrompt,
8466
+ isGitRepoSync,
8467
+ committedInWindow,
8468
+ computeCommittedDuringRun,
8469
+ classifySigtermWithCommit,
8470
+ isFixPlanSlug,
8471
+ classifyDiscoveredFixPlan,
8472
+ resolveIsFixPlan,
8473
+ isFixPlanBeyondDepthCap,
8474
+ MAX_INVESTIGATION_DEPTH,
8475
+ forceTickOutcome,
8476
+ applyPauseCleared,
8477
+ detectNetworkErrorInLog,
8478
+ detectRateLimitInLog,
8479
+ classifyFailureOutcome,
8480
+ commitGuardVerdict,
8481
+ leftoverFieldsFrom,
8482
+ applyLeftoverFields,
8483
+ LEFTOVER_PATHS_CAP,
8484
+ capDirtyPaths,
8485
+ buildForeignWipSection,
8486
+ PRE_RUN_DIRTY_PATHS_CAP,
8487
+ FOREIGN_WIP_DELIMITER,
8488
+ FOREIGN_WIP_END_DELIMITER,
8489
+ TRANSIENT_RETRY_CAP,
8490
+ buildScheduleStatePayload,
8491
+ partitionBootOrphans,
8492
+ applyOrphanOutcome,
8493
+ BOOT_ORPHAN_KILL_GRACE_MS,
8494
+ registerAdminRoutes,
8495
+ notifyOriginatingTab,
8496
+ notifyNeedsReview,
8497
+ isNotifiableTerminalStatus,
8498
+ extractResultTextFromLog,
8499
+ candidatePrdsDirs,
8500
+ candidateArchivedPrdsDirs,
8501
+ resolveArchivedPrdStatus,
8502
+ prdDirForCwd,
8503
+ prdPathForJob,
8504
+ archivedPrdPathForJob,
8505
+ archivedTwinExists,
8506
+ findPrdDir,
8507
+ resolveVerifyPrdPath,
8508
+ resolveFixPlanPath,
8509
+ resolveNotifyPrd,
8510
+ runPrdMigration,
8511
+ consolidateAllFlatPrds,
8512
+ shouldSkipInvestigationForCleanRun,
8513
+ archiveCompletedPrd,
8514
+ retireCompletedSlugs,
8515
+ SCHEDULER_BOOTED_AT,
8516
+ SCHEDULER_CODE_SHA,
8517
+ resetJobFields,
8518
+ executeJob,
8519
+ prdArchivedSkipResult,
8520
+ spawnJob,
8521
+ listPrdsInternal,
8522
+ computeStallSummary,
8523
+ findStaleQuarantinedJobs,
8524
+ QUARANTINE_ESCALATE_MS,
8525
+ applyClearQueueVictims,
8526
+ PIDLESS_SPAWN_GRACE_MS,
8527
+ findStrandedInvestigations,
8528
+ INVESTIGATION_MAX_MS,
8529
+ stashList,
8530
+ parseStashLine,
8531
+ pathsChangedSince,
8532
+ restoreSpecificStash,
8533
+ evaluateSharedTreeGuard,
8534
+ checkSharedTreeGuard,
8535
+ uncommittedChanges,
8536
+ gitHead,
8537
+ selectResumeRecoveryTarget,
8538
+ buildResumeRecoveryPreamble,
8539
+ buildClaudeSpawnArgs,
8540
+ spawnResumeRecovery,
8541
+ selectMechanicalRecoveryTarget,
8542
+ performMechanicalRecovery,
8543
+ MECHANICALLY_RESOLVABLE_VERDICTS,
8544
+ selectLeftoverQuarantineTarget,
8545
+ quarantineLeftovers,
8546
+ performLeftoverQuarantine,
8547
+ spawnInvestigation,
8548
+ computeLaunchHolds,
8549
+ computeDepHistorySatisfaction,
8550
+ handleLaunchFailure,
8551
+ applyLaunchFailure,
8552
+ setPaused,
8553
+ clearPause,
8554
+ tickQueue,
8555
+ runDueJobs,
8556
+ isCooldownSuppressed,
8557
+ nextRapidRateLimitCount,
8558
+ CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
8559
+ RAPID_RATE_LIMIT_WINDOW_MS,
8560
+ MANUAL_PAUSE_COOLDOWN_MS,
8561
+ RUNS_DIR,
8562
+ pickRunDir,
8563
+ };