claude-code-session-manager 0.92.1 → 0.94.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/dist/assets/DataModel-B0LDnnSL.js +1 -0
  2. package/dist/assets/{History-BOR_fJNr.js → History-C685Kytt.js} +2 -2
  3. package/dist/assets/{Hooks-BtD436iL.js → Hooks-CwLnp0Z_.js} +3 -3
  4. package/dist/assets/HostBilko-CwEHEKYk.js +1 -0
  5. package/dist/assets/{Library-DeKdY-tD.js → Library-BU9np05E.js} +25 -25
  6. package/dist/assets/{MarkdownEditor-CbzC-Oqf.js → MarkdownEditor-B74ogK2f.js} +1 -1
  7. package/dist/assets/McpServers-BLM3ftzP.js +2 -0
  8. package/dist/assets/{Memory-B68AJjg2.js → Memory-DAP1t8bA.js} +6 -6
  9. package/dist/assets/{Permissions-Cr0RRvNy.js → Permissions-3pGLJmxQ.js} +3 -3
  10. package/dist/assets/Plugins-Q1KoLEzn.js +2 -0
  11. package/dist/assets/ProvenanceBadge-BnjBQiV9.js +1 -0
  12. package/dist/assets/{SaveBar-87ZJfscX.js → SaveBar-Zq2NfcPf.js} +1 -1
  13. package/dist/assets/Scheduler-D488ebKm.js +16 -0
  14. package/dist/assets/{ScopeSwitcher-CKzpqjJ_.js → ScopeSwitcher-Cnojv22D.js} +1 -1
  15. package/dist/assets/Settings-D5Lhj7ga.js +3 -0
  16. package/dist/assets/{SkillReferenceGraph-D_-wg4r9.js → SkillReferenceGraph-BRDiE6Hv.js} +2 -2
  17. package/dist/assets/Skills-BQDqp-EN.js +3 -0
  18. package/dist/assets/SystemPrompt-tFr19Od6.js +1 -0
  19. package/dist/assets/TagLibrary-BjyrnaRE.js +1 -0
  20. package/dist/assets/{TiptapBody-DYXIp8aG.js → TiptapBody-DGzSg3BO.js} +1 -1
  21. package/dist/assets/{Toggle-CDXR5l3m.js → Toggle-DzoROibf.js} +1 -1
  22. package/dist/assets/index-CCl4tz-u.js +3079 -0
  23. package/dist/assets/{index-D3P_jldk.css → index-CKH5Uxik.css} +1 -1
  24. package/dist/assets/settingsSchema-BYKVe1WM.js +1 -0
  25. package/dist/index.html +2 -2
  26. package/package.json +6 -3
  27. package/scripts/README.md +3 -0
  28. package/src/main/__tests__/agentEffortResolve.test.cjs +117 -0
  29. package/src/main/__tests__/agentLibrary.test.cjs +21 -0
  30. package/src/main/__tests__/agentModelResolve.test.cjs +16 -0
  31. package/src/main/__tests__/agentOverlayWrite.test.cjs +95 -0
  32. package/src/main/__tests__/chat-cancel-terminal.test.cjs +5 -2
  33. package/src/main/__tests__/chat-exit-close-race.test.cjs +8 -2
  34. package/src/main/__tests__/chat-mcp-consent-notice.test.cjs +5 -2
  35. package/src/main/__tests__/chatRunner-session-flag-retry.test.cjs +44 -0
  36. package/src/main/__tests__/intradayRefresh.test.cjs +39 -0
  37. package/src/main/__tests__/openExternalApp-spawn-error.test.cjs +25 -0
  38. package/src/main/__tests__/opsErrorLog.test.cjs +22 -0
  39. package/src/main/__tests__/personaMerge.test.cjs +169 -0
  40. package/src/main/__tests__/prdCreatePlanId.test.cjs +132 -0
  41. package/src/main/__tests__/promptSessionsCreateEpicHandler.test.cjs +13 -0
  42. package/src/main/__tests__/pty-epic-worktree-spawn-cwd.test.cjs +21 -0
  43. package/src/main/__tests__/rateLimitPollerStreak.test.cjs +14 -0
  44. package/src/main/__tests__/runVerify-atomic-verdicts.test.cjs +26 -0
  45. package/src/main/__tests__/runVerify-landed-commit-outranks.test.cjs +11 -1
  46. package/src/main/__tests__/runVerify-transcript-commit-evidence.test.cjs +11 -1
  47. package/src/main/__tests__/runVerify.test.cjs +12 -1
  48. package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +47 -2
  49. package/src/main/__tests__/transcriptsUsageFor.test.cjs +112 -1
  50. package/src/main/agentLibrary.cjs +86 -40
  51. package/src/main/build-info.json +4 -4
  52. package/src/main/chatRunner.cjs +50 -10
  53. package/src/main/git.cjs +9 -2
  54. package/src/main/historyAggregator.cjs +2 -19
  55. package/src/main/index.cjs +17 -8
  56. package/src/main/ipcSchemas.cjs +25 -3
  57. package/src/main/lib/__tests__/childWithLog.test.cjs +180 -0
  58. package/src/main/lib/__tests__/delegationReadiness.test.cjs +19 -0
  59. package/src/main/lib/__tests__/effectiveModelInfo.test.cjs +10 -5
  60. package/src/main/lib/__tests__/gitCacheBound.test.cjs +69 -0
  61. package/src/main/lib/__tests__/modelCatalog.test.cjs +202 -0
  62. package/src/main/lib/agentEffortResolve.cjs +74 -0
  63. package/src/main/lib/agentModelResolve.cjs +52 -59
  64. package/src/main/lib/agentPersonaSchema.cjs +5 -0
  65. package/src/main/lib/childWithLog.cjs +92 -51
  66. package/src/main/lib/delegationReadiness.cjs +3 -1
  67. package/src/main/lib/effectiveModelInfo.cjs +38 -20
  68. package/src/main/lib/epicMint.cjs +5 -2
  69. package/src/main/lib/epicSpawnPlan.cjs +27 -8
  70. package/src/main/lib/epicTranscriptPath.cjs +5 -1
  71. package/src/main/lib/epicWorktreeMerge.cjs +7 -6
  72. package/src/main/lib/headTailBuffer.cjs +43 -0
  73. package/src/main/lib/intradayRefresh.cjs +33 -0
  74. package/src/main/lib/lruCache.cjs +39 -0
  75. package/src/main/lib/modelCatalog.cjs +243 -0
  76. package/src/main/lib/openExternalApp.cjs +27 -9
  77. package/src/main/lib/opsErrorLog.cjs +22 -0
  78. package/src/main/lib/personaMerge.cjs +166 -0
  79. package/src/main/lib/prdCreate.cjs +23 -5
  80. package/src/main/lib/prdDisposition.cjs +46 -0
  81. package/src/main/lib/prdFrontmatter.cjs +6 -2
  82. package/src/main/lib/promptSessionSchema.cjs +5 -0
  83. package/src/main/lib/promptSessionsCreateEpic.cjs +5 -3
  84. package/src/main/lib/rendererRecovery.cjs +141 -0
  85. package/src/main/lib/scheduleJobSchema.cjs +3 -0
  86. package/src/main/runVerify.cjs +3 -1
  87. package/src/main/scheduler/prdParser.cjs +2 -0
  88. package/src/main/scheduler.cjs +381 -262
  89. package/src/main/templates/PRD_AUTHORING.md +4 -0
  90. package/src/main/transcripts.cjs +54 -11
  91. package/src/preload/api.d.ts +56 -3
  92. package/src/preload/index.cjs +4 -0
  93. package/dist/assets/AgentLibrary-Ci6u03bq.js +0 -3
  94. package/dist/assets/DataModel-BIqcjsv2.js +0 -1
  95. package/dist/assets/HostBilko-CMe5cH3H.js +0 -1
  96. package/dist/assets/ListDetail-CtVjrHKB.js +0 -1
  97. package/dist/assets/McpServers-DKN75Shn.js +0 -2
  98. package/dist/assets/Panel-C2YTW-qP.js +0 -1
  99. package/dist/assets/Plugins-DbZkJs5l.js +0 -2
  100. package/dist/assets/ProvenanceBadge-kJ__BOQv.js +0 -1
  101. package/dist/assets/Scheduler-DdEaxku8.js +0 -16
  102. package/dist/assets/Settings-CjywVsNB.js +0 -3
  103. package/dist/assets/Skills-Cd_yPO_l.js +0 -3
  104. package/dist/assets/SystemPrompt-2z6eGPSF.js +0 -1
  105. package/dist/assets/TagLibrary-DcaWQSp4.js +0 -1
  106. package/dist/assets/index-D5H_H5wC.js +0 -3074
  107. package/dist/assets/settingsSchema-JRDfjgHb.js +0 -3
@@ -145,6 +145,7 @@ const queueOps = require('./queueOps.cjs');
145
145
  const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs, deriveProjectCwdFromPrdPath } = require('./lib/prdLocations.cjs');
146
146
  const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
147
147
  const agentModelResolve = require('./lib/agentModelResolve.cjs');
148
+ const { resolveEpicEffort, effortArgs } = require('./lib/agentEffortResolve.cjs');
148
149
  const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
149
150
  const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
150
151
  const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
@@ -181,7 +182,7 @@ const supervisorRecord = require('./lib/jobSupervisorRecord.cjs');
181
182
  const adoptedRunSupervisor = require('./lib/adoptedRunSupervisor.cjs');
182
183
  const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
183
184
  const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
184
- const { computeDispositionRewrite } = require('./lib/prdDisposition.cjs');
185
+ const { computeDispositionRewrite, mintPlanId, resolveInheritedPlanId } = require('./lib/prdDisposition.cjs');
185
186
  const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
186
187
  const { allProjectCwds } = require('./lib/activeSessions.cjs');
187
188
 
@@ -1506,6 +1507,7 @@ function loadSchedulerState() {
1506
1507
  if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
1507
1508
  if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
1508
1509
  if (typeof s.failureStreakWarnedAt === 'number') failureStreakWarnedAt = s.failureStreakWarnedAt;
1510
+ failureStreakWarnedAt = restoreFailureStreakWarnedAt(failureStreakWarned, failureStreakWarnedAt, Date.now());
1509
1511
  if (typeof s.lastEscalationAt === 'number') lastEscalationAtMs = s.lastEscalationAt;
1510
1512
  } catch { /* first boot or corrupt — start fresh */ }
1511
1513
  }
@@ -2811,6 +2813,7 @@ async function reconcile(state) {
2811
2813
  epicId: p.epicId ?? job.epicId ?? null,
2812
2814
  dependsOn: p.dependsOn,
2813
2815
  disposition: p.disposition ?? null,
2816
+ planId: p.planId ?? null,
2814
2817
  quietMachine: p.quietMachine === true,
2815
2818
  budgetExempt: p.budgetExempt === true,
2816
2819
  originSessionId: job.originSessionId
@@ -2933,6 +2936,7 @@ async function reconcile(state) {
2933
2936
  epicId: p.epicId ?? inv.row?.epicId ?? null,
2934
2937
  dependsOn: p.dependsOn,
2935
2938
  disposition: p.disposition ?? null,
2939
+ planId: p.planId ?? null,
2936
2940
  quietMachine: p.quietMachine === true,
2937
2941
  budgetExempt: p.budgetExempt === true,
2938
2942
  originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(repairedCwd, p.epicId ?? p.sourcePromptId),
@@ -3067,6 +3071,7 @@ async function reconcile(state) {
3067
3071
  epicId: p.epicId ?? null,
3068
3072
  dependsOn: p.dependsOn,
3069
3073
  disposition: p.disposition ?? null,
3074
+ planId: p.planId ?? null,
3070
3075
  quietMachine: p.quietMachine === true,
3071
3076
  budgetExempt: p.budgetExempt === true,
3072
3077
  originSessionId: resolveOriginSessionId(discoveredCwd, p.epicId ?? p.sourcePromptId),
@@ -3393,6 +3398,21 @@ function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold =
3393
3398
  return consecutiveFailures >= threshold && !alreadyWarned;
3394
3399
  }
3395
3400
 
3401
+ /**
3402
+ * Pure: `failureStreakWarned === true` must always carry a numeric
3403
+ * `failureStreakWarnedAt` (a state file may hold one without the other), so
3404
+ * the escalation message never renders "after nullm". Backfills `nowMs`.
3405
+ */
3406
+ function restoreFailureStreakWarnedAt(warned, warnedAt, nowMs) {
3407
+ if (!warned) return typeof warnedAt === 'number' ? warnedAt : null;
3408
+ return typeof warnedAt === 'number' ? warnedAt : nowMs;
3409
+ }
3410
+
3411
+ /** Pure: whole minutes a warned streak has persisted; never null/NaN. */
3412
+ function persistedStreakMinutes(warnedAt, nowMs) {
3413
+ return typeof warnedAt === 'number' ? Math.round((nowMs - warnedAt) / 60_000) : 0;
3414
+ }
3415
+
3396
3416
  /**
3397
3417
  * Pure: does a PERSISTING failure streak warrant another escalation (audit
3398
3418
  * event + opsErrorLog line) at `nowMs`? Exported for unit testing. Only
@@ -3434,7 +3454,7 @@ function warnFailureStreakIfNeeded() {
3434
3454
  }
3435
3455
  if (failureStreakWarned && shouldEscalateFailureStreak(consecutiveFailures, lastEscalationAtMs, nowMs)) {
3436
3456
  lastEscalationAtMs = nowMs;
3437
- const persistedMinutes = failureStreakWarnedAt ? Math.round((nowMs - failureStreakWarnedAt) / 60_000) : null;
3457
+ const persistedMinutes = persistedStreakMinutes(failureStreakWarnedAt, nowMs);
3438
3458
  try {
3439
3459
  appendAuditEvent('usage_poller_failure_streak_persists', { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes });
3440
3460
  appendError({
@@ -5249,15 +5269,17 @@ async function performLeftoverQuarantine(job, paths, headBefore = null) {
5249
5269
  * selects `--resume <sessionId>` (reconnect) INSTEAD of `--session-id
5250
5270
  * <sessionId>` (mint) — the two flags are mutually exclusive, never both.
5251
5271
  * `--model` is always explicit (never left to the CLI's drifting default —
5252
- * see conventions.md). `systemPrompt`, when given (the PRD's `agentType`
5272
+ * see conventions.md). `effort`, when a level (persona `effort:` via
5273
+ * agentEffortResolve.cjs), appends `--effort <level>`; null/inherit → no flag. `systemPrompt`, when given (the PRD's `agentType`
5253
5274
  * persona body, resolved by agentModelResolve.cjs's resolvePrdPersonaForSpawn),
5254
5275
  * is passed as `--append-system-prompt` so the executor IS that persona at
5255
5276
  * launch rather than being asked in prose to adopt one.
5256
5277
  */
5257
- function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }) {
5278
+ function buildClaudeSpawnArgs({ prompt, model, effort, sessionId, resume, systemPrompt }) {
5258
5279
  return [
5259
5280
  '-p', prompt,
5260
5281
  '--model', model,
5282
+ ...effortArgs(effort),
5261
5283
  ...(systemPrompt ? ['--append-system-prompt', systemPrompt] : []),
5262
5284
  '--dangerously-skip-permissions',
5263
5285
  '--output-format', 'stream-json',
@@ -5510,7 +5532,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5510
5532
  // a dangling/absent agentType falls back to no persona + FALLBACK_MODEL
5511
5533
  // and is logged once by resolvePrdPersonaForSpawn itself.
5512
5534
  const personaResolution = await agentModelResolve.resolvePrdPersonaForSpawn({ cwd, agentType: job.agentType });
5513
- safeLog(`[scheduler] agentType=${job.agentType || '(none)'} persona=${personaResolution.personaPath || '(fallback — no persona applied)'} model=${personaResolution.model}\n`);
5535
+ const personaEffort = resolveEpicEffort({ cwd, agentType: job.agentType }).effort;
5536
+ safeLog(`[scheduler] agentType=${job.agentType || '(none)'} persona=${personaResolution.personaPath || '(fallback — no persona applied)'} model=${personaResolution.model}${personaEffort ? ` effort=${personaEffort}` : ''}\n`);
5514
5537
 
5515
5538
  return await new Promise((resolve) => {
5516
5539
  const claudeBin = resolveClaudeBin();
@@ -5710,6 +5733,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
5710
5733
  args: buildClaudeSpawnArgs({
5711
5734
  prompt,
5712
5735
  model: personaResolution.model,
5736
+ effort: personaEffort,
5713
5737
  sessionId,
5714
5738
  resume: !!resumeTarget,
5715
5739
  systemPrompt: personaResolution.systemPrompt,
@@ -6588,6 +6612,117 @@ async function handleLaunchFailure({ job, res, runId, runDir, launchKey, launchE
6588
6612
  await broadcast({ flush: true });
6589
6613
  }
6590
6614
 
6615
+ // Scheduler-scoped error sink for failures that would otherwise be invisible
6616
+ // in packaged/npx builds (stdout unread). Never throws.
6617
+ function reportSchedulerError(message, slug, e) {
6618
+ try {
6619
+ logs.writeLine({
6620
+ scope: 'scheduler',
6621
+ level: 'error',
6622
+ message,
6623
+ meta: { slug, error: e?.message || String(e), stack: e?.stack },
6624
+ });
6625
+ } catch { /* logging must never be the thing that fails */ }
6626
+ try {
6627
+ appendAuditEvent('scheduler_error', { slug, message, error: e?.message || String(e), stack: e?.stack });
6628
+ } catch { /* same */ }
6629
+ }
6630
+
6631
+ /**
6632
+ * Salvage, integrate and clean up a job's throwaway worktree once its run has
6633
+ * ended. NEVER throws: any rejection (salvage / integrate / cleanup) is
6634
+ * reported through deps.reportSchedulerError and surfaces as
6635
+ * `worktreeIntegrationFailure`, so spawnJob's finalize mutate always runs and
6636
+ * the job can never be left `running`. The branch is kept on every failure.
6637
+ * @returns {Promise<{worktreeLeftoverDirty: string[], salvagePatch: string|null,
6638
+ * worktreeIntegrationFailure: string|null, worktreeIntegrationDetail: object|null,
6639
+ * mergeAutoResolved: string|null, mergeAutoResolvedPaths: string[]|null}>}
6640
+ */
6641
+ async function finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPaths, deps = {} }) {
6642
+ const jw = deps.jobWorktree || jobWorktree;
6643
+ const uncommitted = deps.uncommittedChanges || uncommittedChanges;
6644
+ const report = deps.reportSchedulerError || reportSchedulerError;
6645
+ let worktreeLeftoverDirty = [];
6646
+ let salvagePatch = null;
6647
+ let worktreeIntegrationFailure = null;
6648
+ let worktreeIntegrationDetail = null;
6649
+ let mergeAutoResolved = null;
6650
+ let mergeAutoResolvedPaths = null;
6651
+ try {
6652
+ worktreeLeftoverDirty = (await uncommitted(worktree.dir)) || [];
6653
+ // Salvage the worktree's full diff (tracked + untracked) to the run
6654
+ // dir BEFORE the checkout is removed below — otherwise a job killed
6655
+ // before its finish-protocol commit loses that work outright, with
6656
+ // no branch, no stash, no patch anywhere. Best-effort: never blocks
6657
+ // integration/cleanup and never changes the job's verdict.
6658
+ if (worktreeLeftoverDirty.length) {
6659
+ const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
6660
+ const salvage = await jw.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
6661
+ if (salvage && salvage.ok) {
6662
+ salvagePatch = salvagePath;
6663
+ console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
6664
+ }
6665
+ }
6666
+ const integration = await jw.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
6667
+ if (integration.ok && integration.reason === 'carried-wip-only') {
6668
+ console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
6669
+ }
6670
+ if (!integration.ok) {
6671
+ worktreeIntegrationFailure = integration.reason;
6672
+ worktreeIntegrationDetail = integration;
6673
+ console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
6674
+ } else if (integration.integrated) {
6675
+ console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
6676
+ if (integration.autoResolved) {
6677
+ mergeAutoResolved = integration.autoResolved;
6678
+ mergeAutoResolvedPaths = integration.resolvedPaths || [];
6679
+ console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
6680
+ }
6681
+ }
6682
+ await jw.cleanupJobWorktree({
6683
+ cwd: guardCwd,
6684
+ dir: worktree.dir,
6685
+ branch: worktree.branch,
6686
+ keepBranch: !integration.ok,
6687
+ });
6688
+ } catch (e) {
6689
+ worktreeIntegrationFailure = e?.message || String(e);
6690
+ report('spawnJob worktree finalize failed', job.slug, e);
6691
+ // Best-effort: release the checkout + worktree-cap slot, keep the branch.
6692
+ try {
6693
+ await jw.cleanupJobWorktree({ cwd: guardCwd, dir: worktree.dir, branch: worktree.branch, keepBranch: true });
6694
+ } catch { /* already reported above */ }
6695
+ }
6696
+ return { worktreeLeftoverDirty, salvagePatch, worktreeIntegrationFailure, worktreeIntegrationDetail, mergeAutoResolved, mergeAutoResolvedPaths };
6697
+ }
6698
+
6699
+ /**
6700
+ * Map a worktree integration failure onto the verifier verdict spawnJob stamps
6701
+ * (pure). Null failure -> null (no override). Always downgrades to needs_review.
6702
+ */
6703
+ function worktreeIntegrationVerdict({ failure, detail, slug }) {
6704
+ if (!failure) return null;
6705
+ return {
6706
+ verdict: 'worktree_integration_failed',
6707
+ reason: detail && detail.failureKind === 'content_conflict'
6708
+ ? `Integration blocked by a content conflict in ${(detail.conflictedPaths || []).join(', ') || 'unknown paths'} — branch ${jobWorktree.branchNameFor(slug)} preserved; needs a manual merge.`
6709
+ : `worktree branch integration failed: ${failure} — branch preserved for manual merge`,
6710
+ downgradeTo: 'needs_review',
6711
+ };
6712
+ }
6713
+
6714
+ /**
6715
+ * Run one interval tick; a throw is reported and swallowed so the interval
6716
+ * keeps firing.
6717
+ */
6718
+ function guardedTick(fn, label) {
6719
+ try {
6720
+ fn();
6721
+ } catch (e) {
6722
+ reportSchedulerError(label, null, e);
6723
+ }
6724
+ }
6725
+
6591
6726
  async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6592
6727
  // Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
6593
6728
  // — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
@@ -6963,42 +7098,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6963
7098
  });
6964
7099
  } finally {
6965
7100
  if (worktree.ok) {
6966
- worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
6967
- // Salvage the worktree's full diff (tracked + untracked) to the run
6968
- // dir BEFORE the checkout is removed below — otherwise a job killed
6969
- // before its finish-protocol commit loses that work outright, with
6970
- // no branch, no stash, no patch anywhere. Best-effort: never blocks
6971
- // integration/cleanup and never changes the job's verdict.
6972
- if (worktreeLeftoverDirty.length) {
6973
- const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
6974
- const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
6975
- if (salvage && salvage.ok) {
6976
- salvagePatch = salvagePath;
6977
- console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
6978
- }
6979
- }
6980
- const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
6981
- if (integration.ok && integration.reason === 'carried-wip-only') {
6982
- console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
6983
- }
6984
- if (!integration.ok) {
6985
- worktreeIntegrationFailure = integration.reason;
6986
- worktreeIntegrationDetail = integration;
6987
- console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
6988
- } else if (integration.integrated) {
6989
- console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
6990
- if (integration.autoResolved) {
6991
- mergeAutoResolved = integration.autoResolved;
6992
- mergeAutoResolvedPaths = integration.resolvedPaths || [];
6993
- console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
6994
- }
6995
- }
6996
- await jobWorktree.cleanupJobWorktree({
6997
- cwd: guardCwd,
6998
- dir: worktree.dir,
6999
- branch: worktree.branch,
7000
- keepBranch: !integration.ok,
7001
- });
7101
+ ({ worktreeLeftoverDirty, salvagePatch, worktreeIntegrationFailure, worktreeIntegrationDetail, mergeAutoResolved, mergeAutoResolvedPaths } =
7102
+ await finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPaths }));
7002
7103
  } else {
7003
7104
  // In-place run (non-git cwd, cap reached, env-disabled, or a carry-over
7004
7105
  // failure) — there is no throwaway checkout to diff, so salvage only
@@ -7271,13 +7372,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
7271
7372
  // guard AC explicitly requires this failure be surfaced as an explicit job
7272
7373
  // outcome, never silently dropped alongside the branch it's stranded on.
7273
7374
  if (worktreeIntegrationFailure) {
7274
- verifyResult = {
7275
- verdict: 'worktree_integration_failed',
7276
- reason: worktreeIntegrationDetail && worktreeIntegrationDetail.failureKind === 'content_conflict'
7277
- ? `Integration blocked by a content conflict in ${(worktreeIntegrationDetail.conflictedPaths || []).join(', ') || 'unknown paths'} — branch ${jobWorktree.branchNameFor(job.slug)} preserved; needs a manual merge.`
7278
- : `worktree branch integration failed: ${worktreeIntegrationFailure} — branch preserved for manual merge`,
7279
- downgradeTo: 'needs_review',
7280
- };
7375
+ verifyResult = worktreeIntegrationVerdict({ failure: worktreeIntegrationFailure, detail: worktreeIntegrationDetail, slug: job.slug });
7281
7376
  }
7282
7377
 
7283
7378
  // Shared-tree stash guard (incident 2026-09-01): only meaningful for an
@@ -7935,6 +8030,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
7935
8030
  }
7936
8031
  } catch (e) {
7937
8032
  console.error('[scheduler] spawnJob error', job.slug, e);
8033
+ reportSchedulerError('spawnJob error', job.slug, e);
7938
8034
  } finally {
7939
8035
  runningSet.delete(job.slug);
7940
8036
  // Slot release notifies subscribed pumps (chat lane) machine-wide.
@@ -8294,7 +8390,7 @@ async function tickBody(gen, { bypassLoadGate }) {
8294
8390
  for (const job of gatedBatch) {
8295
8391
  if (cancelToken.cancelled || stale()) break;
8296
8392
  // spawnJob is fire-and-forget; it calls tickQueue() on completion.
8297
- spawnJob(job, runId, runDir, state.config.defaultCwd).catch(() => {});
8393
+ spawnJob(job, runId, runDir, state.config.defaultCwd).catch((e) => reportSchedulerError('spawnJob dispatch rejected', job.slug, e));
8298
8394
  }
8299
8395
  return recordTick({ fired: true, count: gatedBatch.length, group: gatedBatch[0]?.parallelGroup }, { holds });
8300
8396
  }
@@ -11418,6 +11514,223 @@ function stop() {
11418
11514
  stopDispatchLoop();
11419
11515
  }
11420
11516
 
11517
+ // Body of the 10-minute maintenance interval (self-heal, escalations, restores).
11518
+ // Extracted so a throw is testable through guardedTick.
11519
+ function rescheduleIntervalTick() {
11520
+ rescheduleTimer().catch(() => {});
11521
+ const s = readQueueSync();
11522
+ // Periodic self-heal: re-run the verifier over stale needs_review jobs so a
11523
+ // job whose work actually landed (committed in-window, no FAIL sentinel)
11524
+ // auto-clears WITHOUT waiting for the next app restart. Cheap-guarded by
11525
+ // shouldRunPeriodicReverify, which reuses isRescanCandidate so the guard
11526
+ // and the candidate filter can never drift apart again (they did once —
11527
+ // see that function's comment). Kill-switch:
11528
+ // SM_REVERIFY_PERIODIC_DISABLE=1 (boot reverify above stays always-on).
11529
+ // reverifyNeedsReview's auto-fix loop is capped downstream by
11530
+ // MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
11531
+ // past it), so this interval firing cannot fan out investigations.
11532
+ if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
11533
+ if (shouldRunPeriodicReverify(s.jobs)) {
11534
+ reverifyNeedsReview().catch(() => {});
11535
+ }
11536
+ // A quarantined row only ever promotes to 'pending' through
11537
+ // reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
11538
+ // it re-checks the PRD file's createdVia stamp every pass. broadcast()
11539
+ // already runs reconcile+writeQueue on every normal poll tick, but an
11540
+ // idle queue (nothing pending/running to fire) can back off that
11541
+ // cadence for a long time; this guarantees an adopted-but-still-
11542
+ // quarantined row is re-checked within 10 minutes regardless.
11543
+ if (s.jobs.some((j) => j.status === 'quarantined')) {
11544
+ broadcast().catch(() => {});
11545
+ }
11546
+ }
11547
+ // Age-based escalation (independent of the self-heal kill-switch above —
11548
+ // this is a monitoring signal, not an auto-fix action): a quarantined
11549
+ // row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
11550
+ // warn-logged by project + slug + age so it cannot sit stranded and
11551
+ // silent (the four burrow-project rows this PRD was written against).
11552
+ for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
11553
+ console.warn(
11554
+ `[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
11555
+ + `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
11556
+ + `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
11557
+ );
11558
+ appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
11559
+ }
11560
+
11561
+ // Estimate-relative overrun escalation. Sits in the blind spot between
11562
+ // the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
11563
+ // producing output while looping trips neither, so nothing noticed a PRD
11564
+ // running 9x its own estimate until a human went looking. Escalate loudly;
11565
+ // never kill on an estimate (see JOB_OVERRUN_FACTOR).
11566
+ for (const over of findOverrunningJobs(s.jobs, Date.now())) {
11567
+ console.warn(
11568
+ `[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
11569
+ + `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
11570
+ + `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
11571
+ + `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
11572
+ + `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
11573
+ );
11574
+ appendAuditEvent('job_overrunning_estimate', {
11575
+ slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
11576
+ });
11577
+ // Durable stamp so schedule:state (and therefore the renderer) can see
11578
+ // this without re-deriving it — the console.warn/audit event above are
11579
+ // visible only in the log, never on the row itself. Display-only
11580
+ // advisory field; re-stamped in place every sweep, never appended.
11581
+ mutate((state) => {
11582
+ const j = state.jobs.find((x) => x.slug === over.slug);
11583
+ if (!j) return;
11584
+ j.overrun = {
11585
+ ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
11586
+ };
11587
+ }).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
11588
+ }
11589
+
11590
+ // Stranded-investigation restore. Unlike the two escalations above, this
11591
+ // one ACTS: 'investigating' is a transient status whose restore
11592
+ // (spawnInvestigation's onExit/catch) only runs inside the process that
11593
+ // spawned the probe, so an app restart mid-probe leaves the row frozen
11594
+ // there forever (see findStrandedInvestigations' header, and the
11595
+ // "'investigating' must never be the job's resting state" comment at
11596
+ // spawnInvestigation's onExit). This restores each stranded row to the
11597
+ // exact terminal status it already carried before the probe was
11598
+ // spawned — it never re-runs or re-investigates anything.
11599
+ const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
11600
+ if (stranded.length > 0) {
11601
+ mutate((ms) => {
11602
+ for (const st of stranded) {
11603
+ const j = ms.jobs.find((x) => x.slug === st.slug);
11604
+ if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
11605
+ transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
11606
+ delete j.runtime;
11607
+ console.warn(
11608
+ `[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
11609
+ + `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
11610
+ + `restored to '${st.restoreStatus}'`,
11611
+ );
11612
+ appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
11613
+ }
11614
+ })
11615
+ .then(() => broadcast({ flush: true }))
11616
+ .catch(() => {});
11617
+ }
11618
+
11619
+ // Per-project starvation (PRD 1087): a project with pending work that has
11620
+ // been passed over on every tick while OTHER projects dispatch. Nothing
11621
+ // else distinguishes "no pending work" from "pending work, never
11622
+ // started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
11623
+ // Escalation only, same shape as the quarantine/overrun warnings above.
11624
+ const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
11625
+ for (const sp of starvedProjects) {
11626
+ console.warn(
11627
+ `[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
11628
+ + `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
11629
+ + `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
11630
+ );
11631
+ appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
11632
+ }
11633
+ // Bounded, automated consequence for a starve that outlives the WARN
11634
+ // above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
11635
+ // project_starved rows and zero consequence). STARVE_ESCALATION_MS is
11636
+ // strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
11637
+ // a subset of the rows already reported above — same verdict, no
11638
+ // re-derivation.
11639
+ runStarveEscalationSweep(starvedProjects);
11640
+
11641
+ // Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
11642
+ // escalation now narrowed to only the rows that auto-reset gave up on.
11643
+ // See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
11644
+ // Computed together, acted on in the SAME mutate(...) pass, so the
11645
+ // stuckFailedNotified race guard below and the auto-reset race guard
11646
+ // above it can never observe two different snapshots of the same row.
11647
+ // Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
11648
+ const autoResetTargets = failedAutoResetDisabled()
11649
+ ? []
11650
+ : selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
11651
+ const stuckFailed = stuckFailedEscalationDisabled()
11652
+ ? []
11653
+ : findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
11654
+ // Bounded automatic terminal decision for exhausted needs_review rows
11655
+ // (this PRD): computed alongside the failed-row passes above and acted
11656
+ // on in the SAME mutate(...) pass below, for the same race-guard reason
11657
+ // — a row's exhaustedResolveAttempts counter must never be read from one
11658
+ // snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
11659
+ const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
11660
+ ? []
11661
+ : selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
11662
+ // Bounded automatic exit for quarantined rows (this PRD): computed
11663
+ // alongside the passes above and acted on in the SAME mutate(...) pass
11664
+ // below, for the same race-guard reason — quarantineResolveAttempts must
11665
+ // never be read from one snapshot and written from another, and the
11666
+ // createdVia re-check inside autoResolveQuarantine must happen in the
11667
+ // same turn as the transition it gates. Kill-switch:
11668
+ // SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
11669
+ const quarantineTargets = quarantineAutoResolveDisabled()
11670
+ ? []
11671
+ : selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
11672
+ if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
11673
+ mutate(async (ms) => {
11674
+ for (const target of autoResetTargets) {
11675
+ const j = ms.jobs.find((x) => x.slug === target.slug);
11676
+ if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
11677
+ const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
11678
+ j.failedAutoResetAttempts = attempt;
11679
+ const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
11680
+ // resetJobFields is the same field-clearing list the admin
11681
+ // scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
11682
+ // it rather than inventing a second list. It also sets job.error to
11683
+ // the reason text passed in; we clear that back to null right
11684
+ // after since this is a clean auto-reset, not a recorded error.
11685
+ if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
11686
+ j.error = null;
11687
+ delete j.stuckFailedNotified;
11688
+ console.warn(
11689
+ `[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11690
+ + `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
11691
+ );
11692
+ appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
11693
+ }
11694
+ for (const stuck of stuckFailed) {
11695
+ const j = ms.jobs.find((x) => x.slug === stuck.slug);
11696
+ if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
11697
+ // Still has auto-reset attempts left — it will be (or already was,
11698
+ // earlier this same pass) picked up by the loop above instead.
11699
+ // Never log "reset it by hand" for a row that isn't actually stuck.
11700
+ if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
11701
+ j.stuckFailedNotified = true;
11702
+ console.warn(
11703
+ `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
11704
+ + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
11705
+ + `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
11706
+ );
11707
+ appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
11708
+ }
11709
+ for (const target of exhaustedNeedsReviewTargets) {
11710
+ const j = ms.jobs.find((x) => x.slug === target.slug);
11711
+ const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
11712
+ if (outcome) {
11713
+ console.warn(
11714
+ `[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11715
+ + `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
11716
+ );
11717
+ }
11718
+ }
11719
+ for (const target of quarantineTargets) {
11720
+ const j = ms.jobs.find((x) => x.slug === target.slug);
11721
+ if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
11722
+ const outcome = await autoResolveQuarantine(j, target.ageMs);
11723
+ if (outcome) {
11724
+ console.warn(
11725
+ `[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11726
+ + `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
11727
+ );
11728
+ }
11729
+ }
11730
+ }).catch(() => {});
11731
+ }
11732
+ }
11733
+
11421
11734
  async function init() {
11422
11735
  ensureDirs();
11423
11736
  // Boot phase — reconciliation, migrations, self-heal, first reset probe.
@@ -11618,218 +11931,9 @@ async function init() {
11618
11931
  // resets early or the auth token rotates. Tracked so re-init doesn't leak.
11619
11932
  if (rescheduleInterval) clearInterval(rescheduleInterval);
11620
11933
  rescheduleInterval = setInterval(() => {
11621
- rescheduleTimer().catch(() => {});
11622
- const s = readQueueSync();
11623
- // Periodic self-heal: re-run the verifier over stale needs_review jobs so a
11624
- // job whose work actually landed (committed in-window, no FAIL sentinel)
11625
- // auto-clears WITHOUT waiting for the next app restart. Cheap-guarded by
11626
- // shouldRunPeriodicReverify, which reuses isRescanCandidate so the guard
11627
- // and the candidate filter can never drift apart again (they did once —
11628
- // see that function's comment). Kill-switch:
11629
- // SM_REVERIFY_PERIODIC_DISABLE=1 (boot reverify above stays always-on).
11630
- // reverifyNeedsReview's auto-fix loop is capped downstream by
11631
- // MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
11632
- // past it), so this interval firing cannot fan out investigations.
11633
- if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
11634
- if (shouldRunPeriodicReverify(s.jobs)) {
11635
- reverifyNeedsReview().catch(() => {});
11636
- }
11637
- // A quarantined row only ever promotes to 'pending' through
11638
- // reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
11639
- // it re-checks the PRD file's createdVia stamp every pass. broadcast()
11640
- // already runs reconcile+writeQueue on every normal poll tick, but an
11641
- // idle queue (nothing pending/running to fire) can back off that
11642
- // cadence for a long time; this guarantees an adopted-but-still-
11643
- // quarantined row is re-checked within 10 minutes regardless.
11644
- if (s.jobs.some((j) => j.status === 'quarantined')) {
11645
- broadcast().catch(() => {});
11646
- }
11647
- }
11648
- // Age-based escalation (independent of the self-heal kill-switch above —
11649
- // this is a monitoring signal, not an auto-fix action): a quarantined
11650
- // row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
11651
- // warn-logged by project + slug + age so it cannot sit stranded and
11652
- // silent (the four burrow-project rows this PRD was written against).
11653
- for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
11654
- console.warn(
11655
- `[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
11656
- + `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
11657
- + `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
11658
- );
11659
- appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
11660
- }
11661
-
11662
- // Estimate-relative overrun escalation. Sits in the blind spot between
11663
- // the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
11664
- // producing output while looping trips neither, so nothing noticed a PRD
11665
- // running 9x its own estimate until a human went looking. Escalate loudly;
11666
- // never kill on an estimate (see JOB_OVERRUN_FACTOR).
11667
- for (const over of findOverrunningJobs(s.jobs, Date.now())) {
11668
- console.warn(
11669
- `[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
11670
- + `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
11671
- + `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
11672
- + `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
11673
- + `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
11674
- );
11675
- appendAuditEvent('job_overrunning_estimate', {
11676
- slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
11677
- });
11678
- // Durable stamp so schedule:state (and therefore the renderer) can see
11679
- // this without re-deriving it — the console.warn/audit event above are
11680
- // visible only in the log, never on the row itself. Display-only
11681
- // advisory field; re-stamped in place every sweep, never appended.
11682
- mutate((state) => {
11683
- const j = state.jobs.find((x) => x.slug === over.slug);
11684
- if (!j) return;
11685
- j.overrun = {
11686
- ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
11687
- };
11688
- }).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
11689
- }
11690
-
11691
- // Stranded-investigation restore. Unlike the two escalations above, this
11692
- // one ACTS: 'investigating' is a transient status whose restore
11693
- // (spawnInvestigation's onExit/catch) only runs inside the process that
11694
- // spawned the probe, so an app restart mid-probe leaves the row frozen
11695
- // there forever (see findStrandedInvestigations' header, and the
11696
- // "'investigating' must never be the job's resting state" comment at
11697
- // spawnInvestigation's onExit). This restores each stranded row to the
11698
- // exact terminal status it already carried before the probe was
11699
- // spawned — it never re-runs or re-investigates anything.
11700
- const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
11701
- if (stranded.length > 0) {
11702
- mutate((ms) => {
11703
- for (const st of stranded) {
11704
- const j = ms.jobs.find((x) => x.slug === st.slug);
11705
- if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
11706
- transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
11707
- delete j.runtime;
11708
- console.warn(
11709
- `[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
11710
- + `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
11711
- + `restored to '${st.restoreStatus}'`,
11712
- );
11713
- appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
11714
- }
11715
- })
11716
- .then(() => broadcast({ flush: true }))
11717
- .catch(() => {});
11718
- }
11719
-
11720
- // Per-project starvation (PRD 1087): a project with pending work that has
11721
- // been passed over on every tick while OTHER projects dispatch. Nothing
11722
- // else distinguishes "no pending work" from "pending work, never
11723
- // started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
11724
- // Escalation only, same shape as the quarantine/overrun warnings above.
11725
- const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
11726
- for (const sp of starvedProjects) {
11727
- console.warn(
11728
- `[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
11729
- + `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
11730
- + `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
11731
- );
11732
- appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
11733
- }
11734
- // Bounded, automated consequence for a starve that outlives the WARN
11735
- // above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
11736
- // project_starved rows and zero consequence). STARVE_ESCALATION_MS is
11737
- // strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
11738
- // a subset of the rows already reported above — same verdict, no
11739
- // re-derivation.
11740
- runStarveEscalationSweep(starvedProjects);
11741
-
11742
- // Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
11743
- // escalation now narrowed to only the rows that auto-reset gave up on.
11744
- // See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
11745
- // Computed together, acted on in the SAME mutate(...) pass, so the
11746
- // stuckFailedNotified race guard below and the auto-reset race guard
11747
- // above it can never observe two different snapshots of the same row.
11748
- // Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
11749
- const autoResetTargets = failedAutoResetDisabled()
11750
- ? []
11751
- : selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
11752
- const stuckFailed = stuckFailedEscalationDisabled()
11753
- ? []
11754
- : findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
11755
- // Bounded automatic terminal decision for exhausted needs_review rows
11756
- // (this PRD): computed alongside the failed-row passes above and acted
11757
- // on in the SAME mutate(...) pass below, for the same race-guard reason
11758
- // — a row's exhaustedResolveAttempts counter must never be read from one
11759
- // snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
11760
- const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
11761
- ? []
11762
- : selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
11763
- // Bounded automatic exit for quarantined rows (this PRD): computed
11764
- // alongside the passes above and acted on in the SAME mutate(...) pass
11765
- // below, for the same race-guard reason — quarantineResolveAttempts must
11766
- // never be read from one snapshot and written from another, and the
11767
- // createdVia re-check inside autoResolveQuarantine must happen in the
11768
- // same turn as the transition it gates. Kill-switch:
11769
- // SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
11770
- const quarantineTargets = quarantineAutoResolveDisabled()
11771
- ? []
11772
- : selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
11773
- if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
11774
- mutate(async (ms) => {
11775
- for (const target of autoResetTargets) {
11776
- const j = ms.jobs.find((x) => x.slug === target.slug);
11777
- if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
11778
- const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
11779
- j.failedAutoResetAttempts = attempt;
11780
- const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
11781
- // resetJobFields is the same field-clearing list the admin
11782
- // scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
11783
- // it rather than inventing a second list. It also sets job.error to
11784
- // the reason text passed in; we clear that back to null right
11785
- // after since this is a clean auto-reset, not a recorded error.
11786
- if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
11787
- j.error = null;
11788
- delete j.stuckFailedNotified;
11789
- console.warn(
11790
- `[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11791
- + `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
11792
- );
11793
- appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
11794
- }
11795
- for (const stuck of stuckFailed) {
11796
- const j = ms.jobs.find((x) => x.slug === stuck.slug);
11797
- if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
11798
- // Still has auto-reset attempts left — it will be (or already was,
11799
- // earlier this same pass) picked up by the loop above instead.
11800
- // Never log "reset it by hand" for a row that isn't actually stuck.
11801
- if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
11802
- j.stuckFailedNotified = true;
11803
- console.warn(
11804
- `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
11805
- + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
11806
- + `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
11807
- );
11808
- appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
11809
- }
11810
- for (const target of exhaustedNeedsReviewTargets) {
11811
- const j = ms.jobs.find((x) => x.slug === target.slug);
11812
- const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
11813
- if (outcome) {
11814
- console.warn(
11815
- `[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11816
- + `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
11817
- );
11818
- }
11819
- }
11820
- for (const target of quarantineTargets) {
11821
- const j = ms.jobs.find((x) => x.slug === target.slug);
11822
- if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
11823
- const outcome = await autoResolveQuarantine(j, target.ageMs);
11824
- if (outcome) {
11825
- console.warn(
11826
- `[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11827
- + `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
11828
- );
11829
- }
11830
- }
11831
- }).catch(() => {});
11832
- }
11934
+ // One throwing tick (e.g. readQueueSync on a torn queue.json) must skip
11935
+ // only itself — the interval keeps firing and the failure is logged.
11936
+ guardedTick(rescheduleIntervalTick, 'rescheduleInterval tick failed');
11833
11937
  }, REVERIFY_INTERVAL_MS);
11834
11938
 
11835
11939
  // Self-rescheduling poll loop with exponential backoff. Replaces the
@@ -11946,6 +12050,7 @@ async function listPrdsInternal() {
11946
12050
  dependsOn: parsed.dependsOn ?? null,
11947
12051
  agentType: parsed.agentType ?? null,
11948
12052
  disposition: parsed.disposition ?? null,
12053
+ planId: parsed.planId ?? null,
11949
12054
  mtimeMs: stat.mtimeMs,
11950
12055
  archived,
11951
12056
  };
@@ -12379,7 +12484,14 @@ const remote = {
12379
12484
  const rows = listing.prds ?? [];
12380
12485
  const rewrite = computeDispositionRewrite({ slug, disposition, dependsOn: dependsOn ?? [], rows });
12381
12486
  if (!rewrite.ok) return rewrite;
12382
- return this.updatePrd({ slug, cwd, frontmatter: { dependsOn: rewrite.dependsOn, disposition } });
12487
+ // Keep the durable planId in step with the new relationship: a promoted head
12488
+ // starts its own plan; a re-attached row joins the target chain's plan (cleared
12489
+ // when the target predates the stamp, so the derivation fallback applies).
12490
+ let planId = mintPlanId();
12491
+ if (disposition === 'append') {
12492
+ planId = resolveInheritedPlanId(rewrite.dependsOn, rows).planId;
12493
+ }
12494
+ return this.updatePrd({ slug, cwd, frontmatter: { dependsOn: rewrite.dependsOn, disposition, planId } });
12383
12495
  },
12384
12496
 
12385
12497
  // Cancels a job that hasn't finished yet. A 'running' job's process group
@@ -12493,6 +12605,11 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
12493
12605
  }
12494
12606
 
12495
12607
  module.exports = {
12608
+ reportSchedulerError,
12609
+ finalizeJobWorktree,
12610
+ worktreeIntegrationVerdict,
12611
+ guardedTick,
12612
+ rescheduleIntervalTick,
12496
12613
  classifyQueueStarvation,
12497
12614
  classifyQueueStarvationByProject,
12498
12615
  dispatchIdleMs,
@@ -12526,6 +12643,8 @@ module.exports = {
12526
12643
  nextBackoffMs,
12527
12644
  shouldWarnFailureStreak,
12528
12645
  shouldEscalateFailureStreak,
12646
+ restoreFailureStreakWarnedAt,
12647
+ persistedStreakMinutes,
12529
12648
  computeDegradedBudget,
12530
12649
  healRefusalReason,
12531
12650
  writeQueue,