claude-code-session-manager 0.93.0 → 0.94.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/dist/assets/{DataModel-DDGx3qrn.js → DataModel-B0LDnnSL.js} +1 -1
  2. package/dist/assets/{History-CHBjYZyD.js → History-C685Kytt.js} +2 -2
  3. package/dist/assets/{Hooks-DG4c-Rzz.js → Hooks-CwLnp0Z_.js} +3 -3
  4. package/dist/assets/HostBilko-CwEHEKYk.js +1 -0
  5. package/dist/assets/{Library-CN1h5rAE.js → Library-BU9np05E.js} +25 -25
  6. package/dist/assets/{MarkdownEditor-DmBxrhdn.js → MarkdownEditor-B74ogK2f.js} +1 -1
  7. package/dist/assets/McpServers-BLM3ftzP.js +2 -0
  8. package/dist/assets/{Memory-Dk4uDCjg.js → Memory-DAP1t8bA.js} +6 -6
  9. package/dist/assets/{Permissions-BovobeZw.js → Permissions-3pGLJmxQ.js} +3 -3
  10. package/dist/assets/{Plugins-DvOf_B62.js → Plugins-Q1KoLEzn.js} +2 -2
  11. package/dist/assets/{ProvenanceBadge-Dutz_4OY.js → ProvenanceBadge-BnjBQiV9.js} +1 -1
  12. package/dist/assets/{SaveBar-CT5u6ZCJ.js → SaveBar-Zq2NfcPf.js} +1 -1
  13. package/dist/assets/{Scheduler-B7GSnZj5.js → Scheduler-D488ebKm.js} +9 -9
  14. package/dist/assets/{ScopeSwitcher-BUYDpz2v.js → ScopeSwitcher-Cnojv22D.js} +1 -1
  15. package/dist/assets/{Settings-DIRmr3tL.js → Settings-D5Lhj7ga.js} +3 -3
  16. package/dist/assets/{SkillReferenceGraph-CdGr6d3_.js → SkillReferenceGraph-BRDiE6Hv.js} +2 -2
  17. package/dist/assets/{Skills-BQOheQv4.js → Skills-BQDqp-EN.js} +2 -2
  18. package/dist/assets/{SystemPrompt-5LsimKbX.js → SystemPrompt-tFr19Od6.js} +1 -1
  19. package/dist/assets/{TagLibrary-DreLsT3n.js → TagLibrary-BjyrnaRE.js} +1 -1
  20. package/dist/assets/{TiptapBody-C9kmlBnI.js → TiptapBody-DGzSg3BO.js} +1 -1
  21. package/dist/assets/{Toggle-BKLCBu9A.js → Toggle-DzoROibf.js} +1 -1
  22. package/dist/assets/{index-Db_SK9rW.js → index-CCl4tz-u.js} +821 -821
  23. package/dist/assets/{settingsSchema-B5jBhSUP.js → settingsSchema-BYKVe1WM.js} +1 -1
  24. package/dist/index.html +1 -1
  25. package/package.json +6 -3
  26. package/scripts/README.md +3 -0
  27. package/src/main/__tests__/chat-cancel-terminal.test.cjs +5 -2
  28. package/src/main/__tests__/chat-exit-close-race.test.cjs +8 -2
  29. package/src/main/__tests__/chat-mcp-consent-notice.test.cjs +5 -2
  30. package/src/main/__tests__/intradayRefresh.test.cjs +39 -0
  31. package/src/main/__tests__/openExternalApp-spawn-error.test.cjs +25 -0
  32. package/src/main/__tests__/opsErrorLog.test.cjs +22 -0
  33. package/src/main/__tests__/pty-epic-worktree-spawn-cwd.test.cjs +21 -0
  34. package/src/main/__tests__/rateLimitPollerStreak.test.cjs +14 -0
  35. package/src/main/__tests__/runVerify-atomic-verdicts.test.cjs +26 -0
  36. package/src/main/__tests__/runVerify-landed-commit-outranks.test.cjs +11 -1
  37. package/src/main/__tests__/runVerify-transcript-commit-evidence.test.cjs +11 -1
  38. package/src/main/__tests__/runVerify.test.cjs +12 -1
  39. package/src/main/__tests__/transcriptsUsageFor.test.cjs +112 -1
  40. package/src/main/build-info.json +4 -4
  41. package/src/main/chatRunner.cjs +45 -10
  42. package/src/main/git.cjs +9 -2
  43. package/src/main/historyAggregator.cjs +2 -19
  44. package/src/main/index.cjs +7 -8
  45. package/src/main/ipcSchemas.cjs +13 -3
  46. package/src/main/lib/__tests__/childWithLog.test.cjs +180 -0
  47. package/src/main/lib/__tests__/delegationReadiness.test.cjs +19 -0
  48. package/src/main/lib/__tests__/gitCacheBound.test.cjs +69 -0
  49. package/src/main/lib/agentPersonaSchema.cjs +2 -0
  50. package/src/main/lib/childWithLog.cjs +92 -51
  51. package/src/main/lib/delegationReadiness.cjs +3 -1
  52. package/src/main/lib/epicSpawnPlan.cjs +27 -8
  53. package/src/main/lib/epicTranscriptPath.cjs +5 -1
  54. package/src/main/lib/headTailBuffer.cjs +43 -0
  55. package/src/main/lib/intradayRefresh.cjs +33 -0
  56. package/src/main/lib/lruCache.cjs +39 -0
  57. package/src/main/lib/openExternalApp.cjs +27 -9
  58. package/src/main/lib/opsErrorLog.cjs +22 -0
  59. package/src/main/lib/promptSessionSchema.cjs +2 -0
  60. package/src/main/lib/rendererRecovery.cjs +141 -0
  61. package/src/main/lib/scheduleJobSchema.cjs +3 -0
  62. package/src/main/runVerify.cjs +3 -1
  63. package/src/main/scheduler.cjs +360 -257
  64. package/src/main/transcripts.cjs +54 -11
  65. package/src/preload/api.d.ts +3 -1
  66. package/dist/assets/HostBilko-DHXcpV2c.js +0 -1
  67. package/dist/assets/McpServers-Beb_HD6I.js +0 -2
@@ -1507,6 +1507,7 @@ function loadSchedulerState() {
1507
1507
  if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
1508
1508
  if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
1509
1509
  if (typeof s.failureStreakWarnedAt === 'number') failureStreakWarnedAt = s.failureStreakWarnedAt;
1510
+ failureStreakWarnedAt = restoreFailureStreakWarnedAt(failureStreakWarned, failureStreakWarnedAt, Date.now());
1510
1511
  if (typeof s.lastEscalationAt === 'number') lastEscalationAtMs = s.lastEscalationAt;
1511
1512
  } catch { /* first boot or corrupt — start fresh */ }
1512
1513
  }
@@ -3397,6 +3398,21 @@ function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold =
3397
3398
  return consecutiveFailures >= threshold && !alreadyWarned;
3398
3399
  }
3399
3400
 
3401
+ /**
3402
+ * Pure: `failureStreakWarned === true` must always carry a numeric
3403
+ * `failureStreakWarnedAt` (a state file may hold one without the other), so
3404
+ * the escalation message never renders "after nullm". Backfills `nowMs`.
3405
+ */
3406
+ function restoreFailureStreakWarnedAt(warned, warnedAt, nowMs) {
3407
+ if (!warned) return typeof warnedAt === 'number' ? warnedAt : null;
3408
+ return typeof warnedAt === 'number' ? warnedAt : nowMs;
3409
+ }
3410
+
3411
+ /** Pure: whole minutes a warned streak has persisted; never null/NaN. */
3412
+ function persistedStreakMinutes(warnedAt, nowMs) {
3413
+ return typeof warnedAt === 'number' ? Math.round((nowMs - warnedAt) / 60_000) : 0;
3414
+ }
3415
+
3400
3416
  /**
3401
3417
  * Pure: does a PERSISTING failure streak warrant another escalation (audit
3402
3418
  * event + opsErrorLog line) at `nowMs`? Exported for unit testing. Only
@@ -3438,7 +3454,7 @@ function warnFailureStreakIfNeeded() {
3438
3454
  }
3439
3455
  if (failureStreakWarned && shouldEscalateFailureStreak(consecutiveFailures, lastEscalationAtMs, nowMs)) {
3440
3456
  lastEscalationAtMs = nowMs;
3441
- const persistedMinutes = failureStreakWarnedAt ? Math.round((nowMs - failureStreakWarnedAt) / 60_000) : null;
3457
+ const persistedMinutes = persistedStreakMinutes(failureStreakWarnedAt, nowMs);
3442
3458
  try {
3443
3459
  appendAuditEvent('usage_poller_failure_streak_persists', { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes });
3444
3460
  appendError({
@@ -6596,6 +6612,117 @@ async function handleLaunchFailure({ job, res, runId, runDir, launchKey, launchE
6596
6612
  await broadcast({ flush: true });
6597
6613
  }
6598
6614
 
6615
+ // Scheduler-scoped error sink for failures that would otherwise be invisible
6616
+ // in packaged/npx builds (stdout unread). Never throws.
6617
+ function reportSchedulerError(message, slug, e) {
6618
+ try {
6619
+ logs.writeLine({
6620
+ scope: 'scheduler',
6621
+ level: 'error',
6622
+ message,
6623
+ meta: { slug, error: e?.message || String(e), stack: e?.stack },
6624
+ });
6625
+ } catch { /* logging must never be the thing that fails */ }
6626
+ try {
6627
+ appendAuditEvent('scheduler_error', { slug, message, error: e?.message || String(e), stack: e?.stack });
6628
+ } catch { /* same */ }
6629
+ }
6630
+
6631
+ /**
6632
+ * Salvage, integrate and clean up a job's throwaway worktree once its run has
6633
+ * ended. NEVER throws: any rejection (salvage / integrate / cleanup) is
6634
+ * reported through deps.reportSchedulerError and surfaces as
6635
+ * `worktreeIntegrationFailure`, so spawnJob's finalize mutate always runs and
6636
+ * the job can never be left `running`. The branch is kept on every failure.
6637
+ * @returns {Promise<{worktreeLeftoverDirty: string[], salvagePatch: string|null,
6638
+ * worktreeIntegrationFailure: string|null, worktreeIntegrationDetail: object|null,
6639
+ * mergeAutoResolved: string|null, mergeAutoResolvedPaths: string[]|null}>}
6640
+ */
6641
+ async function finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPaths, deps = {} }) {
6642
+ const jw = deps.jobWorktree || jobWorktree;
6643
+ const uncommitted = deps.uncommittedChanges || uncommittedChanges;
6644
+ const report = deps.reportSchedulerError || reportSchedulerError;
6645
+ let worktreeLeftoverDirty = [];
6646
+ let salvagePatch = null;
6647
+ let worktreeIntegrationFailure = null;
6648
+ let worktreeIntegrationDetail = null;
6649
+ let mergeAutoResolved = null;
6650
+ let mergeAutoResolvedPaths = null;
6651
+ try {
6652
+ worktreeLeftoverDirty = (await uncommitted(worktree.dir)) || [];
6653
+ // Salvage the worktree's full diff (tracked + untracked) to the run
6654
+ // dir BEFORE the checkout is removed below — otherwise a job killed
6655
+ // before its finish-protocol commit loses that work outright, with
6656
+ // no branch, no stash, no patch anywhere. Best-effort: never blocks
6657
+ // integration/cleanup and never changes the job's verdict.
6658
+ if (worktreeLeftoverDirty.length) {
6659
+ const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
6660
+ const salvage = await jw.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
6661
+ if (salvage && salvage.ok) {
6662
+ salvagePatch = salvagePath;
6663
+ console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
6664
+ }
6665
+ }
6666
+ const integration = await jw.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
6667
+ if (integration.ok && integration.reason === 'carried-wip-only') {
6668
+ console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
6669
+ }
6670
+ if (!integration.ok) {
6671
+ worktreeIntegrationFailure = integration.reason;
6672
+ worktreeIntegrationDetail = integration;
6673
+ console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
6674
+ } else if (integration.integrated) {
6675
+ console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
6676
+ if (integration.autoResolved) {
6677
+ mergeAutoResolved = integration.autoResolved;
6678
+ mergeAutoResolvedPaths = integration.resolvedPaths || [];
6679
+ console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
6680
+ }
6681
+ }
6682
+ await jw.cleanupJobWorktree({
6683
+ cwd: guardCwd,
6684
+ dir: worktree.dir,
6685
+ branch: worktree.branch,
6686
+ keepBranch: !integration.ok,
6687
+ });
6688
+ } catch (e) {
6689
+ worktreeIntegrationFailure = e?.message || String(e);
6690
+ report('spawnJob worktree finalize failed', job.slug, e);
6691
+ // Best-effort: release the checkout + worktree-cap slot, keep the branch.
6692
+ try {
6693
+ await jw.cleanupJobWorktree({ cwd: guardCwd, dir: worktree.dir, branch: worktree.branch, keepBranch: true });
6694
+ } catch { /* already reported above */ }
6695
+ }
6696
+ return { worktreeLeftoverDirty, salvagePatch, worktreeIntegrationFailure, worktreeIntegrationDetail, mergeAutoResolved, mergeAutoResolvedPaths };
6697
+ }
6698
+
6699
+ /**
6700
+ * Map a worktree integration failure onto the verifier verdict spawnJob stamps
6701
+ * (pure). Null failure -> null (no override). Always downgrades to needs_review.
6702
+ */
6703
+ function worktreeIntegrationVerdict({ failure, detail, slug }) {
6704
+ if (!failure) return null;
6705
+ return {
6706
+ verdict: 'worktree_integration_failed',
6707
+ reason: detail && detail.failureKind === 'content_conflict'
6708
+ ? `Integration blocked by a content conflict in ${(detail.conflictedPaths || []).join(', ') || 'unknown paths'} — branch ${jobWorktree.branchNameFor(slug)} preserved; needs a manual merge.`
6709
+ : `worktree branch integration failed: ${failure} — branch preserved for manual merge`,
6710
+ downgradeTo: 'needs_review',
6711
+ };
6712
+ }
6713
+
6714
+ /**
6715
+ * Run one interval tick; a throw is reported and swallowed so the interval
6716
+ * keeps firing.
6717
+ */
6718
+ function guardedTick(fn, label) {
6719
+ try {
6720
+ fn();
6721
+ } catch (e) {
6722
+ reportSchedulerError(label, null, e);
6723
+ }
6724
+ }
6725
+
6599
6726
  async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6600
6727
  // Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
6601
6728
  // — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
@@ -6971,42 +7098,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6971
7098
  });
6972
7099
  } finally {
6973
7100
  if (worktree.ok) {
6974
- worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
6975
- // Salvage the worktree's full diff (tracked + untracked) to the run
6976
- // dir BEFORE the checkout is removed below — otherwise a job killed
6977
- // before its finish-protocol commit loses that work outright, with
6978
- // no branch, no stash, no patch anywhere. Best-effort: never blocks
6979
- // integration/cleanup and never changes the job's verdict.
6980
- if (worktreeLeftoverDirty.length) {
6981
- const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
6982
- const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
6983
- if (salvage && salvage.ok) {
6984
- salvagePatch = salvagePath;
6985
- console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
6986
- }
6987
- }
6988
- const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
6989
- if (integration.ok && integration.reason === 'carried-wip-only') {
6990
- console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
6991
- }
6992
- if (!integration.ok) {
6993
- worktreeIntegrationFailure = integration.reason;
6994
- worktreeIntegrationDetail = integration;
6995
- console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
6996
- } else if (integration.integrated) {
6997
- console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
6998
- if (integration.autoResolved) {
6999
- mergeAutoResolved = integration.autoResolved;
7000
- mergeAutoResolvedPaths = integration.resolvedPaths || [];
7001
- console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
7002
- }
7003
- }
7004
- await jobWorktree.cleanupJobWorktree({
7005
- cwd: guardCwd,
7006
- dir: worktree.dir,
7007
- branch: worktree.branch,
7008
- keepBranch: !integration.ok,
7009
- });
7101
+ ({ worktreeLeftoverDirty, salvagePatch, worktreeIntegrationFailure, worktreeIntegrationDetail, mergeAutoResolved, mergeAutoResolvedPaths } =
7102
+ await finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPaths }));
7010
7103
  } else {
7011
7104
  // In-place run (non-git cwd, cap reached, env-disabled, or a carry-over
7012
7105
  // failure) — there is no throwaway checkout to diff, so salvage only
@@ -7279,13 +7372,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
7279
7372
  // guard AC explicitly requires this failure be surfaced as an explicit job
7280
7373
  // outcome, never silently dropped alongside the branch it's stranded on.
7281
7374
  if (worktreeIntegrationFailure) {
7282
- verifyResult = {
7283
- verdict: 'worktree_integration_failed',
7284
- reason: worktreeIntegrationDetail && worktreeIntegrationDetail.failureKind === 'content_conflict'
7285
- ? `Integration blocked by a content conflict in ${(worktreeIntegrationDetail.conflictedPaths || []).join(', ') || 'unknown paths'} — branch ${jobWorktree.branchNameFor(job.slug)} preserved; needs a manual merge.`
7286
- : `worktree branch integration failed: ${worktreeIntegrationFailure} — branch preserved for manual merge`,
7287
- downgradeTo: 'needs_review',
7288
- };
7375
+ verifyResult = worktreeIntegrationVerdict({ failure: worktreeIntegrationFailure, detail: worktreeIntegrationDetail, slug: job.slug });
7289
7376
  }
7290
7377
 
7291
7378
  // Shared-tree stash guard (incident 2026-09-01): only meaningful for an
@@ -7943,6 +8030,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
7943
8030
  }
7944
8031
  } catch (e) {
7945
8032
  console.error('[scheduler] spawnJob error', job.slug, e);
8033
+ reportSchedulerError('spawnJob error', job.slug, e);
7946
8034
  } finally {
7947
8035
  runningSet.delete(job.slug);
7948
8036
  // Slot release notifies subscribed pumps (chat lane) machine-wide.
@@ -8302,7 +8390,7 @@ async function tickBody(gen, { bypassLoadGate }) {
8302
8390
  for (const job of gatedBatch) {
8303
8391
  if (cancelToken.cancelled || stale()) break;
8304
8392
  // spawnJob is fire-and-forget; it calls tickQueue() on completion.
8305
- spawnJob(job, runId, runDir, state.config.defaultCwd).catch(() => {});
8393
+ spawnJob(job, runId, runDir, state.config.defaultCwd).catch((e) => reportSchedulerError('spawnJob dispatch rejected', job.slug, e));
8306
8394
  }
8307
8395
  return recordTick({ fired: true, count: gatedBatch.length, group: gatedBatch[0]?.parallelGroup }, { holds });
8308
8396
  }
@@ -11426,6 +11514,223 @@ function stop() {
11426
11514
  stopDispatchLoop();
11427
11515
  }
11428
11516
 
11517
+ // Body of the 10-minute maintenance interval (self-heal, escalations, restores).
11518
+ // Extracted so a throw is testable through guardedTick.
11519
+ function rescheduleIntervalTick() {
11520
+ rescheduleTimer().catch(() => {});
11521
+ const s = readQueueSync();
11522
+ // Periodic self-heal: re-run the verifier over stale needs_review jobs so a
11523
+ // job whose work actually landed (committed in-window, no FAIL sentinel)
11524
+ // auto-clears WITHOUT waiting for the next app restart. Cheap-guarded by
11525
+ // shouldRunPeriodicReverify, which reuses isRescanCandidate so the guard
11526
+ // and the candidate filter can never drift apart again (they did once —
11527
+ // see that function's comment). Kill-switch:
11528
+ // SM_REVERIFY_PERIODIC_DISABLE=1 (boot reverify above stays always-on).
11529
+ // reverifyNeedsReview's auto-fix loop is capped downstream by
11530
+ // MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
11531
+ // past it), so this interval firing cannot fan out investigations.
11532
+ if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
11533
+ if (shouldRunPeriodicReverify(s.jobs)) {
11534
+ reverifyNeedsReview().catch(() => {});
11535
+ }
11536
+ // A quarantined row only ever promotes to 'pending' through
11537
+ // reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
11538
+ // it re-checks the PRD file's createdVia stamp every pass. broadcast()
11539
+ // already runs reconcile+writeQueue on every normal poll tick, but an
11540
+ // idle queue (nothing pending/running to fire) can back off that
11541
+ // cadence for a long time; this guarantees an adopted-but-still-
11542
+ // quarantined row is re-checked within 10 minutes regardless.
11543
+ if (s.jobs.some((j) => j.status === 'quarantined')) {
11544
+ broadcast().catch(() => {});
11545
+ }
11546
+ }
11547
+ // Age-based escalation (independent of the self-heal kill-switch above —
11548
+ // this is a monitoring signal, not an auto-fix action): a quarantined
11549
+ // row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
11550
+ // warn-logged by project + slug + age so it cannot sit stranded and
11551
+ // silent (the four burrow-project rows this PRD was written against).
11552
+ for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
11553
+ console.warn(
11554
+ `[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
11555
+ + `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
11556
+ + `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
11557
+ );
11558
+ appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
11559
+ }
11560
+
11561
+ // Estimate-relative overrun escalation. Sits in the blind spot between
11562
+ // the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
11563
+ // producing output while looping trips neither, so nothing noticed a PRD
11564
+ // running 9x its own estimate until a human went looking. Escalate loudly;
11565
+ // never kill on an estimate (see JOB_OVERRUN_FACTOR).
11566
+ for (const over of findOverrunningJobs(s.jobs, Date.now())) {
11567
+ console.warn(
11568
+ `[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
11569
+ + `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
11570
+ + `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
11571
+ + `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
11572
+ + `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
11573
+ );
11574
+ appendAuditEvent('job_overrunning_estimate', {
11575
+ slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
11576
+ });
11577
+ // Durable stamp so schedule:state (and therefore the renderer) can see
11578
+ // this without re-deriving it — the console.warn/audit event above are
11579
+ // visible only in the log, never on the row itself. Display-only
11580
+ // advisory field; re-stamped in place every sweep, never appended.
11581
+ mutate((state) => {
11582
+ const j = state.jobs.find((x) => x.slug === over.slug);
11583
+ if (!j) return;
11584
+ j.overrun = {
11585
+ ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
11586
+ };
11587
+ }).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
11588
+ }
11589
+
11590
+ // Stranded-investigation restore. Unlike the two escalations above, this
11591
+ // one ACTS: 'investigating' is a transient status whose restore
11592
+ // (spawnInvestigation's onExit/catch) only runs inside the process that
11593
+ // spawned the probe, so an app restart mid-probe leaves the row frozen
11594
+ // there forever (see findStrandedInvestigations' header, and the
11595
+ // "'investigating' must never be the job's resting state" comment at
11596
+ // spawnInvestigation's onExit). This restores each stranded row to the
11597
+ // exact terminal status it already carried before the probe was
11598
+ // spawned — it never re-runs or re-investigates anything.
11599
+ const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
11600
+ if (stranded.length > 0) {
11601
+ mutate((ms) => {
11602
+ for (const st of stranded) {
11603
+ const j = ms.jobs.find((x) => x.slug === st.slug);
11604
+ if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
11605
+ transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
11606
+ delete j.runtime;
11607
+ console.warn(
11608
+ `[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
11609
+ + `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
11610
+ + `restored to '${st.restoreStatus}'`,
11611
+ );
11612
+ appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
11613
+ }
11614
+ })
11615
+ .then(() => broadcast({ flush: true }))
11616
+ .catch(() => {});
11617
+ }
11618
+
11619
+ // Per-project starvation (PRD 1087): a project with pending work that has
11620
+ // been passed over on every tick while OTHER projects dispatch. Nothing
11621
+ // else distinguishes "no pending work" from "pending work, never
11622
+ // started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
11623
+ // Escalation only, same shape as the quarantine/overrun warnings above.
11624
+ const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
11625
+ for (const sp of starvedProjects) {
11626
+ console.warn(
11627
+ `[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
11628
+ + `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
11629
+ + `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
11630
+ );
11631
+ appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
11632
+ }
11633
+ // Bounded, automated consequence for a starve that outlives the WARN
11634
+ // above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
11635
+ // project_starved rows and zero consequence). STARVE_ESCALATION_MS is
11636
+ // strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
11637
+ // a subset of the rows already reported above — same verdict, no
11638
+ // re-derivation.
11639
+ runStarveEscalationSweep(starvedProjects);
11640
+
11641
+ // Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
11642
+ // escalation now narrowed to only the rows that auto-reset gave up on.
11643
+ // See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
11644
+ // Computed together, acted on in the SAME mutate(...) pass, so the
11645
+ // stuckFailedNotified race guard below and the auto-reset race guard
11646
+ // above it can never observe two different snapshots of the same row.
11647
+ // Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
11648
+ const autoResetTargets = failedAutoResetDisabled()
11649
+ ? []
11650
+ : selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
11651
+ const stuckFailed = stuckFailedEscalationDisabled()
11652
+ ? []
11653
+ : findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
11654
+ // Bounded automatic terminal decision for exhausted needs_review rows
11655
+ // (this PRD): computed alongside the failed-row passes above and acted
11656
+ // on in the SAME mutate(...) pass below, for the same race-guard reason
11657
+ // — a row's exhaustedResolveAttempts counter must never be read from one
11658
+ // snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
11659
+ const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
11660
+ ? []
11661
+ : selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
11662
+ // Bounded automatic exit for quarantined rows (this PRD): computed
11663
+ // alongside the passes above and acted on in the SAME mutate(...) pass
11664
+ // below, for the same race-guard reason — quarantineResolveAttempts must
11665
+ // never be read from one snapshot and written from another, and the
11666
+ // createdVia re-check inside autoResolveQuarantine must happen in the
11667
+ // same turn as the transition it gates. Kill-switch:
11668
+ // SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
11669
+ const quarantineTargets = quarantineAutoResolveDisabled()
11670
+ ? []
11671
+ : selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
11672
+ if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
11673
+ mutate(async (ms) => {
11674
+ for (const target of autoResetTargets) {
11675
+ const j = ms.jobs.find((x) => x.slug === target.slug);
11676
+ if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
11677
+ const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
11678
+ j.failedAutoResetAttempts = attempt;
11679
+ const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
11680
+ // resetJobFields is the same field-clearing list the admin
11681
+ // scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
11682
+ // it rather than inventing a second list. It also sets job.error to
11683
+ // the reason text passed in; we clear that back to null right
11684
+ // after since this is a clean auto-reset, not a recorded error.
11685
+ if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
11686
+ j.error = null;
11687
+ delete j.stuckFailedNotified;
11688
+ console.warn(
11689
+ `[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11690
+ + `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
11691
+ );
11692
+ appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
11693
+ }
11694
+ for (const stuck of stuckFailed) {
11695
+ const j = ms.jobs.find((x) => x.slug === stuck.slug);
11696
+ if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
11697
+ // Still has auto-reset attempts left — it will be (or already was,
11698
+ // earlier this same pass) picked up by the loop above instead.
11699
+ // Never log "reset it by hand" for a row that isn't actually stuck.
11700
+ if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
11701
+ j.stuckFailedNotified = true;
11702
+ console.warn(
11703
+ `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
11704
+ + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
11705
+ + `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
11706
+ );
11707
+ appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
11708
+ }
11709
+ for (const target of exhaustedNeedsReviewTargets) {
11710
+ const j = ms.jobs.find((x) => x.slug === target.slug);
11711
+ const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
11712
+ if (outcome) {
11713
+ console.warn(
11714
+ `[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11715
+ + `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
11716
+ );
11717
+ }
11718
+ }
11719
+ for (const target of quarantineTargets) {
11720
+ const j = ms.jobs.find((x) => x.slug === target.slug);
11721
+ if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
11722
+ const outcome = await autoResolveQuarantine(j, target.ageMs);
11723
+ if (outcome) {
11724
+ console.warn(
11725
+ `[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11726
+ + `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
11727
+ );
11728
+ }
11729
+ }
11730
+ }).catch(() => {});
11731
+ }
11732
+ }
11733
+
11429
11734
  async function init() {
11430
11735
  ensureDirs();
11431
11736
  // Boot phase — reconciliation, migrations, self-heal, first reset probe.
@@ -11626,218 +11931,9 @@ async function init() {
11626
11931
  // resets early or the auth token rotates. Tracked so re-init doesn't leak.
11627
11932
  if (rescheduleInterval) clearInterval(rescheduleInterval);
11628
11933
  rescheduleInterval = setInterval(() => {
11629
- rescheduleTimer().catch(() => {});
11630
- const s = readQueueSync();
11631
- // Periodic self-heal: re-run the verifier over stale needs_review jobs so a
11632
- // job whose work actually landed (committed in-window, no FAIL sentinel)
11633
- // auto-clears WITHOUT waiting for the next app restart. Cheap-guarded by
11634
- // shouldRunPeriodicReverify, which reuses isRescanCandidate so the guard
11635
- // and the candidate filter can never drift apart again (they did once —
11636
- // see that function's comment). Kill-switch:
11637
- // SM_REVERIFY_PERIODIC_DISABLE=1 (boot reverify above stays always-on).
11638
- // reverifyNeedsReview's auto-fix loop is capped downstream by
11639
- // MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
11640
- // past it), so this interval firing cannot fan out investigations.
11641
- if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
11642
- if (shouldRunPeriodicReverify(s.jobs)) {
11643
- reverifyNeedsReview().catch(() => {});
11644
- }
11645
- // A quarantined row only ever promotes to 'pending' through
11646
- // reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
11647
- // it re-checks the PRD file's createdVia stamp every pass. broadcast()
11648
- // already runs reconcile+writeQueue on every normal poll tick, but an
11649
- // idle queue (nothing pending/running to fire) can back off that
11650
- // cadence for a long time; this guarantees an adopted-but-still-
11651
- // quarantined row is re-checked within 10 minutes regardless.
11652
- if (s.jobs.some((j) => j.status === 'quarantined')) {
11653
- broadcast().catch(() => {});
11654
- }
11655
- }
11656
- // Age-based escalation (independent of the self-heal kill-switch above —
11657
- // this is a monitoring signal, not an auto-fix action): a quarantined
11658
- // row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
11659
- // warn-logged by project + slug + age so it cannot sit stranded and
11660
- // silent (the four burrow-project rows this PRD was written against).
11661
- for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
11662
- console.warn(
11663
- `[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
11664
- + `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
11665
- + `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
11666
- );
11667
- appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
11668
- }
11669
-
11670
- // Estimate-relative overrun escalation. Sits in the blind spot between
11671
- // the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
11672
- // producing output while looping trips neither, so nothing noticed a PRD
11673
- // running 9x its own estimate until a human went looking. Escalate loudly;
11674
- // never kill on an estimate (see JOB_OVERRUN_FACTOR).
11675
- for (const over of findOverrunningJobs(s.jobs, Date.now())) {
11676
- console.warn(
11677
- `[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
11678
- + `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
11679
- + `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
11680
- + `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
11681
- + `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
11682
- );
11683
- appendAuditEvent('job_overrunning_estimate', {
11684
- slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
11685
- });
11686
- // Durable stamp so schedule:state (and therefore the renderer) can see
11687
- // this without re-deriving it — the console.warn/audit event above are
11688
- // visible only in the log, never on the row itself. Display-only
11689
- // advisory field; re-stamped in place every sweep, never appended.
11690
- mutate((state) => {
11691
- const j = state.jobs.find((x) => x.slug === over.slug);
11692
- if (!j) return;
11693
- j.overrun = {
11694
- ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
11695
- };
11696
- }).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
11697
- }
11698
-
11699
- // Stranded-investigation restore. Unlike the two escalations above, this
11700
- // one ACTS: 'investigating' is a transient status whose restore
11701
- // (spawnInvestigation's onExit/catch) only runs inside the process that
11702
- // spawned the probe, so an app restart mid-probe leaves the row frozen
11703
- // there forever (see findStrandedInvestigations' header, and the
11704
- // "'investigating' must never be the job's resting state" comment at
11705
- // spawnInvestigation's onExit). This restores each stranded row to the
11706
- // exact terminal status it already carried before the probe was
11707
- // spawned — it never re-runs or re-investigates anything.
11708
- const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
11709
- if (stranded.length > 0) {
11710
- mutate((ms) => {
11711
- for (const st of stranded) {
11712
- const j = ms.jobs.find((x) => x.slug === st.slug);
11713
- if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
11714
- transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
11715
- delete j.runtime;
11716
- console.warn(
11717
- `[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
11718
- + `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
11719
- + `restored to '${st.restoreStatus}'`,
11720
- );
11721
- appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
11722
- }
11723
- })
11724
- .then(() => broadcast({ flush: true }))
11725
- .catch(() => {});
11726
- }
11727
-
11728
- // Per-project starvation (PRD 1087): a project with pending work that has
11729
- // been passed over on every tick while OTHER projects dispatch. Nothing
11730
- // else distinguishes "no pending work" from "pending work, never
11731
- // started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
11732
- // Escalation only, same shape as the quarantine/overrun warnings above.
11733
- const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
11734
- for (const sp of starvedProjects) {
11735
- console.warn(
11736
- `[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
11737
- + `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
11738
- + `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
11739
- );
11740
- appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
11741
- }
11742
- // Bounded, automated consequence for a starve that outlives the WARN
11743
- // above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
11744
- // project_starved rows and zero consequence). STARVE_ESCALATION_MS is
11745
- // strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
11746
- // a subset of the rows already reported above — same verdict, no
11747
- // re-derivation.
11748
- runStarveEscalationSweep(starvedProjects);
11749
-
11750
- // Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
11751
- // escalation now narrowed to only the rows that auto-reset gave up on.
11752
- // See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
11753
- // Computed together, acted on in the SAME mutate(...) pass, so the
11754
- // stuckFailedNotified race guard below and the auto-reset race guard
11755
- // above it can never observe two different snapshots of the same row.
11756
- // Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
11757
- const autoResetTargets = failedAutoResetDisabled()
11758
- ? []
11759
- : selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
11760
- const stuckFailed = stuckFailedEscalationDisabled()
11761
- ? []
11762
- : findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
11763
- // Bounded automatic terminal decision for exhausted needs_review rows
11764
- // (this PRD): computed alongside the failed-row passes above and acted
11765
- // on in the SAME mutate(...) pass below, for the same race-guard reason
11766
- // — a row's exhaustedResolveAttempts counter must never be read from one
11767
- // snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
11768
- const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
11769
- ? []
11770
- : selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
11771
- // Bounded automatic exit for quarantined rows (this PRD): computed
11772
- // alongside the passes above and acted on in the SAME mutate(...) pass
11773
- // below, for the same race-guard reason — quarantineResolveAttempts must
11774
- // never be read from one snapshot and written from another, and the
11775
- // createdVia re-check inside autoResolveQuarantine must happen in the
11776
- // same turn as the transition it gates. Kill-switch:
11777
- // SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
11778
- const quarantineTargets = quarantineAutoResolveDisabled()
11779
- ? []
11780
- : selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
11781
- if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
11782
- mutate(async (ms) => {
11783
- for (const target of autoResetTargets) {
11784
- const j = ms.jobs.find((x) => x.slug === target.slug);
11785
- if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
11786
- const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
11787
- j.failedAutoResetAttempts = attempt;
11788
- const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
11789
- // resetJobFields is the same field-clearing list the admin
11790
- // scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
11791
- // it rather than inventing a second list. It also sets job.error to
11792
- // the reason text passed in; we clear that back to null right
11793
- // after since this is a clean auto-reset, not a recorded error.
11794
- if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
11795
- j.error = null;
11796
- delete j.stuckFailedNotified;
11797
- console.warn(
11798
- `[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11799
- + `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
11800
- );
11801
- appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
11802
- }
11803
- for (const stuck of stuckFailed) {
11804
- const j = ms.jobs.find((x) => x.slug === stuck.slug);
11805
- if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
11806
- // Still has auto-reset attempts left — it will be (or already was,
11807
- // earlier this same pass) picked up by the loop above instead.
11808
- // Never log "reset it by hand" for a row that isn't actually stuck.
11809
- if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
11810
- j.stuckFailedNotified = true;
11811
- console.warn(
11812
- `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
11813
- + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
11814
- + `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
11815
- );
11816
- appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
11817
- }
11818
- for (const target of exhaustedNeedsReviewTargets) {
11819
- const j = ms.jobs.find((x) => x.slug === target.slug);
11820
- const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
11821
- if (outcome) {
11822
- console.warn(
11823
- `[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11824
- + `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
11825
- );
11826
- }
11827
- }
11828
- for (const target of quarantineTargets) {
11829
- const j = ms.jobs.find((x) => x.slug === target.slug);
11830
- if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
11831
- const outcome = await autoResolveQuarantine(j, target.ageMs);
11832
- if (outcome) {
11833
- console.warn(
11834
- `[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11835
- + `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
11836
- );
11837
- }
11838
- }
11839
- }).catch(() => {});
11840
- }
11934
+ // One throwing tick (e.g. readQueueSync on a torn queue.json) must skip
11935
+ // only itself — the interval keeps firing and the failure is logged.
11936
+ guardedTick(rescheduleIntervalTick, 'rescheduleInterval tick failed');
11841
11937
  }, REVERIFY_INTERVAL_MS);
11842
11938
 
11843
11939
  // Self-rescheduling poll loop with exponential backoff. Replaces the
@@ -12509,6 +12605,11 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
12509
12605
  }
12510
12606
 
12511
12607
  module.exports = {
12608
+ reportSchedulerError,
12609
+ finalizeJobWorktree,
12610
+ worktreeIntegrationVerdict,
12611
+ guardedTick,
12612
+ rescheduleIntervalTick,
12512
12613
  classifyQueueStarvation,
12513
12614
  classifyQueueStarvationByProject,
12514
12615
  dispatchIdleMs,
@@ -12542,6 +12643,8 @@ module.exports = {
12542
12643
  nextBackoffMs,
12543
12644
  shouldWarnFailureStreak,
12544
12645
  shouldEscalateFailureStreak,
12646
+ restoreFailureStreakWarnedAt,
12647
+ persistedStreakMinutes,
12545
12648
  computeDegradedBudget,
12546
12649
  healRefusalReason,
12547
12650
  writeQueue,