claude-code-session-manager 0.93.0 → 0.95.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/dist/assets/{DataModel-DDGx3qrn.js → DataModel-DZK9QZpV.js} +1 -1
  2. package/dist/assets/{History-CHBjYZyD.js → History-BDyJUGtR.js} +2 -2
  3. package/dist/assets/{Hooks-DG4c-Rzz.js → Hooks-CmK6l0uf.js} +3 -3
  4. package/dist/assets/HostBilko-D8TIx-Bx.js +1 -0
  5. package/dist/assets/{Library-CN1h5rAE.js → Library-DTxpOjx8.js} +25 -25
  6. package/dist/assets/{MarkdownEditor-DmBxrhdn.js → MarkdownEditor-BPJyrevq.js} +1 -1
  7. package/dist/assets/McpServers-B_rvIYDs.js +2 -0
  8. package/dist/assets/{Memory-Dk4uDCjg.js → Memory-B-9GlHqe.js} +6 -6
  9. package/dist/assets/{Permissions-BovobeZw.js → Permissions-DhMMp27n.js} +3 -3
  10. package/dist/assets/{Plugins-DvOf_B62.js → Plugins-DKfAlmMY.js} +2 -2
  11. package/dist/assets/{ProvenanceBadge-Dutz_4OY.js → ProvenanceBadge-ptlItdCK.js} +1 -1
  12. package/dist/assets/{SaveBar-CT5u6ZCJ.js → SaveBar-CSbNSl2W.js} +1 -1
  13. package/dist/assets/Scheduler--rn1QBYc.js +16 -0
  14. package/dist/assets/{ScopeSwitcher-BUYDpz2v.js → ScopeSwitcher-BJof-GWH.js} +1 -1
  15. package/dist/assets/{Settings-DIRmr3tL.js → Settings-Dsx6XZP5.js} +3 -3
  16. package/dist/assets/{SkillReferenceGraph-CdGr6d3_.js → SkillReferenceGraph-CGl4B89-.js} +2 -2
  17. package/dist/assets/{Skills-BQOheQv4.js → Skills-D522Zfol.js} +2 -2
  18. package/dist/assets/{SystemPrompt-5LsimKbX.js → SystemPrompt-DKj-q39J.js} +1 -1
  19. package/dist/assets/{TagLibrary-DreLsT3n.js → TagLibrary-BH7CG7-6.js} +1 -1
  20. package/dist/assets/{TiptapBody-C9kmlBnI.js → TiptapBody-BbcJL9az.js} +1 -1
  21. package/dist/assets/{Toggle-BKLCBu9A.js → Toggle-DAslvnnv.js} +1 -1
  22. package/dist/assets/{index-Db_SK9rW.js → index-Yf8e7VCB.js} +821 -821
  23. package/dist/assets/{settingsSchema-B5jBhSUP.js → settingsSchema-pzp0TgvN.js} +1 -1
  24. package/dist/index.html +1 -1
  25. package/package.json +8 -3
  26. package/scripts/README.md +6 -0
  27. package/src/main/__tests__/chat-cancel-terminal.test.cjs +5 -2
  28. package/src/main/__tests__/chat-exit-close-race.test.cjs +8 -2
  29. package/src/main/__tests__/chat-mcp-consent-notice.test.cjs +5 -2
  30. package/src/main/__tests__/health-tick-liveness.test.cjs +12 -0
  31. package/src/main/__tests__/intradayRefresh.test.cjs +39 -0
  32. package/src/main/__tests__/openExternalApp-spawn-error.test.cjs +25 -0
  33. package/src/main/__tests__/opsErrorLog.test.cjs +22 -0
  34. package/src/main/__tests__/pty-epic-worktree-spawn-cwd.test.cjs +21 -0
  35. package/src/main/__tests__/rateLimitPollerStreak.test.cjs +14 -0
  36. package/src/main/__tests__/runVerify-atomic-verdicts.test.cjs +26 -0
  37. package/src/main/__tests__/runVerify-landed-commit-outranks.test.cjs +11 -1
  38. package/src/main/__tests__/runVerify-transcript-commit-evidence.test.cjs +11 -1
  39. package/src/main/__tests__/runVerify.test.cjs +12 -1
  40. package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +15 -0
  41. package/src/main/__tests__/scheduler-utilization-hold.test.cjs +89 -0
  42. package/src/main/__tests__/transcriptsUsageFor.test.cjs +112 -1
  43. package/src/main/build-info.json +4 -4
  44. package/src/main/chatRunner.cjs +45 -10
  45. package/src/main/crashDiagnostics.cjs +9 -0
  46. package/src/main/git.cjs +9 -2
  47. package/src/main/health.cjs +25 -1
  48. package/src/main/historyAggregator.cjs +2 -19
  49. package/src/main/index.cjs +7 -8
  50. package/src/main/ipcSchemas.cjs +13 -3
  51. package/src/main/lib/__tests__/childWithLog.test.cjs +180 -0
  52. package/src/main/lib/__tests__/delegationReadiness.test.cjs +19 -0
  53. package/src/main/lib/__tests__/gitCacheBound.test.cjs +69 -0
  54. package/src/main/lib/__tests__/loopDelay.test.cjs +68 -0
  55. package/src/main/lib/__tests__/usageCircuit.test.cjs +86 -17
  56. package/src/main/lib/agentPersonaSchema.cjs +2 -0
  57. package/src/main/lib/childWithLog.cjs +92 -51
  58. package/src/main/lib/delegationReadiness.cjs +3 -1
  59. package/src/main/lib/epicSpawnPlan.cjs +27 -8
  60. package/src/main/lib/epicTranscriptPath.cjs +5 -1
  61. package/src/main/lib/headTailBuffer.cjs +43 -0
  62. package/src/main/lib/intradayRefresh.cjs +33 -0
  63. package/src/main/lib/loopDelay.cjs +44 -0
  64. package/src/main/lib/lruCache.cjs +39 -0
  65. package/src/main/lib/openExternalApp.cjs +27 -9
  66. package/src/main/lib/opsErrorLog.cjs +22 -0
  67. package/src/main/lib/promptSessionSchema.cjs +2 -0
  68. package/src/main/lib/rendererRecovery.cjs +141 -0
  69. package/src/main/lib/scheduleJobSchema.cjs +3 -0
  70. package/src/main/lib/usageCircuit.cjs +40 -12
  71. package/src/main/runVerify.cjs +3 -1
  72. package/src/main/scheduler.cjs +409 -260
  73. package/src/main/transcripts.cjs +54 -11
  74. package/src/main/usage.cjs +9 -1
  75. package/src/preload/api.d.ts +20 -3
  76. package/dist/assets/HostBilko-DHXcpV2c.js +0 -1
  77. package/dist/assets/McpServers-Beb_HD6I.js +0 -2
  78. package/dist/assets/Scheduler-B7GSnZj5.js +0 -16
@@ -955,7 +955,11 @@ const DEFAULT_CONFIG = {
955
955
  // 'on-reset' = fire offsetMinutes after the next 5h reset (legacy).
956
956
  // 'manual' = only fire on explicit Run now click.
957
957
  firePolicy: 'when-available',
958
- // For 'when-available'. Fire only when five_hour utilization < this percent.
958
+ // For 'when-available'. Fire only when BINDING-window utilization < this
959
+ // percent. Binding = the most-consumed unscoped window (five_hour OR weekly):
960
+ // dispatching into an exhausted weekly window would just produce 429s, so the
961
+ // gate holds on it — a hold that can last days, which is why it is surfaced
962
+ // loudly (utilizationHold snapshot field, heartbeat window name, health).
959
963
  utilizationThreshold: 90,
960
964
  schemaVersion: 1,
961
965
  supervisor: {
@@ -1507,6 +1511,7 @@ function loadSchedulerState() {
1507
1511
  if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
1508
1512
  if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
1509
1513
  if (typeof s.failureStreakWarnedAt === 'number') failureStreakWarnedAt = s.failureStreakWarnedAt;
1514
+ failureStreakWarnedAt = restoreFailureStreakWarnedAt(failureStreakWarned, failureStreakWarnedAt, Date.now());
1510
1515
  if (typeof s.lastEscalationAt === 'number') lastEscalationAtMs = s.lastEscalationAt;
1511
1516
  } catch { /* first boot or corrupt — start fresh */ }
1512
1517
  }
@@ -1803,6 +1808,9 @@ function heartbeatTick(deps = {}) {
1803
1808
  quarantinedCwds: (s.unreadableCwds ?? []).map((u) => u.cwd),
1804
1809
  nextReset: cachedNextReset,
1805
1810
  utilization: cachedUtilization,
1811
+ // Which window `utilization` (and nextReset) refer to — a reader of
1812
+ // this log can't otherwise tell a 5h hold from a weekly one.
1813
+ utilizationWindow: cachedBindingWindowName,
1806
1814
  consecutiveFailures,
1807
1815
  // State/consecutiveFailures/degraded-budget snapshot of the shared
1808
1816
  // usage-meter breaker, so a human reading only the heartbeat log can
@@ -3225,6 +3233,10 @@ async function reconcile(state) {
3225
3233
 
3226
3234
  let cachedNextReset = null; // bare ISO string or null
3227
3235
  let cachedUtilization = null; // binding-window utilization %, 0–100, or null if unknown
3236
+ let cachedBindingWindowName = null; // name/kind of the window cachedUtilization reads (e.g. 'five_hour', 'weekly_all')
3237
+ // Non-null while maybeLaunchWhenAvailable is holding the queue on the
3238
+ // utilization gate: { window, percent, threshold, resetsAt }. Snapshot-level.
3239
+ let utilizationHold = null;
3228
3240
  // ms timestamp of the last FRESH reset observation (see recordObservedReset)
3229
3241
  // — distinct from Date.now(), so persistSchedulerState never re-stamps a
3230
3242
  // stale cachedNextReset as "just observed" on every poll cycle.
@@ -3280,6 +3292,7 @@ function computeDegradedBudget() {
3280
3292
  function applyDegradedBudget() {
3281
3293
  const budget = computeDegradedBudget();
3282
3294
  cachedUtilization = budget.utilization;
3295
+ cachedBindingWindowName = bindingWindow(lastGoodUsagePayload).name;
3283
3296
  degradedConcurrencyCapValue = budget.concurrencyCap;
3284
3297
  return budget;
3285
3298
  }
@@ -3289,8 +3302,15 @@ async function refreshNextReset() {
3289
3302
  const r = await billing.fetchUsage();
3290
3303
  if (r.kind !== 'ok') throw new Error(`usage fetch failed (${r.kind}): ${r.message ?? ''}`);
3291
3304
  const window = bindingWindow(r.data?.usage);
3305
+ if (!Number.isFinite(window.utilization)) {
3306
+ // Successful fetch with no readable percent = meter/contract problem, not a reading.
3307
+ billing.usageCircuit.recordFailure('no_utilization');
3308
+ throw new Error('usage fetch ok but payload yielded no finite utilization');
3309
+ }
3292
3310
  recordObservedReset(window.resets_at ?? null);
3293
- cachedUtilization = Number.isFinite(window.utilization) ? window.utilization : cachedUtilization;
3311
+ cachedUtilization = window.utilization;
3312
+ cachedBindingWindowName = window.name;
3313
+ lastGoodUsagePayload = r.data?.usage ?? lastGoodUsagePayload;
3294
3314
  return cachedNextReset;
3295
3315
  }
3296
3316
 
@@ -3397,6 +3417,21 @@ function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold =
3397
3417
  return consecutiveFailures >= threshold && !alreadyWarned;
3398
3418
  }
3399
3419
 
3420
+ /**
3421
+ * Pure: `failureStreakWarned === true` must always carry a numeric
3422
+ * `failureStreakWarnedAt` (a state file may hold one without the other), so
3423
+ * the escalation message never renders "after nullm". Backfills `nowMs`.
3424
+ */
3425
+ function restoreFailureStreakWarnedAt(warned, warnedAt, nowMs) {
3426
+ if (!warned) return typeof warnedAt === 'number' ? warnedAt : null;
3427
+ return typeof warnedAt === 'number' ? warnedAt : nowMs;
3428
+ }
3429
+
3430
+ /** Pure: whole minutes a warned streak has persisted; never null/NaN. */
3431
+ function persistedStreakMinutes(warnedAt, nowMs) {
3432
+ return typeof warnedAt === 'number' ? Math.round((nowMs - warnedAt) / 60_000) : 0;
3433
+ }
3434
+
3400
3435
  /**
3401
3436
  * Pure: does a PERSISTING failure streak warrant another escalation (audit
3402
3437
  * event + opsErrorLog line) at `nowMs`? Exported for unit testing. Only
@@ -3438,7 +3473,7 @@ function warnFailureStreakIfNeeded() {
3438
3473
  }
3439
3474
  if (failureStreakWarned && shouldEscalateFailureStreak(consecutiveFailures, lastEscalationAtMs, nowMs)) {
3440
3475
  lastEscalationAtMs = nowMs;
3441
- const persistedMinutes = failureStreakWarnedAt ? Math.round((nowMs - failureStreakWarnedAt) / 60_000) : null;
3476
+ const persistedMinutes = persistedStreakMinutes(failureStreakWarnedAt, nowMs);
3442
3477
  try {
3443
3478
  appendAuditEvent('usage_poller_failure_streak_persists', { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes });
3444
3479
  appendError({
@@ -3590,6 +3625,8 @@ function buildScheduleStatePayload(state) {
3590
3625
  launchBlocks: state.launchBlocks ?? {},
3591
3626
  launchMitigations: state.launchMitigations ?? {},
3592
3627
  utilization: cachedUtilization,
3628
+ utilizationWindow: cachedBindingWindowName,
3629
+ utilizationHold,
3593
3630
  pollHealth: {
3594
3631
  lastPollAt,
3595
3632
  lastPollOk,
@@ -6596,6 +6633,117 @@ async function handleLaunchFailure({ job, res, runId, runDir, launchKey, launchE
6596
6633
  await broadcast({ flush: true });
6597
6634
  }
6598
6635
 
6636
+ // Scheduler-scoped error sink for failures that would otherwise be invisible
6637
+ // in packaged/npx builds (stdout unread). Never throws.
6638
+ function reportSchedulerError(message, slug, e) {
6639
+ try {
6640
+ logs.writeLine({
6641
+ scope: 'scheduler',
6642
+ level: 'error',
6643
+ message,
6644
+ meta: { slug, error: e?.message || String(e), stack: e?.stack },
6645
+ });
6646
+ } catch { /* logging must never be the thing that fails */ }
6647
+ try {
6648
+ appendAuditEvent('scheduler_error', { slug, message, error: e?.message || String(e), stack: e?.stack });
6649
+ } catch { /* same */ }
6650
+ }
6651
+
6652
+ /**
6653
+ * Salvage, integrate and clean up a job's throwaway worktree once its run has
6654
+ * ended. NEVER throws: any rejection (salvage / integrate / cleanup) is
6655
+ * reported through deps.reportSchedulerError and surfaces as
6656
+ * `worktreeIntegrationFailure`, so spawnJob's finalize mutate always runs and
6657
+ * the job can never be left `running`. The branch is kept on every failure.
6658
+ * @returns {Promise<{worktreeLeftoverDirty: string[], salvagePatch: string|null,
6659
+ * worktreeIntegrationFailure: string|null, worktreeIntegrationDetail: object|null,
6660
+ * mergeAutoResolved: string|null, mergeAutoResolvedPaths: string[]|null}>}
6661
+ */
6662
+ async function finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPaths, deps = {} }) {
6663
+ const jw = deps.jobWorktree || jobWorktree;
6664
+ const uncommitted = deps.uncommittedChanges || uncommittedChanges;
6665
+ const report = deps.reportSchedulerError || reportSchedulerError;
6666
+ let worktreeLeftoverDirty = [];
6667
+ let salvagePatch = null;
6668
+ let worktreeIntegrationFailure = null;
6669
+ let worktreeIntegrationDetail = null;
6670
+ let mergeAutoResolved = null;
6671
+ let mergeAutoResolvedPaths = null;
6672
+ try {
6673
+ worktreeLeftoverDirty = (await uncommitted(worktree.dir)) || [];
6674
+ // Salvage the worktree's full diff (tracked + untracked) to the run
6675
+ // dir BEFORE the checkout is removed below — otherwise a job killed
6676
+ // before its finish-protocol commit loses that work outright, with
6677
+ // no branch, no stash, no patch anywhere. Best-effort: never blocks
6678
+ // integration/cleanup and never changes the job's verdict.
6679
+ if (worktreeLeftoverDirty.length) {
6680
+ const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
6681
+ const salvage = await jw.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
6682
+ if (salvage && salvage.ok) {
6683
+ salvagePatch = salvagePath;
6684
+ console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
6685
+ }
6686
+ }
6687
+ const integration = await jw.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
6688
+ if (integration.ok && integration.reason === 'carried-wip-only') {
6689
+ console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
6690
+ }
6691
+ if (!integration.ok) {
6692
+ worktreeIntegrationFailure = integration.reason;
6693
+ worktreeIntegrationDetail = integration;
6694
+ console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
6695
+ } else if (integration.integrated) {
6696
+ console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
6697
+ if (integration.autoResolved) {
6698
+ mergeAutoResolved = integration.autoResolved;
6699
+ mergeAutoResolvedPaths = integration.resolvedPaths || [];
6700
+ console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
6701
+ }
6702
+ }
6703
+ await jw.cleanupJobWorktree({
6704
+ cwd: guardCwd,
6705
+ dir: worktree.dir,
6706
+ branch: worktree.branch,
6707
+ keepBranch: !integration.ok,
6708
+ });
6709
+ } catch (e) {
6710
+ worktreeIntegrationFailure = e?.message || String(e);
6711
+ report('spawnJob worktree finalize failed', job.slug, e);
6712
+ // Best-effort: release the checkout + worktree-cap slot, keep the branch.
6713
+ try {
6714
+ await jw.cleanupJobWorktree({ cwd: guardCwd, dir: worktree.dir, branch: worktree.branch, keepBranch: true });
6715
+ } catch { /* already reported above */ }
6716
+ }
6717
+ return { worktreeLeftoverDirty, salvagePatch, worktreeIntegrationFailure, worktreeIntegrationDetail, mergeAutoResolved, mergeAutoResolvedPaths };
6718
+ }
6719
+
6720
+ /**
6721
+ * Map a worktree integration failure onto the verifier verdict spawnJob stamps
6722
+ * (pure). Null failure -> null (no override). Always downgrades to needs_review.
6723
+ */
6724
+ function worktreeIntegrationVerdict({ failure, detail, slug }) {
6725
+ if (!failure) return null;
6726
+ return {
6727
+ verdict: 'worktree_integration_failed',
6728
+ reason: detail && detail.failureKind === 'content_conflict'
6729
+ ? `Integration blocked by a content conflict in ${(detail.conflictedPaths || []).join(', ') || 'unknown paths'} — branch ${jobWorktree.branchNameFor(slug)} preserved; needs a manual merge.`
6730
+ : `worktree branch integration failed: ${failure} — branch preserved for manual merge`,
6731
+ downgradeTo: 'needs_review',
6732
+ };
6733
+ }
6734
+
6735
+ /**
6736
+ * Run one interval tick; a throw is reported and swallowed so the interval
6737
+ * keeps firing.
6738
+ */
6739
+ function guardedTick(fn, label) {
6740
+ try {
6741
+ fn();
6742
+ } catch (e) {
6743
+ reportSchedulerError(label, null, e);
6744
+ }
6745
+ }
6746
+
6599
6747
  async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6600
6748
  // Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
6601
6749
  // — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
@@ -6971,42 +7119,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
6971
7119
  });
6972
7120
  } finally {
6973
7121
  if (worktree.ok) {
6974
- worktreeLeftoverDirty = (await uncommittedChanges(worktree.dir)) || [];
6975
- // Salvage the worktree's full diff (tracked + untracked) to the run
6976
- // dir BEFORE the checkout is removed below — otherwise a job killed
6977
- // before its finish-protocol commit loses that work outright, with
6978
- // no branch, no stash, no patch anywhere. Best-effort: never blocks
6979
- // integration/cleanup and never changes the job's verdict.
6980
- if (worktreeLeftoverDirty.length) {
6981
- const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
6982
- const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
6983
- if (salvage && salvage.ok) {
6984
- salvagePatch = salvagePath;
6985
- console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
6986
- }
6987
- }
6988
- const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
6989
- if (integration.ok && integration.reason === 'carried-wip-only') {
6990
- console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
6991
- }
6992
- if (!integration.ok) {
6993
- worktreeIntegrationFailure = integration.reason;
6994
- worktreeIntegrationDetail = integration;
6995
- console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
6996
- } else if (integration.integrated) {
6997
- console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
6998
- if (integration.autoResolved) {
6999
- mergeAutoResolved = integration.autoResolved;
7000
- mergeAutoResolvedPaths = integration.resolvedPaths || [];
7001
- console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
7002
- }
7003
- }
7004
- await jobWorktree.cleanupJobWorktree({
7005
- cwd: guardCwd,
7006
- dir: worktree.dir,
7007
- branch: worktree.branch,
7008
- keepBranch: !integration.ok,
7009
- });
7122
+ ({ worktreeLeftoverDirty, salvagePatch, worktreeIntegrationFailure, worktreeIntegrationDetail, mergeAutoResolved, mergeAutoResolvedPaths } =
7123
+ await finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPaths }));
7010
7124
  } else {
7011
7125
  // In-place run (non-git cwd, cap reached, env-disabled, or a carry-over
7012
7126
  // failure) — there is no throwaway checkout to diff, so salvage only
@@ -7279,13 +7393,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
7279
7393
  // guard AC explicitly requires this failure be surfaced as an explicit job
7280
7394
  // outcome, never silently dropped alongside the branch it's stranded on.
7281
7395
  if (worktreeIntegrationFailure) {
7282
- verifyResult = {
7283
- verdict: 'worktree_integration_failed',
7284
- reason: worktreeIntegrationDetail && worktreeIntegrationDetail.failureKind === 'content_conflict'
7285
- ? `Integration blocked by a content conflict in ${(worktreeIntegrationDetail.conflictedPaths || []).join(', ') || 'unknown paths'} — branch ${jobWorktree.branchNameFor(job.slug)} preserved; needs a manual merge.`
7286
- : `worktree branch integration failed: ${worktreeIntegrationFailure} — branch preserved for manual merge`,
7287
- downgradeTo: 'needs_review',
7288
- };
7396
+ verifyResult = worktreeIntegrationVerdict({ failure: worktreeIntegrationFailure, detail: worktreeIntegrationDetail, slug: job.slug });
7289
7397
  }
7290
7398
 
7291
7399
  // Shared-tree stash guard (incident 2026-09-01): only meaningful for an
@@ -7943,6 +8051,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
7943
8051
  }
7944
8052
  } catch (e) {
7945
8053
  console.error('[scheduler] spawnJob error', job.slug, e);
8054
+ reportSchedulerError('spawnJob error', job.slug, e);
7946
8055
  } finally {
7947
8056
  runningSet.delete(job.slug);
7948
8057
  // Slot release notifies subscribed pumps (chat lane) machine-wide.
@@ -8302,7 +8411,7 @@ async function tickBody(gen, { bypassLoadGate }) {
8302
8411
  for (const job of gatedBatch) {
8303
8412
  if (cancelToken.cancelled || stale()) break;
8304
8413
  // spawnJob is fire-and-forget; it calls tickQueue() on completion.
8305
- spawnJob(job, runId, runDir, state.config.defaultCwd).catch(() => {});
8414
+ spawnJob(job, runId, runDir, state.config.defaultCwd).catch((e) => reportSchedulerError('spawnJob dispatch rejected', job.slug, e));
8306
8415
  }
8307
8416
  return recordTick({ fired: true, count: gatedBatch.length, group: gatedBatch[0]?.parallelGroup }, { holds });
8308
8417
  }
@@ -8361,12 +8470,20 @@ async function runDueJobs({ bypassLoadGate = false } = {}) {
8361
8470
  // ---------- when-available launch logic ----------
8362
8471
 
8363
8472
  async function maybeLaunchWhenAvailable(state) {
8473
+ utilizationHold = null;
8364
8474
  if (state.config.firePolicy !== 'when-available') return;
8365
8475
  if (state.paused) return;
8366
8476
  const pending = state.jobs.filter((j) => j.status === 'pending' && !runningSet.has(j.slug));
8367
8477
  if (pending.length === 0) return;
8368
8478
  if (cachedUtilization === null || cachedUtilization === undefined) return;
8369
8479
  if (cachedUtilization >= state.config.utilizationThreshold) {
8480
+ utilizationHold = {
8481
+ window: cachedBindingWindowName,
8482
+ percent: cachedUtilization,
8483
+ threshold: state.config.utilizationThreshold,
8484
+ resetsAt: cachedNextReset,
8485
+ };
8486
+ console.log(`[scheduler] when-available: utilization-held — ${cachedBindingWindowName ?? 'unknown'} window at ${cachedUtilization}% ≥ ${state.config.utilizationThreshold}%, resets ${cachedNextReset ?? 'unknown'}, ${pending.length} pending — holding, not ticking`);
8370
8487
  await broadcast();
8371
8488
  return;
8372
8489
  }
@@ -9432,10 +9549,26 @@ async function pollLoop() {
9432
9549
 
9433
9550
  const r = await billing.fetchUsage();
9434
9551
 
9552
+ if (r.kind === 'ok' && !Number.isFinite(bindingWindow(r.data?.usage).utilization)) {
9553
+ // Successful fetch but no finite percent: a meter/contract problem, not a
9554
+ // reading. Report it truthfully instead of leaving a stale number that
9555
+ // looks live; carry the conservative degraded budget forward.
9556
+ billing.usageCircuit.recordFailure('no_utilization');
9557
+ applyDegradedBudget();
9558
+ lastPollAt = Date.now();
9559
+ lastPollOk = false;
9560
+ persistSchedulerState();
9561
+ const cur = await readQueue();
9562
+ await maybeLaunchWhenAvailable(cur);
9563
+ await broadcast();
9564
+ return;
9565
+ }
9566
+
9435
9567
  if (r.kind === 'ok') {
9436
9568
  const window = bindingWindow(r.data?.usage);
9437
9569
  recordObservedReset(window.resets_at ?? null);
9438
- cachedUtilization = Number.isFinite(window.utilization) ? window.utilization : cachedUtilization;
9570
+ cachedUtilization = window.utilization;
9571
+ cachedBindingWindowName = window.name;
9439
9572
  lastGoodUsagePayload = r.data?.usage ?? lastGoodUsagePayload;
9440
9573
  degradedConcurrencyCapValue = null;
9441
9574
  consecutiveFailures = 0;
@@ -11426,6 +11559,223 @@ function stop() {
11426
11559
  stopDispatchLoop();
11427
11560
  }
11428
11561
 
11562
+ // Body of the 10-minute maintenance interval (self-heal, escalations, restores).
11563
+ // Extracted so a throw is testable through guardedTick.
11564
+ function rescheduleIntervalTick() {
11565
+ rescheduleTimer().catch(() => {});
11566
+ const s = readQueueSync();
11567
+ // Periodic self-heal: re-run the verifier over stale needs_review jobs so a
11568
+ // job whose work actually landed (committed in-window, no FAIL sentinel)
11569
+ // auto-clears WITHOUT waiting for the next app restart. Cheap-guarded by
11570
+ // shouldRunPeriodicReverify, which reuses isRescanCandidate so the guard
11571
+ // and the candidate filter can never drift apart again (they did once —
11572
+ // see that function's comment). Kill-switch:
11573
+ // SM_REVERIFY_PERIODIC_DISABLE=1 (boot reverify above stays always-on).
11574
+ // reverifyNeedsReview's auto-fix loop is capped downstream by
11575
+ // MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
11576
+ // past it), so this interval firing cannot fan out investigations.
11577
+ if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
11578
+ if (shouldRunPeriodicReverify(s.jobs)) {
11579
+ reverifyNeedsReview().catch(() => {});
11580
+ }
11581
+ // A quarantined row only ever promotes to 'pending' through
11582
+ // reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
11583
+ // it re-checks the PRD file's createdVia stamp every pass. broadcast()
11584
+ // already runs reconcile+writeQueue on every normal poll tick, but an
11585
+ // idle queue (nothing pending/running to fire) can back off that
11586
+ // cadence for a long time; this guarantees an adopted-but-still-
11587
+ // quarantined row is re-checked within 10 minutes regardless.
11588
+ if (s.jobs.some((j) => j.status === 'quarantined')) {
11589
+ broadcast().catch(() => {});
11590
+ }
11591
+ }
11592
+ // Age-based escalation (independent of the self-heal kill-switch above —
11593
+ // this is a monitoring signal, not an auto-fix action): a quarantined
11594
+ // row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
11595
+ // warn-logged by project + slug + age so it cannot sit stranded and
11596
+ // silent (the four burrow-project rows this PRD was written against).
11597
+ for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
11598
+ console.warn(
11599
+ `[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
11600
+ + `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
11601
+ + `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
11602
+ );
11603
+ appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
11604
+ }
11605
+
11606
+ // Estimate-relative overrun escalation. Sits in the blind spot between
11607
+ // the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
11608
+ // producing output while looping trips neither, so nothing noticed a PRD
11609
+ // running 9x its own estimate until a human went looking. Escalate loudly;
11610
+ // never kill on an estimate (see JOB_OVERRUN_FACTOR).
11611
+ for (const over of findOverrunningJobs(s.jobs, Date.now())) {
11612
+ console.warn(
11613
+ `[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
11614
+ + `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
11615
+ + `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
11616
+ + `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
11617
+ + `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
11618
+ );
11619
+ appendAuditEvent('job_overrunning_estimate', {
11620
+ slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
11621
+ });
11622
+ // Durable stamp so schedule:state (and therefore the renderer) can see
11623
+ // this without re-deriving it — the console.warn/audit event above are
11624
+ // visible only in the log, never on the row itself. Display-only
11625
+ // advisory field; re-stamped in place every sweep, never appended.
11626
+ mutate((state) => {
11627
+ const j = state.jobs.find((x) => x.slug === over.slug);
11628
+ if (!j) return;
11629
+ j.overrun = {
11630
+ ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
11631
+ };
11632
+ }).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
11633
+ }
11634
+
11635
+ // Stranded-investigation restore. Unlike the two escalations above, this
11636
+ // one ACTS: 'investigating' is a transient status whose restore
11637
+ // (spawnInvestigation's onExit/catch) only runs inside the process that
11638
+ // spawned the probe, so an app restart mid-probe leaves the row frozen
11639
+ // there forever (see findStrandedInvestigations' header, and the
11640
+ // "'investigating' must never be the job's resting state" comment at
11641
+ // spawnInvestigation's onExit). This restores each stranded row to the
11642
+ // exact terminal status it already carried before the probe was
11643
+ // spawned — it never re-runs or re-investigates anything.
11644
+ const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
11645
+ if (stranded.length > 0) {
11646
+ mutate((ms) => {
11647
+ for (const st of stranded) {
11648
+ const j = ms.jobs.find((x) => x.slug === st.slug);
11649
+ if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
11650
+ transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
11651
+ delete j.runtime;
11652
+ console.warn(
11653
+ `[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
11654
+ + `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
11655
+ + `restored to '${st.restoreStatus}'`,
11656
+ );
11657
+ appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
11658
+ }
11659
+ })
11660
+ .then(() => broadcast({ flush: true }))
11661
+ .catch(() => {});
11662
+ }
11663
+
11664
+ // Per-project starvation (PRD 1087): a project with pending work that has
11665
+ // been passed over on every tick while OTHER projects dispatch. Nothing
11666
+ // else distinguishes "no pending work" from "pending work, never
11667
+ // started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
11668
+ // Escalation only, same shape as the quarantine/overrun warnings above.
11669
+ const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
11670
+ for (const sp of starvedProjects) {
11671
+ console.warn(
11672
+ `[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
11673
+ + `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
11674
+ + `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
11675
+ );
11676
+ appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
11677
+ }
11678
+ // Bounded, automated consequence for a starve that outlives the WARN
11679
+ // above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
11680
+ // project_starved rows and zero consequence). STARVE_ESCALATION_MS is
11681
+ // strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
11682
+ // a subset of the rows already reported above — same verdict, no
11683
+ // re-derivation.
11684
+ runStarveEscalationSweep(starvedProjects);
11685
+
11686
+ // Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
11687
+ // escalation now narrowed to only the rows that auto-reset gave up on.
11688
+ // See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
11689
+ // Computed together, acted on in the SAME mutate(...) pass, so the
11690
+ // stuckFailedNotified race guard below and the auto-reset race guard
11691
+ // above it can never observe two different snapshots of the same row.
11692
+ // Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
11693
+ const autoResetTargets = failedAutoResetDisabled()
11694
+ ? []
11695
+ : selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
11696
+ const stuckFailed = stuckFailedEscalationDisabled()
11697
+ ? []
11698
+ : findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
11699
+ // Bounded automatic terminal decision for exhausted needs_review rows
11700
+ // (this PRD): computed alongside the failed-row passes above and acted
11701
+ // on in the SAME mutate(...) pass below, for the same race-guard reason
11702
+ // — a row's exhaustedResolveAttempts counter must never be read from one
11703
+ // snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
11704
+ const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
11705
+ ? []
11706
+ : selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
11707
+ // Bounded automatic exit for quarantined rows (this PRD): computed
11708
+ // alongside the passes above and acted on in the SAME mutate(...) pass
11709
+ // below, for the same race-guard reason — quarantineResolveAttempts must
11710
+ // never be read from one snapshot and written from another, and the
11711
+ // createdVia re-check inside autoResolveQuarantine must happen in the
11712
+ // same turn as the transition it gates. Kill-switch:
11713
+ // SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
11714
+ const quarantineTargets = quarantineAutoResolveDisabled()
11715
+ ? []
11716
+ : selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
11717
+ if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
11718
+ mutate(async (ms) => {
11719
+ for (const target of autoResetTargets) {
11720
+ const j = ms.jobs.find((x) => x.slug === target.slug);
11721
+ if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
11722
+ const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
11723
+ j.failedAutoResetAttempts = attempt;
11724
+ const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
11725
+ // resetJobFields is the same field-clearing list the admin
11726
+ // scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
11727
+ // it rather than inventing a second list. It also sets job.error to
11728
+ // the reason text passed in; we clear that back to null right
11729
+ // after since this is a clean auto-reset, not a recorded error.
11730
+ if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
11731
+ j.error = null;
11732
+ delete j.stuckFailedNotified;
11733
+ console.warn(
11734
+ `[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11735
+ + `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
11736
+ );
11737
+ appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
11738
+ }
11739
+ for (const stuck of stuckFailed) {
11740
+ const j = ms.jobs.find((x) => x.slug === stuck.slug);
11741
+ if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
11742
+ // Still has auto-reset attempts left — it will be (or already was,
11743
+ // earlier this same pass) picked up by the loop above instead.
11744
+ // Never log "reset it by hand" for a row that isn't actually stuck.
11745
+ if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
11746
+ j.stuckFailedNotified = true;
11747
+ console.warn(
11748
+ `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
11749
+ + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
11750
+ + `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
11751
+ );
11752
+ appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
11753
+ }
11754
+ for (const target of exhaustedNeedsReviewTargets) {
11755
+ const j = ms.jobs.find((x) => x.slug === target.slug);
11756
+ const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
11757
+ if (outcome) {
11758
+ console.warn(
11759
+ `[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11760
+ + `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
11761
+ );
11762
+ }
11763
+ }
11764
+ for (const target of quarantineTargets) {
11765
+ const j = ms.jobs.find((x) => x.slug === target.slug);
11766
+ if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
11767
+ const outcome = await autoResolveQuarantine(j, target.ageMs);
11768
+ if (outcome) {
11769
+ console.warn(
11770
+ `[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11771
+ + `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
11772
+ );
11773
+ }
11774
+ }
11775
+ }).catch(() => {});
11776
+ }
11777
+ }
11778
+
11429
11779
  async function init() {
11430
11780
  ensureDirs();
11431
11781
  // Boot phase — reconciliation, migrations, self-heal, first reset probe.
@@ -11626,218 +11976,9 @@ async function init() {
11626
11976
  // resets early or the auth token rotates. Tracked so re-init doesn't leak.
11627
11977
  if (rescheduleInterval) clearInterval(rescheduleInterval);
11628
11978
  rescheduleInterval = setInterval(() => {
11629
- rescheduleTimer().catch(() => {});
11630
- const s = readQueueSync();
11631
- // Periodic self-heal: re-run the verifier over stale needs_review jobs so a
11632
- // job whose work actually landed (committed in-window, no FAIL sentinel)
11633
- // auto-clears WITHOUT waiting for the next app restart. Cheap-guarded by
11634
- // shouldRunPeriodicReverify, which reuses isRescanCandidate so the guard
11635
- // and the candidate filter can never drift apart again (they did once —
11636
- // see that function's comment). Kill-switch:
11637
- // SM_REVERIFY_PERIODIC_DISABLE=1 (boot reverify above stays always-on).
11638
- // reverifyNeedsReview's auto-fix loop is capped downstream by
11639
- // MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
11640
- // past it), so this interval firing cannot fan out investigations.
11641
- if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
11642
- if (shouldRunPeriodicReverify(s.jobs)) {
11643
- reverifyNeedsReview().catch(() => {});
11644
- }
11645
- // A quarantined row only ever promotes to 'pending' through
11646
- // reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
11647
- // it re-checks the PRD file's createdVia stamp every pass. broadcast()
11648
- // already runs reconcile+writeQueue on every normal poll tick, but an
11649
- // idle queue (nothing pending/running to fire) can back off that
11650
- // cadence for a long time; this guarantees an adopted-but-still-
11651
- // quarantined row is re-checked within 10 minutes regardless.
11652
- if (s.jobs.some((j) => j.status === 'quarantined')) {
11653
- broadcast().catch(() => {});
11654
- }
11655
- }
11656
- // Age-based escalation (independent of the self-heal kill-switch above —
11657
- // this is a monitoring signal, not an auto-fix action): a quarantined
11658
- // row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
11659
- // warn-logged by project + slug + age so it cannot sit stranded and
11660
- // silent (the four burrow-project rows this PRD was written against).
11661
- for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
11662
- console.warn(
11663
- `[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
11664
- + `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
11665
- + `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
11666
- );
11667
- appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
11668
- }
11669
-
11670
- // Estimate-relative overrun escalation. Sits in the blind spot between
11671
- // the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
11672
- // producing output while looping trips neither, so nothing noticed a PRD
11673
- // running 9x its own estimate until a human went looking. Escalate loudly;
11674
- // never kill on an estimate (see JOB_OVERRUN_FACTOR).
11675
- for (const over of findOverrunningJobs(s.jobs, Date.now())) {
11676
- console.warn(
11677
- `[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
11678
- + `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
11679
- + `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
11680
- + `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
11681
- + `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
11682
- );
11683
- appendAuditEvent('job_overrunning_estimate', {
11684
- slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
11685
- });
11686
- // Durable stamp so schedule:state (and therefore the renderer) can see
11687
- // this without re-deriving it — the console.warn/audit event above are
11688
- // visible only in the log, never on the row itself. Display-only
11689
- // advisory field; re-stamped in place every sweep, never appended.
11690
- mutate((state) => {
11691
- const j = state.jobs.find((x) => x.slug === over.slug);
11692
- if (!j) return;
11693
- j.overrun = {
11694
- ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
11695
- };
11696
- }).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
11697
- }
11698
-
11699
- // Stranded-investigation restore. Unlike the two escalations above, this
11700
- // one ACTS: 'investigating' is a transient status whose restore
11701
- // (spawnInvestigation's onExit/catch) only runs inside the process that
11702
- // spawned the probe, so an app restart mid-probe leaves the row frozen
11703
- // there forever (see findStrandedInvestigations' header, and the
11704
- // "'investigating' must never be the job's resting state" comment at
11705
- // spawnInvestigation's onExit). This restores each stranded row to the
11706
- // exact terminal status it already carried before the probe was
11707
- // spawned — it never re-runs or re-investigates anything.
11708
- const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
11709
- if (stranded.length > 0) {
11710
- mutate((ms) => {
11711
- for (const st of stranded) {
11712
- const j = ms.jobs.find((x) => x.slug === st.slug);
11713
- if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
11714
- transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
11715
- delete j.runtime;
11716
- console.warn(
11717
- `[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
11718
- + `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
11719
- + `restored to '${st.restoreStatus}'`,
11720
- );
11721
- appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
11722
- }
11723
- })
11724
- .then(() => broadcast({ flush: true }))
11725
- .catch(() => {});
11726
- }
11727
-
11728
- // Per-project starvation (PRD 1087): a project with pending work that has
11729
- // been passed over on every tick while OTHER projects dispatch. Nothing
11730
- // else distinguishes "no pending work" from "pending work, never
11731
- // started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
11732
- // Escalation only, same shape as the quarantine/overrun warnings above.
11733
- const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
11734
- for (const sp of starvedProjects) {
11735
- console.warn(
11736
- `[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
11737
- + `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
11738
- + `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
11739
- );
11740
- appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
11741
- }
11742
- // Bounded, automated consequence for a starve that outlives the WARN
11743
- // above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
11744
- // project_starved rows and zero consequence). STARVE_ESCALATION_MS is
11745
- // strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
11746
- // a subset of the rows already reported above — same verdict, no
11747
- // re-derivation.
11748
- runStarveEscalationSweep(starvedProjects);
11749
-
11750
- // Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
11751
- // escalation now narrowed to only the rows that auto-reset gave up on.
11752
- // See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
11753
- // Computed together, acted on in the SAME mutate(...) pass, so the
11754
- // stuckFailedNotified race guard below and the auto-reset race guard
11755
- // above it can never observe two different snapshots of the same row.
11756
- // Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
11757
- const autoResetTargets = failedAutoResetDisabled()
11758
- ? []
11759
- : selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
11760
- const stuckFailed = stuckFailedEscalationDisabled()
11761
- ? []
11762
- : findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
11763
- // Bounded automatic terminal decision for exhausted needs_review rows
11764
- // (this PRD): computed alongside the failed-row passes above and acted
11765
- // on in the SAME mutate(...) pass below, for the same race-guard reason
11766
- // — a row's exhaustedResolveAttempts counter must never be read from one
11767
- // snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
11768
- const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
11769
- ? []
11770
- : selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
11771
- // Bounded automatic exit for quarantined rows (this PRD): computed
11772
- // alongside the passes above and acted on in the SAME mutate(...) pass
11773
- // below, for the same race-guard reason — quarantineResolveAttempts must
11774
- // never be read from one snapshot and written from another, and the
11775
- // createdVia re-check inside autoResolveQuarantine must happen in the
11776
- // same turn as the transition it gates. Kill-switch:
11777
- // SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
11778
- const quarantineTargets = quarantineAutoResolveDisabled()
11779
- ? []
11780
- : selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
11781
- if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
11782
- mutate(async (ms) => {
11783
- for (const target of autoResetTargets) {
11784
- const j = ms.jobs.find((x) => x.slug === target.slug);
11785
- if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
11786
- const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
11787
- j.failedAutoResetAttempts = attempt;
11788
- const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
11789
- // resetJobFields is the same field-clearing list the admin
11790
- // scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
11791
- // it rather than inventing a second list. It also sets job.error to
11792
- // the reason text passed in; we clear that back to null right
11793
- // after since this is a clean auto-reset, not a recorded error.
11794
- if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
11795
- j.error = null;
11796
- delete j.stuckFailedNotified;
11797
- console.warn(
11798
- `[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11799
- + `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
11800
- );
11801
- appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
11802
- }
11803
- for (const stuck of stuckFailed) {
11804
- const j = ms.jobs.find((x) => x.slug === stuck.slug);
11805
- if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
11806
- // Still has auto-reset attempts left — it will be (or already was,
11807
- // earlier this same pass) picked up by the loop above instead.
11808
- // Never log "reset it by hand" for a row that isn't actually stuck.
11809
- if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
11810
- j.stuckFailedNotified = true;
11811
- console.warn(
11812
- `[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
11813
- + `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
11814
- + `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
11815
- );
11816
- appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
11817
- }
11818
- for (const target of exhaustedNeedsReviewTargets) {
11819
- const j = ms.jobs.find((x) => x.slug === target.slug);
11820
- const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
11821
- if (outcome) {
11822
- console.warn(
11823
- `[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11824
- + `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
11825
- );
11826
- }
11827
- }
11828
- for (const target of quarantineTargets) {
11829
- const j = ms.jobs.find((x) => x.slug === target.slug);
11830
- if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
11831
- const outcome = await autoResolveQuarantine(j, target.ageMs);
11832
- if (outcome) {
11833
- console.warn(
11834
- `[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
11835
- + `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
11836
- );
11837
- }
11838
- }
11839
- }).catch(() => {});
11840
- }
11979
+ // One throwing tick (e.g. readQueueSync on a torn queue.json) must skip
11980
+ // only itself — the interval keeps firing and the failure is logged.
11981
+ guardedTick(rescheduleIntervalTick, 'rescheduleInterval tick failed');
11841
11982
  }, REVERIFY_INTERVAL_MS);
11842
11983
 
11843
11984
  // Self-rescheduling poll loop with exponential backoff. Replaces the
@@ -12509,6 +12650,11 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
12509
12650
  }
12510
12651
 
12511
12652
  module.exports = {
12653
+ reportSchedulerError,
12654
+ finalizeJobWorktree,
12655
+ worktreeIntegrationVerdict,
12656
+ guardedTick,
12657
+ rescheduleIntervalTick,
12512
12658
  classifyQueueStarvation,
12513
12659
  classifyQueueStarvationByProject,
12514
12660
  dispatchIdleMs,
@@ -12542,6 +12688,8 @@ module.exports = {
12542
12688
  nextBackoffMs,
12543
12689
  shouldWarnFailureStreak,
12544
12690
  shouldEscalateFailureStreak,
12691
+ restoreFailureStreakWarnedAt,
12692
+ persistedStreakMinutes,
12545
12693
  computeDegradedBudget,
12546
12694
  healRefusalReason,
12547
12695
  writeQueue,
@@ -12640,6 +12788,7 @@ module.exports = {
12640
12788
  FOREIGN_WIP_END_DELIMITER,
12641
12789
  TRANSIENT_RETRY_CAP,
12642
12790
  buildScheduleStatePayload,
12791
+ refreshNextReset,
12643
12792
  partitionBootOrphans,
12644
12793
  applyOrphanOutcome,
12645
12794
  registerAdminRoutes,