claude-code-session-manager 0.40.1 → 0.40.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.html CHANGED
@@ -7,7 +7,7 @@
7
7
  <link rel="preconnect" href="https://fonts.googleapis.com">
8
8
  <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
9
9
  <link href="https://fonts.googleapis.com/css2?family=Newsreader:ital,opsz,wght@0,6..72,400;0,6..72,500;0,6..72,600;0,6..72,700;1,6..72,400&family=Geist:wght@300;400;500;600;700&family=IBM+Plex+Mono:wght@400;500;600&display=swap" rel="stylesheet">
10
- <script type="module" crossorigin src="./assets/index-BbI9slj8.js"></script>
10
+ <script type="module" crossorigin src="./assets/index-QRU8uTeq.js"></script>
11
11
  <link rel="modulepreload" crossorigin href="./assets/monaco-editor-BW5C4Iv1.js">
12
12
  <link rel="stylesheet" crossorigin href="./assets/monaco-editor-BTnBOi8r.css">
13
13
  <link rel="stylesheet" crossorigin href="./assets/index-BxVBtmjA.css">
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-code-session-manager",
3
- "version": "0.40.1",
3
+ "version": "0.40.3",
4
4
  "description": "Local cockpit for the Claude Code CLI — multi-tab terminal, full config surface, scheduler, voice dictation, and live observability.",
5
5
  "type": "module",
6
6
  "main": "src/main/index.cjs",
@@ -341,12 +341,30 @@ function getConcurrencyCap() {
341
341
  // await full teardown instead of firing SIGTERM and returning immediately.
342
342
  const inFlight = new Map();
343
343
  const waiting = []; // [{ tabId, sessionId, prompt, cwd, resume, silent, onSilentResult }]
344
+ // tabIds picked by pump() but not yet registered in `inFlight` by executeRun.
345
+ // Closes the microtask-wide window in which per-tab exclusivity would
346
+ // otherwise not hold. See pump().
347
+ const dispatching = new Set();
344
348
  let activeCount = 0;
345
349
 
346
350
  // Indirection so tests can stub the spawn without launching claude.
347
351
  let executor = executeRun;
348
352
  function __setExecutor(fn) { executor = fn || executeRun; }
349
353
 
354
+ /**
355
+ * Test hook: drop all lane state. The module is a singleton shared across
356
+ * every test in a file, and a spec whose fake child never emits 'exit' leaves
357
+ * its run permanently in flight — holding a session slot and a lane seat that
358
+ * silently starve every later test. Also releases the shared slot pool.
359
+ */
360
+ function __resetQueueForTests() {
361
+ inFlight.clear();
362
+ dispatching.clear();
363
+ waiting.length = 0;
364
+ activeCount = 0;
365
+ sessionSlots.__resetForTests();
366
+ }
367
+
350
368
  // ─── Window reference (set by attachWindow) ────────────────────────────────
351
369
 
352
370
  let mainWindow = null;
@@ -375,27 +393,56 @@ function broadcast(channel, payload) {
375
393
  * @param {{ tabId: string, sessionId: string, prompt: string, cwd: string, resume: boolean, silent?: boolean, onSilentResult?: (text: string) => void, promptId?: string }} opts
376
394
  */
377
395
  function run(opts) {
378
- // Per-tab exclusivity guard — unrelated to the cross-tab cap; must hold for
379
- // BOTH manual and silent runs so a manual send can't race a /context probe
380
- // for the same tab against the same --resume sessionId.
381
- if (inFlight.has(opts.tabId) || waiting.some((w) => w.tabId === opts.tabId)) return;
396
+ // Per-tab exclusivity still holds — two `claude -p --resume` against ONE
397
+ // sessionId must never overlap — but it is now enforced in pump() (which
398
+ // refuses to start a run for a tab that already has one live) rather than by
399
+ // discarding the submission here.
400
+ //
401
+ // Dropping was silent and unrecoverable: `chat:run` still resolved
402
+ // { ok: true }, so the renderer sat at running: true forever with no toast
403
+ // and no terminal event. A `silent` /context probe holding the tabId was
404
+ // enough to swallow a user's message. User-initiated sends are now QUEUED
405
+ // behind whatever holds the tab and dispatched in FIFO order.
406
+ // `dispatching` counts as busy too: between pump() picking a run and
407
+ // executeRun registering it in `inFlight`, the tab is committed but not yet
408
+ // visible in `inFlight` — the same window pump() guards against.
409
+ const tabBusy = inFlight.has(opts.tabId)
410
+ || dispatching.has(opts.tabId)
411
+ || waiting.some((w) => w.tabId === opts.tabId);
412
+ if (tabBusy && opts.silent) {
413
+ // Silent probes stay best-effort and invisible: a probe that collides is
414
+ // still dropped, since queueing one would delay real work to no benefit.
415
+ return { accepted: false, reason: 'tab-busy-silent' };
416
+ }
382
417
 
383
418
  waiting.push(opts);
384
419
  pump();
420
+ return { accepted: true, queued: tabBusy };
385
421
  }
386
422
 
387
423
  // Fill open lanes FIFO up to CONCURRENCY_CAP, then announce queue positions for
388
424
  // the remainder. O(n) over the waiting list (bounded by open tabs).
389
425
  function pump() {
390
426
  while (activeCount < getConcurrencyCap() && waiting.length > 0) {
427
+ // Per-tab exclusivity: skip past any waiting run whose tab already has one
428
+ // live, rather than head-blocking the whole lane on it. `dispatching`
429
+ // covers the gap between picking a job here and executeRun() registering
430
+ // it in `inFlight` (which happens a microtask later) — without it, one
431
+ // synchronous sweep of this loop could start two runs for the same tab and
432
+ // race two --resume processes against a single sessionId.
433
+ const idx = waiting.findIndex((w) => !inFlight.has(w.tabId) && !dispatching.has(w.tabId));
434
+ // Everything queued is blocked behind its own tab's live run; the settle()
435
+ // of that run re-pumps.
436
+ if (idx === -1) break;
391
437
  // The chat lane cap is a POLICY bound; actual capacity comes from the
392
438
  // Session-Manager-wide slot pool shared with the scheduler
393
439
  // (lib/sessionSlots.cjs). No slot → everyone keeps waiting; the next
394
440
  // settle() in either subsystem re-pumps.
395
- const slotToken = sessionSlots.acquire(`chat:${waiting[0].tabId}`);
441
+ const slotToken = sessionSlots.acquire(`chat:${waiting[idx].tabId}`);
396
442
  if (!slotToken) break;
397
- const job = waiting.shift();
443
+ const [job] = waiting.splice(idx, 1);
398
444
  activeCount += 1;
445
+ dispatching.add(job.tabId);
399
446
  Promise.resolve()
400
447
  .then(() => {
401
448
  const donePromise = executor(job);
@@ -404,7 +451,12 @@ function pump() {
404
451
  return donePromise;
405
452
  })
406
453
  .catch(() => { /* executeRun never rejects; defensive */ })
407
- .finally(() => { sessionSlots.release(slotToken); activeCount -= 1; pump(); });
454
+ .finally(() => {
455
+ dispatching.delete(job.tabId);
456
+ sessionSlots.release(slotToken);
457
+ activeCount -= 1;
458
+ pump();
459
+ });
408
460
  }
409
461
  // Anyone still waiting gets a 1-based position update. Silent (automated
410
462
  // probe) runs must stay invisible to the renderer, same as every other
@@ -815,8 +867,12 @@ function registerChatHandlers() {
815
867
  const { schemas, validated } = require('./ipcSchemas.cjs');
816
868
 
817
869
  ipcMain.handle('chat:run', validated(schemas.chatRun, async ({ tabId, sessionId, prompt, cwd, resume, promptId }) => {
818
- run({ tabId, sessionId, prompt, cwd, resume: !!resume, promptId });
819
- return { ok: true };
870
+ // Report what actually happened instead of an unconditional { ok: true }.
871
+ // A user send is always accepted now (queued behind the tab's live run if
872
+ // one holds it), but the renderer gets the truth either way rather than a
873
+ // success it can't verify.
874
+ const outcome = run({ tabId, sessionId, prompt, cwd, resume: !!resume, promptId });
875
+ return { ok: outcome.accepted !== false, queued: !!outcome.queued, reason: outcome.reason };
820
876
  }));
821
877
 
822
878
  // Classification for a queued PromptTicket reaching the front of its
@@ -841,6 +897,7 @@ function registerChatHandlers() {
841
897
  module.exports = {
842
898
  run,
843
899
  cancel,
900
+ __resetQueueForTests,
844
901
  attachWindow,
845
902
  registerChatHandlers,
846
903
  parseStopSignal,
@@ -84,6 +84,11 @@ function evaluateTickLiveness(queueState, heartbeat, now, runningCount) {
84
84
  if (queueState.paused) return { stalled: false, reason: 'paused' };
85
85
  if (config.enabled === false) return { stalled: false, reason: 'disabled' };
86
86
  if (running >= concurrencyCap) return { stalled: false, reason: 'at-capacity' };
87
+ // 'manual' means the operator fires batches by hand — the scheduler is not
88
+ // supposed to pick these up on its own, so pending work sitting with free
89
+ // capacity is the configured behaviour, not a stall. Without this, any queue
90
+ // under a manual policy reported RED forever and drowned out real stalls.
91
+ if (config.firePolicy === 'manual') return { stalled: false, reason: 'manual-fire-policy' };
87
92
 
88
93
  const lastRunAt = queueState.lastRunAt ? Date.parse(queueState.lastRunAt) : null;
89
94
  // No lastRunAt at all (fresh install, never ticked) — nothing to measure
package/src/main/pty.cjs CHANGED
@@ -25,6 +25,11 @@ const { sendIfAlive } = require('./lib/sendToRenderer.cjs');
25
25
  // the remediation message so the user can cd there and rebuild.
26
26
  const PKG_DIR = path.join(__dirname, '..', '..');
27
27
 
28
+ // Per-session replay buffer ceiling. Big enough that switching away from an
29
+ // Epic and back shows a meaningful working history, small enough that a dozen
30
+ // chatty sessions can't balloon main-process memory.
31
+ const REPLAY_BUFFER_MAX = 256 * 1024;
32
+
28
33
  /**
29
34
  * ANSI-formatted terminal text explaining a native-module / immediate-exit
30
35
  * failure and exactly how to fix it. Written straight into the tab's output so
@@ -60,6 +65,13 @@ class PtyManager {
60
65
  constructor() {
61
66
  this.sessions = new Map(); // tabId -> { proc, cwd, created }
62
67
  this.killed = new Set(); // tabIds explicitly killed — suppress their exit events
68
+ // tabId -> recent output, replayed verbatim on reattach. An Epic's
69
+ // Terminal pane unmounts whenever the user views a DIFFERENT Epic while
70
+ // its claude keeps running (Epic ⇄ claude-session is 1:1 and must survive
71
+ // browsing), so "pre-reattach output is lost" — tolerable for a dev
72
+ // reload — would read as the session having been wiped. Bounded ring so a
73
+ // long-lived session can't grow this without limit.
74
+ this.buffers = new Map();
63
75
  this.window = null;
64
76
  }
65
77
 
@@ -104,6 +116,15 @@ class PtyManager {
104
116
  } catch {
105
117
  /* pty may have exited between the check and the resize */
106
118
  }
119
+ // Replay recent output so a reattached viewport isn't blank. The caller
120
+ // registers its pty:data listener BEFORE invoking spawn (see
121
+ // EpicTerminalPane / Terminal), and this fires on the next tick, so the
122
+ // listener is always in place by the time the replay lands. Sent as one
123
+ // chunk ahead of any live data, preserving ordering.
124
+ const replay = this.buffers.get(tabId);
125
+ if (replay) {
126
+ setImmediate(() => sendIfAlive(this.window, `pty:data:${tabId}`, replay));
127
+ }
107
128
  // If the process has already exited but its session wasn't cleaned up,
108
129
  // fire a synthetic exit after the renderer re-registers its onExit handler.
109
130
  if (existing.proc.exitCode != null) {
@@ -154,6 +175,7 @@ class PtyManager {
154
175
 
155
176
  proc.onData((data) => {
156
177
  gotData = true;
178
+ this.#appendReplay(tabId, data);
157
179
  sendIfAlive(this.window, `pty:data:${tabId}`, data);
158
180
  });
159
181
 
@@ -164,6 +186,7 @@ class PtyManager {
164
186
  if (this.killed.delete(tabId)) {
165
187
  console.log('[pty] suppressed exit broadcast for killed tabId=', tabId);
166
188
  this.sessions.delete(tabId);
189
+ this.buffers.delete(tabId);
167
190
  return;
168
191
  }
169
192
  // Fast-exit detector: a shell that dies in <1.2s with a non-zero status
@@ -176,12 +199,22 @@ class PtyManager {
176
199
  }
177
200
  sendIfAlive(this.window, `pty:exit:${tabId}`, { exitCode, signal });
178
201
  this.sessions.delete(tabId);
202
+ this.buffers.delete(tabId);
179
203
  });
180
204
 
181
205
  this.sessions.set(tabId, { proc, cwd, created: spawnedAt });
182
206
  return { pid: proc.pid, cwd, reattached: false };
183
207
  }
184
208
 
209
+ /** Append to a session's bounded replay buffer, trimming oldest output. */
210
+ #appendReplay(tabId, data) {
211
+ const next = (this.buffers.get(tabId) ?? '') + data;
212
+ this.buffers.set(
213
+ tabId,
214
+ next.length > REPLAY_BUFFER_MAX ? next.slice(next.length - REPLAY_BUFFER_MAX) : next,
215
+ );
216
+ }
217
+
185
218
  write({ tabId, data }) {
186
219
  const s = this.sessions.get(tabId);
187
220
  if (!s) {
@@ -225,6 +258,7 @@ class PtyManager {
225
258
  /* already dead */
226
259
  }
227
260
  this.sessions.delete(tabId);
261
+ this.buffers.delete(tabId);
228
262
  }
229
263
  }
230
264
 
@@ -3822,116 +3822,137 @@ function registerScheduleHandlers() {
3822
3822
 
3823
3823
  async function init() {
3824
3824
  ensureDirs();
3825
- // A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
3826
- // batch — advance the queue without waiting for the next 60s poll.
3827
- sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
3828
- // Retire the global queue.json: split its rows into per-project shards
3829
- // BEFORE the first read below, so boot reconciliation sees the shards.
3825
+ // Boot phase — reconciliation, migrations, self-heal, first reset probe.
3826
+ // Guarded as a unit: everything below installs the timers that ARE the
3827
+ // running scheduler (poll loop, heartbeat, supervisor). A throw in here
3828
+ // used to reject init() and silently skip all of them, leaving an app
3829
+ // that looks up and holds the ownership lock but never ticks and never
3830
+ // writes a heartbeat — indistinguishable from a hung queue. Most of this
3831
+ // work is best-effort recovery; none of it is worth trading the scheduler
3832
+ // itself for. rescheduleTimer() in particular reaches the billing API,
3833
+ // which fails whenever the OAuth token is stale.
3830
3834
  try {
3831
- const m = await queueStore.migrateLegacyGlobalQueue(DEFAULT_PROJECT_CWD);
3832
- if (m.migrated) {
3833
- console.log(`[scheduler] legacy global queue retired: ${m.moved} row(s) split across ${m.projects} project shard(s)`);
3835
+ // A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
3836
+ // batch — advance the queue without waiting for the next 60s poll.
3837
+ sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
3838
+ // Retire the global queue.json: split its rows into per-project shards
3839
+ // BEFORE the first read below, so boot reconciliation sees the shards.
3840
+ try {
3841
+ const m = await queueStore.migrateLegacyGlobalQueue(DEFAULT_PROJECT_CWD);
3842
+ if (m.migrated) {
3843
+ console.log(`[scheduler] legacy global queue retired: ${m.moved} row(s) split across ${m.projects} project shard(s)`);
3844
+ }
3845
+ } catch (e) {
3846
+ console.error('[scheduler] legacy queue split failed', e?.message);
3834
3847
  }
3835
- } catch (e) {
3836
- console.error('[scheduler] legacy queue split failed', e?.message);
3837
- }
3838
- await runPrdMigration();
3839
- sweepQueueBackups().catch((e) => console.warn('[scheduler] backup sweep failed', e?.message));
3840
-
3841
- // Hydrate cached state from the sidecar before any scheduling decisions.
3842
- loadSchedulerState();
3843
- bootedAt = Date.now();
3844
-
3845
- // Boot reconciliation: finalize any job that was 'running' when the app died.
3846
- // Check the run log first — a job that emitted result/success before the crash
3847
- // should be marked 'completed', not 'failed', so it doesn't wedge the queue
3848
- // via the failure-gate. Also kill any still-live orphan claude child to prevent
3849
- // it from continuing to write to the project unsupervised (2026-05-21 incident).
3850
- //
3851
- // classifyRunOutcome calls readTail → fs.readFileSync (up to 64 KB per job).
3852
- // Pre-compute all outcomes BEFORE entering the mutate lock so the blocking I/O
3853
- // does not stall the event loop or hold the mutateTail chain during startup.
3854
- //
3855
- // Jobs whose recorded pid is still alive are deferred (not classified here) —
3856
- // see partitionBootOrphans. Everything else (dead pid or no pid) is safe to
3857
- // classify immediately below.
3858
- const bootSnap = readQueueSync();
3859
- const { immediate: immediateSlugs, deferred: deferredSlugs } = partitionBootOrphans(bootSnap.jobs);
3860
- const bootOutcomes = new Map();
3861
- for (const j of bootSnap.jobs) {
3862
- if (!immediateSlugs.includes(j.slug)) continue;
3863
- const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
3864
- bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
3865
- }
3866
- const bootReconciledCompletions = [];
3867
- await mutate((state) => {
3868
- for (const j of state.jobs) {
3869
- if (j.status !== 'running' || !immediateSlugs.includes(j.slug)) continue;
3870
- const outcome = bootOutcomes.get(j.slug) ?? 'unknown';
3871
- const pid = j.runtime?.pid;
3872
- const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
3873
- applyOrphanOutcome(j, outcome, killNote);
3874
- if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
3875
- console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
3848
+ await runPrdMigration();
3849
+ sweepQueueBackups().catch((e) => console.warn('[scheduler] backup sweep failed', e?.message));
3850
+
3851
+ // Hydrate cached state from the sidecar before any scheduling decisions.
3852
+ loadSchedulerState();
3853
+ bootedAt = Date.now();
3854
+
3855
+ // Boot reconciliation: finalize any job that was 'running' when the app died.
3856
+ // Check the run log first — a job that emitted result/success before the crash
3857
+ // should be marked 'completed', not 'failed', so it doesn't wedge the queue
3858
+ // via the failure-gate. Also kill any still-live orphan claude child to prevent
3859
+ // it from continuing to write to the project unsupervised (2026-05-21 incident).
3860
+ //
3861
+ // classifyRunOutcome calls readTail → fs.readFileSync (up to 64 KB per job).
3862
+ // Pre-compute all outcomes BEFORE entering the mutate lock so the blocking I/O
3863
+ // does not stall the event loop or hold the mutateTail chain during startup.
3864
+ //
3865
+ // Jobs whose recorded pid is still alive are deferred (not classified here) —
3866
+ // see partitionBootOrphans. Everything else (dead pid or no pid) is safe to
3867
+ // classify immediately below.
3868
+ const bootSnap = readQueueSync();
3869
+ const { immediate: immediateSlugs, deferred: deferredSlugs } = partitionBootOrphans(bootSnap.jobs);
3870
+ const bootOutcomes = new Map();
3871
+ for (const j of bootSnap.jobs) {
3872
+ if (!immediateSlugs.includes(j.slug)) continue;
3873
+ const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
3874
+ bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
3875
+ }
3876
+ const bootReconciledCompletions = [];
3877
+ await mutate((state) => {
3878
+ for (const j of state.jobs) {
3879
+ if (j.status !== 'running' || !immediateSlugs.includes(j.slug)) continue;
3880
+ const outcome = bootOutcomes.get(j.slug) ?? 'unknown';
3881
+ const pid = j.runtime?.pid;
3882
+ const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
3883
+ applyOrphanOutcome(j, outcome, killNote);
3884
+ if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
3885
+ console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
3886
+ }
3887
+ });
3888
+ for (const { slug, cwd } of bootReconciledCompletions) {
3889
+ await archiveCompletedPrd(slug, cwd);
3876
3890
  }
3877
- });
3878
- for (const { slug, cwd } of bootReconciledCompletions) {
3879
- await archiveCompletedPrd(slug, cwd);
3880
- }
3881
3891
 
3882
- // Still-alive orphans: SIGTERM (+ killOrphanClaudePid's own deferred SIGKILL
3883
- // follow-up) now, but classification waits until BOOT_ORPHAN_KILL_GRACE_MS
3884
- // later — reading the log while the orphan might still be writing to it
3885
- // could misclassify an about-to-succeed run as no_result and double-run the
3886
- // same PRD (2026-05-21 incident this guard exists for).
3887
- for (const slug of deferredSlugs) {
3888
- const j = bootSnap.jobs.find((x) => x.slug === slug);
3889
- const pid = j?.runtime?.pid;
3890
- const bootRunId = j?.runId ?? null; // captured now — guards against reconciling a DIFFERENT later run of the same slug
3891
- if (!pid) continue;
3892
- const result = killOrphanClaudePid(pid);
3893
- const killNote = ` (orphan pid=${pid}: ${result})`;
3894
- if (result === 'killed') {
3895
- console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
3892
+ // Still-alive orphans: SIGTERM (+ killOrphanClaudePid's own deferred SIGKILL
3893
+ // follow-up) now, but classification waits until BOOT_ORPHAN_KILL_GRACE_MS
3894
+ // later — reading the log while the orphan might still be writing to it
3895
+ // could misclassify an about-to-succeed run as no_result and double-run the
3896
+ // same PRD (2026-05-21 incident this guard exists for).
3897
+ for (const slug of deferredSlugs) {
3898
+ const j = bootSnap.jobs.find((x) => x.slug === slug);
3899
+ const pid = j?.runtime?.pid;
3900
+ const bootRunId = j?.runId ?? null; // captured now — guards against reconciling a DIFFERENT later run of the same slug
3901
+ if (!pid) continue;
3902
+ const result = killOrphanClaudePid(pid);
3903
+ const killNote = ` (orphan pid=${pid}: ${result})`;
3904
+ if (result === 'killed') {
3905
+ console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
3906
+ }
3907
+ setTimeout(() => {
3908
+ const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
3909
+ const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
3910
+ let deferredCompletedCwd;
3911
+ mutate((state) => {
3912
+ const cur = state.jobs.find((x) => x.slug === slug);
3913
+ // Race guard: bail if the job already resolved, OR if it's already been
3914
+ // re-picked into a NEW run (different runId) within the grace window —
3915
+ // that new run is not the boot orphan we SIGTERM'd and must not be
3916
+ // touched by this stale classification.
3917
+ if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
3918
+ applyOrphanOutcome(cur, outcome, killNote);
3919
+ console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
3920
+ deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
3921
+ }).then(() => {
3922
+ if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
3923
+ }).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
3924
+ }, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
3896
3925
  }
3897
- setTimeout(() => {
3898
- const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
3899
- const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
3900
- let deferredCompletedCwd;
3901
- mutate((state) => {
3902
- const cur = state.jobs.find((x) => x.slug === slug);
3903
- // Race guard: bail if the job already resolved, OR if it's already been
3904
- // re-picked into a NEW run (different runId) within the grace window —
3905
- // that new run is not the boot orphan we SIGTERM'd and must not be
3906
- // touched by this stale classification.
3907
- if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
3908
- applyOrphanOutcome(cur, outcome, killNote);
3909
- console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
3910
- deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
3911
- }).then(() => {
3912
- if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
3913
- }).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
3914
- }, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
3915
- }
3916
3926
 
3917
- // If we boot up while paused with a resumeAt in the past, clear it. This
3918
- // happens when the app was closed across the reset window.
3919
- const boot = await readQueue();
3920
- if (boot.paused && boot.paused.resumeAt && new Date(boot.paused.resumeAt).getTime() <= Date.now()) {
3921
- await clearPause('boot-elapsed');
3922
- } else if (boot.paused && boot.paused.resumeAt) {
3923
- // Re-arm the resume timer (lost across restart).
3924
- await setPaused(boot.paused.reason, boot.paused.resumeAt);
3925
- }
3927
+ // If we boot up while paused with a resumeAt in the past, clear it. This
3928
+ // happens when the app was closed across the reset window.
3929
+ const boot = await readQueue();
3930
+ if (boot.paused && boot.paused.resumeAt && new Date(boot.paused.resumeAt).getTime() <= Date.now()) {
3931
+ await clearPause('boot-elapsed');
3932
+ } else if (boot.paused && boot.paused.resumeAt) {
3933
+ // Re-arm the resume timer (lost across restart).
3934
+ await setPaused(boot.paused.reason, boot.paused.resumeAt);
3935
+ }
3926
3936
 
3927
- // Self-heal stale needs_review flags using the current verifier (see
3928
- // reverifyNeedsReview). Runs once on boot so a shipped verifier fix clears
3929
- // its own historical false positives without manual retagging.
3930
- await reverifyNeedsReview().catch((e) => {
3931
- console.error(`[scheduler] boot reverify failed: ${e?.message ?? e}`);
3932
- });
3937
+ // Self-heal stale needs_review flags using the current verifier (see
3938
+ // reverifyNeedsReview). Runs once on boot so a shipped verifier fix clears
3939
+ // its own historical false positives without manual retagging.
3940
+ await reverifyNeedsReview().catch((e) => {
3941
+ console.error(`[scheduler] boot reverify failed: ${e?.message ?? e}`);
3942
+ });
3933
3943
 
3934
- await rescheduleTimer();
3944
+ await rescheduleTimer();
3945
+ } catch (e) {
3946
+ console.error('[scheduler] boot phase failed — starting timers anyway:', e?.message);
3947
+ try {
3948
+ require('./logs.cjs').writeLine({
3949
+ scope: 'scheduler',
3950
+ level: 'error',
3951
+ message: 'boot phase failed; timers started anyway',
3952
+ meta: { error: e?.message },
3953
+ });
3954
+ } catch { /* logging must never be the thing that stops the scheduler */ }
3955
+ }
3935
3956
  // Refresh next-reset every 10 minutes — billing window can shift if usage
3936
3957
  // resets early or the auth token rotates. Tracked so re-init doesn't leak.
3937
3958
  if (rescheduleInterval) clearInterval(rescheduleInterval);