claude-code-session-manager 0.40.1 → 0.40.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{TiptapBody-DWdlsmg9.js → TiptapBody-DweEtHks.js} +1 -1
- package/dist/assets/{index-BbI9slj8.js → index-QRU8uTeq.js} +815 -814
- package/dist/index.html +1 -1
- package/package.json +1 -1
- package/src/main/chatRunner.cjs +66 -9
- package/src/main/health.cjs +5 -0
- package/src/main/pty.cjs +34 -0
- package/src/main/scheduler.cjs +123 -102
package/dist/index.html
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
<link rel="preconnect" href="https://fonts.googleapis.com">
|
|
8
8
|
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
|
9
9
|
<link href="https://fonts.googleapis.com/css2?family=Newsreader:ital,opsz,wght@0,6..72,400;0,6..72,500;0,6..72,600;0,6..72,700;1,6..72,400&family=Geist:wght@300;400;500;600;700&family=IBM+Plex+Mono:wght@400;500;600&display=swap" rel="stylesheet">
|
|
10
|
-
<script type="module" crossorigin src="./assets/index-
|
|
10
|
+
<script type="module" crossorigin src="./assets/index-QRU8uTeq.js"></script>
|
|
11
11
|
<link rel="modulepreload" crossorigin href="./assets/monaco-editor-BW5C4Iv1.js">
|
|
12
12
|
<link rel="stylesheet" crossorigin href="./assets/monaco-editor-BTnBOi8r.css">
|
|
13
13
|
<link rel="stylesheet" crossorigin href="./assets/index-BxVBtmjA.css">
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-code-session-manager",
|
|
3
|
-
"version": "0.40.
|
|
3
|
+
"version": "0.40.3",
|
|
4
4
|
"description": "Local cockpit for the Claude Code CLI — multi-tab terminal, full config surface, scheduler, voice dictation, and live observability.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/main/index.cjs",
|
package/src/main/chatRunner.cjs
CHANGED
|
@@ -341,12 +341,30 @@ function getConcurrencyCap() {
|
|
|
341
341
|
// await full teardown instead of firing SIGTERM and returning immediately.
|
|
342
342
|
const inFlight = new Map();
|
|
343
343
|
const waiting = []; // [{ tabId, sessionId, prompt, cwd, resume, silent, onSilentResult }]
|
|
344
|
+
// tabIds picked by pump() but not yet registered in `inFlight` by executeRun.
|
|
345
|
+
// Closes the microtask-wide window in which per-tab exclusivity would
|
|
346
|
+
// otherwise not hold. See pump().
|
|
347
|
+
const dispatching = new Set();
|
|
344
348
|
let activeCount = 0;
|
|
345
349
|
|
|
346
350
|
// Indirection so tests can stub the spawn without launching claude.
|
|
347
351
|
let executor = executeRun;
|
|
348
352
|
function __setExecutor(fn) { executor = fn || executeRun; }
|
|
349
353
|
|
|
354
|
+
/**
|
|
355
|
+
* Test hook: drop all lane state. The module is a singleton shared across
|
|
356
|
+
* every test in a file, and a spec whose fake child never emits 'exit' leaves
|
|
357
|
+
* its run permanently in flight — holding a session slot and a lane seat that
|
|
358
|
+
* silently starve every later test. Also releases the shared slot pool.
|
|
359
|
+
*/
|
|
360
|
+
function __resetQueueForTests() {
|
|
361
|
+
inFlight.clear();
|
|
362
|
+
dispatching.clear();
|
|
363
|
+
waiting.length = 0;
|
|
364
|
+
activeCount = 0;
|
|
365
|
+
sessionSlots.__resetForTests();
|
|
366
|
+
}
|
|
367
|
+
|
|
350
368
|
// ─── Window reference (set by attachWindow) ────────────────────────────────
|
|
351
369
|
|
|
352
370
|
let mainWindow = null;
|
|
@@ -375,27 +393,56 @@ function broadcast(channel, payload) {
|
|
|
375
393
|
* @param {{ tabId: string, sessionId: string, prompt: string, cwd: string, resume: boolean, silent?: boolean, onSilentResult?: (text: string) => void, promptId?: string }} opts
|
|
376
394
|
*/
|
|
377
395
|
function run(opts) {
|
|
378
|
-
// Per-tab exclusivity
|
|
379
|
-
//
|
|
380
|
-
// for
|
|
381
|
-
|
|
396
|
+
// Per-tab exclusivity still holds — two `claude -p --resume` against ONE
|
|
397
|
+
// sessionId must never overlap — but it is now enforced in pump() (which
|
|
398
|
+
// refuses to start a run for a tab that already has one live) rather than by
|
|
399
|
+
// discarding the submission here.
|
|
400
|
+
//
|
|
401
|
+
// Dropping was silent and unrecoverable: `chat:run` still resolved
|
|
402
|
+
// { ok: true }, so the renderer sat at running: true forever with no toast
|
|
403
|
+
// and no terminal event. A `silent` /context probe holding the tabId was
|
|
404
|
+
// enough to swallow a user's message. User-initiated sends are now QUEUED
|
|
405
|
+
// behind whatever holds the tab and dispatched in FIFO order.
|
|
406
|
+
// `dispatching` counts as busy too: between pump() picking a run and
|
|
407
|
+
// executeRun registering it in `inFlight`, the tab is committed but not yet
|
|
408
|
+
// visible in `inFlight` — the same window pump() guards against.
|
|
409
|
+
const tabBusy = inFlight.has(opts.tabId)
|
|
410
|
+
|| dispatching.has(opts.tabId)
|
|
411
|
+
|| waiting.some((w) => w.tabId === opts.tabId);
|
|
412
|
+
if (tabBusy && opts.silent) {
|
|
413
|
+
// Silent probes stay best-effort and invisible: a probe that collides is
|
|
414
|
+
// still dropped, since queueing one would delay real work to no benefit.
|
|
415
|
+
return { accepted: false, reason: 'tab-busy-silent' };
|
|
416
|
+
}
|
|
382
417
|
|
|
383
418
|
waiting.push(opts);
|
|
384
419
|
pump();
|
|
420
|
+
return { accepted: true, queued: tabBusy };
|
|
385
421
|
}
|
|
386
422
|
|
|
387
423
|
// Fill open lanes FIFO up to CONCURRENCY_CAP, then announce queue positions for
|
|
388
424
|
// the remainder. O(n) over the waiting list (bounded by open tabs).
|
|
389
425
|
function pump() {
|
|
390
426
|
while (activeCount < getConcurrencyCap() && waiting.length > 0) {
|
|
427
|
+
// Per-tab exclusivity: skip past any waiting run whose tab already has one
|
|
428
|
+
// live, rather than head-blocking the whole lane on it. `dispatching`
|
|
429
|
+
// covers the gap between picking a job here and executeRun() registering
|
|
430
|
+
// it in `inFlight` (which happens a microtask later) — without it, one
|
|
431
|
+
// synchronous sweep of this loop could start two runs for the same tab and
|
|
432
|
+
// race two --resume processes against a single sessionId.
|
|
433
|
+
const idx = waiting.findIndex((w) => !inFlight.has(w.tabId) && !dispatching.has(w.tabId));
|
|
434
|
+
// Everything queued is blocked behind its own tab's live run; the settle()
|
|
435
|
+
// of that run re-pumps.
|
|
436
|
+
if (idx === -1) break;
|
|
391
437
|
// The chat lane cap is a POLICY bound; actual capacity comes from the
|
|
392
438
|
// Session-Manager-wide slot pool shared with the scheduler
|
|
393
439
|
// (lib/sessionSlots.cjs). No slot → everyone keeps waiting; the next
|
|
394
440
|
// settle() in either subsystem re-pumps.
|
|
395
|
-
const slotToken = sessionSlots.acquire(`chat:${waiting[
|
|
441
|
+
const slotToken = sessionSlots.acquire(`chat:${waiting[idx].tabId}`);
|
|
396
442
|
if (!slotToken) break;
|
|
397
|
-
const job = waiting.
|
|
443
|
+
const [job] = waiting.splice(idx, 1);
|
|
398
444
|
activeCount += 1;
|
|
445
|
+
dispatching.add(job.tabId);
|
|
399
446
|
Promise.resolve()
|
|
400
447
|
.then(() => {
|
|
401
448
|
const donePromise = executor(job);
|
|
@@ -404,7 +451,12 @@ function pump() {
|
|
|
404
451
|
return donePromise;
|
|
405
452
|
})
|
|
406
453
|
.catch(() => { /* executeRun never rejects; defensive */ })
|
|
407
|
-
.finally(() => {
|
|
454
|
+
.finally(() => {
|
|
455
|
+
dispatching.delete(job.tabId);
|
|
456
|
+
sessionSlots.release(slotToken);
|
|
457
|
+
activeCount -= 1;
|
|
458
|
+
pump();
|
|
459
|
+
});
|
|
408
460
|
}
|
|
409
461
|
// Anyone still waiting gets a 1-based position update. Silent (automated
|
|
410
462
|
// probe) runs must stay invisible to the renderer, same as every other
|
|
@@ -815,8 +867,12 @@ function registerChatHandlers() {
|
|
|
815
867
|
const { schemas, validated } = require('./ipcSchemas.cjs');
|
|
816
868
|
|
|
817
869
|
ipcMain.handle('chat:run', validated(schemas.chatRun, async ({ tabId, sessionId, prompt, cwd, resume, promptId }) => {
|
|
818
|
-
|
|
819
|
-
|
|
870
|
+
// Report what actually happened instead of an unconditional { ok: true }.
|
|
871
|
+
// A user send is always accepted now (queued behind the tab's live run if
|
|
872
|
+
// one holds it), but the renderer gets the truth either way rather than a
|
|
873
|
+
// success it can't verify.
|
|
874
|
+
const outcome = run({ tabId, sessionId, prompt, cwd, resume: !!resume, promptId });
|
|
875
|
+
return { ok: outcome.accepted !== false, queued: !!outcome.queued, reason: outcome.reason };
|
|
820
876
|
}));
|
|
821
877
|
|
|
822
878
|
// Classification for a queued PromptTicket reaching the front of its
|
|
@@ -841,6 +897,7 @@ function registerChatHandlers() {
|
|
|
841
897
|
module.exports = {
|
|
842
898
|
run,
|
|
843
899
|
cancel,
|
|
900
|
+
__resetQueueForTests,
|
|
844
901
|
attachWindow,
|
|
845
902
|
registerChatHandlers,
|
|
846
903
|
parseStopSignal,
|
package/src/main/health.cjs
CHANGED
|
@@ -84,6 +84,11 @@ function evaluateTickLiveness(queueState, heartbeat, now, runningCount) {
|
|
|
84
84
|
if (queueState.paused) return { stalled: false, reason: 'paused' };
|
|
85
85
|
if (config.enabled === false) return { stalled: false, reason: 'disabled' };
|
|
86
86
|
if (running >= concurrencyCap) return { stalled: false, reason: 'at-capacity' };
|
|
87
|
+
// 'manual' means the operator fires batches by hand — the scheduler is not
|
|
88
|
+
// supposed to pick these up on its own, so pending work sitting with free
|
|
89
|
+
// capacity is the configured behaviour, not a stall. Without this, any queue
|
|
90
|
+
// under a manual policy reported RED forever and drowned out real stalls.
|
|
91
|
+
if (config.firePolicy === 'manual') return { stalled: false, reason: 'manual-fire-policy' };
|
|
87
92
|
|
|
88
93
|
const lastRunAt = queueState.lastRunAt ? Date.parse(queueState.lastRunAt) : null;
|
|
89
94
|
// No lastRunAt at all (fresh install, never ticked) — nothing to measure
|
package/src/main/pty.cjs
CHANGED
|
@@ -25,6 +25,11 @@ const { sendIfAlive } = require('./lib/sendToRenderer.cjs');
|
|
|
25
25
|
// the remediation message so the user can cd there and rebuild.
|
|
26
26
|
const PKG_DIR = path.join(__dirname, '..', '..');
|
|
27
27
|
|
|
28
|
+
// Per-session replay buffer ceiling. Big enough that switching away from an
|
|
29
|
+
// Epic and back shows a meaningful working history, small enough that a dozen
|
|
30
|
+
// chatty sessions can't balloon main-process memory.
|
|
31
|
+
const REPLAY_BUFFER_MAX = 256 * 1024;
|
|
32
|
+
|
|
28
33
|
/**
|
|
29
34
|
* ANSI-formatted terminal text explaining a native-module / immediate-exit
|
|
30
35
|
* failure and exactly how to fix it. Written straight into the tab's output so
|
|
@@ -60,6 +65,13 @@ class PtyManager {
|
|
|
60
65
|
constructor() {
|
|
61
66
|
this.sessions = new Map(); // tabId -> { proc, cwd, created }
|
|
62
67
|
this.killed = new Set(); // tabIds explicitly killed — suppress their exit events
|
|
68
|
+
// tabId -> recent output, replayed verbatim on reattach. An Epic's
|
|
69
|
+
// Terminal pane unmounts whenever the user views a DIFFERENT Epic while
|
|
70
|
+
// its claude keeps running (Epic ⇄ claude-session is 1:1 and must survive
|
|
71
|
+
// browsing), so "pre-reattach output is lost" — tolerable for a dev
|
|
72
|
+
// reload — would read as the session having been wiped. Bounded ring so a
|
|
73
|
+
// long-lived session can't grow this without limit.
|
|
74
|
+
this.buffers = new Map();
|
|
63
75
|
this.window = null;
|
|
64
76
|
}
|
|
65
77
|
|
|
@@ -104,6 +116,15 @@ class PtyManager {
|
|
|
104
116
|
} catch {
|
|
105
117
|
/* pty may have exited between the check and the resize */
|
|
106
118
|
}
|
|
119
|
+
// Replay recent output so a reattached viewport isn't blank. The caller
|
|
120
|
+
// registers its pty:data listener BEFORE invoking spawn (see
|
|
121
|
+
// EpicTerminalPane / Terminal), and this fires on the next tick, so the
|
|
122
|
+
// listener is always in place by the time the replay lands. Sent as one
|
|
123
|
+
// chunk ahead of any live data, preserving ordering.
|
|
124
|
+
const replay = this.buffers.get(tabId);
|
|
125
|
+
if (replay) {
|
|
126
|
+
setImmediate(() => sendIfAlive(this.window, `pty:data:${tabId}`, replay));
|
|
127
|
+
}
|
|
107
128
|
// If the process has already exited but its session wasn't cleaned up,
|
|
108
129
|
// fire a synthetic exit after the renderer re-registers its onExit handler.
|
|
109
130
|
if (existing.proc.exitCode != null) {
|
|
@@ -154,6 +175,7 @@ class PtyManager {
|
|
|
154
175
|
|
|
155
176
|
proc.onData((data) => {
|
|
156
177
|
gotData = true;
|
|
178
|
+
this.#appendReplay(tabId, data);
|
|
157
179
|
sendIfAlive(this.window, `pty:data:${tabId}`, data);
|
|
158
180
|
});
|
|
159
181
|
|
|
@@ -164,6 +186,7 @@ class PtyManager {
|
|
|
164
186
|
if (this.killed.delete(tabId)) {
|
|
165
187
|
console.log('[pty] suppressed exit broadcast for killed tabId=', tabId);
|
|
166
188
|
this.sessions.delete(tabId);
|
|
189
|
+
this.buffers.delete(tabId);
|
|
167
190
|
return;
|
|
168
191
|
}
|
|
169
192
|
// Fast-exit detector: a shell that dies in <1.2s with a non-zero status
|
|
@@ -176,12 +199,22 @@ class PtyManager {
|
|
|
176
199
|
}
|
|
177
200
|
sendIfAlive(this.window, `pty:exit:${tabId}`, { exitCode, signal });
|
|
178
201
|
this.sessions.delete(tabId);
|
|
202
|
+
this.buffers.delete(tabId);
|
|
179
203
|
});
|
|
180
204
|
|
|
181
205
|
this.sessions.set(tabId, { proc, cwd, created: spawnedAt });
|
|
182
206
|
return { pid: proc.pid, cwd, reattached: false };
|
|
183
207
|
}
|
|
184
208
|
|
|
209
|
+
/** Append to a session's bounded replay buffer, trimming oldest output. */
|
|
210
|
+
#appendReplay(tabId, data) {
|
|
211
|
+
const next = (this.buffers.get(tabId) ?? '') + data;
|
|
212
|
+
this.buffers.set(
|
|
213
|
+
tabId,
|
|
214
|
+
next.length > REPLAY_BUFFER_MAX ? next.slice(next.length - REPLAY_BUFFER_MAX) : next,
|
|
215
|
+
);
|
|
216
|
+
}
|
|
217
|
+
|
|
185
218
|
write({ tabId, data }) {
|
|
186
219
|
const s = this.sessions.get(tabId);
|
|
187
220
|
if (!s) {
|
|
@@ -225,6 +258,7 @@ class PtyManager {
|
|
|
225
258
|
/* already dead */
|
|
226
259
|
}
|
|
227
260
|
this.sessions.delete(tabId);
|
|
261
|
+
this.buffers.delete(tabId);
|
|
228
262
|
}
|
|
229
263
|
}
|
|
230
264
|
|
package/src/main/scheduler.cjs
CHANGED
|
@@ -3822,116 +3822,137 @@ function registerScheduleHandlers() {
|
|
|
3822
3822
|
|
|
3823
3823
|
async function init() {
|
|
3824
3824
|
ensureDirs();
|
|
3825
|
-
//
|
|
3826
|
-
//
|
|
3827
|
-
|
|
3828
|
-
//
|
|
3829
|
-
//
|
|
3825
|
+
// Boot phase — reconciliation, migrations, self-heal, first reset probe.
|
|
3826
|
+
// Guarded as a unit: everything below installs the timers that ARE the
|
|
3827
|
+
// running scheduler (poll loop, heartbeat, supervisor). A throw in here
|
|
3828
|
+
// used to reject init() and silently skip all of them, leaving an app
|
|
3829
|
+
// that looks up and holds the ownership lock but never ticks and never
|
|
3830
|
+
// writes a heartbeat — indistinguishable from a hung queue. Most of this
|
|
3831
|
+
// work is best-effort recovery; none of it is worth trading the scheduler
|
|
3832
|
+
// itself for. rescheduleTimer() in particular reaches the billing API,
|
|
3833
|
+
// which fails whenever the OAuth token is stale.
|
|
3830
3834
|
try {
|
|
3831
|
-
|
|
3832
|
-
|
|
3833
|
-
|
|
3835
|
+
// A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
|
|
3836
|
+
// batch — advance the queue without waiting for the next 60s poll.
|
|
3837
|
+
sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
|
|
3838
|
+
// Retire the global queue.json: split its rows into per-project shards
|
|
3839
|
+
// BEFORE the first read below, so boot reconciliation sees the shards.
|
|
3840
|
+
try {
|
|
3841
|
+
const m = await queueStore.migrateLegacyGlobalQueue(DEFAULT_PROJECT_CWD);
|
|
3842
|
+
if (m.migrated) {
|
|
3843
|
+
console.log(`[scheduler] legacy global queue retired: ${m.moved} row(s) split across ${m.projects} project shard(s)`);
|
|
3844
|
+
}
|
|
3845
|
+
} catch (e) {
|
|
3846
|
+
console.error('[scheduler] legacy queue split failed', e?.message);
|
|
3834
3847
|
}
|
|
3835
|
-
|
|
3836
|
-
console.
|
|
3837
|
-
|
|
3838
|
-
|
|
3839
|
-
|
|
3840
|
-
|
|
3841
|
-
|
|
3842
|
-
|
|
3843
|
-
|
|
3844
|
-
|
|
3845
|
-
|
|
3846
|
-
|
|
3847
|
-
|
|
3848
|
-
|
|
3849
|
-
|
|
3850
|
-
|
|
3851
|
-
|
|
3852
|
-
|
|
3853
|
-
|
|
3854
|
-
|
|
3855
|
-
|
|
3856
|
-
|
|
3857
|
-
|
|
3858
|
-
|
|
3859
|
-
|
|
3860
|
-
|
|
3861
|
-
|
|
3862
|
-
|
|
3863
|
-
const
|
|
3864
|
-
|
|
3865
|
-
|
|
3866
|
-
|
|
3867
|
-
|
|
3868
|
-
|
|
3869
|
-
|
|
3870
|
-
|
|
3871
|
-
|
|
3872
|
-
|
|
3873
|
-
|
|
3874
|
-
|
|
3875
|
-
|
|
3848
|
+
await runPrdMigration();
|
|
3849
|
+
sweepQueueBackups().catch((e) => console.warn('[scheduler] backup sweep failed', e?.message));
|
|
3850
|
+
|
|
3851
|
+
// Hydrate cached state from the sidecar before any scheduling decisions.
|
|
3852
|
+
loadSchedulerState();
|
|
3853
|
+
bootedAt = Date.now();
|
|
3854
|
+
|
|
3855
|
+
// Boot reconciliation: finalize any job that was 'running' when the app died.
|
|
3856
|
+
// Check the run log first — a job that emitted result/success before the crash
|
|
3857
|
+
// should be marked 'completed', not 'failed', so it doesn't wedge the queue
|
|
3858
|
+
// via the failure-gate. Also kill any still-live orphan claude child to prevent
|
|
3859
|
+
// it from continuing to write to the project unsupervised (2026-05-21 incident).
|
|
3860
|
+
//
|
|
3861
|
+
// classifyRunOutcome calls readTail → fs.readFileSync (up to 64 KB per job).
|
|
3862
|
+
// Pre-compute all outcomes BEFORE entering the mutate lock so the blocking I/O
|
|
3863
|
+
// does not stall the event loop or hold the mutateTail chain during startup.
|
|
3864
|
+
//
|
|
3865
|
+
// Jobs whose recorded pid is still alive are deferred (not classified here) —
|
|
3866
|
+
// see partitionBootOrphans. Everything else (dead pid or no pid) is safe to
|
|
3867
|
+
// classify immediately below.
|
|
3868
|
+
const bootSnap = readQueueSync();
|
|
3869
|
+
const { immediate: immediateSlugs, deferred: deferredSlugs } = partitionBootOrphans(bootSnap.jobs);
|
|
3870
|
+
const bootOutcomes = new Map();
|
|
3871
|
+
for (const j of bootSnap.jobs) {
|
|
3872
|
+
if (!immediateSlugs.includes(j.slug)) continue;
|
|
3873
|
+
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
3874
|
+
bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
|
|
3875
|
+
}
|
|
3876
|
+
const bootReconciledCompletions = [];
|
|
3877
|
+
await mutate((state) => {
|
|
3878
|
+
for (const j of state.jobs) {
|
|
3879
|
+
if (j.status !== 'running' || !immediateSlugs.includes(j.slug)) continue;
|
|
3880
|
+
const outcome = bootOutcomes.get(j.slug) ?? 'unknown';
|
|
3881
|
+
const pid = j.runtime?.pid;
|
|
3882
|
+
const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
|
|
3883
|
+
applyOrphanOutcome(j, outcome, killNote);
|
|
3884
|
+
if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
|
|
3885
|
+
console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
|
|
3886
|
+
}
|
|
3887
|
+
});
|
|
3888
|
+
for (const { slug, cwd } of bootReconciledCompletions) {
|
|
3889
|
+
await archiveCompletedPrd(slug, cwd);
|
|
3876
3890
|
}
|
|
3877
|
-
});
|
|
3878
|
-
for (const { slug, cwd } of bootReconciledCompletions) {
|
|
3879
|
-
await archiveCompletedPrd(slug, cwd);
|
|
3880
|
-
}
|
|
3881
3891
|
|
|
3882
|
-
|
|
3883
|
-
|
|
3884
|
-
|
|
3885
|
-
|
|
3886
|
-
|
|
3887
|
-
|
|
3888
|
-
|
|
3889
|
-
|
|
3890
|
-
|
|
3891
|
-
|
|
3892
|
-
|
|
3893
|
-
|
|
3894
|
-
|
|
3895
|
-
|
|
3892
|
+
// Still-alive orphans: SIGTERM (+ killOrphanClaudePid's own deferred SIGKILL
|
|
3893
|
+
// follow-up) now, but classification waits until BOOT_ORPHAN_KILL_GRACE_MS
|
|
3894
|
+
// later — reading the log while the orphan might still be writing to it
|
|
3895
|
+
// could misclassify an about-to-succeed run as no_result and double-run the
|
|
3896
|
+
// same PRD (2026-05-21 incident this guard exists for).
|
|
3897
|
+
for (const slug of deferredSlugs) {
|
|
3898
|
+
const j = bootSnap.jobs.find((x) => x.slug === slug);
|
|
3899
|
+
const pid = j?.runtime?.pid;
|
|
3900
|
+
const bootRunId = j?.runId ?? null; // captured now — guards against reconciling a DIFFERENT later run of the same slug
|
|
3901
|
+
if (!pid) continue;
|
|
3902
|
+
const result = killOrphanClaudePid(pid);
|
|
3903
|
+
const killNote = ` (orphan pid=${pid}: ${result})`;
|
|
3904
|
+
if (result === 'killed') {
|
|
3905
|
+
console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
|
|
3906
|
+
}
|
|
3907
|
+
setTimeout(() => {
|
|
3908
|
+
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
3909
|
+
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
3910
|
+
let deferredCompletedCwd;
|
|
3911
|
+
mutate((state) => {
|
|
3912
|
+
const cur = state.jobs.find((x) => x.slug === slug);
|
|
3913
|
+
// Race guard: bail if the job already resolved, OR if it's already been
|
|
3914
|
+
// re-picked into a NEW run (different runId) within the grace window —
|
|
3915
|
+
// that new run is not the boot orphan we SIGTERM'd and must not be
|
|
3916
|
+
// touched by this stale classification.
|
|
3917
|
+
if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
|
|
3918
|
+
applyOrphanOutcome(cur, outcome, killNote);
|
|
3919
|
+
console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
|
|
3920
|
+
deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
|
|
3921
|
+
}).then(() => {
|
|
3922
|
+
if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
|
|
3923
|
+
}).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
|
|
3924
|
+
}, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
|
|
3896
3925
|
}
|
|
3897
|
-
setTimeout(() => {
|
|
3898
|
-
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
3899
|
-
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
3900
|
-
let deferredCompletedCwd;
|
|
3901
|
-
mutate((state) => {
|
|
3902
|
-
const cur = state.jobs.find((x) => x.slug === slug);
|
|
3903
|
-
// Race guard: bail if the job already resolved, OR if it's already been
|
|
3904
|
-
// re-picked into a NEW run (different runId) within the grace window —
|
|
3905
|
-
// that new run is not the boot orphan we SIGTERM'd and must not be
|
|
3906
|
-
// touched by this stale classification.
|
|
3907
|
-
if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
|
|
3908
|
-
applyOrphanOutcome(cur, outcome, killNote);
|
|
3909
|
-
console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
|
|
3910
|
-
deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
|
|
3911
|
-
}).then(() => {
|
|
3912
|
-
if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
|
|
3913
|
-
}).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
|
|
3914
|
-
}, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
|
|
3915
|
-
}
|
|
3916
3926
|
|
|
3917
|
-
|
|
3918
|
-
|
|
3919
|
-
|
|
3920
|
-
|
|
3921
|
-
|
|
3922
|
-
|
|
3923
|
-
|
|
3924
|
-
|
|
3925
|
-
|
|
3927
|
+
// If we boot up while paused with a resumeAt in the past, clear it. This
|
|
3928
|
+
// happens when the app was closed across the reset window.
|
|
3929
|
+
const boot = await readQueue();
|
|
3930
|
+
if (boot.paused && boot.paused.resumeAt && new Date(boot.paused.resumeAt).getTime() <= Date.now()) {
|
|
3931
|
+
await clearPause('boot-elapsed');
|
|
3932
|
+
} else if (boot.paused && boot.paused.resumeAt) {
|
|
3933
|
+
// Re-arm the resume timer (lost across restart).
|
|
3934
|
+
await setPaused(boot.paused.reason, boot.paused.resumeAt);
|
|
3935
|
+
}
|
|
3926
3936
|
|
|
3927
|
-
|
|
3928
|
-
|
|
3929
|
-
|
|
3930
|
-
|
|
3931
|
-
|
|
3932
|
-
|
|
3937
|
+
// Self-heal stale needs_review flags using the current verifier (see
|
|
3938
|
+
// reverifyNeedsReview). Runs once on boot so a shipped verifier fix clears
|
|
3939
|
+
// its own historical false positives without manual retagging.
|
|
3940
|
+
await reverifyNeedsReview().catch((e) => {
|
|
3941
|
+
console.error(`[scheduler] boot reverify failed: ${e?.message ?? e}`);
|
|
3942
|
+
});
|
|
3933
3943
|
|
|
3934
|
-
|
|
3944
|
+
await rescheduleTimer();
|
|
3945
|
+
} catch (e) {
|
|
3946
|
+
console.error('[scheduler] boot phase failed — starting timers anyway:', e?.message);
|
|
3947
|
+
try {
|
|
3948
|
+
require('./logs.cjs').writeLine({
|
|
3949
|
+
scope: 'scheduler',
|
|
3950
|
+
level: 'error',
|
|
3951
|
+
message: 'boot phase failed; timers started anyway',
|
|
3952
|
+
meta: { error: e?.message },
|
|
3953
|
+
});
|
|
3954
|
+
} catch { /* logging must never be the thing that stops the scheduler */ }
|
|
3955
|
+
}
|
|
3935
3956
|
// Refresh next-reset every 10 minutes — billing window can shift if usage
|
|
3936
3957
|
// resets early or the auth token rotates. Tracked so re-init doesn't leak.
|
|
3937
3958
|
if (rescheduleInterval) clearInterval(rescheduleInterval);
|