claude-code-session-manager 0.39.2 → 0.39.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -85,8 +85,13 @@ const queueOps = require('./queueOps.cjs');
85
85
  // match ROOT/QUEUE_PATH below since both resolve the same ~/.claude/session-manager
86
86
  // home-dir layout.
87
87
  const { sweep: sweepFeedback } = require('../../scripts/lib/watchdogHelpers.cjs');
88
- const { resolvePrdsDirs, resolvePrdWriteDir } = require('./lib/prdLocations.cjs');
89
- const { migratePrds } = require('./lib/prdMigration.cjs');
88
+ const { resolvePrdsDirs, resolvePrdWriteDir, listEpicPrdDirs } = require('./lib/prdLocations.cjs');
89
+ const { ensureEpic, appendPrdCreatedEvent } = require('./lib/epicMint.cjs');
90
+ const sessionSlots = require('./lib/sessionSlots.cjs');
91
+ const queueStore = require('./lib/queueStore.cjs');
92
+ const { splitFrontmatter } = require('./lib/prdFrontmatter.cjs');
93
+ const { migratePrds, consolidateFlatPrds } = require('./lib/prdMigration.cjs');
94
+ const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
90
95
 
91
96
  // Captured once at module load so every run's meta sidecar can record how
92
97
  // stale the running process is relative to on-disk source (incident: PRD
@@ -469,6 +474,44 @@ function prdPathForJob(job) {
469
474
  return path.join(prdDirForCwd(job && job.cwd), `${job && job.slug}.md`);
470
475
  }
471
476
 
477
+ /** Absolute path to the sibling `prds-archived/<slug>.md` twin of a job's PRD. */
478
+ function archivedPrdPathForJob(job) {
479
+ return path.join(prdDirForCwd(job && job.cwd), '..', 'prds-archived', `${job && job.slug}.md`);
480
+ }
481
+
482
+ /**
483
+ * True if a job's PRD has already been archived (sibling `prds-archived/<slug>.md`
484
+ * exists). A queue entry whose PRD moved there is stale — the work already shipped
485
+ * — not a genuine missing-PRD failure.
486
+ */
487
+ async function archivedTwinExists(job) {
488
+ try {
489
+ await fsp.access(archivedPrdPathForJob(job));
490
+ return true;
491
+ } catch {
492
+ return false;
493
+ }
494
+ }
495
+
496
+ /**
497
+ * Build the non-failure result + run meta for a job whose PRD has already
498
+ * been archived (work shipped, queue entry is stale). Shared by both
499
+ * PRD-read failure exits in executeJob so the stale-skip logic isn't
500
+ * duplicated.
501
+ */
502
+ function prdArchivedSkipResult(job, cwd, sessionId, startedAt, safeLog, closeFd, metaPath) {
503
+ const archivedTwin = archivedPrdPathForJob(job);
504
+ const msg = `PRD already archived (${archivedTwin}) — work shipped; retiring stale queue entry`;
505
+ safeLog(`[scheduler] ${msg}\n`);
506
+ closeFd();
507
+ const finishedAt = Date.now();
508
+ config.writeJsonSync(metaPath, {
509
+ slug: job.slug, cwd, sessionId, exitCode: 0, skipped: 'prd-archived',
510
+ note: msg, startedAt, finishedAt, durationMs: 0,
511
+ });
512
+ return { exitCode: 0, durationMs: 0, skipped: 'prd-archived', note: msg, sessionId };
513
+ }
514
+
472
515
  /**
473
516
  * Search every candidate PRD dir for `<slug>.md` (legacy dir first, then
474
517
  * each active project's dir). Returns the containing dir, or null if the
@@ -538,6 +581,34 @@ async function archiveCompletedPrd(slug, cwd) {
538
581
  }
539
582
  }
540
583
 
584
+ /**
585
+ * Mark any still-runnable (pending/running) queue job for the given slugs as
586
+ * completed. Called after a PRD's .md is manually archived (queueOps.cjs's
587
+ * `schedule:archive-prd`) so a stale queue entry can never survive to fire
588
+ * against a PRD that no longer exists in the live prds/ dir — the same
589
+ * ENOENT-avoidance archivedTwinExists provides in executeJob, applied at the
590
+ * archiving source instead of at fire-time. auto-archived slugs never need
591
+ * this (selectAutoArchivable in queueOps.cjs only selects already-completed
592
+ * jobs), so this is exercised only by the manual archive path.
593
+ */
594
+ async function retireCompletedSlugs(slugs) {
595
+ const list = Array.isArray(slugs) ? slugs.filter(Boolean) : [];
596
+ if (list.length === 0) return;
597
+ const slugSet = new Set(list);
598
+ await mutate((s) => {
599
+ for (const j of s.jobs) {
600
+ if (!j || !slugSet.has(j.slug)) continue;
601
+ if (j.status !== 'pending' && j.status !== 'running') continue;
602
+ j.status = 'completed';
603
+ j.finishedAt = new Date().toISOString();
604
+ j.exitCode = 0;
605
+ j.error = null;
606
+ delete j.runtime;
607
+ }
608
+ });
609
+ await broadcast({ flush: true });
610
+ }
611
+
541
612
  // Bundled authoring guide seeded into the scheduler dir so the session-manager-dev
542
613
  // plugin's /develop and /prd skills — which reference this stable `~`-absolute
543
614
  // path — work on any user's machine, not just the author's.
@@ -570,7 +641,7 @@ async function runPrdMigration() {
570
641
  result = await migratePrds(PRDS_DIR);
571
642
  } catch (e) {
572
643
  logs.writeLine({ level: 'error', scope: 'scheduler', message: 'PRD migration failed', meta: { error: e?.message } });
573
- return;
644
+ return null;
574
645
  }
575
646
  console.log(`[scheduler] PRD migration: moved ${result.moved}, skipped ${result.skipped}`);
576
647
  if (result.unresolved.length > 0) {
@@ -584,6 +655,30 @@ async function runPrdMigration() {
584
655
  console.warn(`[scheduler] PRD migration: left ${u.file} in legacy dir (${u.reason})`);
585
656
  }
586
657
  }
658
+
659
+ // Phase 2 (2026-07-31 domain-model decision): the flat per-project
660
+ // `scheduler/prds/` dir is retired — new PRDs are epic-scoped, and anything
661
+ // still sitting flat consolidates into `prds-archived/` for later special
662
+ // processing. Queue rows for moved files are reaped by the archived-twin
663
+ // retirement. Idempotent per project; failures are logged, never fatal.
664
+ for (const cwd of allProjectCwds()) {
665
+ try {
666
+ const c = await consolidateFlatPrds(cwd);
667
+ if (c.moved > 0) {
668
+ console.log(`[scheduler] flat-PRD consolidation: archived ${c.moved} file(s) in ${cwd}`);
669
+ }
670
+ for (const f of c.failed) {
671
+ logs.writeLine({
672
+ level: 'warn', scope: 'scheduler',
673
+ message: `flat-PRD consolidation: could not archive ${f.file}`,
674
+ meta: { cwd, reason: f.reason },
675
+ });
676
+ }
677
+ } catch (e) {
678
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'flat-PRD consolidation failed', meta: { cwd, error: e?.message } });
679
+ }
680
+ }
681
+ return result;
587
682
  }
588
683
 
589
684
  // Matches only numbered timestamp backups (queue.json.bak-<epoch>), not the
@@ -698,74 +793,62 @@ function appendHeartbeat(entry) {
698
793
  // So: a MISSING file is a legitimately empty queue (first boot). A file that
699
794
  // exists but won't read or parse is `unreadable` — a poison state that must
700
795
  // never reach reconcile() or writeQueue(). Callers get the flag, not a lie.
701
- const EMPTY_QUEUE = () => ({
702
- config: { ...DEFAULT_CONFIG }, jobs: [], scheduledFor: null, lastRunAt: null, paused: null,
703
- });
704
-
705
- function shapeQueue(raw) {
706
- const data = JSON.parse(raw);
707
- return {
708
- config: { ...DEFAULT_CONFIG, ...(data.config || {}) },
709
- jobs: Array.isArray(data.jobs) ? data.jobs : [],
710
- scheduledFor: data.scheduledFor ?? null,
711
- lastRunAt: data.lastRunAt ?? null,
712
- paused: data.paused ?? null,
713
- };
714
- }
796
+ // (The single-file EMPTY_QUEUE/shapeQueue readers were retired with the
797
+ // global queue.json — queueStore.cjs's merged readers own the shape now.)
715
798
 
716
- // Quarantine a corrupt queue.json alongside itself (once per process — the
717
- // first copy is the one that matters; later ticks would just overwrite it
718
- // with the same bytes) so a human can diff it against the .bak-* snapshots.
799
+ // Quarantine a corrupt shard alongside itself (once per process — the first
800
+ // copy is the one that matters; later ticks would just overwrite it with the
801
+ // same bytes) so a human can diff it against the .bak-* snapshots.
719
802
  let quarantined = false;
720
- function unreadableQueue(e) {
721
- const state = EMPTY_QUEUE();
722
- state.unreadable = e?.message ?? 'queue.json read failed';
723
- if (!quarantined) {
803
+ function flagUnreadable(state) {
804
+ if (!state.unreadable) return state;
805
+ if (!quarantined && state.unreadablePath) {
724
806
  quarantined = true;
725
807
  try {
726
- fs.copyFileSync(QUEUE_PATH, `${QUEUE_PATH}.corrupt-${Date.now()}`);
808
+ fs.copyFileSync(state.unreadablePath, `${state.unreadablePath}.corrupt-${Date.now()}`);
727
809
  } catch { /* best-effort: the read already failed, the copy may too */ }
728
810
  }
729
- console.error(`[scheduler] queue.json unreadable — refusing to treat as empty: ${state.unreadable}`);
811
+ console.error(`[scheduler] queue state unreadable — refusing to treat as empty: ${state.unreadable}`);
730
812
  logs.writeLine({
731
813
  level: 'error', scope: 'scheduler',
732
- message: 'queue.json unreadable — scheduling halted until it reads clean',
733
- meta: { path: QUEUE_PATH, error: state.unreadable },
814
+ message: 'queue state unreadable — scheduling halted until it reads clean',
815
+ meta: { path: state.unreadablePath, error: state.unreadable },
734
816
  });
735
817
  return state;
736
818
  }
737
819
 
820
+ // Storage is FEDERATED (lib/queueStore.cjs, 2026-07-31): per-project job
821
+ // shards under `<cwd>/session-manager-operations/scheduler/state/queue.json`
822
+ // plus one machine-runtime file (config/paused/lastRunAt). Reads merge every
823
+ // shard into the single state object all downstream logic already expects;
824
+ // writes split it back. The old global scheduled-plans/queue.json is retired
825
+ // (split at boot by queueStore.migrateLegacyGlobalQueue).
826
+
738
827
  // Sync queue read — passed to the supervisor module (which calls it from
739
828
  // supervisorTick / applyAction with no await) and the heartbeat interval.
740
829
  // IPC handlers and mutate() use readQueue (async) below.
741
830
  function readQueueSync() {
742
- try {
743
- return shapeQueue(fs.readFileSync(QUEUE_PATH, 'utf8'));
744
- } catch (e) {
745
- if (e?.code === 'ENOENT') return EMPTY_QUEUE();
746
- return unreadableQueue(e);
747
- }
831
+ const s = queueStore.readMergedSync();
832
+ s.config = { ...DEFAULT_CONFIG, ...(s.config || {}) };
833
+ return flagUnreadable(s);
748
834
  }
749
835
 
750
836
  // Async queue read — used on all IPC hot paths. Reading queue.json sync was
751
- // blocking the main thread inside ipcMain.handle callbacks; awaiting fsp.readFile
752
- // hands control back to the renderer while the kernel paginates the file.
837
+ // blocking the main thread inside ipcMain.handle callbacks; awaiting the
838
+ // shard reads hands control back to the renderer between files.
753
839
  async function readQueue() {
754
- try {
755
- return shapeQueue(await fsp.readFile(QUEUE_PATH, 'utf8'));
756
- } catch (e) {
757
- if (e?.code === 'ENOENT') return EMPTY_QUEUE();
758
- return unreadableQueue(e);
759
- }
840
+ const s = await queueStore.readMerged();
841
+ s.config = { ...DEFAULT_CONFIG, ...(s.config || {}) };
842
+ return flagUnreadable(s);
760
843
  }
761
844
 
762
845
  async function writeQueue(state) {
763
846
  // Last line of defence: never persist a state derived from a failed read.
764
847
  if (state && state.unreadable) {
765
- throw new Error(`refusing to write queue.json from an unreadable read (${state.unreadable})`);
848
+ throw new Error(`refusing to write queue state from an unreadable read (${state.unreadable})`);
766
849
  }
767
850
  ensureDirs();
768
- await config.writeJson(QUEUE_PATH, state);
851
+ await queueStore.writeSplit(state, state.config?.defaultCwd ?? DEFAULT_PROJECT_CWD);
769
852
  }
770
853
 
771
854
  // ---------- serialized mutation queue ----------
@@ -1219,7 +1302,7 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
1219
1302
  },
1220
1303
  };
1221
1304
  if (withPaths) {
1222
- payload.paths = { root: ROOT, prds: PRDS_DIR, runs: RUNS_DIR, queue: QUEUE_PATH };
1305
+ payload.paths = { root: ROOT, prds: PRDS_DIR, runs: RUNS_DIR, queue: queueStore.MACHINE_STATE_PATH };
1223
1306
  }
1224
1307
  return payload;
1225
1308
  }
@@ -1713,11 +1796,17 @@ async function executeJob(job, runDir, defaultCwd, onPid) {
1713
1796
  prompt = parsed.body + FINISH_PROTOCOL;
1714
1797
  prdPath = fallbackPath;
1715
1798
  } catch (e2) {
1799
+ if (await archivedTwinExists(job)) {
1800
+ return prdArchivedSkipResult(job, cwd, sessionId, startedAt, safeLog, closeFd, metaPath);
1801
+ }
1716
1802
  safeLog(`[scheduler] failed to read PRD: ${e2?.message}\n`);
1717
1803
  closeFd();
1718
1804
  return { exitCode: -1, durationMs: 0, error: e2?.message };
1719
1805
  }
1720
1806
  } else {
1807
+ if (await archivedTwinExists(job)) {
1808
+ return prdArchivedSkipResult(job, cwd, sessionId, startedAt, safeLog, closeFd, metaPath);
1809
+ }
1721
1810
  safeLog(`[scheduler] failed to read PRD: ${e?.message}\n`);
1722
1811
  closeFd();
1723
1812
  return { exitCode: -1, durationMs: 0, error: e?.message };
@@ -2252,6 +2341,14 @@ async function spawnInvestigation(failedJob, runDir) {
2252
2341
  }
2253
2342
 
2254
2343
  async function spawnJob(job, runId, runDir, defaultCwd) {
2344
+ // Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
2345
+ // — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
2346
+ // leaves the job pending; the next tick retries when a slot frees up.
2347
+ const slotToken = sessionSlots.acquire(`scheduler:${job.slug}`);
2348
+ if (!slotToken) {
2349
+ console.log(`[scheduler] no session slot free for ${job.slug} — deferring (${JSON.stringify(sessionSlots.snapshot().holders.map((h) => h.owner))})`);
2350
+ return;
2351
+ }
2255
2352
  runningSet.add(job.slug);
2256
2353
  try {
2257
2354
  await mutate((s) => {
@@ -2286,6 +2383,25 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2286
2383
  await setPaused('rate_limit', resetIso);
2287
2384
  }
2288
2385
 
2386
+ // Stale queue entry: the PRD already shipped and was archived before this
2387
+ // run fired (see archivedTwinExists in executeJob). Treat it as a plain
2388
+ // completion — no verify pass, no commit guard, no RCA feedback — since
2389
+ // there is no real transcript/commit to check.
2390
+ if (res.skipped === 'prd-archived') {
2391
+ await mutate((s) => {
2392
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
2393
+ if (idx >= 0) {
2394
+ s.jobs[idx].status = 'completed';
2395
+ s.jobs[idx].finishedAt = new Date().toISOString();
2396
+ s.jobs[idx].exitCode = 0;
2397
+ s.jobs[idx].error = null;
2398
+ delete s.jobs[idx].runtime;
2399
+ }
2400
+ });
2401
+ await broadcast({ flush: true });
2402
+ return;
2403
+ }
2404
+
2289
2405
  // Post-run verification: for exit=0 runs, scan the transcript and check
2290
2406
  // dependency prerequisites before stamping 'completed'. This catches the
2291
2407
  // false-positive class where an agent exits cleanly while leaving failures
@@ -2665,6 +2781,8 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2665
2781
  console.error('[scheduler] spawnJob error', job.slug, e);
2666
2782
  } finally {
2667
2783
  runningSet.delete(job.slug);
2784
+ // Slot release notifies subscribed pumps (chat lane) machine-wide.
2785
+ sessionSlots.release(slotToken);
2668
2786
  // Each job completion is a signal to advance the queue.
2669
2787
  tickQueue().catch(() => {});
2670
2788
  }
@@ -2722,7 +2840,18 @@ function tickQueue() {
2722
2840
  // job is never started into the host's own headroom (that path OOM-kills
2723
2841
  // Electron and SIGHUPs every pty — 2026-06-16 incident).
2724
2842
  const jobBudgetMb = availableForJobs(availableMb, RESERVED_HOST_MB);
2725
- const allowed = memoryLimitedBatchSize(jobBudgetMb, MIN_FREE_MB_PER_JOB, runningSet.size, batch.length);
2843
+ // Session-Manager's machine-wide slot pool is the outer bound: chat runs
2844
+ // and scheduler jobs share it, so a busy chat lane shrinks this batch.
2845
+ const slotAllowed = sessionSlots.available();
2846
+ if (slotAllowed === 0) {
2847
+ const snap = sessionSlots.snapshot();
2848
+ console.log(`[scheduler] slot gate: 0 of ${snap.total} session slots free (${snap.holders.map((h) => h.owner).join(', ')}) — deferring ${batch.length} job(s)`);
2849
+ return { fired: false, reason: 'slots-exhausted', deferredCount: batch.length, holders: snap.holders };
2850
+ }
2851
+ const allowed = Math.min(
2852
+ slotAllowed,
2853
+ memoryLimitedBatchSize(jobBudgetMb, MIN_FREE_MB_PER_JOB, runningSet.size, batch.length),
2854
+ );
2726
2855
  if (allowed === 0) {
2727
2856
  const threshold = RESERVED_HOST_MB + MIN_FREE_MB_PER_JOB * (runningSet.size + 1);
2728
2857
  console.log(`[scheduler] memory gate: available=${availableMb} MB < threshold=${threshold} MB (host reserve ${RESERVED_HOST_MB} + ${MIN_FREE_MB_PER_JOB}/job × ${runningSet.size + 1}) — deferring ${batch.length} job(s)`);
@@ -2775,6 +2904,8 @@ function forceTickOutcome(result) {
2775
2904
  return { ok: true, kind: 'warn', message: 'Batch cancelled — try again' };
2776
2905
  case 'memory-deferred':
2777
2906
  return { ok: true, kind: 'warn', message: `Deferred ${result.deferredCount} job(s) — low memory (${result.availableMb} MB available, need ${result.threshold} MB)` };
2907
+ case 'slots-exhausted':
2908
+ return { ok: true, kind: 'warn', message: `Deferred ${result.deferredCount} job(s) — all session slots in use (${(result.holders ?? []).map((h) => h.owner).join(', ')})` };
2778
2909
  case 'held': {
2779
2910
  const detail = String(result.detail ?? '').replace(/^\[scheduler\]\s*[\w-]+\s*(?:\[[^\]]*\])?:\s*/, '');
2780
2911
  return { ok: true, kind: 'warn', message: detail || 'Batch held' };
@@ -3367,6 +3498,11 @@ function registerScheduleHandlers() {
3367
3498
  return buildScheduleStatePayload(state, { withPaths: true });
3368
3499
  });
3369
3500
 
3501
+ // Session-Manager-wide claude -p slot pool (lib/sessionSlots.cjs) —
3502
+ // read-only diagnostic surface for the Home widget and the global
3503
+ // configuration tab.
3504
+ ipcMain.handle('schedule:session-slots', () => sessionSlots.snapshot());
3505
+
3370
3506
  ipcMain.handle('schedule:health', async () => {
3371
3507
  const state = await readQueue();
3372
3508
  const runningJobs = [];
@@ -3623,6 +3759,19 @@ function registerScheduleHandlers() {
3623
3759
 
3624
3760
  async function init() {
3625
3761
  ensureDirs();
3762
+ // A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
3763
+ // batch — advance the queue without waiting for the next 60s poll.
3764
+ sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
3765
+ // Retire the global queue.json: split its rows into per-project shards
3766
+ // BEFORE the first read below, so boot reconciliation sees the shards.
3767
+ try {
3768
+ const m = await queueStore.migrateLegacyGlobalQueue(DEFAULT_PROJECT_CWD);
3769
+ if (m.migrated) {
3770
+ console.log(`[scheduler] legacy global queue retired: ${m.moved} row(s) split across ${m.projects} project shard(s)`);
3771
+ }
3772
+ } catch (e) {
3773
+ console.error('[scheduler] legacy queue split failed', e?.message);
3774
+ }
3626
3775
  await runPrdMigration();
3627
3776
  sweepQueueBackups().catch((e) => console.warn('[scheduler] backup sweep failed', e?.message));
3628
3777
 
@@ -3822,7 +3971,19 @@ const remote = {
3822
3971
  // nothing for findPrdDir to search for); the renderer's slug-only IPC
3823
3972
  // path (editing an already-queued PRD) omits it and relies on findPrdDir.
3824
3973
  async readPrd(slug, cwd) {
3825
- const dir = cwd ? prdDirForCwd(cwd) : await findPrdDir(slug);
3974
+ let dir;
3975
+ if (cwd) {
3976
+ // The slug may live in the legacy flat dir or any Epic's prds/ under
3977
+ // this project; probe local dirs, defaulting to the flat dir (callers
3978
+ // use a miss there as the "doesn't exist yet" signal on create).
3979
+ const localDirs = [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)];
3980
+ dir = localDirs.find((d) => {
3981
+ const p = safeSlugPathIn(d, slug);
3982
+ return p && fs.existsSync(p);
3983
+ }) ?? prdDirForCwd(cwd);
3984
+ } else {
3985
+ dir = await findPrdDir(slug);
3986
+ }
3826
3987
  if (!dir) return { ok: false, error: 'invalid slug' };
3827
3988
  const filePath = safeSlugPathIn(dir, slug);
3828
3989
  if (!filePath) return { ok: false, error: 'invalid slug' };
@@ -3864,8 +4025,35 @@ const remote = {
3864
4025
  // doesn't exist yet, so findPrdDir would return nothing to write into).
3865
4026
  async writePrd(slug, body, cwd) {
3866
4027
  let dir;
4028
+ let epicTrace = null;
3867
4029
  if (cwd) {
3868
- dir = prdDirForCwd(cwd);
4030
+ // Edit-in-place if this slug already lives anywhere under this project
4031
+ // (legacy flat dir or any Epic's prds/); otherwise this is a CREATE,
4032
+ // and every new PRD belongs to an Epic (CLAUDE.md domain model) — mint
4033
+ // one from the body's frontmatter title/tag.
4034
+ const localDirs = [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)];
4035
+ for (const d of localDirs) {
4036
+ const candidate = safeSlugPathIn(d, slug);
4037
+ if (candidate && fs.existsSync(candidate)) { dir = d; break; }
4038
+ }
4039
+ if (!dir) {
4040
+ try {
4041
+ const { fm } = splitFrontmatter(body);
4042
+ const epic = ensureEpic(cwd, {
4043
+ goalText: fm.title || slug,
4044
+ tag: fm.tag,
4045
+ // An Epic-conversation dispatch already has its Epic — join it.
4046
+ epicId: fm.sourcePromptId,
4047
+ });
4048
+ dir = epic.prdDir;
4049
+ epicTrace = epic.epicId;
4050
+ } catch (e) {
4051
+ // Epic mint must never block a PRD write — fall back to the
4052
+ // legacy flat dir and log loudly.
4053
+ console.error(`[scheduler] ensureEpic failed for ${slug}: ${e?.message}`);
4054
+ dir = prdDirForCwd(cwd);
4055
+ }
4056
+ }
3869
4057
  await fsp.mkdir(dir, { recursive: true });
3870
4058
  } else {
3871
4059
  dir = (await findPrdDir(slug)) ?? PRDS_DIR;
@@ -3890,6 +4078,10 @@ const remote = {
3890
4078
  }
3891
4079
  await config.writeTextAtomic(resolved, body);
3892
4080
  const stat = await fsp.stat(resolved);
4081
+ if (epicTrace) {
4082
+ // Best-effort: record the dispatch on the minted Epic's event chain.
4083
+ try { appendPrdCreatedEvent(cwd, epicTrace, slug); } catch { /* trace only */ }
4084
+ }
3893
4085
  return { ok: true, bytesWritten: stat.size };
3894
4086
  } catch (e) {
3895
4087
  return { ok: false, error: e?.message ?? 'write failed' };
@@ -3986,4 +4178,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
3986
4178
  });
3987
4179
  }
3988
4180
 
3989
- module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, writeQueue, reconcile, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, feedbackSweepDue, FEEDBACK_SWEEP_TICK_INTERVAL, sweepFeedback, registerAdminRoutes, notifyOriginatingTab, isNotifiableTerminalStatus, candidatePrdsDirs, prdDirForCwd, prdPathForJob, findPrdDir, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields };
4181
+ module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, writeQueue, reconcile, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, feedbackSweepDue, FEEDBACK_SWEEP_TICK_INTERVAL, sweepFeedback, registerAdminRoutes, notifyOriginatingTab, isNotifiableTerminalStatus, candidatePrdsDirs, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields };
@@ -1277,6 +1277,8 @@ export interface SessionManagerAPI {
1277
1277
  };
1278
1278
  schedule: {
1279
1279
  state: () => Promise<ScheduleStateSnapshot>;
1280
+ /** Machine-wide claude -p slot pool (sessionSlots.cjs): total/inUse/holders. */
1281
+ sessionSlots: () => Promise<{ total: number; inUse: number; holders: { owner: string; at: string }[] }>;
1280
1282
  setConfig: (partial: Partial<ScheduleConfig & { supervisor?: Partial<SupervisorConfig> }>) => Promise<{ ok: boolean; config: ScheduleConfig }>;
1281
1283
  resetJob: (slug: string) => Promise<{ ok: boolean; error?: string }>;
1282
1284
  runNow: () => Promise<{ ok: boolean }>;
@@ -247,6 +247,7 @@ contextBridge.exposeInMainWorld('api', {
247
247
  },
248
248
  schedule: {
249
249
  state: () => ipcRenderer.invoke('schedule:state'),
250
+ sessionSlots: () => ipcRenderer.invoke('schedule:session-slots'),
250
251
  setConfig: (partial) => ipcRenderer.invoke('schedule:set-config', partial),
251
252
  resetJob: (slug) => ipcRenderer.invoke('schedule:reset-job', { slug }),
252
253
  runNow: () => ipcRenderer.invoke('schedule:run-now'),