claude-code-session-manager 0.39.3 → 0.40.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/dist/assets/{TiptapBody-B90xy18x.js → TiptapBody-CNr1qGoi.js} +1 -1
  2. package/dist/assets/{index-DVlD8N1X.css → index-CR-WdMgm.css} +1 -1
  3. package/dist/assets/index-oMNYp73N.js +3194 -0
  4. package/dist/index.html +2 -2
  5. package/package.json +1 -1
  6. package/plugins/session-manager-dev/skills/develop/SKILL.md +34 -27
  7. package/scripts/lib/watchdogHelpers.cjs +57 -21
  8. package/src/main/__tests__/health-prd-migration.test.cjs +37 -0
  9. package/src/main/__tests__/prdCreate.test.cjs +54 -15
  10. package/src/main/__tests__/prdSourcePromptIdBackfill.test.cjs +118 -0
  11. package/src/main/__tests__/queueHistory.test.cjs +9 -4
  12. package/src/main/__tests__/queueOpsAutoArchive.test.cjs +12 -0
  13. package/src/main/__tests__/scheduler-archived-twin-guard.test.cjs +92 -0
  14. package/src/main/__tests__/scheduler-unreadable-queue-guard.test.cjs +1 -1
  15. package/src/main/__tests__/uniquePrdNumbers.test.cjs +119 -0
  16. package/src/main/chatRunner.cjs +11 -1
  17. package/src/main/config.cjs +10 -0
  18. package/src/main/health.cjs +46 -10
  19. package/src/main/index.cjs +28 -9
  20. package/src/main/ipcSchemas.cjs +10 -0
  21. package/src/main/lib/__tests__/instanceLock.test.cjs +87 -0
  22. package/src/main/lib/__tests__/sessionSlots.test.cjs +55 -0
  23. package/src/main/lib/epicMint.cjs +144 -0
  24. package/src/main/lib/instanceLock.cjs +100 -0
  25. package/src/main/lib/prdCreate.cjs +19 -6
  26. package/src/main/lib/prdLocations.cjs +82 -1
  27. package/src/main/lib/prdMigration.cjs +47 -1
  28. package/src/main/lib/queueHistory.cjs +85 -37
  29. package/src/main/lib/queueStore.cjs +299 -0
  30. package/src/main/lib/schedulerBatch.cjs +36 -4
  31. package/src/main/lib/sessionSlots.cjs +85 -0
  32. package/src/main/pty.cjs +9 -0
  33. package/src/main/queueOps.cjs +20 -1
  34. package/src/main/scheduler/prdParser.cjs +28 -4
  35. package/src/main/scheduler.cjs +333 -52
  36. package/src/main/templates/PRD_AUTHORING.md +9 -5
  37. package/src/preload/api.d.ts +18 -3
  38. package/src/preload/index.cjs +2 -0
  39. package/dist/assets/index-BUuhV6vT.js +0 -3180
@@ -85,8 +85,31 @@ const queueOps = require('./queueOps.cjs');
85
85
  // match ROOT/QUEUE_PATH below since both resolve the same ~/.claude/session-manager
86
86
  // home-dir layout.
87
87
  const { sweep: sweepFeedback } = require('../../scripts/lib/watchdogHelpers.cjs');
88
- const { resolvePrdsDirs, resolvePrdWriteDir } = require('./lib/prdLocations.cjs');
89
- const { migratePrds } = require('./lib/prdMigration.cjs');
88
+ const { resolvePrdsDirs, resolvePrdWriteDir, listEpicPrdDirs } = require('./lib/prdLocations.cjs');
89
+ const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
90
+
91
+ // ---------- origin session resolution (PRD 832) ----------
92
+ // An Epic IS a tagged claude session — job rows carry the originating
93
+ // claudeSessionId alongside sourcePromptId so every PRD stays traceable to
94
+ // the session that spawned it. active-index.json is tiny; a short TTL cache
95
+ // keeps reconcile (every 60s, N jobs) at one read per project per pass.
96
+ const originIndexCache = new Map(); // cwd -> { at, sessions }
97
+ const ORIGIN_CACHE_TTL_MS = 30_000;
98
+ function resolveOriginSessionId(cwd, epicId) {
99
+ if (!cwd || !epicId) return null;
100
+ let entry = originIndexCache.get(cwd);
101
+ if (!entry || Date.now() - entry.at > ORIGIN_CACHE_TTL_MS) {
102
+ entry = { at: Date.now(), sessions: readActiveIndex(cwd).sessions };
103
+ originIndexCache.set(cwd, entry);
104
+ }
105
+ const session = entry.sessions[epicId];
106
+ return session && typeof session.claudeSessionId === 'string' ? session.claudeSessionId : null;
107
+ }
108
+ const sessionSlots = require('./lib/sessionSlots.cjs');
109
+ const queueStore = require('./lib/queueStore.cjs');
110
+ const { splitFrontmatter } = require('./lib/prdFrontmatter.cjs');
111
+ const { migratePrds, consolidateFlatPrds } = require('./lib/prdMigration.cjs');
112
+ const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
90
113
 
91
114
  // Captured once at module load so every run's meta sidecar can record how
92
115
  // stale the running process is relative to on-disk source (incident: PRD
@@ -469,6 +492,44 @@ function prdPathForJob(job) {
469
492
  return path.join(prdDirForCwd(job && job.cwd), `${job && job.slug}.md`);
470
493
  }
471
494
 
495
+ /** Absolute path to the sibling `prds-archived/<slug>.md` twin of a job's PRD. */
496
+ function archivedPrdPathForJob(job) {
497
+ return path.join(prdDirForCwd(job && job.cwd), '..', 'prds-archived', `${job && job.slug}.md`);
498
+ }
499
+
500
+ /**
501
+ * True if a job's PRD has already been archived (sibling `prds-archived/<slug>.md`
502
+ * exists). A queue entry whose PRD moved there is stale — the work already shipped
503
+ * — not a genuine missing-PRD failure.
504
+ */
505
+ async function archivedTwinExists(job) {
506
+ try {
507
+ await fsp.access(archivedPrdPathForJob(job));
508
+ return true;
509
+ } catch {
510
+ return false;
511
+ }
512
+ }
513
+
514
+ /**
515
+ * Build the non-failure result + run meta for a job whose PRD has already
516
+ * been archived (work shipped, queue entry is stale). Shared by both
517
+ * PRD-read failure exits in executeJob so the stale-skip logic isn't
518
+ * duplicated.
519
+ */
520
+ function prdArchivedSkipResult(job, cwd, sessionId, startedAt, safeLog, closeFd, metaPath) {
521
+ const archivedTwin = archivedPrdPathForJob(job);
522
+ const msg = `PRD already archived (${archivedTwin}) — work shipped; retiring stale queue entry`;
523
+ safeLog(`[scheduler] ${msg}\n`);
524
+ closeFd();
525
+ const finishedAt = Date.now();
526
+ config.writeJsonSync(metaPath, {
527
+ slug: job.slug, cwd, sessionId, exitCode: 0, skipped: 'prd-archived',
528
+ note: msg, startedAt, finishedAt, durationMs: 0,
529
+ });
530
+ return { exitCode: 0, durationMs: 0, skipped: 'prd-archived', note: msg, sessionId };
531
+ }
532
+
472
533
  /**
473
534
  * Search every candidate PRD dir for `<slug>.md` (legacy dir first, then
474
535
  * each active project's dir). Returns the containing dir, or null if the
@@ -538,6 +599,34 @@ async function archiveCompletedPrd(slug, cwd) {
538
599
  }
539
600
  }
540
601
 
602
+ /**
603
+ * Mark any still-runnable (pending/running) queue job for the given slugs as
604
+ * completed. Called after a PRD's .md is manually archived (queueOps.cjs's
605
+ * `schedule:archive-prd`) so a stale queue entry can never survive to fire
606
+ * against a PRD that no longer exists in the live prds/ dir — the same
607
+ * ENOENT-avoidance archivedTwinExists provides in executeJob, applied at the
608
+ * archiving source instead of at fire-time. auto-archived slugs never need
609
+ * this (selectAutoArchivable in queueOps.cjs only selects already-completed
610
+ * jobs), so this is exercised only by the manual archive path.
611
+ */
612
+ async function retireCompletedSlugs(slugs) {
613
+ const list = Array.isArray(slugs) ? slugs.filter(Boolean) : [];
614
+ if (list.length === 0) return;
615
+ const slugSet = new Set(list);
616
+ await mutate((s) => {
617
+ for (const j of s.jobs) {
618
+ if (!j || !slugSet.has(j.slug)) continue;
619
+ if (j.status !== 'pending' && j.status !== 'running') continue;
620
+ j.status = 'completed';
621
+ j.finishedAt = new Date().toISOString();
622
+ j.exitCode = 0;
623
+ j.error = null;
624
+ delete j.runtime;
625
+ }
626
+ });
627
+ await broadcast({ flush: true });
628
+ }
629
+
541
630
  // Bundled authoring guide seeded into the scheduler dir so the session-manager-dev
542
631
  // plugin's /develop and /prd skills — which reference this stable `~`-absolute
543
632
  // path — work on any user's machine, not just the author's.
@@ -570,7 +659,7 @@ async function runPrdMigration() {
570
659
  result = await migratePrds(PRDS_DIR);
571
660
  } catch (e) {
572
661
  logs.writeLine({ level: 'error', scope: 'scheduler', message: 'PRD migration failed', meta: { error: e?.message } });
573
- return;
662
+ return null;
574
663
  }
575
664
  console.log(`[scheduler] PRD migration: moved ${result.moved}, skipped ${result.skipped}`);
576
665
  if (result.unresolved.length > 0) {
@@ -584,6 +673,30 @@ async function runPrdMigration() {
584
673
  console.warn(`[scheduler] PRD migration: left ${u.file} in legacy dir (${u.reason})`);
585
674
  }
586
675
  }
676
+
677
+ // Phase 2 (2026-07-31 domain-model decision): the flat per-project
678
+ // `scheduler/prds/` dir is retired — new PRDs are epic-scoped, and anything
679
+ // still sitting flat consolidates into `prds-archived/` for later special
680
+ // processing. Queue rows for moved files are reaped by the archived-twin
681
+ // retirement. Idempotent per project; failures are logged, never fatal.
682
+ for (const cwd of allProjectCwds()) {
683
+ try {
684
+ const c = await consolidateFlatPrds(cwd);
685
+ if (c.moved > 0) {
686
+ console.log(`[scheduler] flat-PRD consolidation: archived ${c.moved} file(s) in ${cwd}`);
687
+ }
688
+ for (const f of c.failed) {
689
+ logs.writeLine({
690
+ level: 'warn', scope: 'scheduler',
691
+ message: `flat-PRD consolidation: could not archive ${f.file}`,
692
+ meta: { cwd, reason: f.reason },
693
+ });
694
+ }
695
+ } catch (e) {
696
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'flat-PRD consolidation failed', meta: { cwd, error: e?.message } });
697
+ }
698
+ }
699
+ return result;
587
700
  }
588
701
 
589
702
  // Matches only numbered timestamp backups (queue.json.bak-<epoch>), not the
@@ -698,74 +811,62 @@ function appendHeartbeat(entry) {
698
811
  // So: a MISSING file is a legitimately empty queue (first boot). A file that
699
812
  // exists but won't read or parse is `unreadable` — a poison state that must
700
813
  // never reach reconcile() or writeQueue(). Callers get the flag, not a lie.
701
- const EMPTY_QUEUE = () => ({
702
- config: { ...DEFAULT_CONFIG }, jobs: [], scheduledFor: null, lastRunAt: null, paused: null,
703
- });
704
-
705
- function shapeQueue(raw) {
706
- const data = JSON.parse(raw);
707
- return {
708
- config: { ...DEFAULT_CONFIG, ...(data.config || {}) },
709
- jobs: Array.isArray(data.jobs) ? data.jobs : [],
710
- scheduledFor: data.scheduledFor ?? null,
711
- lastRunAt: data.lastRunAt ?? null,
712
- paused: data.paused ?? null,
713
- };
714
- }
814
+ // (The single-file EMPTY_QUEUE/shapeQueue readers were retired with the
815
+ // global queue.json — queueStore.cjs's merged readers own the shape now.)
715
816
 
716
- // Quarantine a corrupt queue.json alongside itself (once per process — the
717
- // first copy is the one that matters; later ticks would just overwrite it
718
- // with the same bytes) so a human can diff it against the .bak-* snapshots.
817
+ // Quarantine a corrupt shard alongside itself (once per process — the first
818
+ // copy is the one that matters; later ticks would just overwrite it with the
819
+ // same bytes) so a human can diff it against the .bak-* snapshots.
719
820
  let quarantined = false;
720
- function unreadableQueue(e) {
721
- const state = EMPTY_QUEUE();
722
- state.unreadable = e?.message ?? 'queue.json read failed';
723
- if (!quarantined) {
821
+ function flagUnreadable(state) {
822
+ if (!state.unreadable) return state;
823
+ if (!quarantined && state.unreadablePath) {
724
824
  quarantined = true;
725
825
  try {
726
- fs.copyFileSync(QUEUE_PATH, `${QUEUE_PATH}.corrupt-${Date.now()}`);
826
+ fs.copyFileSync(state.unreadablePath, `${state.unreadablePath}.corrupt-${Date.now()}`);
727
827
  } catch { /* best-effort: the read already failed, the copy may too */ }
728
828
  }
729
- console.error(`[scheduler] queue.json unreadable — refusing to treat as empty: ${state.unreadable}`);
829
+ console.error(`[scheduler] queue state unreadable — refusing to treat as empty: ${state.unreadable}`);
730
830
  logs.writeLine({
731
831
  level: 'error', scope: 'scheduler',
732
- message: 'queue.json unreadable — scheduling halted until it reads clean',
733
- meta: { path: QUEUE_PATH, error: state.unreadable },
832
+ message: 'queue state unreadable — scheduling halted until it reads clean',
833
+ meta: { path: state.unreadablePath, error: state.unreadable },
734
834
  });
735
835
  return state;
736
836
  }
737
837
 
838
+ // Storage is FEDERATED (lib/queueStore.cjs, 2026-07-31): per-project job
839
+ // shards under `<cwd>/session-manager-operations/scheduler/state/queue.json`
840
+ // plus one machine-runtime file (config/paused/lastRunAt). Reads merge every
841
+ // shard into the single state object all downstream logic already expects;
842
+ // writes split it back. The old global scheduled-plans/queue.json is retired
843
+ // (split at boot by queueStore.migrateLegacyGlobalQueue).
844
+
738
845
  // Sync queue read — passed to the supervisor module (which calls it from
739
846
  // supervisorTick / applyAction with no await) and the heartbeat interval.
740
847
  // IPC handlers and mutate() use readQueue (async) below.
741
848
  function readQueueSync() {
742
- try {
743
- return shapeQueue(fs.readFileSync(QUEUE_PATH, 'utf8'));
744
- } catch (e) {
745
- if (e?.code === 'ENOENT') return EMPTY_QUEUE();
746
- return unreadableQueue(e);
747
- }
849
+ const s = queueStore.readMergedSync();
850
+ s.config = { ...DEFAULT_CONFIG, ...(s.config || {}) };
851
+ return flagUnreadable(s);
748
852
  }
749
853
 
750
854
  // Async queue read — used on all IPC hot paths. Reading queue.json sync was
751
- // blocking the main thread inside ipcMain.handle callbacks; awaiting fsp.readFile
752
- // hands control back to the renderer while the kernel paginates the file.
855
+ // blocking the main thread inside ipcMain.handle callbacks; awaiting the
856
+ // shard reads hands control back to the renderer between files.
753
857
  async function readQueue() {
754
- try {
755
- return shapeQueue(await fsp.readFile(QUEUE_PATH, 'utf8'));
756
- } catch (e) {
757
- if (e?.code === 'ENOENT') return EMPTY_QUEUE();
758
- return unreadableQueue(e);
759
- }
858
+ const s = await queueStore.readMerged();
859
+ s.config = { ...DEFAULT_CONFIG, ...(s.config || {}) };
860
+ return flagUnreadable(s);
760
861
  }
761
862
 
762
863
  async function writeQueue(state) {
763
864
  // Last line of defence: never persist a state derived from a failed read.
764
865
  if (state && state.unreadable) {
765
- throw new Error(`refusing to write queue.json from an unreadable read (${state.unreadable})`);
866
+ throw new Error(`refusing to write queue state from an unreadable read (${state.unreadable})`);
766
867
  }
767
868
  ensureDirs();
768
- await config.writeJson(QUEUE_PATH, state);
869
+ await queueStore.writeSplit(state, state.config?.defaultCwd ?? DEFAULT_PROJECT_CWD);
769
870
  }
770
871
 
771
872
  // ---------- serialized mutation queue ----------
@@ -828,7 +929,23 @@ async function listPrdFiles() {
828
929
  async function allocateParallelGroup(cwd) {
829
930
  const dir = prdDirForCwd(cwd);
830
931
  await fsp.mkdir(dir, { recursive: true });
831
- return prdParser.allocateParallelGroup(dir);
932
+ // PRD 832: numbers are unique across the WHOLE project, not just the
933
+ // allocator's bookkeeping dir — scan every Epic prds/ dir plus the
934
+ // archive so a number used anywhere (even by a hand-authored or archived
935
+ // PRD) is never reissued. The reservation markers + high-water sidecar
936
+ // stay in `dir`; the cross-dir max only raises the floor.
937
+ const targetCwd = cwd || DEFAULT_PROJECT_CWD;
938
+ const extraDirs = [
939
+ ...listEpicPrdDirs(targetCwd),
940
+ path.join(prdDirForCwd(targetCwd), '..', 'prds-archived'),
941
+ ];
942
+ let extraFloor = 0;
943
+ for (const d of extraDirs) {
944
+ try {
945
+ extraFloor = Math.max(extraFloor, await prdParser.maxParallelGroupInUse(d));
946
+ } catch { /* missing dir — nothing allocated there */ }
947
+ }
948
+ return prdParser.allocateParallelGroup(dir, { extraFloor });
832
949
  }
833
950
 
834
951
  /**
@@ -908,6 +1025,20 @@ function validatePromptForSpawn(body, srcLabel) {
908
1025
 
909
1026
  // ---------- queue reconciliation ----------
910
1027
 
1028
+ /**
1029
+ * sourcePromptId backfill (PRD 830) is pending-only: a pending row always
1030
+ * takes the freshly-parsed value (explicit frontmatter, or the dir-derived
1031
+ * epic id parsePrd falls back to when frontmatter has none) since it hasn't
1032
+ * started executing yet. A running/completed row's sourcePromptId is left
1033
+ * exactly as it was minted at dispatch time — reconcile must not rewrite
1034
+ * linkage on work already in flight or finished.
1035
+ */
1036
+ function reconcileSourcePromptId(job, parsedSourcePromptId) {
1037
+ return job.status === 'pending'
1038
+ ? (parsedSourcePromptId ?? job.sourcePromptId ?? null)
1039
+ : job.sourcePromptId;
1040
+ }
1041
+
911
1042
  /**
912
1043
  * Walk prds/, ensure every .md has a queue entry. Drop entries whose .md
913
1044
  * is gone. Refresh title/cwd/parallelGroup from disk every reconcile so
@@ -950,6 +1081,16 @@ async function reconcile(state) {
950
1081
  // or mid-move. "I can't see it" is not "the user deleted it", so the
951
1082
  // row survives — worst case it re-resolves on the next pass.
952
1083
  if (job.status === 'pending' || job.status === 'running') {
1084
+ // Exception: a PENDING row whose PRD has an archived twin was
1085
+ // retired on purpose (work landed by other means — e.g. implemented
1086
+ // inline — and the source .md moved to prds-archived/). Keeping it
1087
+ // would show a phantom "scheduled" job forever; firing it would just
1088
+ // hit executeJob's archived-twin skip anyway. Running rows are left
1089
+ // alone — the reaper owns their lifecycle.
1090
+ if (job.status === 'pending' && (await archivedTwinExists(job))) {
1091
+ console.log(`[scheduler] reconcile: retiring pending job ${job.slug} — PRD already archived (work landed elsewhere)`);
1092
+ continue;
1093
+ }
953
1094
  seen.add(job.slug);
954
1095
  next.push({ ...job });
955
1096
  console.warn(`[scheduler] reconcile: keeping ${job.status} job ${job.slug} — PRD source not visible in any candidate dir`);
@@ -963,8 +1104,11 @@ async function reconcile(state) {
963
1104
  cwd: p.cwd,
964
1105
  parallelGroup: p.parallelGroup,
965
1106
  estimateMinutes: p.estimateMinutes,
966
- sourcePromptId: p.sourcePromptId,
1107
+ sourcePromptId: reconcileSourcePromptId(job, p.sourcePromptId),
967
1108
  sourceTabId: p.sourceTabId,
1109
+ dependsOn: p.dependsOn,
1110
+ originSessionId: job.originSessionId
1111
+ ?? resolveOriginSessionId(p.cwd, reconcileSourcePromptId(job, p.sourcePromptId)),
968
1112
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
969
1113
  });
970
1114
  }
@@ -1023,6 +1167,8 @@ async function reconcile(state) {
1023
1167
  estimateMinutes: p.estimateMinutes,
1024
1168
  sourcePromptId: p.sourcePromptId,
1025
1169
  sourceTabId: p.sourceTabId,
1170
+ dependsOn: p.dependsOn,
1171
+ originSessionId: resolveOriginSessionId(p.cwd, p.sourcePromptId),
1026
1172
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1027
1173
  status: 'pending',
1028
1174
  runId: null,
@@ -1219,7 +1365,7 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
1219
1365
  },
1220
1366
  };
1221
1367
  if (withPaths) {
1222
- payload.paths = { root: ROOT, prds: PRDS_DIR, runs: RUNS_DIR, queue: QUEUE_PATH };
1368
+ payload.paths = { root: ROOT, prds: PRDS_DIR, runs: RUNS_DIR, queue: queueStore.MACHINE_STATE_PATH };
1223
1369
  }
1224
1370
  return payload;
1225
1371
  }
@@ -1713,11 +1859,17 @@ async function executeJob(job, runDir, defaultCwd, onPid) {
1713
1859
  prompt = parsed.body + FINISH_PROTOCOL;
1714
1860
  prdPath = fallbackPath;
1715
1861
  } catch (e2) {
1862
+ if (await archivedTwinExists(job)) {
1863
+ return prdArchivedSkipResult(job, cwd, sessionId, startedAt, safeLog, closeFd, metaPath);
1864
+ }
1716
1865
  safeLog(`[scheduler] failed to read PRD: ${e2?.message}\n`);
1717
1866
  closeFd();
1718
1867
  return { exitCode: -1, durationMs: 0, error: e2?.message };
1719
1868
  }
1720
1869
  } else {
1870
+ if (await archivedTwinExists(job)) {
1871
+ return prdArchivedSkipResult(job, cwd, sessionId, startedAt, safeLog, closeFd, metaPath);
1872
+ }
1721
1873
  safeLog(`[scheduler] failed to read PRD: ${e?.message}\n`);
1722
1874
  closeFd();
1723
1875
  return { exitCode: -1, durationMs: 0, error: e?.message };
@@ -2252,6 +2404,14 @@ async function spawnInvestigation(failedJob, runDir) {
2252
2404
  }
2253
2405
 
2254
2406
  async function spawnJob(job, runId, runDir, defaultCwd) {
2407
+ // Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
2408
+ // — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
2409
+ // leaves the job pending; the next tick retries when a slot frees up.
2410
+ const slotToken = sessionSlots.acquire(`scheduler:${job.slug}`);
2411
+ if (!slotToken) {
2412
+ console.log(`[scheduler] no session slot free for ${job.slug} — deferring (${JSON.stringify(sessionSlots.snapshot().holders.map((h) => h.owner))})`);
2413
+ return;
2414
+ }
2255
2415
  runningSet.add(job.slug);
2256
2416
  try {
2257
2417
  await mutate((s) => {
@@ -2286,6 +2446,25 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2286
2446
  await setPaused('rate_limit', resetIso);
2287
2447
  }
2288
2448
 
2449
+ // Stale queue entry: the PRD already shipped and was archived before this
2450
+ // run fired (see archivedTwinExists in executeJob). Treat it as a plain
2451
+ // completion — no verify pass, no commit guard, no RCA feedback — since
2452
+ // there is no real transcript/commit to check.
2453
+ if (res.skipped === 'prd-archived') {
2454
+ await mutate((s) => {
2455
+ const idx = s.jobs.findIndex((x) => x.slug === job.slug);
2456
+ if (idx >= 0) {
2457
+ s.jobs[idx].status = 'completed';
2458
+ s.jobs[idx].finishedAt = new Date().toISOString();
2459
+ s.jobs[idx].exitCode = 0;
2460
+ s.jobs[idx].error = null;
2461
+ delete s.jobs[idx].runtime;
2462
+ }
2463
+ });
2464
+ await broadcast({ flush: true });
2465
+ return;
2466
+ }
2467
+
2289
2468
  // Post-run verification: for exit=0 runs, scan the transcript and check
2290
2469
  // dependency prerequisites before stamping 'completed'. This catches the
2291
2470
  // false-positive class where an agent exits cleanly while leaving failures
@@ -2665,6 +2844,8 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2665
2844
  console.error('[scheduler] spawnJob error', job.slug, e);
2666
2845
  } finally {
2667
2846
  runningSet.delete(job.slug);
2847
+ // Slot release notifies subscribed pumps (chat lane) machine-wide.
2848
+ sessionSlots.release(slotToken);
2668
2849
  // Each job completion is a signal to advance the queue.
2669
2850
  tickQueue().catch(() => {});
2670
2851
  }
@@ -2722,7 +2903,18 @@ function tickQueue() {
2722
2903
  // job is never started into the host's own headroom (that path OOM-kills
2723
2904
  // Electron and SIGHUPs every pty — 2026-06-16 incident).
2724
2905
  const jobBudgetMb = availableForJobs(availableMb, RESERVED_HOST_MB);
2725
- const allowed = memoryLimitedBatchSize(jobBudgetMb, MIN_FREE_MB_PER_JOB, runningSet.size, batch.length);
2906
+ // Session-Manager's machine-wide slot pool is the outer bound: chat runs
2907
+ // and scheduler jobs share it, so a busy chat lane shrinks this batch.
2908
+ const slotAllowed = sessionSlots.available();
2909
+ if (slotAllowed === 0) {
2910
+ const snap = sessionSlots.snapshot();
2911
+ console.log(`[scheduler] slot gate: 0 of ${snap.total} session slots free (${snap.holders.map((h) => h.owner).join(', ')}) — deferring ${batch.length} job(s)`);
2912
+ return { fired: false, reason: 'slots-exhausted', deferredCount: batch.length, holders: snap.holders };
2913
+ }
2914
+ const allowed = Math.min(
2915
+ slotAllowed,
2916
+ memoryLimitedBatchSize(jobBudgetMb, MIN_FREE_MB_PER_JOB, runningSet.size, batch.length),
2917
+ );
2726
2918
  if (allowed === 0) {
2727
2919
  const threshold = RESERVED_HOST_MB + MIN_FREE_MB_PER_JOB * (runningSet.size + 1);
2728
2920
  console.log(`[scheduler] memory gate: available=${availableMb} MB < threshold=${threshold} MB (host reserve ${RESERVED_HOST_MB} + ${MIN_FREE_MB_PER_JOB}/job × ${runningSet.size + 1}) — deferring ${batch.length} job(s)`);
@@ -2775,6 +2967,8 @@ function forceTickOutcome(result) {
2775
2967
  return { ok: true, kind: 'warn', message: 'Batch cancelled — try again' };
2776
2968
  case 'memory-deferred':
2777
2969
  return { ok: true, kind: 'warn', message: `Deferred ${result.deferredCount} job(s) — low memory (${result.availableMb} MB available, need ${result.threshold} MB)` };
2970
+ case 'slots-exhausted':
2971
+ return { ok: true, kind: 'warn', message: `Deferred ${result.deferredCount} job(s) — all session slots in use (${(result.holders ?? []).map((h) => h.owner).join(', ')})` };
2778
2972
  case 'held': {
2779
2973
  const detail = String(result.detail ?? '').replace(/^\[scheduler\]\s*[\w-]+\s*(?:\[[^\]]*\])?:\s*/, '');
2780
2974
  return { ok: true, kind: 'warn', message: detail || 'Batch held' };
@@ -3367,6 +3561,11 @@ function registerScheduleHandlers() {
3367
3561
  return buildScheduleStatePayload(state, { withPaths: true });
3368
3562
  });
3369
3563
 
3564
+ // Session-Manager-wide claude -p slot pool (lib/sessionSlots.cjs) —
3565
+ // read-only diagnostic surface for the Home widget and the global
3566
+ // configuration tab.
3567
+ ipcMain.handle('schedule:session-slots', () => sessionSlots.snapshot());
3568
+
3370
3569
  ipcMain.handle('schedule:health', async () => {
3371
3570
  const state = await readQueue();
3372
3571
  const runningJobs = [];
@@ -3623,6 +3822,19 @@ function registerScheduleHandlers() {
3623
3822
 
3624
3823
  async function init() {
3625
3824
  ensureDirs();
3825
+ // A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
3826
+ // batch — advance the queue without waiting for the next 60s poll.
3827
+ sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
3828
+ // Retire the global queue.json: split its rows into per-project shards
3829
+ // BEFORE the first read below, so boot reconciliation sees the shards.
3830
+ try {
3831
+ const m = await queueStore.migrateLegacyGlobalQueue(DEFAULT_PROJECT_CWD);
3832
+ if (m.migrated) {
3833
+ console.log(`[scheduler] legacy global queue retired: ${m.moved} row(s) split across ${m.projects} project shard(s)`);
3834
+ }
3835
+ } catch (e) {
3836
+ console.error('[scheduler] legacy queue split failed', e?.message);
3837
+ }
3626
3838
  await runPrdMigration();
3627
3839
  sweepQueueBackups().catch((e) => console.warn('[scheduler] backup sweep failed', e?.message));
3628
3840
 
@@ -3822,7 +4034,19 @@ const remote = {
3822
4034
  // nothing for findPrdDir to search for); the renderer's slug-only IPC
3823
4035
  // path (editing an already-queued PRD) omits it and relies on findPrdDir.
3824
4036
  async readPrd(slug, cwd) {
3825
- const dir = cwd ? prdDirForCwd(cwd) : await findPrdDir(slug);
4037
+ let dir;
4038
+ if (cwd) {
4039
+ // The slug may live in the legacy flat dir or any Epic's prds/ under
4040
+ // this project; probe local dirs, defaulting to the flat dir (callers
4041
+ // use a miss there as the "doesn't exist yet" signal on create).
4042
+ const localDirs = [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)];
4043
+ dir = localDirs.find((d) => {
4044
+ const p = safeSlugPathIn(d, slug);
4045
+ return p && fs.existsSync(p);
4046
+ }) ?? prdDirForCwd(cwd);
4047
+ } else {
4048
+ dir = await findPrdDir(slug);
4049
+ }
3826
4050
  if (!dir) return { ok: false, error: 'invalid slug' };
3827
4051
  const filePath = safeSlugPathIn(dir, slug);
3828
4052
  if (!filePath) return { ok: false, error: 'invalid slug' };
@@ -3864,15 +4088,65 @@ const remote = {
3864
4088
  // doesn't exist yet, so findPrdDir would return nothing to write into).
3865
4089
  async writePrd(slug, body, cwd) {
3866
4090
  let dir;
4091
+ let epicTrace = null;
4092
+ let epicCreated = false;
4093
+ let epicId = null;
3867
4094
  if (cwd) {
3868
- dir = prdDirForCwd(cwd);
4095
+ // Edit-in-place if this slug already lives anywhere under this project
4096
+ // (legacy flat dir or any Epic's prds/); otherwise this is a CREATE,
4097
+ // and every new PRD belongs to an Epic (CLAUDE.md domain model) — mint
4098
+ // one from the body's frontmatter title/tag.
4099
+ const localDirs = [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)];
4100
+ for (const d of localDirs) {
4101
+ const candidate = safeSlugPathIn(d, slug);
4102
+ if (candidate && fs.existsSync(candidate)) { dir = d; break; }
4103
+ }
4104
+ if (!dir) {
4105
+ try {
4106
+ const { fm } = splitFrontmatter(body);
4107
+ const epic = ensureEpic(cwd, {
4108
+ goalText: fm.title || slug,
4109
+ tag: fm.tag,
4110
+ // An Epic-conversation dispatch already has its Epic — join it.
4111
+ epicId: fm.sourcePromptId,
4112
+ });
4113
+ dir = epic.prdDir;
4114
+ epicTrace = epic.epicId;
4115
+ epicCreated = epic.created === true;
4116
+ epicId = epic.epicId;
4117
+ } catch (e) {
4118
+ // Epic mint must never block a PRD write — fall back to the
4119
+ // legacy flat dir and log loudly.
4120
+ console.error(`[scheduler] ensureEpic failed for ${slug}: ${e?.message}`);
4121
+ dir = prdDirForCwd(cwd);
4122
+ }
4123
+ }
3869
4124
  await fsp.mkdir(dir, { recursive: true });
3870
4125
  } else {
3871
4126
  dir = (await findPrdDir(slug)) ?? PRDS_DIR;
3872
4127
  if (dir === PRDS_DIR) ensureDirs();
3873
4128
  }
4129
+
4130
+ // PRD 825: if this call minted a brand-new Epic (ensureEpic's `created`)
4131
+ // and the write below never lands, don't strand an empty Epic dir —
4132
+ // best-effort remove `<epic>/prds` then `<epic>` itself, only when empty.
4133
+ const cleanupEmptyMintedEpic = async () => {
4134
+ if (!epicCreated || !epicId) return;
4135
+ try {
4136
+ const entries = await fsp.readdir(dir);
4137
+ if (entries.length > 0) return;
4138
+ await fsp.rmdir(dir);
4139
+ const epicRootDir = path.dirname(dir);
4140
+ const epicRootEntries = await fsp.readdir(epicRootDir);
4141
+ if (epicRootEntries.length === 0) await fsp.rmdir(epicRootDir);
4142
+ } catch { /* best-effort only */ }
4143
+ };
4144
+
3874
4145
  const resolved = safeSlugPathIn(dir, slug);
3875
- if (!resolved) return { ok: false, error: 'invalid slug' };
4146
+ if (!resolved) {
4147
+ await cleanupEmptyMintedEpic();
4148
+ return { ok: false, error: 'invalid slug' };
4149
+ }
3876
4150
  try {
3877
4151
  // Symlink defense, matching readPrd/readLog: safeSlugPathIn is lexical
3878
4152
  // and does NOT resolve symlinks, so a rogue job could plant a PRDs-dir
@@ -3882,16 +4156,23 @@ const remote = {
3882
4156
  // symlink.
3883
4157
  const realParent = await fsp.realpath(path.dirname(resolved));
3884
4158
  if (realParent !== dir && !realParent.startsWith(dir + path.sep)) {
4159
+ await cleanupEmptyMintedEpic();
3885
4160
  return { ok: false, error: 'invalid slug' };
3886
4161
  }
3887
4162
  const existing = await fsp.lstat(resolved).catch(() => null);
3888
4163
  if (existing && existing.isSymbolicLink()) {
4164
+ await cleanupEmptyMintedEpic();
3889
4165
  return { ok: false, error: 'invalid slug' };
3890
4166
  }
3891
4167
  await config.writeTextAtomic(resolved, body);
3892
4168
  const stat = await fsp.stat(resolved);
4169
+ if (epicTrace) {
4170
+ // Best-effort: record the dispatch on the minted Epic's event chain.
4171
+ try { appendPrdCreatedEvent(cwd, epicTrace, slug); } catch { /* trace only */ }
4172
+ }
3893
4173
  return { ok: true, bytesWritten: stat.size };
3894
4174
  } catch (e) {
4175
+ await cleanupEmptyMintedEpic();
3895
4176
  return { ok: false, error: e?.message ?? 'write failed' };
3896
4177
  }
3897
4178
  },
@@ -3986,4 +4267,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
3986
4267
  });
3987
4268
  }
3988
4269
 
3989
- module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, writeQueue, reconcile, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, feedbackSweepDue, FEEDBACK_SWEEP_TICK_INTERVAL, sweepFeedback, registerAdminRoutes, notifyOriginatingTab, isNotifiableTerminalStatus, candidatePrdsDirs, prdDirForCwd, prdPathForJob, findPrdDir, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields };
4270
+ module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, feedbackSweepDue, FEEDBACK_SWEEP_TICK_INTERVAL, sweepFeedback, registerAdminRoutes, notifyOriginatingTab, isNotifiableTerminalStatus, candidatePrdsDirs, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields };
@@ -392,17 +392,21 @@ the user may or may not have open.
392
392
 
393
393
  ### Fallback: writing the PRD file directly
394
394
 
395
- When the app is not running, write `<NN>-<slug>.md` by hand into the target repo's own
396
- `<cwd>/session-manager-operations/scheduler/prds/` following the frontmatter rules in §6 and the
395
+ When the app is not running, first mint (or join) an Epic — `node <session-manager-repo>/scripts/mint-epic.cjs <cwd> "<goal>" [feature|bug|discussion]`; its last stdout line is the prds dir — then write `<NN>-<slug>.md` by hand into that
396
+ `<cwd>/session-manager-operations/scheduler/epics/<epic-id>/prds/` dir (the flat `scheduler/prds/` is RETIRED and auto-archived unexecuted at boot), add `sourcePromptId: <epic-id>` to the frontmatter so the job keeps its Epic linkage, following the frontmatter rules in §6 and the
397
397
  body conventions the rest of this guide describes (`# Goal`, `# Acceptance criteria`,
398
398
  `# Implementation notes`, `## Engineering standards` inlined verbatim — see `/develop`'s output
399
399
  for the exact shape). **Trade-off:** this path has no atomic `NN` allocation. The tool's
400
400
  `allocateParallelGroup()` (PRD 548) exists specifically to close a race where two writers pick
401
401
  the same `NN` at once; a hand-written file bypasses that reservation entirely, so if another
402
402
  writer (a human, `/develop`, or another automation) picks the same `NN` around the same time, one
403
- file silently shadows or is shadowed by the other's parallel-group slot. Pick an `NN` by scanning
404
- the existing prds directory for the current max and incrementing, and treat a collision as
405
- possible, not merely theoretical.
403
+ file silently shadows or is shadowed by the other's number. Pick an `NN` by scanning ALL of the
404
+ project's prds dirs (`scheduler/epics/*/prds/` and `prds-archived/`) for the current max and
405
+ incrementing, and treat a collision as possible, not merely theoretical. NN is strictly unique
406
+ per project (PRD 832) — NEVER reuse an existing number to signal "runs in parallel"; that
407
+ convention is retired. Express ordering with `dependsOn: [<slug>, ...]` frontmatter (the job is
408
+ eligible once every listed slug's queue row is completed); independent PRDs omit it and the
409
+ scheduler may run them concurrently.
406
410
 
407
411
  ### Ownership boundary
408
412