claude-code-session-manager 0.40.2 → 0.41.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/dist/assets/{TiptapBody-CFCp4Mz9.js → TiptapBody-BnRle0iw.js} +1 -1
  2. package/dist/assets/{index-BxVBtmjA.css → index-CKY3mHgV.css} +1 -1
  3. package/dist/assets/{index-bdqOSyxG.js → index-H7rwsoKV.js} +650 -643
  4. package/dist/index.html +2 -2
  5. package/package.json +3 -1
  6. package/plugins/session-manager-dev/skills/develop/SKILL.md +5 -5
  7. package/plugins/session-manager-dev/skills/develop/standards.md +1 -1
  8. package/plugins/session-manager-dev/skills/explain-to-me/SKILL.md +2 -2
  9. package/plugins/session-manager-dev/skills/find-opportunity/SKILL.md +1 -1
  10. package/plugins/session-manager-dev/skills/project-status/SKILL.md +20 -34
  11. package/plugins/session-manager-dev/skills/propose-epic/SKILL.md +68 -0
  12. package/scripts/lib/watchdogHelpers.cjs +5 -5
  13. package/scripts/mint-epic.cjs +28 -0
  14. package/scripts/propose-epic.cjs +55 -0
  15. package/src/main/__tests__/epicMint.test.cjs +85 -0
  16. package/src/main/__tests__/prdCreate.test.cjs +54 -2
  17. package/src/main/__tests__/promptSessionEvents.test.cjs +56 -2
  18. package/src/main/__tests__/promptSessionTranscript.test.cjs +0 -0
  19. package/src/main/__tests__/rcaFeedbackHook.test.cjs +43 -211
  20. package/src/main/__tests__/scheduler-archived-twin-guard.test.cjs +63 -0
  21. package/src/main/__tests__/scheduler-notify-originating-tab-transcript.test.cjs +86 -0
  22. package/src/main/__tests__/scheduler-notify-originating-tab.test.cjs +21 -0
  23. package/src/main/__tests__/scheduler-writeprd-epic-rollback.test.cjs +78 -0
  24. package/src/main/browserView.cjs +1 -1
  25. package/src/main/chatRunner.cjs +68 -55
  26. package/src/main/config.cjs +12 -6
  27. package/src/main/docEdit.cjs +3 -4
  28. package/src/main/health.cjs +5 -0
  29. package/src/main/index.cjs +12 -0
  30. package/src/main/ipcSchemas.cjs +51 -11
  31. package/src/main/lib/__tests__/opsOwnership.test.cjs +92 -0
  32. package/src/main/lib/__tests__/projectBriefCore.test.cjs +74 -0
  33. package/src/main/lib/epicMint.cjs +133 -59
  34. package/src/main/lib/opsOwnership.cjs +166 -0
  35. package/src/main/lib/prdCreate.cjs +9 -1
  36. package/src/main/lib/prdLocations.cjs +21 -0
  37. package/src/main/lib/projectBriefCore.cjs +70 -4
  38. package/src/main/lib/queueStore.cjs +3 -0
  39. package/src/main/lib/rcaFeedbackHook.cjs +41 -55
  40. package/src/main/projectBrief.cjs +36 -2
  41. package/src/main/promptSessionEvents.cjs +24 -2
  42. package/src/main/promptSessionTranscript.cjs +0 -0
  43. package/src/main/queueOps.cjs +1 -1
  44. package/src/main/scheduler/prdParser.cjs +8 -0
  45. package/src/main/scheduler.cjs +281 -164
  46. package/src/main/templates/PRD_AUTHORING.md +1 -1
  47. package/src/preload/api.d.ts +76 -3
  48. package/src/preload/index.cjs +21 -3
  49. package/plugins/session-manager-dev/skills/my-feedback/SKILL.md +0 -140
  50. package/plugins/session-manager-dev/skills/optimize-kpi/SKILL.md +0 -290
  51. package/plugins/session-manager-dev/skills/process-feedback/SKILL.md +0 -265
@@ -63,6 +63,7 @@ const prdParser = require('./scheduler/prdParser.cjs');
63
63
  const sessionsStore = require('./sessionsStore.cjs');
64
64
  const { enqueueExternalPrompt } = require('./chatRunner.cjs');
65
65
  const { appendResponseEventIfKnown } = require('./promptSessionEvents.cjs');
66
+ const promptSessionTranscript = require('./promptSessionTranscript.cjs');
66
67
  const { verifyRun } = require('./runVerify.cjs');
67
68
  const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
68
69
  const logs = require('./logs.cjs');
@@ -84,9 +85,8 @@ const queueOps = require('./queueOps.cjs');
84
85
  // Plain Node module, no Electron dependency; queuePath/prdsDir defaults already
85
86
  // match ROOT/QUEUE_PATH below since both resolve the same ~/.claude/session-manager
86
87
  // home-dir layout.
87
- const { sweep: sweepFeedback } = require('../../scripts/lib/watchdogHelpers.cjs');
88
- const { resolvePrdsDirs, resolvePrdWriteDir, listEpicPrdDirs } = require('./lib/prdLocations.cjs');
89
- const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
88
+ const { resolvePrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
89
+ const { ensureEpic, appendPrdCreatedEvent, removeEpic, readActiveIndex } = require('./lib/epicMint.cjs');
90
90
 
91
91
  // ---------- origin session resolution (PRD 832) ----------
92
92
  // An Epic IS a tagged claude session — job rows carry the originating
@@ -144,6 +144,10 @@ const POST_RESULT_GRACE_MS = 90_000;
144
144
  const POST_RESULT_KILL_MS = 30_000;
145
145
  const RESULT_TAIL_POLL_MS = 5_000;
146
146
  const RESULT_TAIL_BYTES = 8 * 1024;
147
+ // Wider than RESULT_TAIL_BYTES (which only needs the subtype near the
148
+ // boundary) — a real `result` string can be a full paragraph, so
149
+ // extractResultTextFromLog scans more of the tail to find it whole.
150
+ const RESULT_TEXT_TAIL_BYTES = 64 * 1024;
147
151
 
148
152
  // Idle-output watchdog: if the log file mtime stops advancing for this long
149
153
  // while the process is still alive, the agent is hung mid-work (network
@@ -492,23 +496,43 @@ function prdPathForJob(job) {
492
496
  return path.join(prdDirForCwd(job && job.cwd), `${job && job.slug}.md`);
493
497
  }
494
498
 
495
- /** Absolute path to the sibling `prds-archived/<slug>.md` twin of a job's PRD. */
499
+ /**
500
+ * Absolute path to a job's archived-twin `<slug>.md`. Resolves to the first
501
+ * archive dir (across the flat legacy layout and every Epic's own sibling
502
+ * archive — see listArchivedPrdDirs) that actually contains the slug, falling
503
+ * back to the flat `prds-archived/<slug>.md` path when none do (this fallback
504
+ * is only ever used for its string value — logging/note text in
505
+ * prdArchivedSkipResult — never as an existence check).
506
+ */
496
507
  function archivedPrdPathForJob(job) {
497
- return path.join(prdDirForCwd(job && job.cwd), '..', 'prds-archived', `${job && job.slug}.md`);
508
+ const slug = job && job.slug;
509
+ const flatPath = path.join(prdDirForCwd(job && job.cwd), '..', 'prds-archived', `${slug}.md`);
510
+ for (const dir of listArchivedPrdDirs((job && job.cwd) || DEFAULT_PROJECT_CWD)) {
511
+ const candidate = safeSlugPathIn(dir, slug);
512
+ if (candidate && fs.existsSync(candidate)) return candidate;
513
+ }
514
+ return flatPath;
498
515
  }
499
516
 
500
517
  /**
501
- * True if a job's PRD has already been archived (sibling `prds-archived/<slug>.md`
502
- * exists). A queue entry whose PRD moved there is stale — the work already shipped
503
- * — not a genuine missing-PRD failure.
518
+ * True if a job's PRD has already been archived — in the flat legacy
519
+ * `prds-archived/` dir OR in the Epic-scoped sibling archive dir every new
520
+ * PRD actually lands in (archiveCompletedPrd archives into the source PRD's
521
+ * OWN parent's prds-archived/, which for an Epic PRD is inside that Epic).
522
+ * A queue entry whose PRD moved to either is stale — the work already
523
+ * shipped — not a genuine missing-PRD failure.
504
524
  */
505
525
  async function archivedTwinExists(job) {
506
- try {
507
- await fsp.access(archivedPrdPathForJob(job));
508
- return true;
509
- } catch {
510
- return false;
526
+ const slug = job && job.slug;
527
+ for (const dir of listArchivedPrdDirs((job && job.cwd) || DEFAULT_PROJECT_CWD)) {
528
+ const candidate = safeSlugPathIn(dir, slug);
529
+ if (!candidate) continue;
530
+ try {
531
+ await fsp.access(candidate);
532
+ return true;
533
+ } catch { /* not here — try the next archive dir */ }
511
534
  }
535
+ return false;
512
536
  }
513
537
 
514
538
  /**
@@ -1106,9 +1130,13 @@ async function reconcile(state) {
1106
1130
  estimateMinutes: p.estimateMinutes,
1107
1131
  sourcePromptId: reconcileSourcePromptId(job, p.sourcePromptId),
1108
1132
  sourceTabId: p.sourceTabId,
1133
+ // Unlike sourcePromptId (frozen once a row leaves 'pending'), epicId is
1134
+ // refreshed from disk every pass: the PRD's directory IS its Epic
1135
+ // membership, so moving the file between Epic dirs must re-point the row.
1136
+ epicId: p.epicId ?? job.epicId ?? null,
1109
1137
  dependsOn: p.dependsOn,
1110
1138
  originSessionId: job.originSessionId
1111
- ?? resolveOriginSessionId(p.cwd, reconcileSourcePromptId(job, p.sourcePromptId)),
1139
+ ?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
1112
1140
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1113
1141
  });
1114
1142
  }
@@ -1167,8 +1195,9 @@ async function reconcile(state) {
1167
1195
  estimateMinutes: p.estimateMinutes,
1168
1196
  sourcePromptId: p.sourcePromptId,
1169
1197
  sourceTabId: p.sourceTabId,
1198
+ epicId: p.epicId ?? null,
1170
1199
  dependsOn: p.dependsOn,
1171
- originSessionId: resolveOriginSessionId(p.cwd, p.sourcePromptId),
1200
+ originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
1172
1201
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1173
1202
  status: 'pending',
1174
1203
  runId: null,
@@ -1267,17 +1296,11 @@ let resumeTimer = null;
1267
1296
  let pollLoopTimer = null;
1268
1297
  let rescheduleInterval = null;
1269
1298
  let heartbeatInterval = null;
1270
- // Feedback sweep piggybacks on the 60s heartbeat tick but runs far less often —
1271
- // every Nth tick — since it's a readdir-per-active-project scan, not free.
1272
- // 5 ticks = 5 minutes; well inside sweep()/activeProjectCwds()'s 90-minute
1273
- // active-session window, so no active project can be missed between sweeps.
1274
- const FEEDBACK_SWEEP_TICK_INTERVAL = 5;
1275
- let feedbackSweepTickCount = 0;
1276
- /** Pure gate for the tick counter above — exported so the wiring is testable
1277
- * without waiting on real setInterval timers. */
1278
- function feedbackSweepDue(tickCount, interval = FEEDBACK_SWEEP_TICK_INTERVAL) {
1279
- return tickCount >= interval;
1280
- }
1299
+ // (The 5-minute feedback sweep that used to piggyback on this heartbeat is
1300
+ // gone: it scanned each active project's session-manager-operations/feedback/
1301
+ // and auto-queued a /process-feedback PRD. Both the folder and that skill are
1302
+ // retired — agents now file a `proposed` Epic that waits for human approval,
1303
+ // so there is nothing to sweep for. See lib/rcaFeedbackHook.cjs.)
1281
1304
  // In-memory set of slugs currently spawned in this process. Prevents
1282
1305
  // double-spawn when runDueJobs() is called while jobs are in flight.
1283
1306
  const runningSet = new Set();
@@ -1623,6 +1646,34 @@ function isNotifiableTerminalStatus(effectiveStatus) {
1623
1646
  return effectiveStatus === 'completed' || effectiveStatus === 'failed';
1624
1647
  }
1625
1648
 
1649
+ /**
1650
+ * Scans a run log's tail for the last `{"type":"result",...}` JSONL event
1651
+ * (the claude harness's final message) and returns its `result` string, or
1652
+ * null if the log is missing, unreadable, or has no parseable result line.
1653
+ * Never throws — a missing/torn log must not break notifyOriginatingTab.
1654
+ */
1655
+ function extractResultTextFromLog(logPath) {
1656
+ if (!logPath) return null;
1657
+ try {
1658
+ const tail = readTail(logPath, RESULT_TEXT_TAIL_BYTES);
1659
+ if (!tail) return null;
1660
+ const lines = tail.split('\n');
1661
+ for (let i = lines.length - 1; i >= 0; i--) {
1662
+ const line = lines[i].trim();
1663
+ if (!line || !line.includes('"type":"result"')) continue;
1664
+ try {
1665
+ const parsed = JSON.parse(line);
1666
+ if (parsed && typeof parsed.result === 'string') return parsed.result;
1667
+ } catch {
1668
+ // torn/partial line — keep scanning backward
1669
+ }
1670
+ }
1671
+ return null;
1672
+ } catch {
1673
+ return null;
1674
+ }
1675
+ }
1676
+
1626
1677
  /**
1627
1678
  * notifyOriginatingTab(job) → void
1628
1679
  *
@@ -1630,31 +1681,62 @@ function isNotifiableTerminalStatus(effectiveStatus) {
1630
1681
  * rateLimited auto-pause, which resets the job to pending instead), publish
1631
1682
  * a short status notification for the PRD that queued this job.
1632
1683
  *
1633
- * Resolution order (PRD 814): (1) if the PRD's `sourcePromptId` resolves to
1634
- * a known, still-active PromptSession (minted for a dev-work dispatch, PRD
1635
- * 813) under the job's cwd, append a 'response' PromptSessionEvent to THAT
1636
- * session's own event chain — its own scoped PromptSessionConversation, not
1637
- * whatever tab happens to be active — and stop; (2) otherwise, fall back to
1638
- * today's behavior: push a short status prompt into the chat tab that queued
1639
- * this PRD via enqueueExternalPrompt (PRD 753), resolved via the PRD's own
1640
- * `sourceTabId` frontmatter, then the first open tab (per sessionsStore's
1641
- * persisted tabs.json) whose cwd matches the job's cwd — first match only,
1642
- * no fan-out to multiple matching tabs; (3) no-op. Never throws to the
1643
- * caller (fire-and-forget from spawnJob). Deps are injectable (mirrors
1644
- * partitionBootOrphans's isAlive param) so unit tests can exercise the
1645
- * resolution logic without touching disk/electron.
1684
+ * Resolution order (PRD 814, extended by PRD 854): (1) if the PRD's
1685
+ * `sourcePromptId` resolves to a known, still-active PromptSession (minted
1686
+ * for a dev-work dispatch, PRD 813) under the job's cwd, append a 'response'
1687
+ * PromptSessionEvent to THAT session's own event chain — its own scoped Epic
1688
+ * conversation (EpicDetail.tsx), not whatever tab happens to be active — and
1689
+ * stop; (2) otherwise, fall back to pushing a short status prompt via
1690
+ * enqueueExternalPrompt (PRD 753) at a target id resolved from, in order,
1691
+ * the PRD's own `sourceTabId` frontmatter, then `sourcePromptId` again (a PRD
1692
+ * dispatched straight from an Epic's composer, dispatchPromptSessionToPrd,
1693
+ * never sets sourceTabId — sourcePromptId IS the Epic id and is a valid
1694
+ * chat:external-send target too, which the renderer now resolves against
1695
+ * both open tabs and known Epics, refusing a completed one), then the first
1696
+ * open tab (per sessionsStore's persisted tabs.json) whose cwd matches the
1697
+ * job's cwd — first match only, no fan-out to multiple matching tabs; (3)
1698
+ * no-op. The cwd-match step only runs when NEITHER sourceTabId nor
1699
+ * sourcePromptId is present in frontmatter at all — `sendPrompt` (
1700
+ * enqueueExternalPrompt) is fire-and-forget IPC with no ack, so once step
1701
+ * (2) picks either id it is not retried even if the renderer can't resolve
1702
+ * it either (e.g. a stale/garbage id) — same non-guaranteed-delivery
1703
+ * tradeoff this function already accepted for sourceTabId pre-PRD-854.
1704
+ * Never throws to the caller (fire-and-forget from spawnJob). Deps are
1705
+ * injectable (mirrors partitionBootOrphans's isAlive param) so unit tests
1706
+ * can exercise the resolution logic without touching disk/electron.
1646
1707
  */
1647
1708
  async function notifyOriginatingTab(job, {
1648
1709
  parsePrdRaw = prdParser.parsePrdRaw,
1649
1710
  loadSessions = sessionsStore.load,
1650
1711
  sendPrompt = enqueueExternalPrompt,
1651
1712
  appendResponseEvent = appendResponseEventIfKnown,
1713
+ appendTranscriptTurn = promptSessionTranscript.appendTurn,
1714
+ readResultFromLog = extractResultTextFromLog,
1652
1715
  } = {}) {
1653
1716
  try {
1654
1717
  const prdPath = prdPathForJob(job);
1655
1718
  const prd = await parsePrdRaw(prdPath).catch(() => null);
1656
1719
  const message = `PRD ${job.slug} finished: ${job.status}. Check Scheduler for details.`;
1657
1720
 
1721
+ // Persist the job's real result text (not just the short status chip
1722
+ // above) to the durable per-Epic transcript, keyed off whichever id
1723
+ // notifyOriginatingTab would otherwise notify. Best-effort: a missing
1724
+ // cwd/epic id, an unreadable run log, or an IPC error here must never
1725
+ // block the notification below.
1726
+ const epicIdForTranscript = prd?.sourcePromptId || prd?.sourceTabId || null;
1727
+ if (epicIdForTranscript && job.cwd) {
1728
+ try {
1729
+ const logPath = job.runId ? path.join(RUNS_DIR, job.runId, `${job.slug}.log`) : null;
1730
+ const resultText = readResultFromLog(logPath);
1731
+ await appendTranscriptTurn(job.cwd, epicIdForTranscript, {
1732
+ role: 'assistant',
1733
+ text: resultText || message,
1734
+ });
1735
+ } catch (e) {
1736
+ console.error('[scheduler] notifyOriginatingTab transcript append error', job?.slug, e);
1737
+ }
1738
+ }
1739
+
1658
1740
  if (prd?.sourcePromptId) {
1659
1741
  const routed = await appendResponseEvent(job.cwd || null, prd.sourcePromptId, message).catch((e) => {
1660
1742
  console.error('[scheduler] notifyOriginatingTab appendResponseEvent error', job?.slug, e);
@@ -1663,7 +1745,13 @@ async function notifyOriginatingTab(job, {
1663
1745
  if (routed) return;
1664
1746
  }
1665
1747
 
1666
- let targetTabId = prd?.sourceTabId || null;
1748
+ // appendResponseEvent already refused above (unknown id, completed
1749
+ // Epic, or disk error) — the sourcePromptId fallback below re-sends to
1750
+ // the SAME id via chat:external-send, which the renderer independently
1751
+ // re-checks against its own live PromptSession store. Deliberate
1752
+ // defense-in-depth (main's disk-backed check vs. the renderer's
1753
+ // in-memory one can disagree/race), not a redundant duplicate to prune.
1754
+ let targetTabId = prd?.sourceTabId || prd?.sourcePromptId || null;
1667
1755
  if (!targetTabId) {
1668
1756
  const jobCwd = job.cwd || null;
1669
1757
  if (jobCwd) {
@@ -1673,7 +1761,7 @@ async function notifyOriginatingTab(job, {
1673
1761
  }
1674
1762
  }
1675
1763
  if (!targetTabId) {
1676
- console.log(`[scheduler] notifyOriginatingTab: no target tab for ${job.slug}, skipping`);
1764
+ console.log(`[scheduler] notifyOriginatingTab: no open tab and no Epic for ${job.slug}, skipping`);
1677
1765
  return;
1678
1766
  }
1679
1767
 
@@ -1838,8 +1926,13 @@ async function executeJob(job, runDir, defaultCwd, onPid) {
1838
1926
  }
1839
1927
 
1840
1928
  // Read full PRD body fresh from disk (queue stored only the preview).
1929
+ // Resolve through findPrdDir's full candidate search (legacy flat dir +
1930
+ // every project's Epic-scoped dirs) first, so the common case — a live
1931
+ // Epic-scoped PRD — is a first-try hit instead of probing the retired flat
1932
+ // dir and only then falling back.
1841
1933
  let prompt;
1842
- let prdPath = prdPathForJob(job);
1934
+ const resolvedDir = await findPrdDir(job.slug);
1935
+ let prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
1843
1936
  try {
1844
1937
  const parsed = await parsePrd(prdPath);
1845
1938
  // Centrally enforce the review → security-review → verify → commit finish
@@ -3757,7 +3850,7 @@ function registerScheduleHandlers() {
3757
3850
  const resolved = safeSlugPathIn(dir, data.slug);
3758
3851
  if (!resolved) return { ok: false, error: 'invalid slug' };
3759
3852
  try {
3760
- await config.writeTextAtomic(resolved, data.body);
3853
+ await config.writeTextAtomic(resolved, data.body, { writer: 'scheduler' });
3761
3854
  } catch (e) {
3762
3855
  return { ok: false, error: e?.message ?? 'write failed' };
3763
3856
  }
@@ -3795,6 +3888,7 @@ function registerScheduleHandlers() {
3795
3888
  cwd: parsed.cwd || '',
3796
3889
  estimateMinutes: parsed.estimateMinutes,
3797
3890
  sourcePromptId: parsed.sourcePromptId,
3891
+ epicId: parsed.epicId ?? null,
3798
3892
  mtimeMs: stat.mtimeMs,
3799
3893
  });
3800
3894
  } catch (e) {
@@ -3822,116 +3916,137 @@ function registerScheduleHandlers() {
3822
3916
 
3823
3917
  async function init() {
3824
3918
  ensureDirs();
3825
- // A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
3826
- // batch — advance the queue without waiting for the next 60s poll.
3827
- sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
3828
- // Retire the global queue.json: split its rows into per-project shards
3829
- // BEFORE the first read below, so boot reconciliation sees the shards.
3919
+ // Boot phase — reconciliation, migrations, self-heal, first reset probe.
3920
+ // Guarded as a unit: everything below installs the timers that ARE the
3921
+ // running scheduler (poll loop, heartbeat, supervisor). A throw in here
3922
+ // used to reject init() and silently skip all of them, leaving an app
3923
+ // that looks up and holds the ownership lock but never ticks and never
3924
+ // writes a heartbeat — indistinguishable from a hung queue. Most of this
3925
+ // work is best-effort recovery; none of it is worth trading the scheduler
3926
+ // itself for. rescheduleTimer() in particular reaches the billing API,
3927
+ // which fails whenever the OAuth token is stale.
3830
3928
  try {
3831
- const m = await queueStore.migrateLegacyGlobalQueue(DEFAULT_PROJECT_CWD);
3832
- if (m.migrated) {
3833
- console.log(`[scheduler] legacy global queue retired: ${m.moved} row(s) split across ${m.projects} project shard(s)`);
3929
+ // A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
3930
+ // batch — advance the queue without waiting for the next 60s poll.
3931
+ sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
3932
+ // Retire the global queue.json: split its rows into per-project shards
3933
+ // BEFORE the first read below, so boot reconciliation sees the shards.
3934
+ try {
3935
+ const m = await queueStore.migrateLegacyGlobalQueue(DEFAULT_PROJECT_CWD);
3936
+ if (m.migrated) {
3937
+ console.log(`[scheduler] legacy global queue retired: ${m.moved} row(s) split across ${m.projects} project shard(s)`);
3938
+ }
3939
+ } catch (e) {
3940
+ console.error('[scheduler] legacy queue split failed', e?.message);
3834
3941
  }
3835
- } catch (e) {
3836
- console.error('[scheduler] legacy queue split failed', e?.message);
3837
- }
3838
- await runPrdMigration();
3839
- sweepQueueBackups().catch((e) => console.warn('[scheduler] backup sweep failed', e?.message));
3840
-
3841
- // Hydrate cached state from the sidecar before any scheduling decisions.
3842
- loadSchedulerState();
3843
- bootedAt = Date.now();
3844
-
3845
- // Boot reconciliation: finalize any job that was 'running' when the app died.
3846
- // Check the run log first — a job that emitted result/success before the crash
3847
- // should be marked 'completed', not 'failed', so it doesn't wedge the queue
3848
- // via the failure-gate. Also kill any still-live orphan claude child to prevent
3849
- // it from continuing to write to the project unsupervised (2026-05-21 incident).
3850
- //
3851
- // classifyRunOutcome calls readTail → fs.readFileSync (up to 64 KB per job).
3852
- // Pre-compute all outcomes BEFORE entering the mutate lock so the blocking I/O
3853
- // does not stall the event loop or hold the mutateTail chain during startup.
3854
- //
3855
- // Jobs whose recorded pid is still alive are deferred (not classified here) —
3856
- // see partitionBootOrphans. Everything else (dead pid or no pid) is safe to
3857
- // classify immediately below.
3858
- const bootSnap = readQueueSync();
3859
- const { immediate: immediateSlugs, deferred: deferredSlugs } = partitionBootOrphans(bootSnap.jobs);
3860
- const bootOutcomes = new Map();
3861
- for (const j of bootSnap.jobs) {
3862
- if (!immediateSlugs.includes(j.slug)) continue;
3863
- const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
3864
- bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
3865
- }
3866
- const bootReconciledCompletions = [];
3867
- await mutate((state) => {
3868
- for (const j of state.jobs) {
3869
- if (j.status !== 'running' || !immediateSlugs.includes(j.slug)) continue;
3870
- const outcome = bootOutcomes.get(j.slug) ?? 'unknown';
3871
- const pid = j.runtime?.pid;
3872
- const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
3873
- applyOrphanOutcome(j, outcome, killNote);
3874
- if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
3875
- console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
3942
+ await runPrdMigration();
3943
+ sweepQueueBackups().catch((e) => console.warn('[scheduler] backup sweep failed', e?.message));
3944
+
3945
+ // Hydrate cached state from the sidecar before any scheduling decisions.
3946
+ loadSchedulerState();
3947
+ bootedAt = Date.now();
3948
+
3949
+ // Boot reconciliation: finalize any job that was 'running' when the app died.
3950
+ // Check the run log first — a job that emitted result/success before the crash
3951
+ // should be marked 'completed', not 'failed', so it doesn't wedge the queue
3952
+ // via the failure-gate. Also kill any still-live orphan claude child to prevent
3953
+ // it from continuing to write to the project unsupervised (2026-05-21 incident).
3954
+ //
3955
+ // classifyRunOutcome calls readTail → fs.readFileSync (up to 64 KB per job).
3956
+ // Pre-compute all outcomes BEFORE entering the mutate lock so the blocking I/O
3957
+ // does not stall the event loop or hold the mutateTail chain during startup.
3958
+ //
3959
+ // Jobs whose recorded pid is still alive are deferred (not classified here) —
3960
+ // see partitionBootOrphans. Everything else (dead pid or no pid) is safe to
3961
+ // classify immediately below.
3962
+ const bootSnap = readQueueSync();
3963
+ const { immediate: immediateSlugs, deferred: deferredSlugs } = partitionBootOrphans(bootSnap.jobs);
3964
+ const bootOutcomes = new Map();
3965
+ for (const j of bootSnap.jobs) {
3966
+ if (!immediateSlugs.includes(j.slug)) continue;
3967
+ const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
3968
+ bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
3969
+ }
3970
+ const bootReconciledCompletions = [];
3971
+ await mutate((state) => {
3972
+ for (const j of state.jobs) {
3973
+ if (j.status !== 'running' || !immediateSlugs.includes(j.slug)) continue;
3974
+ const outcome = bootOutcomes.get(j.slug) ?? 'unknown';
3975
+ const pid = j.runtime?.pid;
3976
+ const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
3977
+ applyOrphanOutcome(j, outcome, killNote);
3978
+ if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
3979
+ console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
3980
+ }
3981
+ });
3982
+ for (const { slug, cwd } of bootReconciledCompletions) {
3983
+ await archiveCompletedPrd(slug, cwd);
3876
3984
  }
3877
- });
3878
- for (const { slug, cwd } of bootReconciledCompletions) {
3879
- await archiveCompletedPrd(slug, cwd);
3880
- }
3881
3985
 
3882
- // Still-alive orphans: SIGTERM (+ killOrphanClaudePid's own deferred SIGKILL
3883
- // follow-up) now, but classification waits until BOOT_ORPHAN_KILL_GRACE_MS
3884
- // later — reading the log while the orphan might still be writing to it
3885
- // could misclassify an about-to-succeed run as no_result and double-run the
3886
- // same PRD (2026-05-21 incident this guard exists for).
3887
- for (const slug of deferredSlugs) {
3888
- const j = bootSnap.jobs.find((x) => x.slug === slug);
3889
- const pid = j?.runtime?.pid;
3890
- const bootRunId = j?.runId ?? null; // captured now — guards against reconciling a DIFFERENT later run of the same slug
3891
- if (!pid) continue;
3892
- const result = killOrphanClaudePid(pid);
3893
- const killNote = ` (orphan pid=${pid}: ${result})`;
3894
- if (result === 'killed') {
3895
- console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
3986
+ // Still-alive orphans: SIGTERM (+ killOrphanClaudePid's own deferred SIGKILL
3987
+ // follow-up) now, but classification waits until BOOT_ORPHAN_KILL_GRACE_MS
3988
+ // later — reading the log while the orphan might still be writing to it
3989
+ // could misclassify an about-to-succeed run as no_result and double-run the
3990
+ // same PRD (2026-05-21 incident this guard exists for).
3991
+ for (const slug of deferredSlugs) {
3992
+ const j = bootSnap.jobs.find((x) => x.slug === slug);
3993
+ const pid = j?.runtime?.pid;
3994
+ const bootRunId = j?.runId ?? null; // captured now — guards against reconciling a DIFFERENT later run of the same slug
3995
+ if (!pid) continue;
3996
+ const result = killOrphanClaudePid(pid);
3997
+ const killNote = ` (orphan pid=${pid}: ${result})`;
3998
+ if (result === 'killed') {
3999
+ console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
4000
+ }
4001
+ setTimeout(() => {
4002
+ const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
4003
+ const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
4004
+ let deferredCompletedCwd;
4005
+ mutate((state) => {
4006
+ const cur = state.jobs.find((x) => x.slug === slug);
4007
+ // Race guard: bail if the job already resolved, OR if it's already been
4008
+ // re-picked into a NEW run (different runId) within the grace window —
4009
+ // that new run is not the boot orphan we SIGTERM'd and must not be
4010
+ // touched by this stale classification.
4011
+ if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
4012
+ applyOrphanOutcome(cur, outcome, killNote);
4013
+ console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
4014
+ deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
4015
+ }).then(() => {
4016
+ if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
4017
+ }).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
4018
+ }, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
3896
4019
  }
3897
- setTimeout(() => {
3898
- const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
3899
- const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
3900
- let deferredCompletedCwd;
3901
- mutate((state) => {
3902
- const cur = state.jobs.find((x) => x.slug === slug);
3903
- // Race guard: bail if the job already resolved, OR if it's already been
3904
- // re-picked into a NEW run (different runId) within the grace window —
3905
- // that new run is not the boot orphan we SIGTERM'd and must not be
3906
- // touched by this stale classification.
3907
- if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
3908
- applyOrphanOutcome(cur, outcome, killNote);
3909
- console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
3910
- deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
3911
- }).then(() => {
3912
- if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
3913
- }).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
3914
- }, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
3915
- }
3916
4020
 
3917
- // If we boot up while paused with a resumeAt in the past, clear it. This
3918
- // happens when the app was closed across the reset window.
3919
- const boot = await readQueue();
3920
- if (boot.paused && boot.paused.resumeAt && new Date(boot.paused.resumeAt).getTime() <= Date.now()) {
3921
- await clearPause('boot-elapsed');
3922
- } else if (boot.paused && boot.paused.resumeAt) {
3923
- // Re-arm the resume timer (lost across restart).
3924
- await setPaused(boot.paused.reason, boot.paused.resumeAt);
3925
- }
4021
+ // If we boot up while paused with a resumeAt in the past, clear it. This
4022
+ // happens when the app was closed across the reset window.
4023
+ const boot = await readQueue();
4024
+ if (boot.paused && boot.paused.resumeAt && new Date(boot.paused.resumeAt).getTime() <= Date.now()) {
4025
+ await clearPause('boot-elapsed');
4026
+ } else if (boot.paused && boot.paused.resumeAt) {
4027
+ // Re-arm the resume timer (lost across restart).
4028
+ await setPaused(boot.paused.reason, boot.paused.resumeAt);
4029
+ }
3926
4030
 
3927
- // Self-heal stale needs_review flags using the current verifier (see
3928
- // reverifyNeedsReview). Runs once on boot so a shipped verifier fix clears
3929
- // its own historical false positives without manual retagging.
3930
- await reverifyNeedsReview().catch((e) => {
3931
- console.error(`[scheduler] boot reverify failed: ${e?.message ?? e}`);
3932
- });
4031
+ // Self-heal stale needs_review flags using the current verifier (see
4032
+ // reverifyNeedsReview). Runs once on boot so a shipped verifier fix clears
4033
+ // its own historical false positives without manual retagging.
4034
+ await reverifyNeedsReview().catch((e) => {
4035
+ console.error(`[scheduler] boot reverify failed: ${e?.message ?? e}`);
4036
+ });
3933
4037
 
3934
- await rescheduleTimer();
4038
+ await rescheduleTimer();
4039
+ } catch (e) {
4040
+ console.error('[scheduler] boot phase failed — starting timers anyway:', e?.message);
4041
+ try {
4042
+ require('./logs.cjs').writeLine({
4043
+ scope: 'scheduler',
4044
+ level: 'error',
4045
+ message: 'boot phase failed; timers started anyway',
4046
+ meta: { error: e?.message },
4047
+ });
4048
+ } catch { /* logging must never be the thing that stops the scheduler */ }
4049
+ }
3935
4050
  // Refresh next-reset every 10 minutes — billing window can shift if usage
3936
4051
  // resets early or the auth token rotates. Tracked so re-init doesn't leak.
3937
4052
  if (rescheduleInterval) clearInterval(rescheduleInterval);
@@ -3985,16 +4100,6 @@ async function init() {
3985
4100
  utilization: cachedUtilization,
3986
4101
  consecutiveFailures,
3987
4102
  });
3988
-
3989
- feedbackSweepTickCount++;
3990
- if (feedbackSweepDue(feedbackSweepTickCount, FEEDBACK_SWEEP_TICK_INTERVAL)) {
3991
- feedbackSweepTickCount = 0;
3992
- try {
3993
- sweepFeedback();
3994
- } catch (e) {
3995
- console.warn('[scheduler] feedback sweep failed', e?.message);
3996
- }
3997
- }
3998
4103
  }, 60_000);
3999
4104
  if (heartbeatInterval.unref) heartbeatInterval.unref();
4000
4105
 
@@ -4104,10 +4209,16 @@ const remote = {
4104
4209
  if (!dir) {
4105
4210
  try {
4106
4211
  const { fm } = splitFrontmatter(body);
4107
- const epic = ensureEpic(cwd, {
4212
+ const epic = await ensureEpic(cwd, {
4108
4213
  goalText: fm.title || slug,
4109
4214
  tag: fm.tag,
4110
4215
  // An Epic-conversation dispatch already has its Epic — join it.
4216
+ // fm.sourcePromptId must be an existing Epic's promptSessionId
4217
+ // (== its active-index.json sessions key), NOT a
4218
+ // PromptTicket.id — epicMint.cjs's ensureEpic looks it up via
4219
+ // `index.sessions[explicitEpicId]` (see epicMint.cjs ~line 73),
4220
+ // a literal-equality join. Any other id (e.g. a PromptTicket.id)
4221
+ // simply won't match and mints a sibling Epic instead of joining.
4111
4222
  epicId: fm.sourcePromptId,
4112
4223
  });
4113
4224
  dir = epic.prdDir;
@@ -4130,16 +4241,22 @@ const remote = {
4130
4241
  // PRD 825: if this call minted a brand-new Epic (ensureEpic's `created`)
4131
4242
  // and the write below never lands, don't strand an empty Epic dir —
4132
4243
  // best-effort remove `<epic>/prds` then `<epic>` itself, only when empty.
4244
+ // PRD 851: also drop the Epic's active-index.json entry (sessions/events)
4245
+ // so an orphaned seed-prompt-only Epic doesn't linger in the Epics nav.
4246
+ // Gated on epicCreated (never fires when ensureEpic joined an existing
4247
+ // Epic) so a pre-existing Epic's history is never touched.
4133
4248
  const cleanupEmptyMintedEpic = async () => {
4134
4249
  if (!epicCreated || !epicId) return;
4135
4250
  try {
4136
4251
  const entries = await fsp.readdir(dir);
4137
- if (entries.length > 0) return;
4138
- await fsp.rmdir(dir);
4139
- const epicRootDir = path.dirname(dir);
4140
- const epicRootEntries = await fsp.readdir(epicRootDir);
4141
- if (epicRootEntries.length === 0) await fsp.rmdir(epicRootDir);
4252
+ if (entries.length === 0) {
4253
+ await fsp.rmdir(dir);
4254
+ const epicRootDir = path.dirname(dir);
4255
+ const epicRootEntries = await fsp.readdir(epicRootDir);
4256
+ if (epicRootEntries.length === 0) await fsp.rmdir(epicRootDir);
4257
+ }
4142
4258
  } catch { /* best-effort only */ }
4259
+ try { removeEpic(cwd, epicId); } catch { /* best-effort only */ }
4143
4260
  };
4144
4261
 
4145
4262
  const resolved = safeSlugPathIn(dir, slug);
@@ -4164,11 +4281,11 @@ const remote = {
4164
4281
  await cleanupEmptyMintedEpic();
4165
4282
  return { ok: false, error: 'invalid slug' };
4166
4283
  }
4167
- await config.writeTextAtomic(resolved, body);
4284
+ await config.writeTextAtomic(resolved, body, { writer: 'scheduler' });
4168
4285
  const stat = await fsp.stat(resolved);
4169
4286
  if (epicTrace) {
4170
4287
  // Best-effort: record the dispatch on the minted Epic's event chain.
4171
- try { appendPrdCreatedEvent(cwd, epicTrace, slug); } catch { /* trace only */ }
4288
+ try { await appendPrdCreatedEvent(cwd, epicTrace, slug); } catch { /* trace only */ }
4172
4289
  }
4173
4290
  return { ok: true, bytesWritten: stat.size };
4174
4291
  } catch (e) {
@@ -4267,4 +4384,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
4267
4384
  });
4268
4385
  }
4269
4386
 
4270
- module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, feedbackSweepDue, FEEDBACK_SWEEP_TICK_INTERVAL, sweepFeedback, registerAdminRoutes, notifyOriginatingTab, isNotifiableTerminalStatus, candidatePrdsDirs, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields };
4387
+ module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields };
@@ -437,7 +437,7 @@ not be re-proposed:
437
437
 
438
438
  ### Reference implementation of this exact loop
439
439
 
440
- `/process-feedback`'s step 0b (source → triage → `/develop` → queue) is a working example of this
440
+ The proposal → approve → `/develop` → queue path is a working example of this
441
441
  shape, landed 2026-07-14 (`feat(process-feedback): sync open GitHub issues into the feedback
442
442
  intake`, commit `352b89c`). It syncs open GitHub issues into `session-manager-operations/feedback/`
443
443
  (deduped on a `gh-issue-<N>` token), then processes each item through the same triage → `/develop`