claude-code-session-manager 0.65.0 → 0.67.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/dist/assets/AgentLibrary-CiqimwWl.js +3 -0
  2. package/dist/assets/{History-DB-9-zwc.js → History-DbzjII2Z.js} +2 -2
  3. package/dist/assets/{Hooks-DnxqMRrR.js → Hooks-C0PTplI_.js} +3 -3
  4. package/dist/assets/{HostBilko-Cw6JocV7.js → HostBilko-vGc1FYLD.js} +1 -1
  5. package/dist/assets/{Library-BYSB0dmY.js → Library-4tU4rg2h.js} +1 -1
  6. package/dist/assets/{ListDetail-BfKqnL0r.js → ListDetail-BGIvkCwZ.js} +1 -1
  7. package/dist/assets/{MarkdownEditor-Cs-l5JOv.js → MarkdownEditor-ChFdpCam.js} +1 -1
  8. package/dist/assets/{McpServers-bBqj3hGg.js → McpServers-DsRWWpU3.js} +2 -2
  9. package/dist/assets/{Memory-DzPGXT5J.js → Memory-DmxoxxxY.js} +6 -6
  10. package/dist/assets/{Panel-BX6UZ-W0.js → Panel-KC-rv3jS.js} +1 -1
  11. package/dist/assets/{Permissions-Db7S4T3Z.js → Permissions-DrDjCRFl.js} +3 -3
  12. package/dist/assets/{Plugins-CIuey9R3.js → Plugins-KexSQwk0.js} +2 -2
  13. package/dist/assets/{ProvenanceBadge-FPWNWc_-.js → ProvenanceBadge-Rg-jL94y.js} +1 -1
  14. package/dist/assets/SaveBar-BbEo8U2e.js +1 -0
  15. package/dist/assets/Scheduler-C4Ti9TCc.js +14 -0
  16. package/dist/assets/ScopeSwitcher-Di9BmYze.js +1 -0
  17. package/dist/assets/Settings-CxyhPuYQ.js +3 -0
  18. package/dist/assets/{SkillReferenceGraph-CprLcOCe.js → SkillReferenceGraph-0q26RHrn.js} +1 -1
  19. package/dist/assets/Skills-BXkVyCo5.js +3 -0
  20. package/dist/assets/SystemPrompt-Bwe3NTbi.js +1 -0
  21. package/dist/assets/{TagLibrary-q66s4N3i.js → TagLibrary-CnwEf6EI.js} +1 -1
  22. package/dist/assets/{TiptapBody-I22aArc2.js → TiptapBody-faCtLj9L.js} +1 -1
  23. package/dist/assets/{Toggle-D9sapSAv.js → Toggle-Bw7G_-RR.js} +1 -1
  24. package/dist/assets/{index-BRwaw_1W.js → index-B6dJ2CsU.js} +666 -665
  25. package/dist/assets/{index-LlWpj2VJ.css → index-CEnMgeQU.css} +1 -1
  26. package/dist/assets/{settingsSchema-CTc4qelV.js → settingsSchema-B5C9hZoS.js} +1 -1
  27. package/dist/index.html +2 -2
  28. package/package.json +1 -1
  29. package/plugins/session-manager-dev/skills/develop/SKILL.md +73 -15
  30. package/plugins/session-manager-dev/skills/ops-sweep/SKILL.md +10 -0
  31. package/src/main/__tests__/agentLibrary.test.cjs +40 -0
  32. package/src/main/__tests__/develop-skill-failure-modes.test.cjs +70 -0
  33. package/src/main/__tests__/flatPrdTickSweep.test.cjs +110 -0
  34. package/src/main/__tests__/health-per-project-stall.test.cjs +82 -0
  35. package/src/main/__tests__/prdAdminRouteParity.test.cjs +68 -0
  36. package/src/main/__tests__/prdAdminRoutes.test.cjs +311 -0
  37. package/src/main/__tests__/prdCreate.test.cjs +7 -2
  38. package/src/main/__tests__/prdMigration.test.cjs +17 -0
  39. package/src/main/__tests__/prdMigrationLegacyAdopt.test.cjs +91 -0
  40. package/src/main/__tests__/reconcileFlatPrdSweep.test.cjs +109 -0
  41. package/src/main/__tests__/scheduleJobSchema.test.cjs +127 -0
  42. package/src/main/__tests__/scheduleJobStatusDrift.test.cjs +65 -0
  43. package/src/main/__tests__/scheduleJobTransitions.test.cjs +152 -0
  44. package/src/main/__tests__/scheduleJobTransitionsGrep.test.cjs +59 -0
  45. package/src/main/__tests__/scheduler-reconcile-invalid-repair.test.cjs +203 -0
  46. package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +196 -0
  47. package/src/main/__tests__/scheduler-stall-per-project.test.cjs +108 -0
  48. package/src/main/agentLibrary.cjs +40 -2
  49. package/src/main/health.cjs +97 -2
  50. package/src/main/index.cjs +2 -0
  51. package/src/main/ipcSchemas.cjs +60 -0
  52. package/src/main/lib/localAdminHttp.cjs +10 -3
  53. package/src/main/lib/prdAdminRoutes.cjs +175 -0
  54. package/src/main/lib/prdCreate.cjs +37 -2
  55. package/src/main/lib/prdFrontmatter.cjs +179 -1
  56. package/src/main/lib/prdMigration.cjs +82 -5
  57. package/src/main/lib/queueStore.cjs +41 -7
  58. package/src/main/lib/scheduleJobSchema.cjs +114 -0
  59. package/src/main/lib/scheduleJobTransitions.cjs +164 -0
  60. package/src/main/lib/schedulerConfig.cjs +7 -0
  61. package/src/main/scheduler/prdParser.cjs +7 -0
  62. package/src/main/scheduler.cjs +769 -138
  63. package/src/preload/api.d.ts +54 -2
  64. package/src/preload/index.cjs +12 -0
  65. package/dist/assets/AgentLibrary-DYriNDGf.js +0 -1
  66. package/dist/assets/Scheduler-X5y252Qw.js +0 -14
  67. package/dist/assets/ScopeSwitcher-DU7M_q5-.js +0 -1
  68. package/dist/assets/Settings-DEptXcEY.js +0 -3
  69. package/dist/assets/Skills-CGN56X1i.js +0 -3
  70. package/dist/assets/SystemPrompt-DIEK7FpJ.js +0 -1
@@ -75,7 +75,11 @@ const {
75
75
  USAGE_REFRESH_INTERVAL_MS,
76
76
  MAX_JOB_DURATION_MS,
77
77
  BROADCAST_COALESCE_MS,
78
+ QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
78
79
  } = require('./lib/schedulerConfig.cjs');
80
+ const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
81
+ ? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
82
+ : QUARANTINE_ESCALATE_MS_DEFAULT;
79
83
  const { pickForProject, pickNextBatch, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
80
84
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
81
85
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
@@ -88,7 +92,10 @@ const queueOps = require('./queueOps.cjs');
88
92
  // home-dir layout.
89
93
  const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
90
94
  const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
95
+ const { transitionJob } = require('./lib/scheduleJobTransitions.cjs');
91
96
  const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
97
+ const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
98
+ const { appendAuditEvent } = require('./lib/auditLog.cjs');
92
99
 
93
100
  // ---------- origin session resolution (PRD 832) ----------
94
101
  // An Epic IS a tagged claude session — job rows carry the originating
@@ -110,8 +117,8 @@ function resolveOriginSessionId(cwd, epicId) {
110
117
  const sessionSlots = require('./lib/sessionSlots.cjs');
111
118
  const jobWorktree = require('./lib/jobWorktree.cjs');
112
119
  const queueStore = require('./lib/queueStore.cjs');
113
- const { splitFrontmatter } = require('./lib/prdFrontmatter.cjs');
114
- const { migratePrds, consolidateFlatPrds } = require('./lib/prdMigration.cjs');
120
+ const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
121
+ const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
115
122
  const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
116
123
 
117
124
  // Captured once at module load so every run's meta sidecar can record how
@@ -720,7 +727,7 @@ async function retireCompletedSlugs(slugs) {
720
727
  for (const j of s.jobs) {
721
728
  if (!j || !slugSet.has(j.slug)) continue;
722
729
  if (j.status !== 'pending' && j.status !== 'running') continue;
723
- j.status = 'completed';
730
+ if (!transitionJob(j, 'completed', { reason: 'manual archive of an already-shipped PRD', source: 'retireCompletedSlugs' })) continue;
724
731
  j.finishedAt = new Date().toISOString();
725
732
  j.exitCode = 0;
726
733
  j.error = null;
@@ -756,6 +763,50 @@ function ensureDirs() {
756
763
  * unparseable cwd, cwd not on disk) are left in place and logged as a
757
764
  * warning — never silently dropped — so a human can fix the frontmatter.
758
765
  */
766
+ /**
767
+ * consolidateAllFlatPrds(cwds) — run consolidateFlatPrds() over every given
768
+ * project cwd, logging outcomes. Called from TWO places: once at boot (over
769
+ * every historical project, via runPrdMigration below) AND at the top of
770
+ * every reconcile() call (over every project reconcile itself would
771
+ * otherwise scan), BEFORE reconcile scans the flat dir for PRD sources. The
772
+ * reconcile()-level call is what makes "anything written to the retired flat
773
+ * prds/ dir is swept into prds-archived/ without being executed" actually
774
+ * true regardless of which of reconcile's several callers (tickQueue's poll,
775
+ * job completion, the schedule:state/schedule:rescan IPC handlers,
776
+ * rescheduleTimer) triggers the pass: a PRD dropped in the flat dir has no
777
+ * queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
778
+ * sweep archives it before reconcile can ever turn it into a pending job.
779
+ */
780
+ async function consolidateAllFlatPrds(cwds) {
781
+ for (const cwd of cwds) {
782
+ try {
783
+ const c = await consolidateFlatPrds(cwd);
784
+ if (c.moved > 0) {
785
+ console.log(`[scheduler] flat-PRD consolidation: archived ${c.moved} file(s) in ${cwd}`);
786
+ }
787
+ for (const f of c.failed) {
788
+ logs.writeLine({
789
+ level: 'warn', scope: 'scheduler',
790
+ message: `flat-PRD consolidation: could not archive ${f.file}`,
791
+ meta: { cwd, reason: f.reason },
792
+ });
793
+ }
794
+ // Deliberately left behind because a live job still points at them
795
+ // (PRD 992). Logged so a permanently-stuck flat PRD is visible rather
796
+ // than looking like a clean consolidation.
797
+ for (const s of c.skipped ?? []) {
798
+ logs.writeLine({
799
+ level: 'info', scope: 'scheduler',
800
+ message: `flat-PRD consolidation: left ${s.file} in place`,
801
+ meta: { cwd, reason: s.reason },
802
+ });
803
+ }
804
+ } catch (e) {
805
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'flat-PRD consolidation failed', meta: { cwd, error: e?.message } });
806
+ }
807
+ }
808
+ }
809
+
759
810
  async function runPrdMigration() {
760
811
  let result;
761
812
  try {
@@ -782,33 +833,30 @@ async function runPrdMigration() {
782
833
  // still sitting flat consolidates into `prds-archived/` for later special
783
834
  // processing. Queue rows for moved files are reaped by the archived-twin
784
835
  // retirement. Idempotent per project; failures are logged, never fatal.
785
- for (const cwd of allProjectCwds()) {
786
- try {
787
- const c = await consolidateFlatPrds(cwd);
788
- if (c.moved > 0) {
789
- console.log(`[scheduler] flat-PRD consolidation: archived ${c.moved} file(s) in ${cwd}`);
790
- }
791
- for (const f of c.failed) {
792
- logs.writeLine({
793
- level: 'warn', scope: 'scheduler',
794
- message: `flat-PRD consolidation: could not archive ${f.file}`,
795
- meta: { cwd, reason: f.reason },
796
- });
797
- }
798
- // Deliberately left behind because a live job still points at them
799
- // (PRD 992). Logged so a permanently-stuck flat PRD is visible rather
800
- // than looking like a clean consolidation.
801
- for (const s of c.skipped ?? []) {
802
- logs.writeLine({
803
- level: 'info', scope: 'scheduler',
804
- message: `flat-PRD consolidation: left ${s.file} in place`,
805
- meta: { cwd, reason: s.reason },
806
- });
807
- }
808
- } catch (e) {
809
- logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'flat-PRD consolidation failed', meta: { cwd, error: e?.message } });
836
+ // (This boot-time pass is redundant with the one reconcile() now also runs
837
+ // on every pass, but stays here so a fresh boot's very first log line
838
+ // still reports the initial sweep — see consolidateAllFlatPrds's own
839
+ // comment for why reconcile() is the load-bearing call site.)
840
+ await consolidateAllFlatPrds(allProjectCwds());
841
+
842
+ // Rollout migration for the PRD-authoring-lockdown feature: stamp every
843
+ // pre-existing PRD as legacy-adopted BEFORE reconcile() ever runs its
844
+ // provenance gate against it. Must run every boot (idempotent, cheap
845
+ // scan-and-skip) rather than once — a project opened for the first time
846
+ // after this shipped still has pre-existing unstamped PRDs the very first
847
+ // time reconcile() sees them.
848
+ try {
849
+ const adopted = await legacyAdoptExistingPrds();
850
+ if (adopted.stamped > 0) {
851
+ console.log(`[scheduler] legacy-adopt migration: stamped ${adopted.stamped} pre-existing PRD(s) as createdVia=legacy-adopted`);
852
+ }
853
+ for (const f of adopted.failed) {
854
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'legacy-adopt migration: could not stamp PRD', meta: f });
810
855
  }
856
+ } catch (e) {
857
+ logs.writeLine({ level: 'error', scope: 'scheduler', message: 'legacy-adopt migration failed', meta: { error: e?.message } });
811
858
  }
859
+
812
860
  return result;
813
861
  }
814
862
 
@@ -913,6 +961,88 @@ function appendHeartbeat(entry) {
913
961
  }
914
962
  }
915
963
 
964
+ /**
965
+ * computeStallSummary(state) → { stalled, total, running, pending, byProject }
966
+ *
967
+ * Pure, no IO. `state` is a merged queue-store read ({ jobs, invalidJobs,
968
+ * paused }). The engine (reconcile/reaper/auto-fix/reverify) already
969
+ * operates machine-wide via queueStore's stateCwds() — this function is
970
+ * MONITORING, and monitoring must not collapse per-project reality into one
971
+ * boolean. `stalled` (top-level) is the pre-existing machine-wide roll-up:
972
+ * the queue holds work — valid rows OR rows quarantined for an invalid
973
+ * status — but nothing anywhere is running or pending and the scheduler
974
+ * isn't paused. The 2026-08-07 incident sat exactly in this state for 4+
975
+ * hours: 2 jobs, 0 running, 0 pending, and the only visible symptom was a
976
+ * heartbeat `counts` object that had silently minted a `queued` bucket
977
+ * instead of reporting anything actionable.
978
+ *
979
+ * `byProject[cwd].stalled` is the PER-PROJECT verdict added for the
980
+ * "burrow went dark while other projects were busy" gap: a project can hold
981
+ * jobs (including ones parked `quarantined`) with 0 running and 0 pending
982
+ * while the machine-wide `stalled` above reads false because a different
983
+ * project has running/pending work. Each project's own status counts
984
+ * (`byProject[cwd][status]`) already summed to a total before this — the
985
+ * fix is only the boolean, not the counting.
986
+ */
987
+ function computeStallSummary(state) {
988
+ const jobs = Array.isArray(state?.jobs) ? state.jobs : [];
989
+ const invalidJobs = Array.isArray(state?.invalidJobs) ? state.invalidJobs : [];
990
+ let running = 0;
991
+ let pending = 0;
992
+ const byProject = {};
993
+ for (const j of jobs) {
994
+ if (j.status === 'running') running += 1;
995
+ if (j.status === 'pending') pending += 1;
996
+ const key = j.cwd || '(unknown)';
997
+ byProject[key] = byProject[key] || {};
998
+ byProject[key][j.status] = (byProject[key][j.status] || 0) + 1;
999
+ }
1000
+ for (const inv of invalidJobs) {
1001
+ const key = inv.row?.cwd || '(unknown)';
1002
+ byProject[key] = byProject[key] || {};
1003
+ byProject[key].invalid = (byProject[key].invalid || 0) + 1;
1004
+ }
1005
+ const total = jobs.length + invalidJobs.length;
1006
+ const stalled = total > 0 && running === 0 && pending === 0 && !state?.paused;
1007
+ for (const key of Object.keys(byProject)) {
1008
+ const counts = byProject[key];
1009
+ const projRunning = counts.running || 0;
1010
+ const projPending = counts.pending || 0;
1011
+ const projTotal = Object.keys(counts)
1012
+ .filter((k) => k !== 'stalled')
1013
+ .reduce((sum, k) => sum + counts[k], 0);
1014
+ counts.stalled = projTotal > 0 && projRunning === 0 && projPending === 0 && !state?.paused;
1015
+ }
1016
+ return { stalled, total, running, pending, byProject };
1017
+ }
1018
+
1019
+ /**
1020
+ * findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
1021
+ *
1022
+ * Pure, no IO. A 'quarantined' row (no createdVia provenance) can otherwise
1023
+ * sit forever with nothing looking at it — quarantine only ever clears via a
1024
+ * human adopting or archiving it. This is the escalation half of that gate:
1025
+ * any quarantined row whose recorded quarantine timestamp (statusHistory's
1026
+ * `to === 'quarantined'` entry — stamped at creation, or backfilled from the
1027
+ * PRD file's mtime by reconcile() for rows quarantined before that stamp
1028
+ * existed) is older than `thresholdMs` is reported so the caller can
1029
+ * warn-log and surface it distinctly. A row with no recoverable timestamp is
1030
+ * skipped rather than guessed at.
1031
+ */
1032
+ function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
1033
+ const stale = [];
1034
+ for (const j of jobs ?? []) {
1035
+ if (j.status !== 'quarantined') continue;
1036
+ const entry = (j.statusHistory || []).find((h) => h.to === 'quarantined');
1037
+ if (!entry) continue;
1038
+ const since = Date.parse(entry.at);
1039
+ if (Number.isNaN(since)) continue;
1040
+ const ageMs = now - since;
1041
+ if (ageMs >= thresholdMs) stale.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs });
1042
+ }
1043
+ return stale;
1044
+ }
1045
+
916
1046
  // An empty queue and an unreadable queue are NOT the same thing, and
917
1047
  // conflating them is destructive: reconcile() treats every PRD .md with no
918
1048
  // matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
@@ -1172,6 +1302,15 @@ async function reconcile(state) {
1172
1302
  if (state && state.unreadable) {
1173
1303
  throw new Error(`reconcile skipped: queue.json unreadable (${state.unreadable})`);
1174
1304
  }
1305
+ // Sweep the retired flat prds/ dir BEFORE scanning it below. reconcile()
1306
+ // has several callers besides tickQueue's ~60s poll (broadcast,
1307
+ // rescheduleTimer, the schedule:state IPC handler, schedule:rescan) — this
1308
+ // lives here, not in any one caller, so the "a hand-written PRD in the flat
1309
+ // dir is swept before it can become a job" guarantee holds regardless of
1310
+ // which caller triggers this reconcile pass. A freshly hand-written file
1311
+ // has no queue row yet, so it is never "live" and gets archived here
1312
+ // instead of ever reaching the onDisk scan below.
1313
+ await consolidateAllFlatPrds(allProjectCwds());
1175
1314
  const files = await listPrdFiles();
1176
1315
  const onDisk = new Map();
1177
1316
  for (const f of files) {
@@ -1213,13 +1352,16 @@ async function reconcile(state) {
1213
1352
  // file may be unreadable, on a project whose dir failed to enumerate,
1214
1353
  // or mid-move. "I can't see it" is not "the user deleted it", so the
1215
1354
  // row survives — worst case it re-resolves on the next pass.
1216
- if (job.status === 'pending' || job.status === 'running') {
1355
+ if (job.status === 'pending' || job.status === 'running' || job.status === 'quarantined') {
1217
1356
  // Exception: a PENDING row whose PRD has an archived twin was
1218
1357
  // retired on purpose (work landed by other means — e.g. implemented
1219
1358
  // inline — and the source .md moved to prds-archived/). Keeping it
1220
1359
  // would show a phantom "scheduled" job forever; firing it would just
1221
1360
  // hit executeJob's archived-twin skip anyway. Running rows are left
1222
- // alone — the reaper owns their lifecycle.
1361
+ // alone — the reaper owns their lifecycle. A quarantined row's file
1362
+ // going merely-not-visible must survive too — quarantine is meant to
1363
+ // be loud and reversible, never a silent drop (see this function's
1364
+ // header comment on the 2026-08-01 outage a silent skip caused).
1223
1365
  if (job.status === 'pending' && (await archivedTwinExists(job))) {
1224
1366
  console.log(`[scheduler] reconcile: retiring pending job ${job.slug} — PRD already archived (work landed elsewhere)`);
1225
1367
  continue;
@@ -1233,7 +1375,7 @@ async function reconcile(state) {
1233
1375
  continue;
1234
1376
  }
1235
1377
  seen.add(job.slug);
1236
- next.push({
1378
+ const updatedJob = {
1237
1379
  ...job,
1238
1380
  title: p.title,
1239
1381
  cwd: p.cwd,
@@ -1249,7 +1391,38 @@ async function reconcile(state) {
1249
1391
  originSessionId: job.originSessionId
1250
1392
  ?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
1251
1393
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1252
- });
1394
+ };
1395
+ // Adopt path: a row parked 'quarantined' (no createdVia provenance when
1396
+ // discovered) whose PRD file now carries a stamp — written via the
1397
+ // update-prd API's adopt patch, either the Scheduler tab's one-click
1398
+ // "adopt PRD" action or a manual scheduler_update_prd call — promotes to
1399
+ // 'pending' the very next reconcile pass. This is the ONLY way a
1400
+ // quarantined row becomes runnable; nothing else in reconcile() clears
1401
+ // that status.
1402
+ if (updatedJob.status === 'quarantined' && p.createdVia) {
1403
+ transitionJob(updatedJob, 'pending', {
1404
+ reason: `adopted via API (createdVia=${p.createdVia})`,
1405
+ source: 'reconcile-adopt',
1406
+ });
1407
+ console.log(`[scheduler] reconcile: adopted quarantined PRD ${job.slug} — createdVia=${p.createdVia}`);
1408
+ appendAuditEvent('scheduler_prd_adopted', { slug: job.slug, cwd: p.cwd, createdVia: p.createdVia, source: 'reconcile' });
1409
+ }
1410
+ // Backfill a quarantine timestamp for rows quarantined before the
1411
+ // statusHistory stamp below existed (e.g. the burrow-project rows
1412
+ // quarantined under the PRD-authoring lockdown) — findStaleQuarantinedJobs
1413
+ // needs SOME timestamp to escalate an un-adopted row past its age
1414
+ // threshold, and the PRD file's own mtime is the best available proxy
1415
+ // for "when this file first showed up unstamped" for a row that has
1416
+ // never been touched since.
1417
+ if (updatedJob.status === 'quarantined' && !(updatedJob.statusHistory || []).some((h) => h.to === 'quarantined')) {
1418
+ try {
1419
+ const at = new Date(fs.statSync(p.path).mtimeMs).toISOString();
1420
+ const history = Array.isArray(updatedJob.statusHistory) ? [...updatedJob.statusHistory] : [];
1421
+ history.push({ from: null, to: 'quarantined', reason: 'backfilled from PRD file mtime', source: 'reconcile-backfill', at });
1422
+ updatedJob.statusHistory = history;
1423
+ } catch { /* best-effort only — a missing/unreadable file just skips the backfill */ }
1424
+ }
1425
+ next.push(updatedJob);
1253
1426
  }
1254
1427
  // Slugs on disk with no matching state.jobs row are normally brand-new
1255
1428
  // PRDs — but once queueHistory.partitionJobs (above, later this same
@@ -1264,7 +1437,11 @@ async function reconcile(state) {
1264
1437
  for (const [slug] of onDisk) {
1265
1438
  if (!seen.has(slug)) unmatchedSlugs.push(slug);
1266
1439
  }
1267
- const historyBySlug = (unmatchedSlugs.length > 0 || terminalDroppedNeedingHistoryCheck.length > 0)
1440
+ // Rows quarantined by queueStore.shapeJobs because their `status` failed
1441
+ // ScheduleJobSchema (e.g. the 1021/1022 incident's `"status": "queued"`) —
1442
+ // see the repair pass below, right after historyBySlug is available.
1443
+ const invalidJobs = Array.isArray(state.invalidJobs) ? state.invalidJobs : [];
1444
+ const historyBySlug = (unmatchedSlugs.length > 0 || terminalDroppedNeedingHistoryCheck.length > 0 || invalidJobs.length > 0)
1268
1445
  ? await queueHistory.historyTerminalBySlug()
1269
1446
  : new Map();
1270
1447
 
@@ -1282,6 +1459,79 @@ async function reconcile(state) {
1282
1459
  }
1283
1460
  }
1284
1461
 
1462
+ // Repair pass: an invalid row must self-heal within this one tick, not
1463
+ // wait for its slug to also drop out of `seen` via some unrelated code
1464
+ // path. Before this pass, reconcile was add-only (`if (seen.has(slug))
1465
+ // continue` below) — a quarantined row simply vanished from state.jobs
1466
+ // with no log of what its bad status actually was and no repair, which is
1467
+ // how the 1021/1022 rows sat invisible for 4+ hours (2026-08-07).
1468
+ let repairedInvalidCount = 0;
1469
+ for (const inv of invalidJobs) {
1470
+ if (seen.has(inv.slug)) continue; // a valid row for this slug already exists
1471
+ const oldStatus = inv.row?.status;
1472
+ const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir: RUNS_DIR });
1473
+ if (hist) {
1474
+ // Never resurrect: this slug already has a durable terminal record
1475
+ // elsewhere (history.jsonl or a run sidecar) — repairing its corrupted
1476
+ // row back to 'pending' would re-execute already-shipped work. Drop
1477
+ // the row (its real outcome is recorded elsewhere), loudly.
1478
+ console.warn(`[scheduler] reconcile: dropping invalid queue row ${inv.slug} (status was ${JSON.stringify(oldStatus)}) — already terminal (${hist.status}) in history/run sidecar, not resurrecting`);
1479
+ appendAuditEvent('scheduler_row_repaired', {
1480
+ slug: inv.slug, cwd: inv.row?.cwd ?? null, oldStatus: oldStatus ?? null,
1481
+ action: 'dropped-already-terminal', terminalStatus: hist.status, issues: inv.issues,
1482
+ });
1483
+ continue;
1484
+ }
1485
+ const p = onDisk.get(inv.slug);
1486
+ if (!p) {
1487
+ // PRD file also gone with no terminal record anywhere — nothing to
1488
+ // repair against. queueStore already logged the quarantine once.
1489
+ continue;
1490
+ }
1491
+ const job = {
1492
+ ...inv.row,
1493
+ slug: inv.slug,
1494
+ title: p.title,
1495
+ cwd: p.cwd,
1496
+ parallelGroup: p.parallelGroup,
1497
+ estimateMinutes: p.estimateMinutes,
1498
+ sourcePromptId: p.sourcePromptId ?? inv.row?.sourcePromptId ?? null,
1499
+ sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
1500
+ epicId: p.epicId ?? inv.row?.epicId ?? null,
1501
+ dependsOn: p.dependsOn,
1502
+ originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
1503
+ bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1504
+ };
1505
+ const reason = `reconcile: repaired invalid status ${JSON.stringify(oldStatus)}`;
1506
+ // A repair is not a lifecycle transition — the corrupted `status` was
1507
+ // never a legal predecessor to check against LEGAL_TRANSITIONS, so this
1508
+ // goes through transitionJob's allowAnyFrom escape hatch (still gets the
1509
+ // normal mutation/statusHistory/audit trail, just skips the legality
1510
+ // gate on `from`) rather than a bare field assignment.
1511
+ transitionJob(job, 'pending', { reason, source: 'reconcile-repair', allowAnyFrom: true });
1512
+ if (job.runId || job.startedAt || job.runtime) {
1513
+ // This row had actually begun executing before its status got
1514
+ // corrupted.
1515
+ job.runId = null;
1516
+ job.startedAt = null;
1517
+ job.finishedAt = null;
1518
+ job.exitCode = null;
1519
+ delete job.runtime;
1520
+ delete job.verifierVerdict;
1521
+ }
1522
+ job.error = null;
1523
+ seen.add(inv.slug);
1524
+ next.push(job);
1525
+ repairedInvalidCount += 1;
1526
+ console.warn(`[scheduler] reconcile: repaired invalid queue row ${inv.slug} — status was ${JSON.stringify(oldStatus)}, reset to 'pending' (${inv.issues})`);
1527
+ appendAuditEvent('scheduler_row_repaired', {
1528
+ slug: inv.slug, cwd: p.cwd, oldStatus: oldStatus ?? null, newStatus: 'pending', issues: inv.issues,
1529
+ });
1530
+ }
1531
+ if (repairedInvalidCount > 0) {
1532
+ console.warn(`[scheduler] reconcile: repaired ${repairedInvalidCount} invalid queue row(s) this pass`);
1533
+ }
1534
+
1285
1535
  // Terminal-in-history slugs whose .md file is still on disk: fed into the
1286
1536
  // auto-archive selection pass below (as synthetic completed entries) so
1287
1537
  // their file can still be swept, without ever creating a live job row
@@ -1302,6 +1552,7 @@ async function reconcile(state) {
1302
1552
  return idx.sessions[epicId]?.status ?? null;
1303
1553
  }
1304
1554
 
1555
+ let staleNewDiscoveryCount = 0;
1305
1556
  for (const [slug, p] of onDisk) {
1306
1557
  if (seen.has(slug)) continue;
1307
1558
  // Security gate: a PRD's file location IS its Epic membership
@@ -1385,8 +1636,59 @@ async function reconcile(state) {
1385
1636
  const parent = healTargetForFix(slug, state.jobs);
1386
1637
  entry.investigationDepth = parent ? (parent.investigationDepth ?? 1) + 1 : 2;
1387
1638
  }
1639
+ // Provenance gate (PRD-authoring lockdown): a PRD discovered with no
1640
+ // `createdVia` stamp was never written through scheduler_create_prd/
1641
+ // chat:create-prd (prdCreate.cjs always stamps 'scheduler-api') or the
1642
+ // legacy-adopt boot migration ('legacy-adopted') — it bypassed the
1643
+ // sanctioned API, most likely via a raw Write/Edit tool call the
1644
+ // guard-prd-writes.cjs PreToolUse hook should have denied. Fix-plan PRDs
1645
+ // are exempt: spawnInvestigation's own probe writes them directly by
1646
+ // design (a trusted, scheduler-spawned internal loop, not an
1647
+ // agent/human authoring a PRD), matching the isFixPlanSlug convention
1648
+ // used everywhere else this distinction matters.
1649
+ //
1650
+ // Quarantine is loud and reversible, never a silent skip (see the
1651
+ // 2026-08-01 23-PRD outage this file's header references for what a
1652
+ // SILENT skip costs): logged at warn, audited, and surfaced in the
1653
+ // Scheduler tab's Quarantined filter with a one-click adopt action
1654
+ // (schedule:adopt-prd) that stamps the file via the same update-prd API
1655
+ // route the MCP tool uses — reconcile()'s adopt path above promotes it
1656
+ // to 'pending' on the very next pass, within one tick of being stamped.
1657
+ if (!p.createdVia && !isFixPlanSlug(slug)) {
1658
+ entry.status = 'quarantined';
1659
+ // Stamped at creation (not via transitionJob, since this is a
1660
+ // brand-new row minted directly at 'quarantined' rather than
1661
+ // transitioning through 'pending') so findStaleQuarantinedJobs has a
1662
+ // real quarantine timestamp to escalate against, instead of only the
1663
+ // reconcile-backfill fallback above.
1664
+ entry.statusHistory = [{
1665
+ from: null,
1666
+ to: 'quarantined',
1667
+ reason: 'missing createdVia provenance frontmatter',
1668
+ source: 'reconcile',
1669
+ at: new Date().toISOString(),
1670
+ }];
1671
+ console.warn(`[scheduler] reconcile: quarantining unstamped PRD ${slug} (${p.path}) — no createdVia provenance; adopt it from the Scheduler tab's Quarantined filter or via scheduler_update_prd to make it runnable`);
1672
+ appendAuditEvent('prd_quarantined', { slug, cwd: p.cwd, path: p.path, reason: 'missing createdVia provenance frontmatter' });
1673
+ }
1674
+ // A PRD with no queue row and no terminal record is normally a
1675
+ // brand-new file — but one whose mtime already predates a full poll
1676
+ // interval means it sat unpicked (a prior reconcile pass should have
1677
+ // caught it, or it's arriving from a source that bypassed the app's
1678
+ // normal write path). Report it rather than silently treating "first
1679
+ // seen this pass" as "just created".
1680
+ try {
1681
+ const ageMs = Date.now() - fs.statSync(p.path).mtimeMs;
1682
+ if (ageMs > POLL_INTERVAL_MS) {
1683
+ staleNewDiscoveryCount += 1;
1684
+ console.warn(`[scheduler] reconcile: discovered PRD ${slug} with no queue row and no terminal record — file is ${Math.round(ageMs / 1000)}s old, only first seen this pass`);
1685
+ }
1686
+ } catch { /* stat is best-effort reporting only */ }
1388
1687
  next.push(entry);
1389
1688
  }
1689
+ if (staleNewDiscoveryCount > 0) {
1690
+ console.warn(`[scheduler] reconcile: ${staleNewDiscoveryCount} PRD(s) discovered this pass were already older than one poll interval with no prior queue row`);
1691
+ }
1390
1692
  const sorted = next.sort((a, b) => b.slug.localeCompare(a.slug));
1391
1693
 
1392
1694
  // Move terminal jobs past the retention window out to history.jsonl so
@@ -1467,6 +1769,16 @@ let resumeTimer = null;
1467
1769
  let pollLoopTimer = null;
1468
1770
  let rescheduleInterval = null;
1469
1771
  let heartbeatInterval = null;
1772
+ // Stall-detector state (computeStallSummary), read/written only inside the
1773
+ // heartbeat interval below. Keyed per-project cwd (never a single value) —
1774
+ // a single module-level flag would let one busy project's activity clear or
1775
+ // suppress another stalled project's alert. stallSince.get(cwd): wall-clock
1776
+ // ms that project's stalled condition was first observed, absent when clear.
1777
+ // stallToasted.get(cwd): rate-limits that project's error-log + toast to
1778
+ // once per stall episode (cleared the moment that project stops being
1779
+ // stalled) rather than every 60s heartbeat tick.
1780
+ let stallSince = new Map();
1781
+ let stallToasted = new Map();
1470
1782
  // (The 5-minute feedback sweep that used to piggyback on this heartbeat is
1471
1783
  // gone: it scanned each active project's session-manager-operations/feedback/
1472
1784
  // and auto-queued a /process-feedback PRD. Both the folder and that skill are
@@ -1738,7 +2050,7 @@ async function clearPause(source) {
1738
2050
  */
1739
2051
  function resetJobFields(job, errorMsg, opts = {}) {
1740
2052
  if (job.status === 'completed' && opts.force !== true) return false;
1741
- job.status = 'pending';
2053
+ if (!transitionJob(job, 'pending', { reason: errorMsg ?? 'reset to pending', source: opts.source ?? 'resetJobFields' })) return false;
1742
2054
  job.runId = null;
1743
2055
  job.startedAt = null;
1744
2056
  job.finishedAt = null;
@@ -1799,13 +2111,13 @@ function partitionBootOrphans(jobs, isAlive = claudePidAlive) {
1799
2111
  function applyOrphanOutcome(job, outcome, killNote = '') {
1800
2112
  const now = new Date().toISOString();
1801
2113
  if (outcome === 'success') {
1802
- job.status = 'completed';
2114
+ transitionJob(job, 'completed', { reason: 'boot orphan reconciliation: run succeeded', source: 'applyOrphanOutcome' });
1803
2115
  job.exitCode = 0;
1804
2116
  job.error = null;
1805
2117
  job.finishedAt = now;
1806
2118
  delete job.runtime;
1807
2119
  } else if (outcome === 'failed') {
1808
- job.status = 'failed';
2120
+ transitionJob(job, 'failed', { reason: `orphaned: app restarted while running${killNote}`, source: 'applyOrphanOutcome' });
1809
2121
  job.exitCode = job.exitCode ?? 1;
1810
2122
  job.error = `orphaned: app restarted while running${killNote}`;
1811
2123
  job.finishedAt = now;
@@ -1813,10 +2125,10 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
1813
2125
  } else {
1814
2126
  const tries = job.orphanRetries ?? 0;
1815
2127
  if (tries < ORPHAN_REQUEUE_CAP) {
1816
- resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}`);
2128
+ resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}`, { source: 'applyOrphanOutcome' });
1817
2129
  job.orphanRetries = tries + 1;
1818
2130
  } else {
1819
- job.status = 'failed';
2131
+ transitionJob(job, 'failed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`, source: 'applyOrphanOutcome' });
1820
2132
  job.exitCode = job.exitCode ?? 1;
1821
2133
  job.error = `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`;
1822
2134
  job.finishedAt = now;
@@ -2860,7 +3172,7 @@ async function spawnInvestigation(failedJob, runDir) {
2860
3172
  // "nothing is happening" even though an Opus process was actively running.
2861
3173
  await mutate((s) => {
2862
3174
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
2863
- if (j) j.status = 'investigating';
3175
+ if (j) transitionJob(j, 'investigating', { reason: 'spawning investigation probe', source: 'spawnInvestigation:start' });
2864
3176
  });
2865
3177
  await broadcast({ flush: true });
2866
3178
 
@@ -2905,7 +3217,7 @@ async function spawnInvestigation(failedJob, runDir) {
2905
3217
  // 'investigating' must never be the job's resting state.
2906
3218
  mutate((s) => {
2907
3219
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
2908
- if (j && j.status === 'investigating') j.status = failedJob.status || 'failed';
3220
+ if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation probe exited — restoring prior status', source: 'spawnInvestigation:onExit' });
2909
3221
  })
2910
3222
  .then(() => broadcast({ flush: true }))
2911
3223
  .catch(() => {});
@@ -2960,7 +3272,7 @@ async function spawnInvestigation(failedJob, runDir) {
2960
3272
  releaseSlot();
2961
3273
  mutate((s) => {
2962
3274
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
2963
- if (j && j.status === 'investigating') j.status = failedJob.status || 'failed';
3275
+ if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation spawn threw before exiting — restoring prior status', source: 'spawnInvestigation:catch' });
2964
3276
  })
2965
3277
  .then(() => broadcast({ flush: true }))
2966
3278
  .catch(() => {});
@@ -2982,7 +3294,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2982
3294
  await mutate((s) => {
2983
3295
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
2984
3296
  if (idx >= 0) {
2985
- s.jobs[idx].status = 'running';
3297
+ transitionJob(s.jobs[idx], 'running', { reason: 'dispatched for execution', source: 'spawnJob:dispatch' });
2986
3298
  s.jobs[idx].runId = runId;
2987
3299
  s.jobs[idx].startedAt = new Date().toISOString();
2988
3300
  }
@@ -3069,7 +3381,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3069
3381
  await mutate((s) => {
3070
3382
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
3071
3383
  if (idx >= 0) {
3072
- s.jobs[idx].status = 'completed';
3384
+ transitionJob(s.jobs[idx], 'completed', { reason: res.note ?? 'PRD archived or missing — treated as already-shipped', source: 'spawnJob:skip-archived' });
3073
3385
  s.jobs[idx].finishedAt = new Date().toISOString();
3074
3386
  s.jobs[idx].exitCode = 0;
3075
3387
  s.jobs[idx].error = null;
@@ -3239,10 +3551,19 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3239
3551
  const newlyCompletedPrds = [];
3240
3552
  await mutate((s) => {
3241
3553
  const i2 = s.jobs.findIndex((x) => x.slug === job.slug);
3554
+ // A job already moved off 'running' by someone else (namely
3555
+ // remote.cancelJob, PRD 1024 — it SIGTERMs the process then finalizes
3556
+ // the row to 'failed' before this exit handler necessarily runs) is
3557
+ // not this run's to finalize: doing so anyway could re-legalize the
3558
+ // row via a legal failed->completed/needs_review edge (see
3559
+ // scheduleJobTransitions.cjs's LEGAL_TRANSITIONS) and silently
3560
+ // undo the cancellation. Skip — the row already reflects its real
3561
+ // terminal state.
3562
+ if (i2 >= 0 && s.jobs[i2].status !== 'running') return;
3242
3563
  if (i2 >= 0) {
3243
3564
  const treatAsPending = res.rateLimited || (s.paused && s.paused.reason === 'rate_limit');
3244
3565
  if (treatAsPending) {
3245
- resetJobFields(s.jobs[i2], res.rateLimited ? 'paused: rate limit' : 'paused: queue halted');
3566
+ resetJobFields(s.jobs[i2], res.rateLimited ? 'paused: rate limit' : 'paused: queue halted', { source: 'spawnJob:halt-reset' });
3246
3567
  } else {
3247
3568
  // Determine effective status, applying the verifier verdict for exit=0 runs.
3248
3569
  let effectiveStatus;
@@ -3262,14 +3583,14 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3262
3583
  effectiveStatus = 'completed';
3263
3584
  } else if (verifyResult.downgradeTo === 'pending') {
3264
3585
  // HALT or deps_unmet: reset to pending so the job re-fires.
3265
- resetJobFields(s.jobs[i2], verifyResult.reason);
3586
+ resetJobFields(s.jobs[i2], verifyResult.reason, { source: 'spawnJob:verify-downgrade' });
3266
3587
  return; // job already mutated by resetJobFields; skip the rest
3267
3588
  } else {
3268
3589
  // transcript_errors or verify_unavailable: escalate to needs_review.
3269
3590
  effectiveStatus = 'needs_review';
3270
3591
  }
3271
3592
 
3272
- s.jobs[i2].status = effectiveStatus;
3593
+ transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
3273
3594
  s.jobs[i2].finishedAt = new Date().toISOString();
3274
3595
  s.jobs[i2].exitCode = res.exitCode;
3275
3596
  s.jobs[i2].error = effectiveStatus === 'needs_review'
@@ -3356,7 +3677,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3356
3677
  if (orig) {
3357
3678
  const priorStatus = orig.status;
3358
3679
  console.log(`[scheduler] auto-promote: ${orig.slug} (${priorStatus}) → completed because ${job.slug} succeeded`);
3359
- orig.status = 'completed';
3680
+ transitionJob(orig, 'completed', { reason: `auto-promoted: fix plan ${job.slug} succeeded`, source: 'spawnJob:auto-promote' });
3360
3681
  orig.exitCode = 0;
3361
3682
  orig.error = null;
3362
3683
  orig.completedBy = job.slug;
@@ -3444,7 +3765,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3444
3765
  await mutate((s) => {
3445
3766
  const i = s.jobs.findIndex((x) => x.slug === job.slug);
3446
3767
  if (i >= 0) {
3447
- resetJobFields(s.jobs[i], null);
3768
+ resetJobFields(s.jobs[i], null, { source: 'spawnJob:transient-retry' });
3448
3769
  s.jobs[i].transientRetries = decision.retries + 1;
3449
3770
  }
3450
3771
  });
@@ -3454,7 +3775,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3454
3775
  await mutate((s) => {
3455
3776
  const i = s.jobs.findIndex((x) => x.slug === job.slug);
3456
3777
  if (i >= 0) {
3457
- s.jobs[i].status = 'failed';
3778
+ transitionJob(s.jobs[i], 'failed', { reason: `transient failure (${decision.transientKind}) left uncommitted work — not auto-requeued`, source: 'spawnJob:fail-dirty' });
3458
3779
  s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample}) — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
3459
3780
  }
3460
3781
  });
@@ -3515,6 +3836,9 @@ function tickQueue() {
3515
3836
  }
3516
3837
  if (cancelToken.cancelled) return { fired: false, reason: 'cancelled' };
3517
3838
 
3839
+ // The retired-flat-dir sweep now lives inside reconcile() itself (see its
3840
+ // own comment) so every caller of reconcile — not just this tick — gets
3841
+ // the guarantee.
3518
3842
  await reconcile(state);
3519
3843
  // Session-Manager's machine-wide slot pool is the ONLY concurrency limit
3520
3844
  // the picker answers to (plus the memory gate below). The scheduler used
@@ -3702,7 +4026,7 @@ async function reapDeadRunningJobs() {
3702
4026
  const idx = s.jobs.findIndex((x) => x.slug === slug);
3703
4027
  if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
3704
4028
  const success = outcome === 'success';
3705
- s.jobs[idx].status = success ? 'completed' : 'failed';
4029
+ transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: `reaped: process gone (outcome=${outcome})`, source: 'reapDeadRunningJobs' });
3706
4030
  s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
3707
4031
  s.jobs[idx].finishedAt = new Date().toISOString();
3708
4032
  s.jobs[idx].error = success ? null : `reaped: process gone, no success result in log (${outcome})`;
@@ -4125,7 +4449,7 @@ async function reverifyNeedsReview() {
4125
4449
  await mutate((s) => {
4126
4450
  for (const j of s.jobs) {
4127
4451
  if (j.status === 'needs_review' && healSet.has(j.slug)) {
4128
- j.status = 'completed';
4452
+ transitionJob(j, 'completed', { reason: 'boot reverify: stale needs_review healed', source: 'reverifyNeedsReview:heal' });
4129
4453
  j.error = null;
4130
4454
  delete j.verifierVerdict;
4131
4455
  healedPrds.push({ slug: j.slug, cwd: j.cwd });
@@ -4163,7 +4487,7 @@ async function reverifyNeedsReview() {
4163
4487
  const orig = healTargetForFix(job.slug, s.jobs);
4164
4488
  if (!orig) continue;
4165
4489
  const priorStatus = orig.status;
4166
- orig.status = 'completed';
4490
+ transitionJob(orig, 'completed', { reason: `auto-promoted: fix plan ${job.slug} already completed`, source: 'reverifyNeedsReview:auto-promote' });
4167
4491
  orig.exitCode = 0;
4168
4492
  orig.error = null;
4169
4493
  orig.completedBy = job.slug;
@@ -4341,7 +4665,7 @@ function registerScheduleHandlers() {
4341
4665
  // Guard is in resetJobFields: refuses to reset an already-'completed'
4342
4666
  // job, which would otherwise re-fire a PRD whose deliverable already
4343
4667
  // landed (see resetJobFields' doc comment for the incident).
4344
- return resetJobFields(state.jobs[idx]) ? 'ok' : 'refused';
4668
+ return resetJobFields(state.jobs[idx], null, { source: 'ipc:schedule:reset-job' }) ? 'ok' : 'refused';
4345
4669
  });
4346
4670
  if (outcome === 'not-found') return { ok: false, error: 'not found' };
4347
4671
  if (outcome === 'refused') {
@@ -4354,6 +4678,36 @@ function registerScheduleHandlers() {
4354
4678
  return { ok: true };
4355
4679
  }));
4356
4680
 
4681
+ // Renderer-facing counterpart to prdCreate.cjs's chat:create-prd handler
4682
+ // (index.cjs): calls the SAME remote.updatePrd the admin HTTP route/MCP
4683
+ // tool use, so "stamps it through the API" holds for the Scheduler tab's
4684
+ // one-click adopt action too, not just a direct fs write. Only a
4685
+ // 'quarantined' row is eligible — see reconcile()'s provenance gate.
4686
+ ipcMain.handle('schedule:adopt-prd', validated(schemas.scheduleSlug, async ({ slug }) => {
4687
+ if (!(await safeSlugPath(slug))) return { ok: false, kind: 'error', message: 'invalid slug' };
4688
+ const state = await readQueue();
4689
+ const job = state.jobs.find((j) => j.slug === slug);
4690
+ if (!job) return { ok: false, kind: 'error', message: 'not found' };
4691
+ if (job.status !== 'quarantined') {
4692
+ return { ok: false, kind: 'error', message: `job status is "${job.status}" — only a quarantined PRD may be adopted` };
4693
+ }
4694
+ const result = await remote.updatePrd({
4695
+ slug,
4696
+ cwd: job.cwd,
4697
+ frontmatter: { createdVia: 'legacy-adopted', issuedAt: new Date().toISOString() },
4698
+ });
4699
+ if (!result.ok) return { ok: false, kind: 'error', message: result.error ?? 'adopt failed' };
4700
+ appendAuditEvent('scheduler_prd_adopted', { slug, cwd: job.cwd ?? null, source: 'ipc:schedule:adopt-prd' });
4701
+ // Promote the row to 'pending' immediately rather than waiting for the
4702
+ // next poll tick — the Scheduler tab's "adopt PRD" click should be
4703
+ // visibly effective within this one round-trip.
4704
+ const freshState = await readQueue();
4705
+ await reconcile(freshState);
4706
+ await writeQueue(freshState);
4707
+ await broadcast({ flush: true });
4708
+ return { ok: true, kind: 'info', message: `Adopted ${slug} — it will run as a normal pending job` };
4709
+ }));
4710
+
4357
4711
  ipcMain.handle('schedule:run-now', async () => {
4358
4712
  // Manual run-now overrides any auto-pause. Clear it first.
4359
4713
  await clearPause('run-now');
@@ -4483,85 +4837,7 @@ function registerScheduleHandlers() {
4483
4837
  }
4484
4838
  }));
4485
4839
 
4486
- ipcMain.handle('schedule:list-prds', async () => {
4487
- ensureDirs();
4488
- const out = [];
4489
- const seenSlugs = new Set();
4490
-
4491
- async function readDirInto(dir, { archived }) {
4492
- let entries;
4493
- try {
4494
- entries = await fsp.readdir(dir);
4495
- } catch (e) {
4496
- if (e?.code !== 'ENOENT') {
4497
- logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: readdir failed', meta: { dir, error: e?.message } });
4498
- }
4499
- return;
4500
- }
4501
- for (const name of entries) {
4502
- if (!name.endsWith('.md') || name.startsWith('.')) continue;
4503
- const filePath = path.join(dir, name);
4504
- try {
4505
- const parsed = await parsePrd(filePath);
4506
- // A slug can't be both live and archived at once, but a duplicate
4507
- // slug found in two archive dirs (shouldn't happen — archiving is
4508
- // a single rename — but is cheap to guard) is skipped rather than
4509
- // double-counted.
4510
- if (seenSlugs.has(parsed.slug)) continue;
4511
- seenSlugs.add(parsed.slug);
4512
- const stat = await fsp.stat(filePath);
4513
- const entry = {
4514
- slug: parsed.slug,
4515
- parallelGroup: parsed.parallelGroup,
4516
- title: parsed.title,
4517
- cwd: parsed.cwd || '',
4518
- estimateMinutes: parsed.estimateMinutes,
4519
- sourcePromptId: parsed.sourcePromptId,
4520
- epicId: parsed.epicId ?? null,
4521
- mtimeMs: stat.mtimeMs,
4522
- archived,
4523
- };
4524
- out.push(entry);
4525
- } catch (e) {
4526
- logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: skipping unparseable file', meta: { name, error: e?.message } });
4527
- }
4528
- }
4529
- }
4530
-
4531
- // Live PRDs first, so an archived duplicate (shouldn't exist, but a
4532
- // stale rename copy is possible) never shadows the still-runnable live
4533
- // entry.
4534
- for (const dir of candidatePrdsDirs()) {
4535
- await readDirInto(dir, { archived: false });
4536
- }
4537
-
4538
- const archivedStart = out.length;
4539
- for (const dir of candidateArchivedPrdsDirs()) {
4540
- await readDirInto(dir, { archived: true });
4541
- }
4542
-
4543
- // Archived PRDs need a status: archiveCompletedPrd (scheduler.cjs) only
4544
- // ever archives a job whose effective status is 'completed' — a 'failed'
4545
- // job's PRD source stays in the live prds/ dir (still visible/countable
4546
- // there already). Still resolve the real job status defensively (live
4547
- // queue row, falling back to history.jsonl) rather than hard-coding
4548
- // 'completed', so this stays correct if that archiving invariant ever
4549
- // changes.
4550
- if (out.length > archivedStart) {
4551
- const [state, histBySlug] = await Promise.all([
4552
- readQueue(),
4553
- queueHistory.historyTerminalBySlug().catch(() => new Map()),
4554
- ]);
4555
- const liveStatusBySlug = new Map(state.jobs.map((j) => [j.slug, j.status]));
4556
- for (let i = archivedStart; i < out.length; i++) {
4557
- const entry = out[i];
4558
- entry.archivedStatus = resolveArchivedPrdStatus(entry.slug, liveStatusBySlug, histBySlug);
4559
- }
4560
- }
4561
-
4562
- out.sort((a, b) => a.slug.localeCompare(b.slug, undefined, { numeric: true }));
4563
- return out;
4564
- });
4840
+ ipcMain.handle('schedule:list-prds', async () => listPrdsInternal());
4565
4841
 
4566
4842
  // Return last N completed/failed jobs from queue.json, newest first.
4567
4843
  // Purely additive: no schema change, no archive-folder read needed.
@@ -4729,6 +5005,7 @@ async function init() {
4729
5005
  if (rescheduleInterval) clearInterval(rescheduleInterval);
4730
5006
  rescheduleInterval = setInterval(() => {
4731
5007
  rescheduleTimer().catch(() => {});
5008
+ const s = readQueueSync();
4732
5009
  // Periodic self-heal: re-run the verifier over stale needs_review jobs so a
4733
5010
  // job whose work actually landed (committed in-window, no FAIL sentinel)
4734
5011
  // auto-clears WITHOUT waiting for the next app restart. Cheap-guarded — the
@@ -4738,10 +5015,32 @@ async function init() {
4738
5015
  // MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
4739
5016
  // past it), so this interval firing cannot fan out investigations.
4740
5017
  if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
4741
- const s = readQueueSync();
4742
5018
  if (s.jobs.some((j) => j.status === 'needs_review')) {
4743
5019
  reverifyNeedsReview().catch(() => {});
4744
5020
  }
5021
+ // A quarantined row only ever promotes to 'pending' through
5022
+ // reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
5023
+ // it re-checks the PRD file's createdVia stamp every pass. broadcast()
5024
+ // already runs reconcile+writeQueue on every normal poll tick, but an
5025
+ // idle queue (nothing pending/running to fire) can back off that
5026
+ // cadence for a long time; this guarantees an adopted-but-still-
5027
+ // quarantined row is re-checked within 10 minutes regardless.
5028
+ if (s.jobs.some((j) => j.status === 'quarantined')) {
5029
+ broadcast().catch(() => {});
5030
+ }
5031
+ }
5032
+ // Age-based escalation (independent of the self-heal kill-switch above —
5033
+ // this is a monitoring signal, not an auto-fix action): a quarantined
5034
+ // row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
5035
+ // warn-logged by project + slug + age so it cannot sit stranded and
5036
+ // silent (the four burrow-project rows this PRD was written against).
5037
+ for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
5038
+ console.warn(
5039
+ `[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
5040
+ + `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
5041
+ + `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
5042
+ );
5043
+ appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
4745
5044
  }
4746
5045
  }, 10 * 60_000);
4747
5046
 
@@ -4766,12 +5065,69 @@ async function init() {
4766
5065
  if (heartbeatInterval) clearInterval(heartbeatInterval);
4767
5066
  heartbeatInterval = setInterval(() => {
4768
5067
  const s = readQueueSync();
4769
- const counts = { pending: 0, running: 0, completed: 0, failed: 0 };
4770
- for (const j of s.jobs) counts[j.status] = (counts[j.status] || 0) + 1;
5068
+ // Initialise from the real status union (scheduleJobSchema.cjs) rather
5069
+ // than a hand-maintained subset — the old `{ pending, running, completed,
5070
+ // failed }` literal silently minted a NEW key for any other value
5071
+ // (`counts[j.status] = (counts[j.status]||0)+1`), which is exactly how a
5072
+ // heartbeat with a `queued: 2` bucket looked like "normal" 24h
5073
+ // visibility instead of the alarm it should have been. Any row whose
5074
+ // status isn't in JOB_STATUSES (shouldn't happen post-quarantine, but
5075
+ // this is the last line of defence) routes into `unknown`, never a
5076
+ // freshly-minted key.
5077
+ const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
5078
+ counts.unknown = 0;
5079
+ for (const j of s.jobs) {
5080
+ if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
5081
+ counts[j.status] += 1;
5082
+ } else {
5083
+ counts.unknown += 1;
5084
+ }
5085
+ }
5086
+
5087
+ const stall = computeStallSummary(s);
5088
+ // Per-project alerting (see computeStallSummary's header): a project
5089
+ // stalled while others are busy must still fire, and one project
5090
+ // recovering must not clear or suppress another's still-open episode —
5091
+ // that is exactly what a single module-level stallSince/stallToasted
5092
+ // flag masked before (the burrow-vs-others incident this PRD fixes).
5093
+ const now = Date.now();
5094
+ const stalledCwds = Object.keys(stall.byProject).filter((cwd) => stall.byProject[cwd].stalled);
5095
+ for (const cwd of [...stallSince.keys()]) {
5096
+ if (!stalledCwds.includes(cwd)) {
5097
+ stallSince.delete(cwd);
5098
+ stallToasted.delete(cwd);
5099
+ }
5100
+ }
5101
+ const toAlert = [];
5102
+ for (const cwd of stalledCwds) {
5103
+ if (!stallSince.has(cwd)) stallSince.set(cwd, now);
5104
+ if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
5105
+ stallToasted.set(cwd, true);
5106
+ toAlert.push(cwd);
5107
+ }
5108
+ }
5109
+ if (toAlert.length > 0) {
5110
+ console.error(
5111
+ `[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
5112
+ + `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
5113
+ stall.byProject,
5114
+ );
5115
+ appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: stall.total, byProject: stall.byProject });
5116
+ if (mainWindow && !mainWindow.isDestroyed()) {
5117
+ sendIfAlive(mainWindow, 'schedule:stall', {
5118
+ message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
5119
+ projects: toAlert,
5120
+ total: stall.total,
5121
+ byProject: stall.byProject,
5122
+ });
5123
+ }
5124
+ }
5125
+
4771
5126
  appendHeartbeat({
4772
5127
  ts: Date.now(),
4773
5128
  pid: process.pid,
4774
5129
  counts,
5130
+ stall: { stalled: stall.stalled, total: stall.total },
4775
5131
  paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
4776
5132
  nextReset: cachedNextReset,
4777
5133
  utilization: cachedUtilization,
@@ -4802,6 +5158,102 @@ async function init() {
4802
5158
  }
4803
5159
  }
4804
5160
 
5161
+ /**
5162
+ * listPrdsInternal() → every live + archived PRD across every project,
5163
+ * with each entry's real job status folded in (`status`: the live queue
5164
+ * row's status, or the resolved terminal status for an archived entry, or
5165
+ * null when no queue row exists yet — e.g. a PRD just written and not yet
5166
+ * picked up by reconcile()). Single source of truth for both the renderer's
5167
+ * `schedule:list-prds` IPC handler and the admin HTTP `GET
5168
+ * /admin/scheduler/prds` route (PRD 1024) — neither re-implements this scan.
5169
+ */
5170
+ async function listPrdsInternal() {
5171
+ ensureDirs();
5172
+ const out = [];
5173
+ const seenSlugs = new Set();
5174
+
5175
+ async function readDirInto(dir, { archived }) {
5176
+ let entries;
5177
+ try {
5178
+ entries = await fsp.readdir(dir);
5179
+ } catch (e) {
5180
+ if (e?.code !== 'ENOENT') {
5181
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: readdir failed', meta: { dir, error: e?.message } });
5182
+ }
5183
+ return;
5184
+ }
5185
+ for (const name of entries) {
5186
+ if (!name.endsWith('.md') || name.startsWith('.')) continue;
5187
+ const filePath = path.join(dir, name);
5188
+ try {
5189
+ const parsed = await parsePrd(filePath);
5190
+ // A slug can't be both live and archived at once, but a duplicate
5191
+ // slug found in two archive dirs (shouldn't happen — archiving is
5192
+ // a single rename — but is cheap to guard) is skipped rather than
5193
+ // double-counted.
5194
+ if (seenSlugs.has(parsed.slug)) continue;
5195
+ seenSlugs.add(parsed.slug);
5196
+ const stat = await fsp.stat(filePath);
5197
+ const entry = {
5198
+ slug: parsed.slug,
5199
+ parallelGroup: parsed.parallelGroup,
5200
+ title: parsed.title,
5201
+ cwd: parsed.cwd || '',
5202
+ estimateMinutes: parsed.estimateMinutes,
5203
+ sourcePromptId: parsed.sourcePromptId,
5204
+ epicId: parsed.epicId ?? null,
5205
+ mtimeMs: stat.mtimeMs,
5206
+ archived,
5207
+ };
5208
+ out.push(entry);
5209
+ } catch (e) {
5210
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: skipping unparseable file', meta: { name, error: e?.message } });
5211
+ }
5212
+ }
5213
+ }
5214
+
5215
+ // Live PRDs first, so an archived duplicate (shouldn't exist, but a
5216
+ // stale rename copy is possible) never shadows the still-runnable live
5217
+ // entry.
5218
+ for (const dir of candidatePrdsDirs()) {
5219
+ await readDirInto(dir, { archived: false });
5220
+ }
5221
+
5222
+ const archivedStart = out.length;
5223
+ for (const dir of candidateArchivedPrdsDirs()) {
5224
+ await readDirInto(dir, { archived: true });
5225
+ }
5226
+
5227
+ // Every entry (live and archived) gets a real job status folded in.
5228
+ // Archived PRDs need one resolved defensively (live queue row, falling
5229
+ // back to history.jsonl) rather than hard-coded 'completed', so this
5230
+ // stays correct if the archive-only-completed invariant ever changes; a
5231
+ // live entry with no queue row yet (just written, not yet reconciled)
5232
+ // gets `status: null`.
5233
+ const [state, histBySlug] = await Promise.all([
5234
+ readQueue(),
5235
+ queueHistory.historyTerminalBySlug().catch(() => new Map()),
5236
+ ]);
5237
+ const liveStatusBySlug = new Map(state.jobs.map((j) => [j.slug, j.status]));
5238
+ for (let i = 0; i < out.length; i++) {
5239
+ const entry = out[i];
5240
+ // `entry` is a freshly-synthesized PRD-listing row, not a persisted
5241
+ // ScheduleJob — assigning its `status` here is not a queue-job status
5242
+ // transition (no queue.json row is mutated, no statusHistory/audit
5243
+ // trail applies), so it is intentionally exempt from the
5244
+ // transitionJob-only rule enforced by scheduleJobTransitionsGrep.test.cjs.
5245
+ if (i < archivedStart) {
5246
+ entry.status = liveStatusBySlug.get(entry.slug) ?? null;
5247
+ } else {
5248
+ entry.archivedStatus = resolveArchivedPrdStatus(entry.slug, liveStatusBySlug, histBySlug);
5249
+ entry.status = entry.archivedStatus;
5250
+ }
5251
+ }
5252
+
5253
+ out.sort((a, b) => a.slug.localeCompare(b.slug, undefined, { numeric: true }));
5254
+ return out;
5255
+ }
5256
+
4805
5257
  // remote — in-process (non-IPC) scheduler accessors, used by prdCreate.cjs
4806
5258
  // and other main-process callers. (Named for the retired web-remote relay,
4807
5259
  // its original consumer; kept because it still has in-process callers.)
@@ -4921,7 +5373,7 @@ const remote = {
4921
5373
  // Best-effort: record the dispatch on the Epic's event chain.
4922
5374
  try { await appendPrdCreatedEvent(cwd, epicTrace, slug); } catch { /* trace only */ }
4923
5375
  }
4924
- return { ok: true, bytesWritten: stat.size };
5376
+ return { ok: true, bytesWritten: stat.size, path: resolved, epicId: epicTrace };
4925
5377
  } catch (e) {
4926
5378
  return { ok: false, error: e?.message ?? 'write failed' };
4927
5379
  }
@@ -4934,7 +5386,7 @@ const remote = {
4934
5386
  if (idx < 0) return { kind: 'not-found' };
4935
5387
  // Terminal-status guard lives in resetJobFields itself; force:true
4936
5388
  // threads through to override it.
4937
- if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true })) {
5389
+ if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true, source: 'remote:resetJob' })) {
4938
5390
  return { kind: 'refused' };
4939
5391
  }
4940
5392
  return { kind: 'ok' };
@@ -4955,6 +5407,185 @@ const remote = {
4955
5407
  return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd }));
4956
5408
  },
4957
5409
 
5410
+ // Single queue row lookup, used by cancelJob/updatePrd's status guards and
5411
+ // the admin GET /admin/scheduler/prds?slug= route (PRD 1024).
5412
+ async getJob(slug) {
5413
+ const state = await readQueue();
5414
+ const job = state.jobs.find((j) => j.slug === slug);
5415
+ return job ? { slug: job.slug, title: job.title, status: job.status, cwd: job.cwd, error: job.error ?? null } : null;
5416
+ },
5417
+
5418
+ // Every live+archived PRD across every project (listPrdsInternal, shared
5419
+ // with the renderer's schedule:list-prds IPC handler), filtered by the
5420
+ // admin route's cwd/epicId/status query params.
5421
+ async listPrds(filter = {}) {
5422
+ const all = await listPrdsInternal();
5423
+ return all.filter((entry) => {
5424
+ if (filter.cwd && entry.cwd !== filter.cwd) return false;
5425
+ if (filter.epicId && entry.epicId !== filter.epicId) return false;
5426
+ if (filter.status && entry.status !== filter.status) return false;
5427
+ return true;
5428
+ });
5429
+ },
5430
+
5431
+ // Full body + parsed frontmatter for one PRD, live or archived. Mirrors
5432
+ // readPrd's dir-search + symlink-defense pattern (see that method's
5433
+ // comment) rather than sharing code with it, since readPrd intentionally
5434
+ // returns raw text only and is a much narrower/hotter path (executeJob's
5435
+ // PRD re-reads) that shouldn't grow a second return shape.
5436
+ async getPrdParsed(slug, cwd) {
5437
+ let dir = null;
5438
+ let filePath = null;
5439
+ if (cwd) {
5440
+ for (const d of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
5441
+ const p = safeSlugPathIn(d, slug);
5442
+ if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
5443
+ }
5444
+ if (!filePath) {
5445
+ for (const d of listArchivedPrdDirs(cwd)) {
5446
+ const p = safeSlugPathIn(d, slug);
5447
+ if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
5448
+ }
5449
+ }
5450
+ } else {
5451
+ dir = await findPrdDir(slug);
5452
+ filePath = dir ? safeSlugPathIn(dir, slug) : null;
5453
+ if (!filePath) {
5454
+ for (const d of candidateArchivedPrdsDirs()) {
5455
+ const p = safeSlugPathIn(d, slug);
5456
+ if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
5457
+ }
5458
+ }
5459
+ }
5460
+ if (!filePath) return { ok: false, error: 'invalid slug' };
5461
+ try {
5462
+ // Symlink defense, matching readPrd/writePrd's comment: safeSlugPathIn
5463
+ // is lexical and does not resolve symlinks.
5464
+ const real = await fsp.realpath(filePath);
5465
+ if (!real.startsWith(dir + path.sep)) return { ok: false, error: 'invalid slug' };
5466
+ const [raw, parsed] = await Promise.all([fsp.readFile(real, 'utf8'), prdParser.parsePrdRaw(real)]);
5467
+ return {
5468
+ ok: true,
5469
+ slug: parsed.slug,
5470
+ frontmatter: {
5471
+ title: parsed.title,
5472
+ cwd: parsed.cwd,
5473
+ estimateMinutes: parsed.estimateMinutes,
5474
+ parallelGroup: parsed.parallelGroup,
5475
+ sourcePromptId: parsed.sourcePromptId,
5476
+ sourceTabId: parsed.sourceTabId,
5477
+ epicId: parsed.epicId,
5478
+ dependsOn: parsed.dependsOn,
5479
+ createdVia: parsed.createdVia,
5480
+ issuedAt: parsed.issuedAt,
5481
+ },
5482
+ body: parsed.body,
5483
+ raw,
5484
+ };
5485
+ } catch (e) {
5486
+ return { ok: false, error: e?.message ?? 'read failed' };
5487
+ }
5488
+ },
5489
+
5490
+ // Edits a NOT-yet-running PRD's frontmatter and/or body in place, refusing
5491
+ // once a queue row exists for it and that row is anything but 'pending'
5492
+ // (running/completed/failed/needs_review — editing the spec under a live
5493
+ // or already-finished executor would silently rewrite history). Reuses
5494
+ // prdFrontmatter.cjs's parsePrdFile/serializePrdFile round-trip pair (PRD
5495
+ // 1024) so unrecognized keys (e.g. dependsOn) and untouched recognized
5496
+ // keys' original line formatting survive unchanged.
5497
+ async updatePrd({ slug, cwd, frontmatter, body }) {
5498
+ const job = await this.getJob(slug);
5499
+ // 'quarantined' is also editable: it's the ONLY way a quarantined PRD's
5500
+ // createdVia stamp gets written (the adopt action below), so refusing it
5501
+ // here would make quarantine irreversible through the API.
5502
+ if (job && job.status !== 'pending' && job.status !== 'quarantined') {
5503
+ return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
5504
+ }
5505
+
5506
+ let dir = null;
5507
+ let filePath = null;
5508
+ if (cwd) {
5509
+ for (const d of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
5510
+ const p = safeSlugPathIn(d, slug);
5511
+ if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
5512
+ }
5513
+ } else {
5514
+ dir = await findPrdDir(slug);
5515
+ filePath = dir ? safeSlugPathIn(dir, slug) : null;
5516
+ }
5517
+ if (!filePath) return { ok: false, error: 'PRD not found' };
5518
+
5519
+ let raw;
5520
+ try {
5521
+ // Symlink defense, matching writePrd's comment: safeSlugPathIn is
5522
+ // lexical and does not resolve symlinks. updatePrd is a WRITE path
5523
+ // (unlike getPrdParsed's read-only realpath check), so also reject a
5524
+ // target that is itself already a symlink — a rogue job could plant
5525
+ // one inside the PRDs dir pointing outside the safe root.
5526
+ const real = await fsp.realpath(filePath);
5527
+ if (!real.startsWith(dir + path.sep)) return { ok: false, error: 'invalid slug' };
5528
+ const existing = await fsp.lstat(filePath).catch(() => null);
5529
+ if (existing && existing.isSymbolicLink()) return { ok: false, error: 'invalid slug' };
5530
+ raw = await fsp.readFile(real, 'utf8');
5531
+ } catch (e) {
5532
+ return { ok: false, error: e?.message ?? 'read failed' };
5533
+ }
5534
+
5535
+ const { frontmatter: fm, body: origBody } = parsePrdFile(raw);
5536
+ if (frontmatter) {
5537
+ for (const key of Object.keys(frontmatter)) {
5538
+ if (frontmatter[key] === undefined) continue;
5539
+ fm[key] = frontmatter[key];
5540
+ }
5541
+ }
5542
+ const newBody = body !== undefined ? body : origBody;
5543
+ const newRaw = serializePrdFile(fm, newBody);
5544
+
5545
+ try {
5546
+ await config.writeTextAtomic(filePath, newRaw, { writer: 'scheduler' });
5547
+ const stat = await fsp.stat(filePath);
5548
+ return { ok: true, slug, bytesWritten: stat.size };
5549
+ } catch (e) {
5550
+ return { ok: false, error: e?.message ?? 'write failed' };
5551
+ }
5552
+ },
5553
+
5554
+ // Cancels a job that hasn't finished yet. A 'running' job's process group
5555
+ // is SIGTERM'd (reusing killOrphanClaudePid — the same kill path boot
5556
+ // reconciliation uses for an orphaned running job) before its queue row is
5557
+ // finalized; a 'pending' job has no process to kill. There is no
5558
+ // 'cancelled' status in the closed job-status set (pending/running/
5559
+ // completed/failed/needs_review — see CLAUDE.md's domain model), so a
5560
+ // cancelled job lands in 'failed' with an error naming the cause,
5561
+ // consistent with every other non-success terminal outcome. Refuses a
5562
+ // slug that's already terminal — nothing left to cancel.
5563
+ async cancelJob(slug) {
5564
+ const state = await readQueue();
5565
+ const job = state.jobs.find((j) => j.slug === slug);
5566
+ if (!job) return { ok: false, error: 'not found' };
5567
+ if (job.status === 'completed' || job.status === 'failed' || job.status === 'needs_review') {
5568
+ return { ok: false, error: `job already terminal (status: "${job.status}") — nothing to cancel` };
5569
+ }
5570
+ const wasRunning = job.status === 'running';
5571
+ const pid = job.runtime?.pid;
5572
+ if (wasRunning && pid) {
5573
+ killOrphanClaudePid(pid);
5574
+ }
5575
+ await mutate((s) => {
5576
+ const idx = s.jobs.findIndex((j) => j.slug === slug);
5577
+ if (idx < 0) return;
5578
+ const j = s.jobs[idx];
5579
+ transitionJob(j, 'failed', { reason: 'cancelled via admin API', source: 'remote:cancelJob' });
5580
+ j.error = 'cancelled via admin API';
5581
+ j.finishedAt = new Date().toISOString();
5582
+ j.exitCode = j.exitCode ?? null;
5583
+ delete j.runtime;
5584
+ });
5585
+ await broadcast({ flush: true });
5586
+ return { ok: true, slug, status: 'failed', wasRunning, cwd: job.cwd ?? null };
5587
+ },
5588
+
4958
5589
  // Exposes the module-level allocateParallelGroup (PRD 548) to callers that
4959
5590
  // only hold the `remote` object (lib/prdCreate.cjs's create-prd route) —
4960
5591
  // reuses the same allocator the file-based /develop authoring path relies
@@ -4995,4 +5626,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
4995
5626
  });
4996
5627
  }
4997
5628
 
4998
- module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveNotifyPrd, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob };
5629
+ module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS };