claude-code-session-manager 0.65.0 → 0.66.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/dist/assets/AgentLibrary-Bzg89D5Y.js +3 -0
  2. package/dist/assets/{History-DB-9-zwc.js → History-CNH9vA0A.js} +2 -2
  3. package/dist/assets/{Hooks-DnxqMRrR.js → Hooks-CyTksPza.js} +3 -3
  4. package/dist/assets/{HostBilko-Cw6JocV7.js → HostBilko-DXQVDNHn.js} +1 -1
  5. package/dist/assets/{Library-BYSB0dmY.js → Library-9E2UOIsm.js} +1 -1
  6. package/dist/assets/{ListDetail-BfKqnL0r.js → ListDetail-CQWU_Yn5.js} +1 -1
  7. package/dist/assets/{MarkdownEditor-Cs-l5JOv.js → MarkdownEditor-DQgfpSef.js} +1 -1
  8. package/dist/assets/{McpServers-bBqj3hGg.js → McpServers-r3qwDIj2.js} +2 -2
  9. package/dist/assets/{Memory-DzPGXT5J.js → Memory-BaxOpj-3.js} +6 -6
  10. package/dist/assets/{Panel-BX6UZ-W0.js → Panel-BhFD8Lqo.js} +1 -1
  11. package/dist/assets/{Permissions-Db7S4T3Z.js → Permissions-Cj-mODQQ.js} +3 -3
  12. package/dist/assets/{Plugins-CIuey9R3.js → Plugins-DEL3Fqng.js} +2 -2
  13. package/dist/assets/{ProvenanceBadge-FPWNWc_-.js → ProvenanceBadge-CIAg6-JQ.js} +1 -1
  14. package/dist/assets/Scheduler-YOuZKkES.js +14 -0
  15. package/dist/assets/{ScopeSwitcher-DU7M_q5-.js → ScopeSwitcher-DZ_3gEus.js} +1 -1
  16. package/dist/assets/{Settings-DEptXcEY.js → Settings-B0x4oflz.js} +3 -3
  17. package/dist/assets/{SkillReferenceGraph-CprLcOCe.js → SkillReferenceGraph-CoIwsol8.js} +1 -1
  18. package/dist/assets/{Skills-CGN56X1i.js → Skills-CN8R6AWn.js} +2 -2
  19. package/dist/assets/{SystemPrompt-DIEK7FpJ.js → SystemPrompt-DceChpAi.js} +1 -1
  20. package/dist/assets/{TagLibrary-q66s4N3i.js → TagLibrary-BKmz2W7B.js} +1 -1
  21. package/dist/assets/{TiptapBody-I22aArc2.js → TiptapBody-W7n5SwPM.js} +1 -1
  22. package/dist/assets/{Toggle-D9sapSAv.js → Toggle-QuxlVHGI.js} +1 -1
  23. package/dist/assets/{index-LlWpj2VJ.css → index-14dBLqE_.css} +1 -1
  24. package/dist/assets/{index-BRwaw_1W.js → index-TejhHSzN.js} +526 -525
  25. package/dist/assets/{settingsSchema-CTc4qelV.js → settingsSchema-DNqx6BKJ.js} +1 -1
  26. package/dist/index.html +2 -2
  27. package/package.json +1 -1
  28. package/plugins/session-manager-dev/skills/develop/SKILL.md +41 -13
  29. package/plugins/session-manager-dev/skills/ops-sweep/SKILL.md +10 -0
  30. package/src/main/__tests__/agentLibrary.test.cjs +40 -0
  31. package/src/main/__tests__/flatPrdTickSweep.test.cjs +110 -0
  32. package/src/main/__tests__/prdAdminRouteParity.test.cjs +68 -0
  33. package/src/main/__tests__/prdAdminRoutes.test.cjs +311 -0
  34. package/src/main/__tests__/prdCreate.test.cjs +7 -2
  35. package/src/main/__tests__/prdMigration.test.cjs +17 -0
  36. package/src/main/__tests__/prdMigrationLegacyAdopt.test.cjs +91 -0
  37. package/src/main/__tests__/reconcileFlatPrdSweep.test.cjs +109 -0
  38. package/src/main/__tests__/scheduleJobSchema.test.cjs +127 -0
  39. package/src/main/__tests__/scheduleJobStatusDrift.test.cjs +65 -0
  40. package/src/main/__tests__/scheduleJobTransitions.test.cjs +152 -0
  41. package/src/main/__tests__/scheduleJobTransitionsGrep.test.cjs +59 -0
  42. package/src/main/__tests__/scheduler-reconcile-invalid-repair.test.cjs +203 -0
  43. package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +196 -0
  44. package/src/main/agentLibrary.cjs +40 -2
  45. package/src/main/index.cjs +2 -0
  46. package/src/main/ipcSchemas.cjs +60 -0
  47. package/src/main/lib/localAdminHttp.cjs +10 -3
  48. package/src/main/lib/prdAdminRoutes.cjs +175 -0
  49. package/src/main/lib/prdCreate.cjs +37 -2
  50. package/src/main/lib/prdFrontmatter.cjs +179 -1
  51. package/src/main/lib/prdMigration.cjs +82 -5
  52. package/src/main/lib/queueStore.cjs +41 -7
  53. package/src/main/lib/scheduleJobSchema.cjs +114 -0
  54. package/src/main/lib/scheduleJobTransitions.cjs +164 -0
  55. package/src/main/scheduler/prdParser.cjs +7 -0
  56. package/src/main/scheduler.cjs +649 -137
  57. package/src/preload/api.d.ts +54 -2
  58. package/src/preload/index.cjs +12 -0
  59. package/dist/assets/AgentLibrary-DYriNDGf.js +0 -1
  60. package/dist/assets/Scheduler-X5y252Qw.js +0 -14
@@ -88,7 +88,10 @@ const queueOps = require('./queueOps.cjs');
88
88
  // home-dir layout.
89
89
  const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
90
90
  const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
91
+ const { transitionJob } = require('./lib/scheduleJobTransitions.cjs');
91
92
  const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
93
+ const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
94
+ const { appendAuditEvent } = require('./lib/auditLog.cjs');
92
95
 
93
96
  // ---------- origin session resolution (PRD 832) ----------
94
97
  // An Epic IS a tagged claude session — job rows carry the originating
@@ -110,8 +113,8 @@ function resolveOriginSessionId(cwd, epicId) {
110
113
  const sessionSlots = require('./lib/sessionSlots.cjs');
111
114
  const jobWorktree = require('./lib/jobWorktree.cjs');
112
115
  const queueStore = require('./lib/queueStore.cjs');
113
- const { splitFrontmatter } = require('./lib/prdFrontmatter.cjs');
114
- const { migratePrds, consolidateFlatPrds } = require('./lib/prdMigration.cjs');
116
+ const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
117
+ const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
115
118
  const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
116
119
 
117
120
  // Captured once at module load so every run's meta sidecar can record how
@@ -720,7 +723,7 @@ async function retireCompletedSlugs(slugs) {
720
723
  for (const j of s.jobs) {
721
724
  if (!j || !slugSet.has(j.slug)) continue;
722
725
  if (j.status !== 'pending' && j.status !== 'running') continue;
723
- j.status = 'completed';
726
+ if (!transitionJob(j, 'completed', { reason: 'manual archive of an already-shipped PRD', source: 'retireCompletedSlugs' })) continue;
724
727
  j.finishedAt = new Date().toISOString();
725
728
  j.exitCode = 0;
726
729
  j.error = null;
@@ -756,6 +759,50 @@ function ensureDirs() {
756
759
  * unparseable cwd, cwd not on disk) are left in place and logged as a
757
760
  * warning — never silently dropped — so a human can fix the frontmatter.
758
761
  */
762
+ /**
763
+ * consolidateAllFlatPrds(cwds) — run consolidateFlatPrds() over every given
764
+ * project cwd, logging outcomes. Called from TWO places: once at boot (over
765
+ * every historical project, via runPrdMigration below) AND at the top of
766
+ * every reconcile() call (over every project reconcile itself would
767
+ * otherwise scan), BEFORE reconcile scans the flat dir for PRD sources. The
768
+ * reconcile()-level call is what makes "anything written to the retired flat
769
+ * prds/ dir is swept into prds-archived/ without being executed" actually
770
+ * true regardless of which of reconcile's several callers (tickQueue's poll,
771
+ * job completion, the schedule:state/schedule:rescan IPC handlers,
772
+ * rescheduleTimer) triggers the pass: a PRD dropped in the flat dir has no
773
+ * queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
774
+ * sweep archives it before reconcile can ever turn it into a pending job.
775
+ */
776
+ async function consolidateAllFlatPrds(cwds) {
777
+ for (const cwd of cwds) {
778
+ try {
779
+ const c = await consolidateFlatPrds(cwd);
780
+ if (c.moved > 0) {
781
+ console.log(`[scheduler] flat-PRD consolidation: archived ${c.moved} file(s) in ${cwd}`);
782
+ }
783
+ for (const f of c.failed) {
784
+ logs.writeLine({
785
+ level: 'warn', scope: 'scheduler',
786
+ message: `flat-PRD consolidation: could not archive ${f.file}`,
787
+ meta: { cwd, reason: f.reason },
788
+ });
789
+ }
790
+ // Deliberately left behind because a live job still points at them
791
+ // (PRD 992). Logged so a permanently-stuck flat PRD is visible rather
792
+ // than looking like a clean consolidation.
793
+ for (const s of c.skipped ?? []) {
794
+ logs.writeLine({
795
+ level: 'info', scope: 'scheduler',
796
+ message: `flat-PRD consolidation: left ${s.file} in place`,
797
+ meta: { cwd, reason: s.reason },
798
+ });
799
+ }
800
+ } catch (e) {
801
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'flat-PRD consolidation failed', meta: { cwd, error: e?.message } });
802
+ }
803
+ }
804
+ }
805
+
759
806
  async function runPrdMigration() {
760
807
  let result;
761
808
  try {
@@ -782,33 +829,30 @@ async function runPrdMigration() {
782
829
  // still sitting flat consolidates into `prds-archived/` for later special
783
830
  // processing. Queue rows for moved files are reaped by the archived-twin
784
831
  // retirement. Idempotent per project; failures are logged, never fatal.
785
- for (const cwd of allProjectCwds()) {
786
- try {
787
- const c = await consolidateFlatPrds(cwd);
788
- if (c.moved > 0) {
789
- console.log(`[scheduler] flat-PRD consolidation: archived ${c.moved} file(s) in ${cwd}`);
790
- }
791
- for (const f of c.failed) {
792
- logs.writeLine({
793
- level: 'warn', scope: 'scheduler',
794
- message: `flat-PRD consolidation: could not archive ${f.file}`,
795
- meta: { cwd, reason: f.reason },
796
- });
797
- }
798
- // Deliberately left behind because a live job still points at them
799
- // (PRD 992). Logged so a permanently-stuck flat PRD is visible rather
800
- // than looking like a clean consolidation.
801
- for (const s of c.skipped ?? []) {
802
- logs.writeLine({
803
- level: 'info', scope: 'scheduler',
804
- message: `flat-PRD consolidation: left ${s.file} in place`,
805
- meta: { cwd, reason: s.reason },
806
- });
807
- }
808
- } catch (e) {
809
- logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'flat-PRD consolidation failed', meta: { cwd, error: e?.message } });
832
+ // (This boot-time pass is redundant with the one reconcile() now also runs
833
+ // on every pass, but stays here so a fresh boot's very first log line
834
+ // still reports the initial sweep — see consolidateAllFlatPrds's own
835
+ // comment for why reconcile() is the load-bearing call site.)
836
+ await consolidateAllFlatPrds(allProjectCwds());
837
+
838
+ // Rollout migration for the PRD-authoring-lockdown feature: stamp every
839
+ // pre-existing PRD as legacy-adopted BEFORE reconcile() ever runs its
840
+ // provenance gate against it. Must run every boot (idempotent, cheap
841
+ // scan-and-skip) rather than once — a project opened for the first time
842
+ // after this shipped still has pre-existing unstamped PRDs the very first
843
+ // time reconcile() sees them.
844
+ try {
845
+ const adopted = await legacyAdoptExistingPrds();
846
+ if (adopted.stamped > 0) {
847
+ console.log(`[scheduler] legacy-adopt migration: stamped ${adopted.stamped} pre-existing PRD(s) as createdVia=legacy-adopted`);
848
+ }
849
+ for (const f of adopted.failed) {
850
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'legacy-adopt migration: could not stamp PRD', meta: f });
810
851
  }
852
+ } catch (e) {
853
+ logs.writeLine({ level: 'error', scope: 'scheduler', message: 'legacy-adopt migration failed', meta: { error: e?.message } });
811
854
  }
855
+
812
856
  return result;
813
857
  }
814
858
 
@@ -913,6 +957,41 @@ function appendHeartbeat(entry) {
913
957
  }
914
958
  }
915
959
 
960
+ /**
961
+ * computeStallSummary(state) → { stalled, total, running, pending, byProject }
962
+ *
963
+ * Pure, no IO. `state` is a merged queue-store read ({ jobs, invalidJobs,
964
+ * paused }). "Stalled" = the queue holds work — valid rows OR rows
965
+ * quarantined for an invalid status — but nothing is running or pending and
966
+ * the scheduler isn't paused. The 2026-08-07 incident sat exactly in this
967
+ * state for 4+ hours: 2 jobs, 0 running, 0 pending, and the only visible
968
+ * symptom was a heartbeat `counts` object that had silently minted a
969
+ * `queued` bucket instead of reporting anything actionable. `byProject`
970
+ * breaks the stalled rows down by cwd for the log line / toast.
971
+ */
972
+ function computeStallSummary(state) {
973
+ const jobs = Array.isArray(state?.jobs) ? state.jobs : [];
974
+ const invalidJobs = Array.isArray(state?.invalidJobs) ? state.invalidJobs : [];
975
+ let running = 0;
976
+ let pending = 0;
977
+ const byProject = {};
978
+ for (const j of jobs) {
979
+ if (j.status === 'running') running += 1;
980
+ if (j.status === 'pending') pending += 1;
981
+ const key = j.cwd || '(unknown)';
982
+ byProject[key] = byProject[key] || {};
983
+ byProject[key][j.status] = (byProject[key][j.status] || 0) + 1;
984
+ }
985
+ for (const inv of invalidJobs) {
986
+ const key = inv.row?.cwd || '(unknown)';
987
+ byProject[key] = byProject[key] || {};
988
+ byProject[key].invalid = (byProject[key].invalid || 0) + 1;
989
+ }
990
+ const total = jobs.length + invalidJobs.length;
991
+ const stalled = total > 0 && running === 0 && pending === 0 && !state?.paused;
992
+ return { stalled, total, running, pending, byProject };
993
+ }
994
+
916
995
  // An empty queue and an unreadable queue are NOT the same thing, and
917
996
  // conflating them is destructive: reconcile() treats every PRD .md with no
918
997
  // matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
@@ -1172,6 +1251,15 @@ async function reconcile(state) {
1172
1251
  if (state && state.unreadable) {
1173
1252
  throw new Error(`reconcile skipped: queue.json unreadable (${state.unreadable})`);
1174
1253
  }
1254
+ // Sweep the retired flat prds/ dir BEFORE scanning it below. reconcile()
1255
+ // has several callers besides tickQueue's ~60s poll (broadcast,
1256
+ // rescheduleTimer, the schedule:state IPC handler, schedule:rescan) — this
1257
+ // lives here, not in any one caller, so the "a hand-written PRD in the flat
1258
+ // dir is swept before it can become a job" guarantee holds regardless of
1259
+ // which caller triggers this reconcile pass. A freshly hand-written file
1260
+ // has no queue row yet, so it is never "live" and gets archived here
1261
+ // instead of ever reaching the onDisk scan below.
1262
+ await consolidateAllFlatPrds(allProjectCwds());
1175
1263
  const files = await listPrdFiles();
1176
1264
  const onDisk = new Map();
1177
1265
  for (const f of files) {
@@ -1213,13 +1301,16 @@ async function reconcile(state) {
1213
1301
  // file may be unreadable, on a project whose dir failed to enumerate,
1214
1302
  // or mid-move. "I can't see it" is not "the user deleted it", so the
1215
1303
  // row survives — worst case it re-resolves on the next pass.
1216
- if (job.status === 'pending' || job.status === 'running') {
1304
+ if (job.status === 'pending' || job.status === 'running' || job.status === 'quarantined') {
1217
1305
  // Exception: a PENDING row whose PRD has an archived twin was
1218
1306
  // retired on purpose (work landed by other means — e.g. implemented
1219
1307
  // inline — and the source .md moved to prds-archived/). Keeping it
1220
1308
  // would show a phantom "scheduled" job forever; firing it would just
1221
1309
  // hit executeJob's archived-twin skip anyway. Running rows are left
1222
- // alone — the reaper owns their lifecycle.
1310
+ // alone — the reaper owns their lifecycle. A quarantined row's file
1311
+ // going merely-not-visible must survive too — quarantine is meant to
1312
+ // be loud and reversible, never a silent drop (see this function's
1313
+ // header comment on the 2026-08-01 outage a silent skip caused).
1223
1314
  if (job.status === 'pending' && (await archivedTwinExists(job))) {
1224
1315
  console.log(`[scheduler] reconcile: retiring pending job ${job.slug} — PRD already archived (work landed elsewhere)`);
1225
1316
  continue;
@@ -1233,7 +1324,7 @@ async function reconcile(state) {
1233
1324
  continue;
1234
1325
  }
1235
1326
  seen.add(job.slug);
1236
- next.push({
1327
+ const updatedJob = {
1237
1328
  ...job,
1238
1329
  title: p.title,
1239
1330
  cwd: p.cwd,
@@ -1249,7 +1340,23 @@ async function reconcile(state) {
1249
1340
  originSessionId: job.originSessionId
1250
1341
  ?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
1251
1342
  bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1252
- });
1343
+ };
1344
+ // Adopt path: a row parked 'quarantined' (no createdVia provenance when
1345
+ // discovered) whose PRD file now carries a stamp — written via the
1346
+ // update-prd API's adopt patch, either the Scheduler tab's one-click
1347
+ // "adopt PRD" action or a manual scheduler_update_prd call — promotes to
1348
+ // 'pending' the very next reconcile pass. This is the ONLY way a
1349
+ // quarantined row becomes runnable; nothing else in reconcile() clears
1350
+ // that status.
1351
+ if (updatedJob.status === 'quarantined' && p.createdVia) {
1352
+ transitionJob(updatedJob, 'pending', {
1353
+ reason: `adopted via API (createdVia=${p.createdVia})`,
1354
+ source: 'reconcile-adopt',
1355
+ });
1356
+ console.log(`[scheduler] reconcile: adopted quarantined PRD ${job.slug} — createdVia=${p.createdVia}`);
1357
+ appendAuditEvent('scheduler_prd_adopted', { slug: job.slug, cwd: p.cwd, createdVia: p.createdVia, source: 'reconcile' });
1358
+ }
1359
+ next.push(updatedJob);
1253
1360
  }
1254
1361
  // Slugs on disk with no matching state.jobs row are normally brand-new
1255
1362
  // PRDs — but once queueHistory.partitionJobs (above, later this same
@@ -1264,7 +1371,11 @@ async function reconcile(state) {
1264
1371
  for (const [slug] of onDisk) {
1265
1372
  if (!seen.has(slug)) unmatchedSlugs.push(slug);
1266
1373
  }
1267
- const historyBySlug = (unmatchedSlugs.length > 0 || terminalDroppedNeedingHistoryCheck.length > 0)
1374
+ // Rows quarantined by queueStore.shapeJobs because their `status` failed
1375
+ // ScheduleJobSchema (e.g. the 1021/1022 incident's `"status": "queued"`) —
1376
+ // see the repair pass below, right after historyBySlug is available.
1377
+ const invalidJobs = Array.isArray(state.invalidJobs) ? state.invalidJobs : [];
1378
+ const historyBySlug = (unmatchedSlugs.length > 0 || terminalDroppedNeedingHistoryCheck.length > 0 || invalidJobs.length > 0)
1268
1379
  ? await queueHistory.historyTerminalBySlug()
1269
1380
  : new Map();
1270
1381
 
@@ -1282,6 +1393,79 @@ async function reconcile(state) {
1282
1393
  }
1283
1394
  }
1284
1395
 
1396
+ // Repair pass: an invalid row must self-heal within this one tick, not
1397
+ // wait for its slug to also drop out of `seen` via some unrelated code
1398
+ // path. Before this pass, reconcile was add-only (`if (seen.has(slug))
1399
+ // continue` below) — a quarantined row simply vanished from state.jobs
1400
+ // with no log of what its bad status actually was and no repair, which is
1401
+ // how the 1021/1022 rows sat invisible for 4+ hours (2026-08-07).
1402
+ let repairedInvalidCount = 0;
1403
+ for (const inv of invalidJobs) {
1404
+ if (seen.has(inv.slug)) continue; // a valid row for this slug already exists
1405
+ const oldStatus = inv.row?.status;
1406
+ const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir: RUNS_DIR });
1407
+ if (hist) {
1408
+ // Never resurrect: this slug already has a durable terminal record
1409
+ // elsewhere (history.jsonl or a run sidecar) — repairing its corrupted
1410
+ // row back to 'pending' would re-execute already-shipped work. Drop
1411
+ // the row (its real outcome is recorded elsewhere), loudly.
1412
+ console.warn(`[scheduler] reconcile: dropping invalid queue row ${inv.slug} (status was ${JSON.stringify(oldStatus)}) — already terminal (${hist.status}) in history/run sidecar, not resurrecting`);
1413
+ appendAuditEvent('scheduler_row_repaired', {
1414
+ slug: inv.slug, cwd: inv.row?.cwd ?? null, oldStatus: oldStatus ?? null,
1415
+ action: 'dropped-already-terminal', terminalStatus: hist.status, issues: inv.issues,
1416
+ });
1417
+ continue;
1418
+ }
1419
+ const p = onDisk.get(inv.slug);
1420
+ if (!p) {
1421
+ // PRD file also gone with no terminal record anywhere — nothing to
1422
+ // repair against. queueStore already logged the quarantine once.
1423
+ continue;
1424
+ }
1425
+ const job = {
1426
+ ...inv.row,
1427
+ slug: inv.slug,
1428
+ title: p.title,
1429
+ cwd: p.cwd,
1430
+ parallelGroup: p.parallelGroup,
1431
+ estimateMinutes: p.estimateMinutes,
1432
+ sourcePromptId: p.sourcePromptId ?? inv.row?.sourcePromptId ?? null,
1433
+ sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
1434
+ epicId: p.epicId ?? inv.row?.epicId ?? null,
1435
+ dependsOn: p.dependsOn,
1436
+ originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
1437
+ bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
1438
+ };
1439
+ const reason = `reconcile: repaired invalid status ${JSON.stringify(oldStatus)}`;
1440
+ // A repair is not a lifecycle transition — the corrupted `status` was
1441
+ // never a legal predecessor to check against LEGAL_TRANSITIONS, so this
1442
+ // goes through transitionJob's allowAnyFrom escape hatch (still gets the
1443
+ // normal mutation/statusHistory/audit trail, just skips the legality
1444
+ // gate on `from`) rather than a bare field assignment.
1445
+ transitionJob(job, 'pending', { reason, source: 'reconcile-repair', allowAnyFrom: true });
1446
+ if (job.runId || job.startedAt || job.runtime) {
1447
+ // This row had actually begun executing before its status got
1448
+ // corrupted.
1449
+ job.runId = null;
1450
+ job.startedAt = null;
1451
+ job.finishedAt = null;
1452
+ job.exitCode = null;
1453
+ delete job.runtime;
1454
+ delete job.verifierVerdict;
1455
+ }
1456
+ job.error = null;
1457
+ seen.add(inv.slug);
1458
+ next.push(job);
1459
+ repairedInvalidCount += 1;
1460
+ console.warn(`[scheduler] reconcile: repaired invalid queue row ${inv.slug} — status was ${JSON.stringify(oldStatus)}, reset to 'pending' (${inv.issues})`);
1461
+ appendAuditEvent('scheduler_row_repaired', {
1462
+ slug: inv.slug, cwd: p.cwd, oldStatus: oldStatus ?? null, newStatus: 'pending', issues: inv.issues,
1463
+ });
1464
+ }
1465
+ if (repairedInvalidCount > 0) {
1466
+ console.warn(`[scheduler] reconcile: repaired ${repairedInvalidCount} invalid queue row(s) this pass`);
1467
+ }
1468
+
1285
1469
  // Terminal-in-history slugs whose .md file is still on disk: fed into the
1286
1470
  // auto-archive selection pass below (as synthetic completed entries) so
1287
1471
  // their file can still be swept, without ever creating a live job row
@@ -1302,6 +1486,7 @@ async function reconcile(state) {
1302
1486
  return idx.sessions[epicId]?.status ?? null;
1303
1487
  }
1304
1488
 
1489
+ let staleNewDiscoveryCount = 0;
1305
1490
  for (const [slug, p] of onDisk) {
1306
1491
  if (seen.has(slug)) continue;
1307
1492
  // Security gate: a PRD's file location IS its Epic membership
@@ -1385,8 +1570,47 @@ async function reconcile(state) {
1385
1570
  const parent = healTargetForFix(slug, state.jobs);
1386
1571
  entry.investigationDepth = parent ? (parent.investigationDepth ?? 1) + 1 : 2;
1387
1572
  }
1573
+ // Provenance gate (PRD-authoring lockdown): a PRD discovered with no
1574
+ // `createdVia` stamp was never written through scheduler_create_prd/
1575
+ // chat:create-prd (prdCreate.cjs always stamps 'scheduler-api') or the
1576
+ // legacy-adopt boot migration ('legacy-adopted') — it bypassed the
1577
+ // sanctioned API, most likely via a raw Write/Edit tool call the
1578
+ // guard-prd-writes.cjs PreToolUse hook should have denied. Fix-plan PRDs
1579
+ // are exempt: spawnInvestigation's own probe writes them directly by
1580
+ // design (a trusted, scheduler-spawned internal loop, not an
1581
+ // agent/human authoring a PRD), matching the isFixPlanSlug convention
1582
+ // used everywhere else this distinction matters.
1583
+ //
1584
+ // Quarantine is loud and reversible, never a silent skip (see the
1585
+ // 2026-08-01 23-PRD outage this file's header references for what a
1586
+ // SILENT skip costs): logged at warn, audited, and surfaced in the
1587
+ // Scheduler tab's Quarantined filter with a one-click adopt action
1588
+ // (schedule:adopt-prd) that stamps the file via the same update-prd API
1589
+ // route the MCP tool uses — reconcile()'s adopt path above promotes it
1590
+ // to 'pending' on the very next pass, within one tick of being stamped.
1591
+ if (!p.createdVia && !isFixPlanSlug(slug)) {
1592
+ entry.status = 'quarantined';
1593
+ console.warn(`[scheduler] reconcile: quarantining unstamped PRD ${slug} (${p.path}) — no createdVia provenance; adopt it from the Scheduler tab's Quarantined filter or via scheduler_update_prd to make it runnable`);
1594
+ appendAuditEvent('prd_quarantined', { slug, cwd: p.cwd, path: p.path, reason: 'missing createdVia provenance frontmatter' });
1595
+ }
1596
+ // A PRD with no queue row and no terminal record is normally a
1597
+ // brand-new file — but one whose mtime already predates a full poll
1598
+ // interval means it sat unpicked (a prior reconcile pass should have
1599
+ // caught it, or it's arriving from a source that bypassed the app's
1600
+ // normal write path). Report it rather than silently treating "first
1601
+ // seen this pass" as "just created".
1602
+ try {
1603
+ const ageMs = Date.now() - fs.statSync(p.path).mtimeMs;
1604
+ if (ageMs > POLL_INTERVAL_MS) {
1605
+ staleNewDiscoveryCount += 1;
1606
+ console.warn(`[scheduler] reconcile: discovered PRD ${slug} with no queue row and no terminal record — file is ${Math.round(ageMs / 1000)}s old, only first seen this pass`);
1607
+ }
1608
+ } catch { /* stat is best-effort reporting only */ }
1388
1609
  next.push(entry);
1389
1610
  }
1611
+ if (staleNewDiscoveryCount > 0) {
1612
+ console.warn(`[scheduler] reconcile: ${staleNewDiscoveryCount} PRD(s) discovered this pass were already older than one poll interval with no prior queue row`);
1613
+ }
1390
1614
  const sorted = next.sort((a, b) => b.slug.localeCompare(a.slug));
1391
1615
 
1392
1616
  // Move terminal jobs past the retention window out to history.jsonl so
@@ -1467,6 +1691,13 @@ let resumeTimer = null;
1467
1691
  let pollLoopTimer = null;
1468
1692
  let rescheduleInterval = null;
1469
1693
  let heartbeatInterval = null;
1694
+ // Stall-detector state (computeStallSummary), read/written only inside the
1695
+ // heartbeat interval below. stallSince: wall-clock ms the stalled condition
1696
+ // was first observed, null when clear. stallToasted: rate-limits the
1697
+ // error-log + toast to once per stall episode (cleared the moment the queue
1698
+ // stops being stalled) rather than every 60s heartbeat tick.
1699
+ let stallSince = null;
1700
+ let stallToasted = false;
1470
1701
  // (The 5-minute feedback sweep that used to piggyback on this heartbeat is
1471
1702
  // gone: it scanned each active project's session-manager-operations/feedback/
1472
1703
  // and auto-queued a /process-feedback PRD. Both the folder and that skill are
@@ -1738,7 +1969,7 @@ async function clearPause(source) {
1738
1969
  */
1739
1970
  function resetJobFields(job, errorMsg, opts = {}) {
1740
1971
  if (job.status === 'completed' && opts.force !== true) return false;
1741
- job.status = 'pending';
1972
+ if (!transitionJob(job, 'pending', { reason: errorMsg ?? 'reset to pending', source: opts.source ?? 'resetJobFields' })) return false;
1742
1973
  job.runId = null;
1743
1974
  job.startedAt = null;
1744
1975
  job.finishedAt = null;
@@ -1799,13 +2030,13 @@ function partitionBootOrphans(jobs, isAlive = claudePidAlive) {
1799
2030
  function applyOrphanOutcome(job, outcome, killNote = '') {
1800
2031
  const now = new Date().toISOString();
1801
2032
  if (outcome === 'success') {
1802
- job.status = 'completed';
2033
+ transitionJob(job, 'completed', { reason: 'boot orphan reconciliation: run succeeded', source: 'applyOrphanOutcome' });
1803
2034
  job.exitCode = 0;
1804
2035
  job.error = null;
1805
2036
  job.finishedAt = now;
1806
2037
  delete job.runtime;
1807
2038
  } else if (outcome === 'failed') {
1808
- job.status = 'failed';
2039
+ transitionJob(job, 'failed', { reason: `orphaned: app restarted while running${killNote}`, source: 'applyOrphanOutcome' });
1809
2040
  job.exitCode = job.exitCode ?? 1;
1810
2041
  job.error = `orphaned: app restarted while running${killNote}`;
1811
2042
  job.finishedAt = now;
@@ -1813,10 +2044,10 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
1813
2044
  } else {
1814
2045
  const tries = job.orphanRetries ?? 0;
1815
2046
  if (tries < ORPHAN_REQUEUE_CAP) {
1816
- resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}`);
2047
+ resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}`, { source: 'applyOrphanOutcome' });
1817
2048
  job.orphanRetries = tries + 1;
1818
2049
  } else {
1819
- job.status = 'failed';
2050
+ transitionJob(job, 'failed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`, source: 'applyOrphanOutcome' });
1820
2051
  job.exitCode = job.exitCode ?? 1;
1821
2052
  job.error = `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`;
1822
2053
  job.finishedAt = now;
@@ -2860,7 +3091,7 @@ async function spawnInvestigation(failedJob, runDir) {
2860
3091
  // "nothing is happening" even though an Opus process was actively running.
2861
3092
  await mutate((s) => {
2862
3093
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
2863
- if (j) j.status = 'investigating';
3094
+ if (j) transitionJob(j, 'investigating', { reason: 'spawning investigation probe', source: 'spawnInvestigation:start' });
2864
3095
  });
2865
3096
  await broadcast({ flush: true });
2866
3097
 
@@ -2905,7 +3136,7 @@ async function spawnInvestigation(failedJob, runDir) {
2905
3136
  // 'investigating' must never be the job's resting state.
2906
3137
  mutate((s) => {
2907
3138
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
2908
- if (j && j.status === 'investigating') j.status = failedJob.status || 'failed';
3139
+ if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation probe exited — restoring prior status', source: 'spawnInvestigation:onExit' });
2909
3140
  })
2910
3141
  .then(() => broadcast({ flush: true }))
2911
3142
  .catch(() => {});
@@ -2960,7 +3191,7 @@ async function spawnInvestigation(failedJob, runDir) {
2960
3191
  releaseSlot();
2961
3192
  mutate((s) => {
2962
3193
  const j = s.jobs.find((x) => x.slug === failedJob.slug);
2963
- if (j && j.status === 'investigating') j.status = failedJob.status || 'failed';
3194
+ if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation spawn threw before exiting — restoring prior status', source: 'spawnInvestigation:catch' });
2964
3195
  })
2965
3196
  .then(() => broadcast({ flush: true }))
2966
3197
  .catch(() => {});
@@ -2982,7 +3213,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2982
3213
  await mutate((s) => {
2983
3214
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
2984
3215
  if (idx >= 0) {
2985
- s.jobs[idx].status = 'running';
3216
+ transitionJob(s.jobs[idx], 'running', { reason: 'dispatched for execution', source: 'spawnJob:dispatch' });
2986
3217
  s.jobs[idx].runId = runId;
2987
3218
  s.jobs[idx].startedAt = new Date().toISOString();
2988
3219
  }
@@ -3069,7 +3300,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3069
3300
  await mutate((s) => {
3070
3301
  const idx = s.jobs.findIndex((x) => x.slug === job.slug);
3071
3302
  if (idx >= 0) {
3072
- s.jobs[idx].status = 'completed';
3303
+ transitionJob(s.jobs[idx], 'completed', { reason: res.note ?? 'PRD archived or missing — treated as already-shipped', source: 'spawnJob:skip-archived' });
3073
3304
  s.jobs[idx].finishedAt = new Date().toISOString();
3074
3305
  s.jobs[idx].exitCode = 0;
3075
3306
  s.jobs[idx].error = null;
@@ -3239,10 +3470,19 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3239
3470
  const newlyCompletedPrds = [];
3240
3471
  await mutate((s) => {
3241
3472
  const i2 = s.jobs.findIndex((x) => x.slug === job.slug);
3473
+ // A job already moved off 'running' by someone else (namely
3474
+ // remote.cancelJob, PRD 1024 — it SIGTERMs the process then finalizes
3475
+ // the row to 'failed' before this exit handler necessarily runs) is
3476
+ // not this run's to finalize: doing so anyway could re-legalize the
3477
+ // row via a legal failed->completed/needs_review edge (see
3478
+ // scheduleJobTransitions.cjs's LEGAL_TRANSITIONS) and silently
3479
+ // undo the cancellation. Skip — the row already reflects its real
3480
+ // terminal state.
3481
+ if (i2 >= 0 && s.jobs[i2].status !== 'running') return;
3242
3482
  if (i2 >= 0) {
3243
3483
  const treatAsPending = res.rateLimited || (s.paused && s.paused.reason === 'rate_limit');
3244
3484
  if (treatAsPending) {
3245
- resetJobFields(s.jobs[i2], res.rateLimited ? 'paused: rate limit' : 'paused: queue halted');
3485
+ resetJobFields(s.jobs[i2], res.rateLimited ? 'paused: rate limit' : 'paused: queue halted', { source: 'spawnJob:halt-reset' });
3246
3486
  } else {
3247
3487
  // Determine effective status, applying the verifier verdict for exit=0 runs.
3248
3488
  let effectiveStatus;
@@ -3262,14 +3502,14 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3262
3502
  effectiveStatus = 'completed';
3263
3503
  } else if (verifyResult.downgradeTo === 'pending') {
3264
3504
  // HALT or deps_unmet: reset to pending so the job re-fires.
3265
- resetJobFields(s.jobs[i2], verifyResult.reason);
3505
+ resetJobFields(s.jobs[i2], verifyResult.reason, { source: 'spawnJob:verify-downgrade' });
3266
3506
  return; // job already mutated by resetJobFields; skip the rest
3267
3507
  } else {
3268
3508
  // transcript_errors or verify_unavailable: escalate to needs_review.
3269
3509
  effectiveStatus = 'needs_review';
3270
3510
  }
3271
3511
 
3272
- s.jobs[i2].status = effectiveStatus;
3512
+ transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
3273
3513
  s.jobs[i2].finishedAt = new Date().toISOString();
3274
3514
  s.jobs[i2].exitCode = res.exitCode;
3275
3515
  s.jobs[i2].error = effectiveStatus === 'needs_review'
@@ -3356,7 +3596,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3356
3596
  if (orig) {
3357
3597
  const priorStatus = orig.status;
3358
3598
  console.log(`[scheduler] auto-promote: ${orig.slug} (${priorStatus}) → completed because ${job.slug} succeeded`);
3359
- orig.status = 'completed';
3599
+ transitionJob(orig, 'completed', { reason: `auto-promoted: fix plan ${job.slug} succeeded`, source: 'spawnJob:auto-promote' });
3360
3600
  orig.exitCode = 0;
3361
3601
  orig.error = null;
3362
3602
  orig.completedBy = job.slug;
@@ -3444,7 +3684,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3444
3684
  await mutate((s) => {
3445
3685
  const i = s.jobs.findIndex((x) => x.slug === job.slug);
3446
3686
  if (i >= 0) {
3447
- resetJobFields(s.jobs[i], null);
3687
+ resetJobFields(s.jobs[i], null, { source: 'spawnJob:transient-retry' });
3448
3688
  s.jobs[i].transientRetries = decision.retries + 1;
3449
3689
  }
3450
3690
  });
@@ -3454,7 +3694,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
3454
3694
  await mutate((s) => {
3455
3695
  const i = s.jobs.findIndex((x) => x.slug === job.slug);
3456
3696
  if (i >= 0) {
3457
- s.jobs[i].status = 'failed';
3697
+ transitionJob(s.jobs[i], 'failed', { reason: `transient failure (${decision.transientKind}) left uncommitted work — not auto-requeued`, source: 'spawnJob:fail-dirty' });
3458
3698
  s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample}) — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
3459
3699
  }
3460
3700
  });
@@ -3515,6 +3755,9 @@ function tickQueue() {
3515
3755
  }
3516
3756
  if (cancelToken.cancelled) return { fired: false, reason: 'cancelled' };
3517
3757
 
3758
+ // The retired-flat-dir sweep now lives inside reconcile() itself (see its
3759
+ // own comment) so every caller of reconcile — not just this tick — gets
3760
+ // the guarantee.
3518
3761
  await reconcile(state);
3519
3762
  // Session-Manager's machine-wide slot pool is the ONLY concurrency limit
3520
3763
  // the picker answers to (plus the memory gate below). The scheduler used
@@ -3702,7 +3945,7 @@ async function reapDeadRunningJobs() {
3702
3945
  const idx = s.jobs.findIndex((x) => x.slug === slug);
3703
3946
  if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
3704
3947
  const success = outcome === 'success';
3705
- s.jobs[idx].status = success ? 'completed' : 'failed';
3948
+ transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: `reaped: process gone (outcome=${outcome})`, source: 'reapDeadRunningJobs' });
3706
3949
  s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
3707
3950
  s.jobs[idx].finishedAt = new Date().toISOString();
3708
3951
  s.jobs[idx].error = success ? null : `reaped: process gone, no success result in log (${outcome})`;
@@ -4125,7 +4368,7 @@ async function reverifyNeedsReview() {
4125
4368
  await mutate((s) => {
4126
4369
  for (const j of s.jobs) {
4127
4370
  if (j.status === 'needs_review' && healSet.has(j.slug)) {
4128
- j.status = 'completed';
4371
+ transitionJob(j, 'completed', { reason: 'boot reverify: stale needs_review healed', source: 'reverifyNeedsReview:heal' });
4129
4372
  j.error = null;
4130
4373
  delete j.verifierVerdict;
4131
4374
  healedPrds.push({ slug: j.slug, cwd: j.cwd });
@@ -4163,7 +4406,7 @@ async function reverifyNeedsReview() {
4163
4406
  const orig = healTargetForFix(job.slug, s.jobs);
4164
4407
  if (!orig) continue;
4165
4408
  const priorStatus = orig.status;
4166
- orig.status = 'completed';
4409
+ transitionJob(orig, 'completed', { reason: `auto-promoted: fix plan ${job.slug} already completed`, source: 'reverifyNeedsReview:auto-promote' });
4167
4410
  orig.exitCode = 0;
4168
4411
  orig.error = null;
4169
4412
  orig.completedBy = job.slug;
@@ -4341,7 +4584,7 @@ function registerScheduleHandlers() {
4341
4584
  // Guard is in resetJobFields: refuses to reset an already-'completed'
4342
4585
  // job, which would otherwise re-fire a PRD whose deliverable already
4343
4586
  // landed (see resetJobFields' doc comment for the incident).
4344
- return resetJobFields(state.jobs[idx]) ? 'ok' : 'refused';
4587
+ return resetJobFields(state.jobs[idx], null, { source: 'ipc:schedule:reset-job' }) ? 'ok' : 'refused';
4345
4588
  });
4346
4589
  if (outcome === 'not-found') return { ok: false, error: 'not found' };
4347
4590
  if (outcome === 'refused') {
@@ -4354,6 +4597,36 @@ function registerScheduleHandlers() {
4354
4597
  return { ok: true };
4355
4598
  }));
4356
4599
 
4600
+ // Renderer-facing counterpart to prdCreate.cjs's chat:create-prd handler
4601
+ // (index.cjs): calls the SAME remote.updatePrd the admin HTTP route/MCP
4602
+ // tool use, so "stamps it through the API" holds for the Scheduler tab's
4603
+ // one-click adopt action too, not just a direct fs write. Only a
4604
+ // 'quarantined' row is eligible — see reconcile()'s provenance gate.
4605
+ ipcMain.handle('schedule:adopt-prd', validated(schemas.scheduleSlug, async ({ slug }) => {
4606
+ if (!(await safeSlugPath(slug))) return { ok: false, kind: 'error', message: 'invalid slug' };
4607
+ const state = await readQueue();
4608
+ const job = state.jobs.find((j) => j.slug === slug);
4609
+ if (!job) return { ok: false, kind: 'error', message: 'not found' };
4610
+ if (job.status !== 'quarantined') {
4611
+ return { ok: false, kind: 'error', message: `job status is "${job.status}" — only a quarantined PRD may be adopted` };
4612
+ }
4613
+ const result = await remote.updatePrd({
4614
+ slug,
4615
+ cwd: job.cwd,
4616
+ frontmatter: { createdVia: 'legacy-adopted', issuedAt: new Date().toISOString() },
4617
+ });
4618
+ if (!result.ok) return { ok: false, kind: 'error', message: result.error ?? 'adopt failed' };
4619
+ appendAuditEvent('scheduler_prd_adopted', { slug, cwd: job.cwd ?? null, source: 'ipc:schedule:adopt-prd' });
4620
+ // Promote the row to 'pending' immediately rather than waiting for the
4621
+ // next poll tick — the Scheduler tab's "adopt PRD" click should be
4622
+ // visibly effective within this one round-trip.
4623
+ const freshState = await readQueue();
4624
+ await reconcile(freshState);
4625
+ await writeQueue(freshState);
4626
+ await broadcast({ flush: true });
4627
+ return { ok: true, kind: 'info', message: `Adopted ${slug} — it will run as a normal pending job` };
4628
+ }));
4629
+
4357
4630
  ipcMain.handle('schedule:run-now', async () => {
4358
4631
  // Manual run-now overrides any auto-pause. Clear it first.
4359
4632
  await clearPause('run-now');
@@ -4483,85 +4756,7 @@ function registerScheduleHandlers() {
4483
4756
  }
4484
4757
  }));
4485
4758
 
4486
- ipcMain.handle('schedule:list-prds', async () => {
4487
- ensureDirs();
4488
- const out = [];
4489
- const seenSlugs = new Set();
4490
-
4491
- async function readDirInto(dir, { archived }) {
4492
- let entries;
4493
- try {
4494
- entries = await fsp.readdir(dir);
4495
- } catch (e) {
4496
- if (e?.code !== 'ENOENT') {
4497
- logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: readdir failed', meta: { dir, error: e?.message } });
4498
- }
4499
- return;
4500
- }
4501
- for (const name of entries) {
4502
- if (!name.endsWith('.md') || name.startsWith('.')) continue;
4503
- const filePath = path.join(dir, name);
4504
- try {
4505
- const parsed = await parsePrd(filePath);
4506
- // A slug can't be both live and archived at once, but a duplicate
4507
- // slug found in two archive dirs (shouldn't happen — archiving is
4508
- // a single rename — but is cheap to guard) is skipped rather than
4509
- // double-counted.
4510
- if (seenSlugs.has(parsed.slug)) continue;
4511
- seenSlugs.add(parsed.slug);
4512
- const stat = await fsp.stat(filePath);
4513
- const entry = {
4514
- slug: parsed.slug,
4515
- parallelGroup: parsed.parallelGroup,
4516
- title: parsed.title,
4517
- cwd: parsed.cwd || '',
4518
- estimateMinutes: parsed.estimateMinutes,
4519
- sourcePromptId: parsed.sourcePromptId,
4520
- epicId: parsed.epicId ?? null,
4521
- mtimeMs: stat.mtimeMs,
4522
- archived,
4523
- };
4524
- out.push(entry);
4525
- } catch (e) {
4526
- logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: skipping unparseable file', meta: { name, error: e?.message } });
4527
- }
4528
- }
4529
- }
4530
-
4531
- // Live PRDs first, so an archived duplicate (shouldn't exist, but a
4532
- // stale rename copy is possible) never shadows the still-runnable live
4533
- // entry.
4534
- for (const dir of candidatePrdsDirs()) {
4535
- await readDirInto(dir, { archived: false });
4536
- }
4537
-
4538
- const archivedStart = out.length;
4539
- for (const dir of candidateArchivedPrdsDirs()) {
4540
- await readDirInto(dir, { archived: true });
4541
- }
4542
-
4543
- // Archived PRDs need a status: archiveCompletedPrd (scheduler.cjs) only
4544
- // ever archives a job whose effective status is 'completed' — a 'failed'
4545
- // job's PRD source stays in the live prds/ dir (still visible/countable
4546
- // there already). Still resolve the real job status defensively (live
4547
- // queue row, falling back to history.jsonl) rather than hard-coding
4548
- // 'completed', so this stays correct if that archiving invariant ever
4549
- // changes.
4550
- if (out.length > archivedStart) {
4551
- const [state, histBySlug] = await Promise.all([
4552
- readQueue(),
4553
- queueHistory.historyTerminalBySlug().catch(() => new Map()),
4554
- ]);
4555
- const liveStatusBySlug = new Map(state.jobs.map((j) => [j.slug, j.status]));
4556
- for (let i = archivedStart; i < out.length; i++) {
4557
- const entry = out[i];
4558
- entry.archivedStatus = resolveArchivedPrdStatus(entry.slug, liveStatusBySlug, histBySlug);
4559
- }
4560
- }
4561
-
4562
- out.sort((a, b) => a.slug.localeCompare(b.slug, undefined, { numeric: true }));
4563
- return out;
4564
- });
4759
+ ipcMain.handle('schedule:list-prds', async () => listPrdsInternal());
4565
4760
 
4566
4761
  // Return last N completed/failed jobs from queue.json, newest first.
4567
4762
  // Purely additive: no schema change, no archive-folder read needed.
@@ -4766,12 +4961,54 @@ async function init() {
4766
4961
  if (heartbeatInterval) clearInterval(heartbeatInterval);
4767
4962
  heartbeatInterval = setInterval(() => {
4768
4963
  const s = readQueueSync();
4769
- const counts = { pending: 0, running: 0, completed: 0, failed: 0 };
4770
- for (const j of s.jobs) counts[j.status] = (counts[j.status] || 0) + 1;
4964
+ // Initialise from the real status union (scheduleJobSchema.cjs) rather
4965
+ // than a hand-maintained subset — the old `{ pending, running, completed,
4966
+ // failed }` literal silently minted a NEW key for any other value
4967
+ // (`counts[j.status] = (counts[j.status]||0)+1`), which is exactly how a
4968
+ // heartbeat with a `queued: 2` bucket looked like "normal" 24h
4969
+ // visibility instead of the alarm it should have been. Any row whose
4970
+ // status isn't in JOB_STATUSES (shouldn't happen post-quarantine, but
4971
+ // this is the last line of defence) routes into `unknown`, never a
4972
+ // freshly-minted key.
4973
+ const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
4974
+ counts.unknown = 0;
4975
+ for (const j of s.jobs) {
4976
+ if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
4977
+ counts[j.status] += 1;
4978
+ } else {
4979
+ counts.unknown += 1;
4980
+ }
4981
+ }
4982
+
4983
+ const stall = computeStallSummary(s);
4984
+ if (stall.stalled) {
4985
+ if (stallSince === null) stallSince = Date.now();
4986
+ if (!stallToasted && Date.now() - stallSince >= POLL_INTERVAL_MS) {
4987
+ stallToasted = true;
4988
+ console.error(
4989
+ `[scheduler] STALL DETECTED: ${stall.total} job(s) queued, 0 running, 0 pending, not paused, `
4990
+ + `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
4991
+ stall.byProject,
4992
+ );
4993
+ appendAuditEvent('scheduler_stall_detected', { total: stall.total, byProject: stall.byProject });
4994
+ if (mainWindow && !mainWindow.isDestroyed()) {
4995
+ sendIfAlive(mainWindow, 'schedule:stall', {
4996
+ message: `Scheduler stall: ${stall.total} job(s) queued but none running or pending. Check the Scheduler tab.`,
4997
+ total: stall.total,
4998
+ byProject: stall.byProject,
4999
+ });
5000
+ }
5001
+ }
5002
+ } else {
5003
+ stallSince = null;
5004
+ stallToasted = false;
5005
+ }
5006
+
4771
5007
  appendHeartbeat({
4772
5008
  ts: Date.now(),
4773
5009
  pid: process.pid,
4774
5010
  counts,
5011
+ stall: { stalled: stall.stalled, total: stall.total },
4775
5012
  paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
4776
5013
  nextReset: cachedNextReset,
4777
5014
  utilization: cachedUtilization,
@@ -4802,6 +5039,102 @@ async function init() {
4802
5039
  }
4803
5040
  }
4804
5041
 
5042
+ /**
5043
+ * listPrdsInternal() → every live + archived PRD across every project,
5044
+ * with each entry's real job status folded in (`status`: the live queue
5045
+ * row's status, or the resolved terminal status for an archived entry, or
5046
+ * null when no queue row exists yet — e.g. a PRD just written and not yet
5047
+ * picked up by reconcile()). Single source of truth for both the renderer's
5048
+ * `schedule:list-prds` IPC handler and the admin HTTP `GET
5049
+ * /admin/scheduler/prds` route (PRD 1024) — neither re-implements this scan.
5050
+ */
5051
+ async function listPrdsInternal() {
5052
+ ensureDirs();
5053
+ const out = [];
5054
+ const seenSlugs = new Set();
5055
+
5056
+ async function readDirInto(dir, { archived }) {
5057
+ let entries;
5058
+ try {
5059
+ entries = await fsp.readdir(dir);
5060
+ } catch (e) {
5061
+ if (e?.code !== 'ENOENT') {
5062
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: readdir failed', meta: { dir, error: e?.message } });
5063
+ }
5064
+ return;
5065
+ }
5066
+ for (const name of entries) {
5067
+ if (!name.endsWith('.md') || name.startsWith('.')) continue;
5068
+ const filePath = path.join(dir, name);
5069
+ try {
5070
+ const parsed = await parsePrd(filePath);
5071
+ // A slug can't be both live and archived at once, but a duplicate
5072
+ // slug found in two archive dirs (shouldn't happen — archiving is
5073
+ // a single rename — but is cheap to guard) is skipped rather than
5074
+ // double-counted.
5075
+ if (seenSlugs.has(parsed.slug)) continue;
5076
+ seenSlugs.add(parsed.slug);
5077
+ const stat = await fsp.stat(filePath);
5078
+ const entry = {
5079
+ slug: parsed.slug,
5080
+ parallelGroup: parsed.parallelGroup,
5081
+ title: parsed.title,
5082
+ cwd: parsed.cwd || '',
5083
+ estimateMinutes: parsed.estimateMinutes,
5084
+ sourcePromptId: parsed.sourcePromptId,
5085
+ epicId: parsed.epicId ?? null,
5086
+ mtimeMs: stat.mtimeMs,
5087
+ archived,
5088
+ };
5089
+ out.push(entry);
5090
+ } catch (e) {
5091
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: skipping unparseable file', meta: { name, error: e?.message } });
5092
+ }
5093
+ }
5094
+ }
5095
+
5096
+ // Live PRDs first, so an archived duplicate (shouldn't exist, but a
5097
+ // stale rename copy is possible) never shadows the still-runnable live
5098
+ // entry.
5099
+ for (const dir of candidatePrdsDirs()) {
5100
+ await readDirInto(dir, { archived: false });
5101
+ }
5102
+
5103
+ const archivedStart = out.length;
5104
+ for (const dir of candidateArchivedPrdsDirs()) {
5105
+ await readDirInto(dir, { archived: true });
5106
+ }
5107
+
5108
+ // Every entry (live and archived) gets a real job status folded in.
5109
+ // Archived PRDs need one resolved defensively (live queue row, falling
5110
+ // back to history.jsonl) rather than hard-coded 'completed', so this
5111
+ // stays correct if the archive-only-completed invariant ever changes; a
5112
+ // live entry with no queue row yet (just written, not yet reconciled)
5113
+ // gets `status: null`.
5114
+ const [state, histBySlug] = await Promise.all([
5115
+ readQueue(),
5116
+ queueHistory.historyTerminalBySlug().catch(() => new Map()),
5117
+ ]);
5118
+ const liveStatusBySlug = new Map(state.jobs.map((j) => [j.slug, j.status]));
5119
+ for (let i = 0; i < out.length; i++) {
5120
+ const entry = out[i];
5121
+ // `entry` is a freshly-synthesized PRD-listing row, not a persisted
5122
+ // ScheduleJob — assigning its `status` here is not a queue-job status
5123
+ // transition (no queue.json row is mutated, no statusHistory/audit
5124
+ // trail applies), so it is intentionally exempt from the
5125
+ // transitionJob-only rule enforced by scheduleJobTransitionsGrep.test.cjs.
5126
+ if (i < archivedStart) {
5127
+ entry.status = liveStatusBySlug.get(entry.slug) ?? null;
5128
+ } else {
5129
+ entry.archivedStatus = resolveArchivedPrdStatus(entry.slug, liveStatusBySlug, histBySlug);
5130
+ entry.status = entry.archivedStatus;
5131
+ }
5132
+ }
5133
+
5134
+ out.sort((a, b) => a.slug.localeCompare(b.slug, undefined, { numeric: true }));
5135
+ return out;
5136
+ }
5137
+
4805
5138
  // remote — in-process (non-IPC) scheduler accessors, used by prdCreate.cjs
4806
5139
  // and other main-process callers. (Named for the retired web-remote relay,
4807
5140
  // its original consumer; kept because it still has in-process callers.)
@@ -4921,7 +5254,7 @@ const remote = {
4921
5254
  // Best-effort: record the dispatch on the Epic's event chain.
4922
5255
  try { await appendPrdCreatedEvent(cwd, epicTrace, slug); } catch { /* trace only */ }
4923
5256
  }
4924
- return { ok: true, bytesWritten: stat.size };
5257
+ return { ok: true, bytesWritten: stat.size, path: resolved, epicId: epicTrace };
4925
5258
  } catch (e) {
4926
5259
  return { ok: false, error: e?.message ?? 'write failed' };
4927
5260
  }
@@ -4934,7 +5267,7 @@ const remote = {
4934
5267
  if (idx < 0) return { kind: 'not-found' };
4935
5268
  // Terminal-status guard lives in resetJobFields itself; force:true
4936
5269
  // threads through to override it.
4937
- if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true })) {
5270
+ if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true, source: 'remote:resetJob' })) {
4938
5271
  return { kind: 'refused' };
4939
5272
  }
4940
5273
  return { kind: 'ok' };
@@ -4955,6 +5288,185 @@ const remote = {
4955
5288
  return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd }));
4956
5289
  },
4957
5290
 
5291
+ // Single queue row lookup, used by cancelJob/updatePrd's status guards and
5292
+ // the admin GET /admin/scheduler/prds?slug= route (PRD 1024).
5293
+ async getJob(slug) {
5294
+ const state = await readQueue();
5295
+ const job = state.jobs.find((j) => j.slug === slug);
5296
+ return job ? { slug: job.slug, title: job.title, status: job.status, cwd: job.cwd, error: job.error ?? null } : null;
5297
+ },
5298
+
5299
+ // Every live+archived PRD across every project (listPrdsInternal, shared
5300
+ // with the renderer's schedule:list-prds IPC handler), filtered by the
5301
+ // admin route's cwd/epicId/status query params.
5302
+ async listPrds(filter = {}) {
5303
+ const all = await listPrdsInternal();
5304
+ return all.filter((entry) => {
5305
+ if (filter.cwd && entry.cwd !== filter.cwd) return false;
5306
+ if (filter.epicId && entry.epicId !== filter.epicId) return false;
5307
+ if (filter.status && entry.status !== filter.status) return false;
5308
+ return true;
5309
+ });
5310
+ },
5311
+
5312
+ // Full body + parsed frontmatter for one PRD, live or archived. Mirrors
5313
+ // readPrd's dir-search + symlink-defense pattern (see that method's
5314
+ // comment) rather than sharing code with it, since readPrd intentionally
5315
+ // returns raw text only and is a much narrower/hotter path (executeJob's
5316
+ // PRD re-reads) that shouldn't grow a second return shape.
5317
+ async getPrdParsed(slug, cwd) {
5318
+ let dir = null;
5319
+ let filePath = null;
5320
+ if (cwd) {
5321
+ for (const d of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
5322
+ const p = safeSlugPathIn(d, slug);
5323
+ if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
5324
+ }
5325
+ if (!filePath) {
5326
+ for (const d of listArchivedPrdDirs(cwd)) {
5327
+ const p = safeSlugPathIn(d, slug);
5328
+ if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
5329
+ }
5330
+ }
5331
+ } else {
5332
+ dir = await findPrdDir(slug);
5333
+ filePath = dir ? safeSlugPathIn(dir, slug) : null;
5334
+ if (!filePath) {
5335
+ for (const d of candidateArchivedPrdsDirs()) {
5336
+ const p = safeSlugPathIn(d, slug);
5337
+ if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
5338
+ }
5339
+ }
5340
+ }
5341
+ if (!filePath) return { ok: false, error: 'invalid slug' };
5342
+ try {
5343
+ // Symlink defense, matching readPrd/writePrd's comment: safeSlugPathIn
5344
+ // is lexical and does not resolve symlinks.
5345
+ const real = await fsp.realpath(filePath);
5346
+ if (!real.startsWith(dir + path.sep)) return { ok: false, error: 'invalid slug' };
5347
+ const [raw, parsed] = await Promise.all([fsp.readFile(real, 'utf8'), prdParser.parsePrdRaw(real)]);
5348
+ return {
5349
+ ok: true,
5350
+ slug: parsed.slug,
5351
+ frontmatter: {
5352
+ title: parsed.title,
5353
+ cwd: parsed.cwd,
5354
+ estimateMinutes: parsed.estimateMinutes,
5355
+ parallelGroup: parsed.parallelGroup,
5356
+ sourcePromptId: parsed.sourcePromptId,
5357
+ sourceTabId: parsed.sourceTabId,
5358
+ epicId: parsed.epicId,
5359
+ dependsOn: parsed.dependsOn,
5360
+ createdVia: parsed.createdVia,
5361
+ issuedAt: parsed.issuedAt,
5362
+ },
5363
+ body: parsed.body,
5364
+ raw,
5365
+ };
5366
+ } catch (e) {
5367
+ return { ok: false, error: e?.message ?? 'read failed' };
5368
+ }
5369
+ },
5370
+
5371
+ // Edits a NOT-yet-running PRD's frontmatter and/or body in place, refusing
5372
+ // once a queue row exists for it and that row is anything but 'pending'
5373
+ // (running/completed/failed/needs_review — editing the spec under a live
5374
+ // or already-finished executor would silently rewrite history). Reuses
5375
+ // prdFrontmatter.cjs's parsePrdFile/serializePrdFile round-trip pair (PRD
5376
+ // 1024) so unrecognized keys (e.g. dependsOn) and untouched recognized
5377
+ // keys' original line formatting survive unchanged.
5378
+ async updatePrd({ slug, cwd, frontmatter, body }) {
5379
+ const job = await this.getJob(slug);
5380
+ // 'quarantined' is also editable: it's the ONLY way a quarantined PRD's
5381
+ // createdVia stamp gets written (the adopt action below), so refusing it
5382
+ // here would make quarantine irreversible through the API.
5383
+ if (job && job.status !== 'pending' && job.status !== 'quarantined') {
5384
+ return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
5385
+ }
5386
+
5387
+ let dir = null;
5388
+ let filePath = null;
5389
+ if (cwd) {
5390
+ for (const d of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
5391
+ const p = safeSlugPathIn(d, slug);
5392
+ if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
5393
+ }
5394
+ } else {
5395
+ dir = await findPrdDir(slug);
5396
+ filePath = dir ? safeSlugPathIn(dir, slug) : null;
5397
+ }
5398
+ if (!filePath) return { ok: false, error: 'PRD not found' };
5399
+
5400
+ let raw;
5401
+ try {
5402
+ // Symlink defense, matching writePrd's comment: safeSlugPathIn is
5403
+ // lexical and does not resolve symlinks. updatePrd is a WRITE path
5404
+ // (unlike getPrdParsed's read-only realpath check), so also reject a
5405
+ // target that is itself already a symlink — a rogue job could plant
5406
+ // one inside the PRDs dir pointing outside the safe root.
5407
+ const real = await fsp.realpath(filePath);
5408
+ if (!real.startsWith(dir + path.sep)) return { ok: false, error: 'invalid slug' };
5409
+ const existing = await fsp.lstat(filePath).catch(() => null);
5410
+ if (existing && existing.isSymbolicLink()) return { ok: false, error: 'invalid slug' };
5411
+ raw = await fsp.readFile(real, 'utf8');
5412
+ } catch (e) {
5413
+ return { ok: false, error: e?.message ?? 'read failed' };
5414
+ }
5415
+
5416
+ const { frontmatter: fm, body: origBody } = parsePrdFile(raw);
5417
+ if (frontmatter) {
5418
+ for (const key of Object.keys(frontmatter)) {
5419
+ if (frontmatter[key] === undefined) continue;
5420
+ fm[key] = frontmatter[key];
5421
+ }
5422
+ }
5423
+ const newBody = body !== undefined ? body : origBody;
5424
+ const newRaw = serializePrdFile(fm, newBody);
5425
+
5426
+ try {
5427
+ await config.writeTextAtomic(filePath, newRaw, { writer: 'scheduler' });
5428
+ const stat = await fsp.stat(filePath);
5429
+ return { ok: true, slug, bytesWritten: stat.size };
5430
+ } catch (e) {
5431
+ return { ok: false, error: e?.message ?? 'write failed' };
5432
+ }
5433
+ },
5434
+
5435
+ // Cancels a job that hasn't finished yet. A 'running' job's process group
5436
+ // is SIGTERM'd (reusing killOrphanClaudePid — the same kill path boot
5437
+ // reconciliation uses for an orphaned running job) before its queue row is
5438
+ // finalized; a 'pending' job has no process to kill. There is no
5439
+ // 'cancelled' status in the closed job-status set (pending/running/
5440
+ // completed/failed/needs_review — see CLAUDE.md's domain model), so a
5441
+ // cancelled job lands in 'failed' with an error naming the cause,
5442
+ // consistent with every other non-success terminal outcome. Refuses a
5443
+ // slug that's already terminal — nothing left to cancel.
5444
+ async cancelJob(slug) {
5445
+ const state = await readQueue();
5446
+ const job = state.jobs.find((j) => j.slug === slug);
5447
+ if (!job) return { ok: false, error: 'not found' };
5448
+ if (job.status === 'completed' || job.status === 'failed' || job.status === 'needs_review') {
5449
+ return { ok: false, error: `job already terminal (status: "${job.status}") — nothing to cancel` };
5450
+ }
5451
+ const wasRunning = job.status === 'running';
5452
+ const pid = job.runtime?.pid;
5453
+ if (wasRunning && pid) {
5454
+ killOrphanClaudePid(pid);
5455
+ }
5456
+ await mutate((s) => {
5457
+ const idx = s.jobs.findIndex((j) => j.slug === slug);
5458
+ if (idx < 0) return;
5459
+ const j = s.jobs[idx];
5460
+ transitionJob(j, 'failed', { reason: 'cancelled via admin API', source: 'remote:cancelJob' });
5461
+ j.error = 'cancelled via admin API';
5462
+ j.finishedAt = new Date().toISOString();
5463
+ j.exitCode = j.exitCode ?? null;
5464
+ delete j.runtime;
5465
+ });
5466
+ await broadcast({ flush: true });
5467
+ return { ok: true, slug, status: 'failed', wasRunning, cwd: job.cwd ?? null };
5468
+ },
5469
+
4958
5470
  // Exposes the module-level allocateParallelGroup (PRD 548) to callers that
4959
5471
  // only hold the `remote` object (lib/prdCreate.cjs's create-prd route) —
4960
5472
  // reuses the same allocator the file-based /develop authoring path relies
@@ -4995,4 +5507,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
4995
5507
  });
4996
5508
  }
4997
5509
 
4998
- module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveNotifyPrd, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob };
5510
+ module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary };