claude-code-session-manager 0.39.1 → 0.39.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/dist/assets/{TiptapBody-CtuLATFR.js → TiptapBody-MCzz_6Zm.js} +1 -1
  2. package/dist/assets/{index-CefVSjmm.js → index-BPbLVT6E.js} +490 -488
  3. package/dist/assets/{index-B6JNrbpD.css → index-DVlD8N1X.css} +2 -2
  4. package/dist/index.html +2 -2
  5. package/package.json +3 -3
  6. package/scripts/lib/activeSessions.cjs +26 -3
  7. package/scripts/lib/watchdogHelpers.cjs +13 -6
  8. package/src/main/__tests__/browserView-destroyed-handler.test.cjs +84 -0
  9. package/src/main/__tests__/prdCreate.test.cjs +6 -0
  10. package/src/main/__tests__/prdLocations.test.cjs +41 -0
  11. package/src/main/__tests__/prdMigration.test.cjs +41 -0
  12. package/src/main/__tests__/promptSessionEvents.test.cjs +106 -0
  13. package/src/main/__tests__/queueHistory.test.cjs +2 -2
  14. package/src/main/__tests__/rcaFeedbackHook.test.cjs +108 -0
  15. package/src/main/__tests__/runVerify.test.cjs +459 -4
  16. package/src/main/__tests__/scheduler-admin-routes.test.cjs +52 -3
  17. package/src/main/__tests__/scheduler-archive-completed-prd.test.cjs +102 -0
  18. package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +126 -0
  19. package/src/main/__tests__/scheduler-committed-in-window.test.cjs +58 -3
  20. package/src/main/__tests__/scheduler-find-prd-dir.test.cjs +42 -0
  21. package/src/main/__tests__/scheduler-investigation-clean-skip.test.cjs +63 -0
  22. package/src/main/__tests__/scheduler-meta-code-sha.test.cjs +23 -0
  23. package/src/main/__tests__/scheduler-notify-originating-tab.test.cjs +68 -0
  24. package/src/main/__tests__/scheduler-reset-job-fields-guard.test.cjs +77 -0
  25. package/src/main/__tests__/scheduler-unreadable-queue-guard.test.cjs +62 -0
  26. package/src/main/browserView.cjs +5 -4
  27. package/src/main/chatRunner.cjs +90 -7
  28. package/src/main/config.cjs +9 -0
  29. package/src/main/ipcSchemas.cjs +1 -1
  30. package/src/main/lib/__tests__/terminalRunOutcome.test.cjs +118 -0
  31. package/src/main/lib/prdLocations.cjs +43 -6
  32. package/src/main/lib/prdMigration.cjs +5 -3
  33. package/src/main/lib/queueHistory.cjs +1 -1
  34. package/src/main/lib/rcaFeedbackHook.cjs +55 -5
  35. package/src/main/lib/terminalRunOutcome.cjs +100 -0
  36. package/src/main/promptSessionEvents.cjs +87 -0
  37. package/src/main/runVerify.cjs +195 -3
  38. package/src/main/scheduler.cjs +436 -80
  39. package/src/preload/api.d.ts +5 -1
@@ -48,7 +48,7 @@ const fsp = require('node:fs/promises');
48
48
  const path = require('node:path');
49
49
  const os = require('node:os');
50
50
  const { randomUUID } = require('node:crypto');
51
- const { execFile } = require('node:child_process');
51
+ const { execFile, execFileSync } = require('node:child_process');
52
52
  const { ipcMain } = require('electron');
53
53
  const billing = require('./usage.cjs');
54
54
  const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
@@ -62,7 +62,9 @@ const { createBroadcastCoalescer } = require('./lib/broadcastCoalescer.cjs');
62
62
  const prdParser = require('./scheduler/prdParser.cjs');
63
63
  const sessionsStore = require('./sessionsStore.cjs');
64
64
  const { enqueueExternalPrompt } = require('./chatRunner.cjs');
65
+ const { appendResponseEventIfKnown } = require('./promptSessionEvents.cjs');
65
66
  const { verifyRun } = require('./runVerify.cjs');
67
+ const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
66
68
  const logs = require('./logs.cjs');
67
69
  const { schemas, validated } = require('./ipcSchemas.cjs');
68
70
  const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
@@ -86,6 +88,24 @@ const { sweep: sweepFeedback } = require('../../scripts/lib/watchdogHelpers.cjs'
86
88
  const { resolvePrdsDirs, resolvePrdWriteDir } = require('./lib/prdLocations.cjs');
87
89
  const { migratePrds } = require('./lib/prdMigration.cjs');
88
90
 
91
+ // Captured once at module load so every run's meta sidecar can record how
92
+ // stale the running process is relative to on-disk source (incident: PRD
93
+ // 812-commit-guard-retry — the scheduler process was booted ~52 min before
94
+ // an exemption it should have applied landed on disk, and nothing in the
95
+ // run record showed that; this is the fix).
96
+ const SCHEDULER_BOOTED_AT = new Date().toISOString();
97
+ const SCHEDULER_CODE_SHA = (() => {
98
+ try {
99
+ return execFileSync('git', ['-C', __dirname, 'rev-parse', '--short', 'HEAD'], {
100
+ timeout: 5000,
101
+ encoding: 'utf8',
102
+ stdio: ['ignore', 'pipe', 'ignore'],
103
+ }).trim();
104
+ } catch {
105
+ return null;
106
+ }
107
+ })();
108
+
89
109
  const MAX_INVESTIGATION_DURATION_MS = 30 * 60_000;
90
110
 
91
111
  // After the agent emits a `result` event in its JSONL stream, the parent
@@ -261,9 +281,23 @@ async function committedInWindow(cwd, startedAt, finishedAt) {
261
281
  // checks out other branches, commits real work on each, then checks its
262
282
  // starting branch back out before exit leaves HEAD unchanged even though
263
283
  // commits landed — the fallback catches that case. Never throws.
284
+ //
285
+ // Both signals are known to race against ref/object visibility at the exact
286
+ // moment of process exit — a job that commits in a throwaway linked worktree
287
+ // and removes it before exiting can have committedInWindow() return false
288
+ // even though the commit is real and already pushed (confirmed incidents:
289
+ // pass-no-commit-worktree-commit-invisible-at-exit, RCA 770-pr269). When both
290
+ // signals say "no commit", wait a short bounded delay and retry once before
291
+ // giving up — a replayed identical call moments later reliably finds it.
292
+ const COMMIT_GUARD_RETRY_DELAY_MS = 2000;
293
+
264
294
  async function computeCommittedDuringRun(cwd, headBefore, headAfter, startedAt, untilIso) {
265
295
  if (headBefore && headAfter && headBefore !== headAfter) return true;
266
- return committedInWindow(cwd, startedAt, untilIso);
296
+ // Call via module.exports (not the bare local binding) so tests can
297
+ // vi.spyOn(scheduler, 'committedInWindow') to drive the retry deterministically.
298
+ if (await module.exports.committedInWindow(cwd, startedAt, untilIso)) return true;
299
+ await new Promise((resolve) => { setTimeout(resolve, COMMIT_GUARD_RETRY_DELAY_MS); });
300
+ return module.exports.committedInWindow(cwd, startedAt, untilIso);
267
301
  }
268
302
 
269
303
  /**
@@ -476,6 +510,34 @@ async function safeSlugPath(slug) {
476
510
  return safeSlugPathIn(dir, slug);
477
511
  }
478
512
 
513
+ /**
514
+ * Move a completed job's `<slug>.md` out of its PRD dir into that dir's
515
+ * sibling `prds-archived/`, so a finished slug can't be re-fired by the
516
+ * scheduler. Sibling-of-source (not the hard-coded legacy PRDS_ARCHIVE_DIR)
517
+ * so a per-project PRD (`<cwd>/session-manager-operations/scheduler/prds/`)
518
+ * archives into that SAME project's `prds-archived/`, not the global legacy
519
+ * one — PRDS_ARCHIVE_DIR only happens to coincide with this for the legacy
520
+ * PRDS_DIR. Mirrors the `schedule:clear-queue` archive logic (same
521
+ * containment check). Non-throwing: a missing source file (already archived
522
+ * or already gone) is a silent no-op, and any other error is logged as a
523
+ * warning — an archive failure must never break job-completion bookkeeping.
524
+ */
525
+ async function archiveCompletedPrd(slug, cwd) {
526
+ try {
527
+ const srcDir = (await findPrdDir(slug)) ?? prdDirForCwd(cwd);
528
+ const src = safeSlugPathIn(srcDir, slug);
529
+ if (!src) return;
530
+ const archiveDir = path.join(srcDir, '..', 'prds-archived');
531
+ await fsp.mkdir(archiveDir, { recursive: true });
532
+ const dst = path.join(archiveDir, `${slug}.md`);
533
+ await fsp.rename(src, dst);
534
+ } catch (e) {
535
+ if (e?.code !== 'ENOENT') {
536
+ logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'archiveCompletedPrd: rename failed', meta: { slug, error: e?.message } });
537
+ }
538
+ }
539
+ }
540
+
479
541
  // Bundled authoring guide seeded into the scheduler dir so the session-manager-dev
480
542
  // plugin's /develop and /prd skills — which reference this stable `~`-absolute
481
543
  // path — work on any user's machine, not just the author's.
@@ -625,22 +687,63 @@ function appendHeartbeat(entry) {
625
687
  }
626
688
  }
627
689
 
690
+ // An empty queue and an unreadable queue are NOT the same thing, and
691
+ // conflating them is destructive: reconcile() treats every PRD .md with no
692
+ // matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
693
+ // single failed read that yields `jobs: []` re-queues the entire archive of
694
+ // already-completed work — then mutate() writes that empty array back and the
695
+ // real statuses are gone. (2026-07-31: 189 completed jobs resurrected and
696
+ // fired into an ENOENT retry storm.)
697
+ //
698
+ // So: a MISSING file is a legitimately empty queue (first boot). A file that
699
+ // exists but won't read or parse is `unreadable` — a poison state that must
700
+ // never reach reconcile() or writeQueue(). Callers get the flag, not a lie.
701
+ const EMPTY_QUEUE = () => ({
702
+ config: { ...DEFAULT_CONFIG }, jobs: [], scheduledFor: null, lastRunAt: null, paused: null,
703
+ });
704
+
705
+ function shapeQueue(raw) {
706
+ const data = JSON.parse(raw);
707
+ return {
708
+ config: { ...DEFAULT_CONFIG, ...(data.config || {}) },
709
+ jobs: Array.isArray(data.jobs) ? data.jobs : [],
710
+ scheduledFor: data.scheduledFor ?? null,
711
+ lastRunAt: data.lastRunAt ?? null,
712
+ paused: data.paused ?? null,
713
+ };
714
+ }
715
+
716
+ // Quarantine a corrupt queue.json alongside itself (once per process — the
717
+ // first copy is the one that matters; later ticks would just overwrite it
718
+ // with the same bytes) so a human can diff it against the .bak-* snapshots.
719
+ let quarantined = false;
720
+ function unreadableQueue(e) {
721
+ const state = EMPTY_QUEUE();
722
+ state.unreadable = e?.message ?? 'queue.json read failed';
723
+ if (!quarantined) {
724
+ quarantined = true;
725
+ try {
726
+ fs.copyFileSync(QUEUE_PATH, `${QUEUE_PATH}.corrupt-${Date.now()}`);
727
+ } catch { /* best-effort: the read already failed, the copy may too */ }
728
+ }
729
+ console.error(`[scheduler] queue.json unreadable — refusing to treat as empty: ${state.unreadable}`);
730
+ logs.writeLine({
731
+ level: 'error', scope: 'scheduler',
732
+ message: 'queue.json unreadable — scheduling halted until it reads clean',
733
+ meta: { path: QUEUE_PATH, error: state.unreadable },
734
+ });
735
+ return state;
736
+ }
737
+
628
738
  // Sync queue read — passed to the supervisor module (which calls it from
629
739
  // supervisorTick / applyAction with no await) and the heartbeat interval.
630
740
  // IPC handlers and mutate() use readQueue (async) below.
631
741
  function readQueueSync() {
632
742
  try {
633
- const raw = fs.readFileSync(QUEUE_PATH, 'utf8');
634
- const data = JSON.parse(raw);
635
- return {
636
- config: { ...DEFAULT_CONFIG, ...(data.config || {}) },
637
- jobs: Array.isArray(data.jobs) ? data.jobs : [],
638
- scheduledFor: data.scheduledFor ?? null,
639
- lastRunAt: data.lastRunAt ?? null,
640
- paused: data.paused ?? null,
641
- };
642
- } catch {
643
- return { config: { ...DEFAULT_CONFIG }, jobs: [], scheduledFor: null, lastRunAt: null, paused: null };
743
+ return shapeQueue(fs.readFileSync(QUEUE_PATH, 'utf8'));
744
+ } catch (e) {
745
+ if (e?.code === 'ENOENT') return EMPTY_QUEUE();
746
+ return unreadableQueue(e);
644
747
  }
645
748
  }
646
749
 
@@ -649,21 +752,18 @@ function readQueueSync() {
649
752
  // hands control back to the renderer while the kernel paginates the file.
650
753
  async function readQueue() {
651
754
  try {
652
- const raw = await fsp.readFile(QUEUE_PATH, 'utf8');
653
- const data = JSON.parse(raw);
654
- return {
655
- config: { ...DEFAULT_CONFIG, ...(data.config || {}) },
656
- jobs: Array.isArray(data.jobs) ? data.jobs : [],
657
- scheduledFor: data.scheduledFor ?? null,
658
- lastRunAt: data.lastRunAt ?? null,
659
- paused: data.paused ?? null,
660
- };
661
- } catch {
662
- return { config: { ...DEFAULT_CONFIG }, jobs: [], scheduledFor: null, lastRunAt: null, paused: null };
755
+ return shapeQueue(await fsp.readFile(QUEUE_PATH, 'utf8'));
756
+ } catch (e) {
757
+ if (e?.code === 'ENOENT') return EMPTY_QUEUE();
758
+ return unreadableQueue(e);
663
759
  }
664
760
  }
665
761
 
666
762
  async function writeQueue(state) {
763
+ // Last line of defence: never persist a state derived from a failed read.
764
+ if (state && state.unreadable) {
765
+ throw new Error(`refusing to write queue.json from an unreadable read (${state.unreadable})`);
766
+ }
667
767
  ensureDirs();
668
768
  await config.writeJson(QUEUE_PATH, state);
669
769
  }
@@ -679,6 +779,12 @@ let mutateTail = Promise.resolve();
679
779
  function mutate(fn) {
680
780
  const next = mutateTail.then(async () => {
681
781
  const state = await readQueue();
782
+ // Bail BEFORE fn runs: a mutator handed an unreadable (therefore empty)
783
+ // state would compute its result from a queue that isn't there, and
784
+ // writeQueue would then persist that fiction over the real file.
785
+ if (state.unreadable) {
786
+ throw new Error(`queue mutation skipped: queue.json unreadable (${state.unreadable})`);
787
+ }
682
788
  const ret = await fn(state);
683
789
  await writeQueue(state);
684
790
  return ret;
@@ -811,6 +917,12 @@ function validatePromptForSpawn(body, srcLabel) {
811
917
  * Newly-discovered PRDs land as `pending`.
812
918
  */
813
919
  async function reconcile(state) {
920
+ // Defence in depth — tickQueue already gates on this, but reconcile is the
921
+ // function that would do the damage (every unmatched PRD .md becomes a
922
+ // fresh 'pending' row), so it refuses the poison state itself.
923
+ if (state && state.unreadable) {
924
+ throw new Error(`reconcile skipped: queue.json unreadable (${state.unreadable})`);
925
+ }
814
926
  const files = await listPrdFiles();
815
927
  const onDisk = new Map();
816
928
  for (const f of files) {
@@ -828,7 +940,22 @@ async function reconcile(state) {
828
940
  const seen = new Set();
829
941
  for (const job of state.jobs) {
830
942
  const p = onDisk.get(job.slug);
831
- if (!p) continue;
943
+ if (!p) {
944
+ // A terminal job whose .md is gone was archived on purpose — dropping
945
+ // its row is the intended end of the auto-archive flow.
946
+ //
947
+ // A PENDING or RUNNING job whose .md is merely not VISIBLE is a
948
+ // different thing entirely, and dropping it destroys queued work: the
949
+ // file may be unreadable, on a project whose dir failed to enumerate,
950
+ // or mid-move. "I can't see it" is not "the user deleted it", so the
951
+ // row survives — worst case it re-resolves on the next pass.
952
+ if (job.status === 'pending' || job.status === 'running') {
953
+ seen.add(job.slug);
954
+ next.push({ ...job });
955
+ console.warn(`[scheduler] reconcile: keeping ${job.status} job ${job.slug} — PRD source not visible in any candidate dir`);
956
+ }
957
+ continue;
958
+ }
832
959
  seen.add(job.slug);
833
960
  next.push({
834
961
  ...job,
@@ -876,6 +1003,18 @@ async function reconcile(state) {
876
1003
  // review and failed PRD files are NEVER auto-archived."
877
1004
  continue;
878
1005
  }
1006
+ // history.jsonl may not exist yet (nothing has crossed HISTORY_RETENTION_MS
1007
+ // since the feature shipped), which leaves historyBySlug empty and the
1008
+ // guard above inert. Fall back to reading the slug's own newest run
1009
+ // sidecars straight off disk — same "don't resurrect an already-terminal
1010
+ // slug" intent, independent of history.jsonl's existence.
1011
+ const fallback = latestTerminalOutcomeForSlug(slug, { runsDir: RUNS_DIR });
1012
+ if (fallback) {
1013
+ if (fallback.status === 'completed') {
1014
+ historyArchiveCandidates.push({ slug, status: fallback.status, finishedAt: fallback.finishedAt });
1015
+ }
1016
+ continue;
1017
+ }
879
1018
  const entry = {
880
1019
  slug,
881
1020
  title: p.title,
@@ -1223,8 +1362,21 @@ async function clearPause(source) {
1223
1362
  if (wasPaused) await broadcast({ flush: true });
1224
1363
  }
1225
1364
 
1226
- /** Mutate a job in place to "pending" with cleared run metadata. */
1227
- function resetJobFields(job, errorMsg) {
1365
+ /**
1366
+ * Mutate a job in place to "pending" with cleared run metadata.
1367
+ *
1368
+ * Refuses (no-ops, returns false) on a job already in a terminal success
1369
+ * state ('completed') unless opts.force is true — resetting a completed job
1370
+ * re-fires the PRD and re-executes already-shipped work (the false-failure
1371
+ * class PRD 812-workbench-review-nits-cleanup demonstrated: a completed job
1372
+ * was reset to pending and re-ran a correct no-op that then got flagged
1373
+ * needs_review). All internal call sites operate on jobs that are still
1374
+ * 'running'/'failed' at the point they call this, so the guard is a no-op
1375
+ * for them; only an external reset request (IPC/admin API) can target an
1376
+ * already-'completed' job, and that path is exactly what this guards.
1377
+ */
1378
+ function resetJobFields(job, errorMsg, opts = {}) {
1379
+ if (job.status === 'completed' && opts.force !== true) return false;
1228
1380
  job.status = 'pending';
1229
1381
  job.runId = null;
1230
1382
  job.startedAt = null;
@@ -1233,6 +1385,10 @@ function resetJobFields(job, errorMsg) {
1233
1385
  job.error = errorMsg ?? null;
1234
1386
  delete job.runtime;
1235
1387
  delete job.verifierVerdict;
1388
+ // Deliberately NOT deleting job.landedCommit: it must outlive a reset so a
1389
+ // re-fired run of this same slug can pass it to verifyRun as
1390
+ // priorLandedCommit (pass_no_commit_prior_run_verified exemption).
1391
+ return true;
1236
1392
  }
1237
1393
 
1238
1394
  // Grace period between a boot orphan's SIGTERM and reading its log to
@@ -1325,24 +1481,41 @@ function isNotifiableTerminalStatus(effectiveStatus) {
1325
1481
  * notifyOriginatingTab(job) → void
1326
1482
  *
1327
1483
  * On a true terminal transition (completed/failed — never the benign
1328
- * rateLimited auto-pause, which resets the job to pending instead), push a
1329
- * short status prompt into the chat tab that queued this PRD via
1330
- * enqueueExternalPrompt (PRD 753). Resolution order: (1) the PRD's own
1331
- * `sourceTabId` frontmatter, captured at creation time; (2) the first open
1332
- * tab (per sessionsStore's persisted tabs.json) whose cwd matches the job's
1333
- * cwd — first match only, no fan-out to multiple matching tabs; (3) no-op.
1334
- * Never throws to the caller (fire-and-forget from spawnJob). Deps are
1335
- * injectable (mirrors partitionBootOrphans's isAlive param) so unit tests can
1336
- * exercise the resolution logic without touching disk/electron.
1484
+ * rateLimited auto-pause, which resets the job to pending instead), publish
1485
+ * a short status notification for the PRD that queued this job.
1486
+ *
1487
+ * Resolution order (PRD 814): (1) if the PRD's `sourcePromptId` resolves to
1488
+ * a known, still-active PromptSession (minted for a dev-work dispatch, PRD
1489
+ * 813) under the job's cwd, append a 'response' PromptSessionEvent to THAT
1490
+ * session's own event chain — its own scoped PromptSessionConversation, not
1491
+ * whatever tab happens to be active — and stop; (2) otherwise, fall back to
1492
+ * today's behavior: push a short status prompt into the chat tab that queued
1493
+ * this PRD via enqueueExternalPrompt (PRD 753), resolved via the PRD's own
1494
+ * `sourceTabId` frontmatter, then the first open tab (per sessionsStore's
1495
+ * persisted tabs.json) whose cwd matches the job's cwd — first match only,
1496
+ * no fan-out to multiple matching tabs; (3) no-op. Never throws to the
1497
+ * caller (fire-and-forget from spawnJob). Deps are injectable (mirrors
1498
+ * partitionBootOrphans's isAlive param) so unit tests can exercise the
1499
+ * resolution logic without touching disk/electron.
1337
1500
  */
1338
1501
  async function notifyOriginatingTab(job, {
1339
1502
  parsePrdRaw = prdParser.parsePrdRaw,
1340
1503
  loadSessions = sessionsStore.load,
1341
1504
  sendPrompt = enqueueExternalPrompt,
1505
+ appendResponseEvent = appendResponseEventIfKnown,
1342
1506
  } = {}) {
1343
1507
  try {
1344
1508
  const prdPath = prdPathForJob(job);
1345
1509
  const prd = await parsePrdRaw(prdPath).catch(() => null);
1510
+ const message = `PRD ${job.slug} finished: ${job.status}. Check Scheduler for details.`;
1511
+
1512
+ if (prd?.sourcePromptId) {
1513
+ const routed = await appendResponseEvent(job.cwd || null, prd.sourcePromptId, message).catch((e) => {
1514
+ console.error('[scheduler] notifyOriginatingTab appendResponseEvent error', job?.slug, e);
1515
+ return false;
1516
+ });
1517
+ if (routed) return;
1518
+ }
1346
1519
 
1347
1520
  let targetTabId = prd?.sourceTabId || null;
1348
1521
  if (!targetTabId) {
@@ -1358,7 +1531,7 @@ async function notifyOriginatingTab(job, {
1358
1531
  return;
1359
1532
  }
1360
1533
 
1361
- sendPrompt(targetTabId, `PRD ${job.slug} finished: ${job.status}. Check Scheduler for details.`);
1534
+ sendPrompt(targetTabId, message);
1362
1535
  } catch (e) {
1363
1536
  console.error('[scheduler] notifyOriginatingTab error', job?.slug, e);
1364
1537
  }
@@ -1442,6 +1615,36 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
1442
1615
  return { action: 'retry', transientKind, retries };
1443
1616
  }
1444
1617
 
1618
+ /**
1619
+ * Commit-guard verdict decision. Pure/no I/O so the false-positive defenses
1620
+ * can be unit-tested directly rather than only through a live spawnJob run.
1621
+ * Returns the flagged verifyResult replacement, or null if the guard should
1622
+ * not fire (any of the four defenses applies).
1623
+ *
1624
+ * The fourth defense (legitimateNoOp) exists because runVerify.cjs's own
1625
+ * pass_no_commit exemptions (COMPLETED_EQUIVALENT_VERDICTS members like
1626
+ * pass_no_commit_already_shipped) already independently proved a truthful
1627
+ * PASS-with-no-commit is correct; without this check the commit-guard
1628
+ * double-punishes that same honest no-op for dirt a concurrent interactive
1629
+ * session left behind (incidents: 655-needs-review-rca-feedback-hook,
1630
+ * 672-fix-feedback-session-manager, 2026-07-31).
1631
+ */
1632
+ function commitGuardVerdict({ newlyDirty, siblingRunning, jobSelfCommitted, legitimateNoOp, verifyResult }) {
1633
+ if (!newlyDirty || newlyDirty.length === 0) return null;
1634
+ if (siblingRunning || jobSelfCommitted || legitimateNoOp) return null;
1635
+ const sample = newlyDirty.slice(0, 3).join(', ');
1636
+ const carried = [...(verifyResult?.annotations ?? [])];
1637
+ if (verifyResult && verifyResult.verdict !== 'clean') {
1638
+ carried.push({ verdict: verifyResult.verdict, reason: verifyResult.reason });
1639
+ }
1640
+ return {
1641
+ verdict: 'uncommitted_changes',
1642
+ reason: `finish protocol incomplete: ${newlyDirty.length} uncommitted file(s) left in working tree (e.g. ${sample})`,
1643
+ downgradeTo: 'needs_review',
1644
+ annotations: carried.length ? carried : undefined,
1645
+ };
1646
+ }
1647
+
1445
1648
  // ---------- execution ----------
1446
1649
 
1447
1650
  function pickRunDir() {
@@ -1490,16 +1693,35 @@ async function executeJob(job, runDir, defaultCwd, onPid) {
1490
1693
 
1491
1694
  // Read full PRD body fresh from disk (queue stored only the preview).
1492
1695
  let prompt;
1493
- const prdPath = prdPathForJob(job);
1696
+ let prdPath = prdPathForJob(job);
1494
1697
  try {
1495
1698
  const parsed = await parsePrd(prdPath);
1496
1699
  // Centrally enforce the review → security-review → verify → commit finish
1497
1700
  // sequence on every job, regardless of what the PRD body says.
1498
1701
  prompt = parsed.body + FINISH_PROTOCOL;
1499
1702
  } catch (e) {
1500
- safeLog(`[scheduler] failed to read PRD: ${e?.message}\n`);
1501
- closeFd();
1502
- return { exitCode: -1, durationMs: 0, error: e?.message };
1703
+ // The project-scoped dir isn't the only place a PRD source can live — a
1704
+ // writer that hasn't migrated to prdLocations.cjs yet (or a not-yet-run
1705
+ // boot migration) can leave it in the legacy global dir. Fall back to
1706
+ // findPrdDir's full candidate search before failing the job outright.
1707
+ const fallbackDir = await findPrdDir(job.slug);
1708
+ if (fallbackDir) {
1709
+ const fallbackPath = path.join(fallbackDir, `${job.slug}.md`);
1710
+ safeLog(`[scheduler] PRD not in project dir; found ${job.slug}.md in ${fallbackDir}\n`);
1711
+ try {
1712
+ const parsed = await parsePrd(fallbackPath);
1713
+ prompt = parsed.body + FINISH_PROTOCOL;
1714
+ prdPath = fallbackPath;
1715
+ } catch (e2) {
1716
+ safeLog(`[scheduler] failed to read PRD: ${e2?.message}\n`);
1717
+ closeFd();
1718
+ return { exitCode: -1, durationMs: 0, error: e2?.message };
1719
+ }
1720
+ } else {
1721
+ safeLog(`[scheduler] failed to read PRD: ${e?.message}\n`);
1722
+ closeFd();
1723
+ return { exitCode: -1, durationMs: 0, error: e?.message };
1724
+ }
1503
1725
  }
1504
1726
 
1505
1727
  const promptCheck = validatePromptForSpawn(prompt, prdPath);
@@ -1655,7 +1877,7 @@ async function executeJob(job, runDir, defaultCwd, onPid) {
1655
1877
  sl(`\n[scheduler] ${errMsg}\n`);
1656
1878
  // Sync write: inside a Promise executor callback; must flush meta
1657
1879
  // before resolve() so the spawnJob mutate() that follows sees it.
1658
- config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs });
1880
+ config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA });
1659
1881
  resolve({ exitCode: -1, durationMs, error: errMsg, sessionId });
1660
1882
  return;
1661
1883
  }
@@ -1685,6 +1907,7 @@ async function executeJob(job, runDir, defaultCwd, onPid) {
1685
1907
  slug: job.slug, cwd, sessionId, exitCode: effectiveCode, rateLimited, networkError,
1686
1908
  startedAt, finishedAt: Date.now(), durationMs,
1687
1909
  agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
1910
+ schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
1688
1911
  });
1689
1912
  resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, sessionId });
1690
1913
  },
@@ -1813,6 +2036,36 @@ ${logTail}
1813
2036
  DO NOT attempt the fix. ONLY write the file. When the file exists, exit immediately.`;
1814
2037
  }
1815
2038
 
2039
+ /**
2040
+ * Pure predicate: a run whose meta.json shows a clean exit (exitCode 0) and
2041
+ * whose verdicts.json verdict is completed-equivalent (clean / the
2042
+ * pass_no_commit exemptions) has nothing left to diagnose — spawning an
2043
+ * Opus investigation for it just manufactures a depth+1 fix-of-a-fix PRD for
2044
+ * already-shipped work. Any missing/malformed input is treated as "don't
2045
+ * skip" (fail-open: never let a missing artifact suppress a real
2046
+ * investigation). Exported for tests.
2047
+ */
2048
+ function shouldSkipInvestigationForCleanRun({ meta, verdicts }) {
2049
+ if (!meta || meta.exitCode !== 0) return false;
2050
+ if (!verdicts || !COMPLETED_EQUIVALENT_VERDICTS.has(verdicts.verdict)) return false;
2051
+ return true;
2052
+ }
2053
+
2054
+ /**
2055
+ * Reads <runDir>/<slug>.meta.json + <slug>.verdicts.json off disk for the
2056
+ * shouldSkipInvestigationForCleanRun guard. Fails safe to {} on any read/parse
2057
+ * error (never suppresses an investigation on a missing artifact).
2058
+ */
2059
+ function readRunOutcomeSidecars(runDir, slug) {
2060
+ const readJson = (p) => {
2061
+ try { return JSON.parse(fs.readFileSync(p, 'utf8')); } catch { return null; }
2062
+ };
2063
+ return {
2064
+ meta: readJson(path.join(runDir, `${slug}.meta.json`)),
2065
+ verdicts: readJson(path.join(runDir, `${slug}.verdicts.json`)),
2066
+ };
2067
+ }
2068
+
1816
2069
  /**
1817
2070
  * Spawn an Opus investigation session for a failed job. The investigator's job
1818
2071
  * is to read the failure log + original PRD, identify the root cause, and write
@@ -1821,13 +2074,22 @@ DO NOT attempt the fix. ONLY write the file. When the file exists, exit immediat
1821
2074
  * run out-of-band, so they don't consume the concurrency cap. They DO consume
1822
2075
  * tokens, which the when-available throttle will reflect on the next poll.
1823
2076
  *
1824
- * Skipped if the failed job is itself a fix-plan (avoids infinite recursion).
2077
+ * Skipped if the failed job is itself a fix-plan (avoids infinite recursion),
2078
+ * or if the run being investigated actually verified clean (nothing to fix —
2079
+ * see shouldSkipInvestigationForCleanRun).
1825
2080
  */
1826
2081
  async function spawnInvestigation(failedJob, runDir) {
1827
2082
  if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth)) {
1828
2083
  console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
1829
2084
  return { deferred: false };
1830
2085
  }
2086
+ {
2087
+ const { meta, verdicts } = readRunOutcomeSidecars(runDir, failedJob.slug);
2088
+ if (shouldSkipInvestigationForCleanRun({ meta, verdicts })) {
2089
+ console.log(`[scheduler] skip investigation: ${failedJob.slug} last run verified ${verdicts.verdict} (exit 0) — nothing to diagnose`);
2090
+ return { deferred: false };
2091
+ }
2092
+ }
1831
2093
  if (investigationsInFlight >= MAX_CONCURRENT_INVESTIGATIONS) {
1832
2094
  // Queue for retry when a slot frees rather than dropping — otherwise a failed
1833
2095
  // job (never 'needs_review', so reverifyNeedsReview won't retry it) would
@@ -2030,6 +2292,11 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2030
2292
  // in its tool output (see incidents: PRD 39, 44, 56 on 2026-05-23→24).
2031
2293
  // Called outside mutate() so the queue lock is not held during I/O.
2032
2294
  let verifyResult = null;
2295
+ // Persisted onto the job row (see the mutate() block below) whenever
2296
+ // this run's own HEAD advances, so a LATER re-fire of the same slug can
2297
+ // pass it back into verifyRun as priorLandedCommit (see the
2298
+ // pass_no_commit_prior_run_verified exemption in runVerify.cjs).
2299
+ let jobLandedCommitThisRun = null;
2033
2300
  if (res.exitCode === 0 && !res.rateLimited) {
2034
2301
  // Detect whether the job self-committed by comparing HEAD before/after.
2035
2302
  // Used by the sentinel override: SCHEDULER_VERDICT: PASS + a landed
@@ -2042,15 +2309,30 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2042
2309
  job.startedAt,
2043
2310
  new Date().toISOString(),
2044
2311
  );
2312
+ if (guardHeadBefore && headAtExit && headAtExit !== guardHeadBefore) {
2313
+ jobLandedCommitThisRun = headAtExit;
2314
+ }
2045
2315
 
2046
2316
  const prdPath = prdPathForJob(job);
2047
2317
  const stateForDeps = await readQueue();
2318
+ // priorLandedCommit: the commit a PREVIOUS run of this same slug landed,
2319
+ // if any — prefer the live jobs[] row (survives a resetJob, see
2320
+ // resetJobFields), fall back to history.jsonl for a slug that already
2321
+ // left jobs[]. Never the commit THIS run just made (committedDuringRun
2322
+ // already covers that case).
2323
+ const liveRow = stateForDeps.jobs.find((j) => j.slug === job.slug);
2324
+ let priorLandedCommit = liveRow?.landedCommit ?? null;
2325
+ if (!priorLandedCommit) {
2326
+ const hist = await queueHistory.historyTerminalBySlug().catch(() => null);
2327
+ priorLandedCommit = hist?.get(job.slug)?.landedCommit ?? null;
2328
+ }
2048
2329
  verifyResult = await verifyRun({
2049
2330
  runDir,
2050
2331
  prdPath,
2051
2332
  queueEntry: job,
2052
2333
  allJobs: stateForDeps.jobs,
2053
2334
  committedDuringRun,
2335
+ priorLandedCommit,
2054
2336
  }).catch((e) => ({
2055
2337
  verdict: 'verify_unavailable',
2056
2338
  reason: `verifier threw: ${e?.message ?? String(e)}`,
@@ -2076,6 +2358,17 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2076
2358
  // deliverable; leftover dirt is presumptively a concurrent external edit
2077
2359
  // (e.g. an interactive session editing the same repo), not the job's
2078
2360
  // unsaved work — so skip rather than false-flag a completed job.
2361
+ // - legitimate-no-op skip: if the verifier already independently proved
2362
+ // this run's PASS-with-no-commit is truthful (verdict is one of
2363
+ // COMPLETED_EQUIVALENT_VERDICTS — pass_no_commit_target_verified,
2364
+ // _prior_run_verified, _already_shipped), the run itself did nothing
2365
+ // wrong; dirt left by a concurrent interactive session (e.g.
2366
+ // /process-feedback writing new PRD .md files into the same repo
2367
+ // while this job's own AC turned out to already be satisfied) is not
2368
+ // this job's unfinished work. Without this skip, runVerify.cjs's
2369
+ // exemption and this guard double-punish the same honest no-op from
2370
+ // two different code paths (incidents: 655-needs-review-rca-feedback-hook,
2371
+ // 672-fix-feedback-session-manager, 2026-07-31).
2079
2372
  // Non-git cwds resolve to null and are skipped (the guard is best-effort).
2080
2373
  //
2081
2374
  // Runs even when a transcript-pattern verdict already fired: the commit-guard
@@ -2086,7 +2379,8 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2086
2379
  // annotation, so a real "finish protocol incomplete" is distinguishable from
2087
2380
  // transcript noise in the queue (feedback 2026-06-10 addendum).
2088
2381
  const guardWillRefire = verifyResult && verifyResult.downgradeTo === 'pending';
2089
- if (res.exitCode === 0 && !res.rateLimited && !guardWillRefire) {
2382
+ const guardIsLegitimateNoOp = verifyResult && COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict);
2383
+ if (res.exitCode === 0 && !res.rateLimited && !guardWillRefire && !guardIsLegitimateNoOp) {
2090
2384
  const after = await uncommittedChanges(guardCwd);
2091
2385
  if (after && after.length > 0) {
2092
2386
  const baseSet = new Set(guardBaseline || []);
@@ -2097,19 +2391,15 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2097
2391
  );
2098
2392
  const guardHeadAfter = await gitHead(guardCwd);
2099
2393
  const jobSelfCommitted = guardHeadBefore && guardHeadAfter && guardHeadAfter !== guardHeadBefore;
2100
- if (newlyDirty.length > 0 && !siblingRunning && !jobSelfCommitted) {
2101
- const sample = newlyDirty.slice(0, 3).join(', ');
2102
- // Carry any prior transcript verdict + its annotations forward as notes.
2103
- const carried = [...(verifyResult?.annotations ?? [])];
2104
- if (verifyResult && verifyResult.verdict !== 'clean') {
2105
- carried.push({ verdict: verifyResult.verdict, reason: verifyResult.reason });
2106
- }
2107
- verifyResult = {
2108
- verdict: 'uncommitted_changes',
2109
- reason: `finish protocol incomplete: ${newlyDirty.length} uncommitted file(s) left in working tree (e.g. ${sample})`,
2110
- downgradeTo: 'needs_review',
2111
- annotations: carried.length ? carried : undefined,
2112
- };
2394
+ const guardVerdict = commitGuardVerdict({
2395
+ newlyDirty,
2396
+ siblingRunning,
2397
+ jobSelfCommitted,
2398
+ legitimateNoOp: guardIsLegitimateNoOp,
2399
+ verifyResult,
2400
+ });
2401
+ if (guardVerdict) {
2402
+ verifyResult = guardVerdict;
2113
2403
  console.log(`[scheduler] commit-guard: ${job.slug} left ${newlyDirty.length} files uncommitted → needs_review`);
2114
2404
  }
2115
2405
  }
@@ -2138,6 +2428,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2138
2428
  let investigationJobSnapshot = null;
2139
2429
  let needsReviewRcaSnapshot = null;
2140
2430
  let terminalNotifySnapshot = null;
2431
+ const newlyCompletedPrds = [];
2141
2432
  await mutate((s) => {
2142
2433
  const i2 = s.jobs.findIndex((x) => x.slug === job.slug);
2143
2434
  if (i2 >= 0) {
@@ -2158,11 +2449,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2158
2449
  effectiveStatus = 'failed';
2159
2450
  } else if (
2160
2451
  !verifyResult
2161
- || verifyResult.verdict === 'clean'
2162
- // pass_no_commit_target_verified: -merge-main postcondition exemption
2163
- // (runVerify.cjs) — an independently gh-confirmed clean merge target,
2164
- // not a plain unsubstantiated PASS. Completed, same as 'clean'.
2165
- || verifyResult.verdict === 'pass_no_commit_target_verified'
2452
+ || COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict)
2166
2453
  ) {
2167
2454
  effectiveStatus = 'completed';
2168
2455
  } else if (verifyResult.downgradeTo === 'pending') {
@@ -2180,6 +2467,13 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2180
2467
  s.jobs[i2].error = effectiveStatus === 'needs_review'
2181
2468
  ? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
2182
2469
  : (res.error || null);
2470
+ // Persist the commit THIS run landed (if HEAD advanced) so a later
2471
+ // re-fire of the same slug can prove its own no-op re-run is
2472
+ // truthful via the pass_no_commit_prior_run_verified exemption.
2473
+ // Survives resetJobFields — see that function's comment.
2474
+ if (jobLandedCommitThisRun) {
2475
+ s.jobs[i2].landedCommit = jobLandedCommitThisRun;
2476
+ }
2183
2477
  // Persist the verifier's verdict string so the renderer can show it.
2184
2478
  if (verifyResult?.verdict && verifyResult.verdict !== 'clean') {
2185
2479
  s.jobs[i2].verifierVerdict = verifyResult.verdict;
@@ -2201,6 +2495,9 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2201
2495
  if (isNotifiableTerminalStatus(effectiveStatus)) {
2202
2496
  terminalNotifySnapshot = { ...s.jobs[i2] };
2203
2497
  }
2498
+ if (effectiveStatus === 'completed') {
2499
+ newlyCompletedPrds.push({ slug: s.jobs[i2].slug, cwd: s.jobs[i2].cwd });
2500
+ }
2204
2501
  if (effectiveStatus === 'failed') {
2205
2502
  actuallyFailed = true;
2206
2503
  failedJobSnapshot = { ...s.jobs[i2] };
@@ -2252,11 +2549,15 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
2252
2549
  if (priorStatus === 'needs_review') {
2253
2550
  delete orig.verifierVerdict;
2254
2551
  }
2552
+ newlyCompletedPrds.push({ slug: orig.slug, cwd: orig.cwd });
2255
2553
  }
2256
2554
  }
2257
2555
  }
2258
2556
  }
2259
2557
  });
2558
+ for (const { slug, cwd } of newlyCompletedPrds) {
2559
+ await archiveCompletedPrd(slug, cwd);
2560
+ }
2260
2561
  await broadcast({ flush: true });
2261
2562
 
2262
2563
  if (terminalNotifySnapshot) {
@@ -2377,6 +2678,12 @@ let tickTail = Promise.resolve();
2377
2678
  function tickQueue() {
2378
2679
  const next = tickTail.then(async () => {
2379
2680
  const state = await readQueue();
2681
+ // Never reconcile against an unreadable queue: reconcile() would see zero
2682
+ // job rows for every PRD on disk and resurrect the lot as 'pending'.
2683
+ if (state.unreadable) {
2684
+ console.error('[scheduler] tickQueue skipped: queue.json unreadable');
2685
+ return { fired: false, reason: 'unreadable' };
2686
+ }
2380
2687
  if (state.paused) {
2381
2688
  console.log('[scheduler] tickQueue skipped: paused');
2382
2689
  return { fired: false, reason: 'paused' };
@@ -2462,6 +2769,8 @@ function forceTickOutcome(result) {
2462
2769
  return { ok: true, kind: 'info', message: `Already running — ${result.runningCount} job(s) in flight` };
2463
2770
  case 'paused':
2464
2771
  return { ok: true, kind: 'warn', message: 'Scheduler is paused' };
2772
+ case 'unreadable':
2773
+ return { ok: false, kind: 'error', message: 'queue.json is unreadable — scheduling halted; a .corrupt-<ts> copy was saved next to it' };
2465
2774
  case 'cancelled':
2466
2775
  return { ok: true, kind: 'warn', message: 'Batch cancelled — try again' };
2467
2776
  case 'memory-deferred':
@@ -2477,6 +2786,10 @@ function forceTickOutcome(result) {
2477
2786
 
2478
2787
  async function runDueJobs() {
2479
2788
  const state = await readQueue();
2789
+ if (state.unreadable) {
2790
+ console.error('[scheduler] runDueJobs skipped: queue.json unreadable');
2791
+ return { fired: false, reason: 'unreadable' };
2792
+ }
2480
2793
  if (state.paused) {
2481
2794
  console.log('[scheduler] runDueJobs skipped: paused');
2482
2795
  return { fired: false, reason: 'paused' };
@@ -2730,7 +3043,7 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
2730
3043
  // investigation jobs correctly found "nothing to fix" but were flagged
2731
3044
  // anyway). For non-fix-plan jobs the exemption never applies, so rescanning
2732
3045
  // their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
2733
- const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'pass_no_commit']);
3046
+ const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'pass_no_commit', 'pass_no_commit_already_shipped']);
2734
3047
 
2735
3048
  // Bounds fix-plan recursion: depth 1 = the original job, depth 2 = its fix
2736
3049
  // (gets exactly one follow-up investigation if it also lands in
@@ -2882,6 +3195,13 @@ async function reverifyNeedsReview() {
2882
3195
  // commit-guard uses gitHead() (before/after HEAD diff); here the run is
2883
3196
  // already over so we query git log filtered to [startedAt, finishedAt+60s].
2884
3197
  const committedDuringRun = await committedInWindow(job.cwd, job.startedAt, job.finishedAt);
3198
+ // priorLandedCommit: same lookup as spawnJob's post-run verify — the live
3199
+ // jobs[] row first (survives a resetJob), else history.jsonl.
3200
+ let priorLandedCommit = job.landedCommit ?? null;
3201
+ if (!priorLandedCommit) {
3202
+ const hist = await queueHistory.historyTerminalBySlug().catch(() => null);
3203
+ priorLandedCommit = hist?.get(job.slug)?.landedCommit ?? null;
3204
+ }
2885
3205
  let v = null;
2886
3206
  try {
2887
3207
  v = await verifyRun({
@@ -2891,12 +3211,10 @@ async function reverifyNeedsReview() {
2891
3211
  allJobs: snap.jobs,
2892
3212
  committedDuringRun,
2893
3213
  allowPreSentinelHeal: true,
3214
+ priorLandedCommit,
2894
3215
  });
2895
3216
  } catch { leftForReview.push({ slug: job.slug, reason: 'verifyRun threw' }); continue; }
2896
- // pass_no_commit_target_verified: -merge-main postcondition exemption
2897
- // (runVerify.cjs) — same "heal it" treatment as 'clean', see spawnJob's
2898
- // effectiveStatus branch above for the primary-path equivalent.
2899
- if (v && (v.verdict === 'clean' || v.verdict === 'pass_no_commit_target_verified')) {
3217
+ if (v && COMPLETED_EQUIVALENT_VERDICTS.has(v.verdict)) {
2900
3218
  healed.push(job.slug);
2901
3219
  } else {
2902
3220
  leftForReview.push({ slug: job.slug, reason: v ? `${v.verdict}: ${v.reason}` : 'null verdict' });
@@ -2904,15 +3222,20 @@ async function reverifyNeedsReview() {
2904
3222
  }
2905
3223
  if (healed.length) {
2906
3224
  const healSet = new Set(healed);
3225
+ const healedPrds = [];
2907
3226
  await mutate((s) => {
2908
3227
  for (const j of s.jobs) {
2909
3228
  if (j.status === 'needs_review' && healSet.has(j.slug)) {
2910
3229
  j.status = 'completed';
2911
3230
  j.error = null;
2912
3231
  delete j.verifierVerdict;
3232
+ healedPrds.push({ slug: j.slug, cwd: j.cwd });
2913
3233
  }
2914
3234
  }
2915
3235
  });
3236
+ for (const { slug, cwd } of healedPrds) {
3237
+ await archiveCompletedPrd(slug, cwd);
3238
+ }
2916
3239
  console.log(`[scheduler] boot reverify: healed ${healed.length} stale needs_review → completed (${healed.join(', ')})`);
2917
3240
  await broadcast();
2918
3241
  }
@@ -2934,6 +3257,7 @@ async function reverifyNeedsReview() {
2934
3257
  // boot/tick, because the promotion only ran inside `if (healed.length)`
2935
3258
  // scoped to that single pass's fresh heals.)
2936
3259
  const promoted = [];
3260
+ const promotedPrds = [];
2937
3261
  await mutate((s) => {
2938
3262
  for (const job of s.jobs) {
2939
3263
  if (job.status !== 'completed' || !isFixPlanSlug(job.slug)) continue;
@@ -2946,8 +3270,12 @@ async function reverifyNeedsReview() {
2946
3270
  orig.completedBy = job.slug;
2947
3271
  if (priorStatus === 'needs_review') delete orig.verifierVerdict;
2948
3272
  promoted.push(`${orig.slug} (was ${priorStatus}, via ${job.slug})`);
3273
+ promotedPrds.push({ slug: orig.slug, cwd: orig.cwd });
2949
3274
  }
2950
3275
  });
3276
+ for (const { slug, cwd } of promotedPrds) {
3277
+ await archiveCompletedPrd(slug, cwd);
3278
+ }
2951
3279
  if (promoted.length) {
2952
3280
  console.log(`[scheduler] boot reverify: auto-promoted ${promoted.length} original(s): ${promoted.join(', ')}`);
2953
3281
  await broadcast();
@@ -3094,13 +3422,21 @@ function registerScheduleHandlers() {
3094
3422
 
3095
3423
  ipcMain.handle('schedule:reset-job', validated(schemas.scheduleSlug, async ({ slug }) => {
3096
3424
  if (!(await safeSlugPath(slug))) return { ok: false, error: 'invalid slug' };
3097
- const found = await mutate((state) => {
3425
+ const outcome = await mutate((state) => {
3098
3426
  const idx = state.jobs.findIndex((j) => j.slug === slug);
3099
- if (idx < 0) return false;
3100
- resetJobFields(state.jobs[idx]);
3101
- return true;
3427
+ if (idx < 0) return 'not-found';
3428
+ // Guard is in resetJobFields: refuses to reset an already-'completed'
3429
+ // job, which would otherwise re-fire a PRD whose deliverable already
3430
+ // landed (see resetJobFields' doc comment for the incident).
3431
+ return resetJobFields(state.jobs[idx]) ? 'ok' : 'refused';
3102
3432
  });
3103
- if (!found) return { ok: false, error: 'not found' };
3433
+ if (outcome === 'not-found') return { ok: false, error: 'not found' };
3434
+ if (outcome === 'refused') {
3435
+ return {
3436
+ ok: false,
3437
+ error: 'job already completed — resetting it would re-execute shipped work; archive the PRD instead',
3438
+ };
3439
+ }
3104
3440
  await broadcast({ flush: true });
3105
3441
  return { ok: true };
3106
3442
  }));
@@ -3315,6 +3651,7 @@ async function init() {
3315
3651
  const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
3316
3652
  bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
3317
3653
  }
3654
+ const bootReconciledCompletions = [];
3318
3655
  await mutate((state) => {
3319
3656
  for (const j of state.jobs) {
3320
3657
  if (j.status !== 'running' || !immediateSlugs.includes(j.slug)) continue;
@@ -3322,9 +3659,13 @@ async function init() {
3322
3659
  const pid = j.runtime?.pid;
3323
3660
  const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
3324
3661
  applyOrphanOutcome(j, outcome, killNote);
3662
+ if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
3325
3663
  console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
3326
3664
  }
3327
3665
  });
3666
+ for (const { slug, cwd } of bootReconciledCompletions) {
3667
+ await archiveCompletedPrd(slug, cwd);
3668
+ }
3328
3669
 
3329
3670
  // Still-alive orphans: SIGTERM (+ killOrphanClaudePid's own deferred SIGKILL
3330
3671
  // follow-up) now, but classification waits until BOOT_ORPHAN_KILL_GRACE_MS
@@ -3344,6 +3685,7 @@ async function init() {
3344
3685
  setTimeout(() => {
3345
3686
  const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
3346
3687
  const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
3688
+ let deferredCompletedCwd;
3347
3689
  mutate((state) => {
3348
3690
  const cur = state.jobs.find((x) => x.slug === slug);
3349
3691
  // Race guard: bail if the job already resolved, OR if it's already been
@@ -3353,6 +3695,9 @@ async function init() {
3353
3695
  if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
3354
3696
  applyOrphanOutcome(cur, outcome, killNote);
3355
3697
  console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
3698
+ deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
3699
+ }).then(() => {
3700
+ if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
3356
3701
  }).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
3357
3702
  }, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
3358
3703
  }
@@ -3551,15 +3896,25 @@ const remote = {
3551
3896
  }
3552
3897
  },
3553
3898
 
3554
- async resetJob(slug) {
3899
+ async resetJob(slug, opts = {}) {
3555
3900
  if (!(await safeSlugPath(slug))) return { ok: false, error: 'invalid slug' };
3556
- const found = await mutate((state) => {
3901
+ const outcome = await mutate((state) => {
3557
3902
  const idx = state.jobs.findIndex((j) => j.slug === slug);
3558
- if (idx < 0) return false;
3559
- resetJobFields(state.jobs[idx]);
3560
- return true;
3903
+ if (idx < 0) return { kind: 'not-found' };
3904
+ // Terminal-status guard lives in resetJobFields itself; force:true
3905
+ // threads through to override it.
3906
+ if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true })) {
3907
+ return { kind: 'refused' };
3908
+ }
3909
+ return { kind: 'ok' };
3561
3910
  });
3562
- if (!found) return { ok: false, error: 'not found' };
3911
+ if (outcome.kind === 'not-found') return { ok: false, error: 'not found' };
3912
+ if (outcome.kind === 'refused') {
3913
+ return {
3914
+ ok: false,
3915
+ error: 'job already completed — resetting it would re-execute shipped work; archive the PRD instead, or pass force:true',
3916
+ };
3917
+ }
3563
3918
  await broadcast({ flush: true });
3564
3919
  return { ok: true, slug, status: 'pending' };
3565
3920
  },
@@ -3625,9 +3980,10 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
3625
3980
  sendJson(res, 400, { ok: false, error: 'missing slug' });
3626
3981
  return;
3627
3982
  }
3628
- const result = await remoteObj.resetJob(slug);
3983
+ const force = parsed.force === true;
3984
+ const result = await remoteObj.resetJob(slug, { force });
3629
3985
  sendJson(res, 200, result);
3630
3986
  });
3631
3987
  }
3632
3988
 
3633
- module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, feedbackSweepDue, FEEDBACK_SWEEP_TICK_INTERVAL, sweepFeedback, registerAdminRoutes, notifyOriginatingTab, isNotifiableTerminalStatus, candidatePrdsDirs, prdDirForCwd, prdPathForJob, findPrdDir, runPrdMigration };
3989
+ module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, writeQueue, reconcile, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, feedbackSweepDue, FEEDBACK_SWEEP_TICK_INTERVAL, sweepFeedback, registerAdminRoutes, notifyOriginatingTab, isNotifiableTerminalStatus, candidatePrdsDirs, prdDirForCwd, prdPathForJob, findPrdDir, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields };