claude-code-session-manager 0.39.1 → 0.39.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{TiptapBody-CtuLATFR.js → TiptapBody-B90xy18x.js} +1 -1
- package/dist/assets/{index-CefVSjmm.js → index-BUuhV6vT.js} +496 -494
- package/dist/assets/{index-B6JNrbpD.css → index-DVlD8N1X.css} +2 -2
- package/dist/index.html +2 -2
- package/package.json +3 -3
- package/scripts/lib/activeSessions.cjs +26 -3
- package/scripts/lib/watchdogHelpers.cjs +13 -6
- package/src/main/__tests__/browserView-destroyed-handler.test.cjs +84 -0
- package/src/main/__tests__/prdCreate.test.cjs +6 -0
- package/src/main/__tests__/prdLocations.test.cjs +41 -0
- package/src/main/__tests__/prdMigration.test.cjs +41 -0
- package/src/main/__tests__/promptSessionEvents.test.cjs +106 -0
- package/src/main/__tests__/queueHistory.test.cjs +2 -2
- package/src/main/__tests__/rcaFeedbackHook.test.cjs +108 -0
- package/src/main/__tests__/runVerify.test.cjs +459 -4
- package/src/main/__tests__/scheduler-admin-routes.test.cjs +52 -3
- package/src/main/__tests__/scheduler-archive-completed-prd.test.cjs +102 -0
- package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +126 -0
- package/src/main/__tests__/scheduler-committed-in-window.test.cjs +58 -3
- package/src/main/__tests__/scheduler-find-prd-dir.test.cjs +42 -0
- package/src/main/__tests__/scheduler-investigation-clean-skip.test.cjs +63 -0
- package/src/main/__tests__/scheduler-meta-code-sha.test.cjs +23 -0
- package/src/main/__tests__/scheduler-notify-originating-tab.test.cjs +68 -0
- package/src/main/__tests__/scheduler-reset-job-fields-guard.test.cjs +77 -0
- package/src/main/__tests__/scheduler-unreadable-queue-guard.test.cjs +62 -0
- package/src/main/browserView.cjs +5 -4
- package/src/main/chatRunner.cjs +90 -7
- package/src/main/config.cjs +9 -0
- package/src/main/ipcSchemas.cjs +1 -1
- package/src/main/lib/__tests__/terminalRunOutcome.test.cjs +118 -0
- package/src/main/lib/prdLocations.cjs +43 -6
- package/src/main/lib/prdMigration.cjs +5 -3
- package/src/main/lib/queueHistory.cjs +1 -1
- package/src/main/lib/rcaFeedbackHook.cjs +55 -5
- package/src/main/lib/terminalRunOutcome.cjs +100 -0
- package/src/main/promptSessionEvents.cjs +87 -0
- package/src/main/runVerify.cjs +195 -3
- package/src/main/scheduler.cjs +436 -80
- package/src/preload/api.d.ts +5 -1
package/src/main/scheduler.cjs
CHANGED
|
@@ -48,7 +48,7 @@ const fsp = require('node:fs/promises');
|
|
|
48
48
|
const path = require('node:path');
|
|
49
49
|
const os = require('node:os');
|
|
50
50
|
const { randomUUID } = require('node:crypto');
|
|
51
|
-
const { execFile } = require('node:child_process');
|
|
51
|
+
const { execFile, execFileSync } = require('node:child_process');
|
|
52
52
|
const { ipcMain } = require('electron');
|
|
53
53
|
const billing = require('./usage.cjs');
|
|
54
54
|
const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
|
|
@@ -62,7 +62,9 @@ const { createBroadcastCoalescer } = require('./lib/broadcastCoalescer.cjs');
|
|
|
62
62
|
const prdParser = require('./scheduler/prdParser.cjs');
|
|
63
63
|
const sessionsStore = require('./sessionsStore.cjs');
|
|
64
64
|
const { enqueueExternalPrompt } = require('./chatRunner.cjs');
|
|
65
|
+
const { appendResponseEventIfKnown } = require('./promptSessionEvents.cjs');
|
|
65
66
|
const { verifyRun } = require('./runVerify.cjs');
|
|
67
|
+
const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
|
|
66
68
|
const logs = require('./logs.cjs');
|
|
67
69
|
const { schemas, validated } = require('./ipcSchemas.cjs');
|
|
68
70
|
const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
|
|
@@ -86,6 +88,24 @@ const { sweep: sweepFeedback } = require('../../scripts/lib/watchdogHelpers.cjs'
|
|
|
86
88
|
const { resolvePrdsDirs, resolvePrdWriteDir } = require('./lib/prdLocations.cjs');
|
|
87
89
|
const { migratePrds } = require('./lib/prdMigration.cjs');
|
|
88
90
|
|
|
91
|
+
// Captured once at module load so every run's meta sidecar can record how
|
|
92
|
+
// stale the running process is relative to on-disk source (incident: PRD
|
|
93
|
+
// 812-commit-guard-retry — the scheduler process was booted ~52 min before
|
|
94
|
+
// an exemption it should have applied landed on disk, and nothing in the
|
|
95
|
+
// run record showed that; this is the fix).
|
|
96
|
+
const SCHEDULER_BOOTED_AT = new Date().toISOString();
|
|
97
|
+
const SCHEDULER_CODE_SHA = (() => {
|
|
98
|
+
try {
|
|
99
|
+
return execFileSync('git', ['-C', __dirname, 'rev-parse', '--short', 'HEAD'], {
|
|
100
|
+
timeout: 5000,
|
|
101
|
+
encoding: 'utf8',
|
|
102
|
+
stdio: ['ignore', 'pipe', 'ignore'],
|
|
103
|
+
}).trim();
|
|
104
|
+
} catch {
|
|
105
|
+
return null;
|
|
106
|
+
}
|
|
107
|
+
})();
|
|
108
|
+
|
|
89
109
|
const MAX_INVESTIGATION_DURATION_MS = 30 * 60_000;
|
|
90
110
|
|
|
91
111
|
// After the agent emits a `result` event in its JSONL stream, the parent
|
|
@@ -261,9 +281,23 @@ async function committedInWindow(cwd, startedAt, finishedAt) {
|
|
|
261
281
|
// checks out other branches, commits real work on each, then checks its
|
|
262
282
|
// starting branch back out before exit leaves HEAD unchanged even though
|
|
263
283
|
// commits landed — the fallback catches that case. Never throws.
|
|
284
|
+
//
|
|
285
|
+
// Both signals are known to race against ref/object visibility at the exact
|
|
286
|
+
// moment of process exit — a job that commits in a throwaway linked worktree
|
|
287
|
+
// and removes it before exiting can have committedInWindow() return false
|
|
288
|
+
// even though the commit is real and already pushed (confirmed incidents:
|
|
289
|
+
// pass-no-commit-worktree-commit-invisible-at-exit, RCA 770-pr269). When both
|
|
290
|
+
// signals say "no commit", wait a short bounded delay and retry once before
|
|
291
|
+
// giving up — a replayed identical call moments later reliably finds it.
|
|
292
|
+
const COMMIT_GUARD_RETRY_DELAY_MS = 2000;
|
|
293
|
+
|
|
264
294
|
async function computeCommittedDuringRun(cwd, headBefore, headAfter, startedAt, untilIso) {
|
|
265
295
|
if (headBefore && headAfter && headBefore !== headAfter) return true;
|
|
266
|
-
|
|
296
|
+
// Call via module.exports (not the bare local binding) so tests can
|
|
297
|
+
// vi.spyOn(scheduler, 'committedInWindow') to drive the retry deterministically.
|
|
298
|
+
if (await module.exports.committedInWindow(cwd, startedAt, untilIso)) return true;
|
|
299
|
+
await new Promise((resolve) => { setTimeout(resolve, COMMIT_GUARD_RETRY_DELAY_MS); });
|
|
300
|
+
return module.exports.committedInWindow(cwd, startedAt, untilIso);
|
|
267
301
|
}
|
|
268
302
|
|
|
269
303
|
/**
|
|
@@ -476,6 +510,34 @@ async function safeSlugPath(slug) {
|
|
|
476
510
|
return safeSlugPathIn(dir, slug);
|
|
477
511
|
}
|
|
478
512
|
|
|
513
|
+
/**
|
|
514
|
+
* Move a completed job's `<slug>.md` out of its PRD dir into that dir's
|
|
515
|
+
* sibling `prds-archived/`, so a finished slug can't be re-fired by the
|
|
516
|
+
* scheduler. Sibling-of-source (not the hard-coded legacy PRDS_ARCHIVE_DIR)
|
|
517
|
+
* so a per-project PRD (`<cwd>/session-manager-operations/scheduler/prds/`)
|
|
518
|
+
* archives into that SAME project's `prds-archived/`, not the global legacy
|
|
519
|
+
* one — PRDS_ARCHIVE_DIR only happens to coincide with this for the legacy
|
|
520
|
+
* PRDS_DIR. Mirrors the `schedule:clear-queue` archive logic (same
|
|
521
|
+
* containment check). Non-throwing: a missing source file (already archived
|
|
522
|
+
* or already gone) is a silent no-op, and any other error is logged as a
|
|
523
|
+
* warning — an archive failure must never break job-completion bookkeeping.
|
|
524
|
+
*/
|
|
525
|
+
async function archiveCompletedPrd(slug, cwd) {
|
|
526
|
+
try {
|
|
527
|
+
const srcDir = (await findPrdDir(slug)) ?? prdDirForCwd(cwd);
|
|
528
|
+
const src = safeSlugPathIn(srcDir, slug);
|
|
529
|
+
if (!src) return;
|
|
530
|
+
const archiveDir = path.join(srcDir, '..', 'prds-archived');
|
|
531
|
+
await fsp.mkdir(archiveDir, { recursive: true });
|
|
532
|
+
const dst = path.join(archiveDir, `${slug}.md`);
|
|
533
|
+
await fsp.rename(src, dst);
|
|
534
|
+
} catch (e) {
|
|
535
|
+
if (e?.code !== 'ENOENT') {
|
|
536
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'archiveCompletedPrd: rename failed', meta: { slug, error: e?.message } });
|
|
537
|
+
}
|
|
538
|
+
}
|
|
539
|
+
}
|
|
540
|
+
|
|
479
541
|
// Bundled authoring guide seeded into the scheduler dir so the session-manager-dev
|
|
480
542
|
// plugin's /develop and /prd skills — which reference this stable `~`-absolute
|
|
481
543
|
// path — work on any user's machine, not just the author's.
|
|
@@ -625,22 +687,63 @@ function appendHeartbeat(entry) {
|
|
|
625
687
|
}
|
|
626
688
|
}
|
|
627
689
|
|
|
690
|
+
// An empty queue and an unreadable queue are NOT the same thing, and
|
|
691
|
+
// conflating them is destructive: reconcile() treats every PRD .md with no
|
|
692
|
+
// matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
|
|
693
|
+
// single failed read that yields `jobs: []` re-queues the entire archive of
|
|
694
|
+
// already-completed work — then mutate() writes that empty array back and the
|
|
695
|
+
// real statuses are gone. (2026-07-31: 189 completed jobs resurrected and
|
|
696
|
+
// fired into an ENOENT retry storm.)
|
|
697
|
+
//
|
|
698
|
+
// So: a MISSING file is a legitimately empty queue (first boot). A file that
|
|
699
|
+
// exists but won't read or parse is `unreadable` — a poison state that must
|
|
700
|
+
// never reach reconcile() or writeQueue(). Callers get the flag, not a lie.
|
|
701
|
+
const EMPTY_QUEUE = () => ({
|
|
702
|
+
config: { ...DEFAULT_CONFIG }, jobs: [], scheduledFor: null, lastRunAt: null, paused: null,
|
|
703
|
+
});
|
|
704
|
+
|
|
705
|
+
function shapeQueue(raw) {
|
|
706
|
+
const data = JSON.parse(raw);
|
|
707
|
+
return {
|
|
708
|
+
config: { ...DEFAULT_CONFIG, ...(data.config || {}) },
|
|
709
|
+
jobs: Array.isArray(data.jobs) ? data.jobs : [],
|
|
710
|
+
scheduledFor: data.scheduledFor ?? null,
|
|
711
|
+
lastRunAt: data.lastRunAt ?? null,
|
|
712
|
+
paused: data.paused ?? null,
|
|
713
|
+
};
|
|
714
|
+
}
|
|
715
|
+
|
|
716
|
+
// Quarantine a corrupt queue.json alongside itself (once per process — the
|
|
717
|
+
// first copy is the one that matters; later ticks would just overwrite it
|
|
718
|
+
// with the same bytes) so a human can diff it against the .bak-* snapshots.
|
|
719
|
+
let quarantined = false;
|
|
720
|
+
function unreadableQueue(e) {
|
|
721
|
+
const state = EMPTY_QUEUE();
|
|
722
|
+
state.unreadable = e?.message ?? 'queue.json read failed';
|
|
723
|
+
if (!quarantined) {
|
|
724
|
+
quarantined = true;
|
|
725
|
+
try {
|
|
726
|
+
fs.copyFileSync(QUEUE_PATH, `${QUEUE_PATH}.corrupt-${Date.now()}`);
|
|
727
|
+
} catch { /* best-effort: the read already failed, the copy may too */ }
|
|
728
|
+
}
|
|
729
|
+
console.error(`[scheduler] queue.json unreadable — refusing to treat as empty: ${state.unreadable}`);
|
|
730
|
+
logs.writeLine({
|
|
731
|
+
level: 'error', scope: 'scheduler',
|
|
732
|
+
message: 'queue.json unreadable — scheduling halted until it reads clean',
|
|
733
|
+
meta: { path: QUEUE_PATH, error: state.unreadable },
|
|
734
|
+
});
|
|
735
|
+
return state;
|
|
736
|
+
}
|
|
737
|
+
|
|
628
738
|
// Sync queue read — passed to the supervisor module (which calls it from
|
|
629
739
|
// supervisorTick / applyAction with no await) and the heartbeat interval.
|
|
630
740
|
// IPC handlers and mutate() use readQueue (async) below.
|
|
631
741
|
function readQueueSync() {
|
|
632
742
|
try {
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
return
|
|
636
|
-
|
|
637
|
-
jobs: Array.isArray(data.jobs) ? data.jobs : [],
|
|
638
|
-
scheduledFor: data.scheduledFor ?? null,
|
|
639
|
-
lastRunAt: data.lastRunAt ?? null,
|
|
640
|
-
paused: data.paused ?? null,
|
|
641
|
-
};
|
|
642
|
-
} catch {
|
|
643
|
-
return { config: { ...DEFAULT_CONFIG }, jobs: [], scheduledFor: null, lastRunAt: null, paused: null };
|
|
743
|
+
return shapeQueue(fs.readFileSync(QUEUE_PATH, 'utf8'));
|
|
744
|
+
} catch (e) {
|
|
745
|
+
if (e?.code === 'ENOENT') return EMPTY_QUEUE();
|
|
746
|
+
return unreadableQueue(e);
|
|
644
747
|
}
|
|
645
748
|
}
|
|
646
749
|
|
|
@@ -649,21 +752,18 @@ function readQueueSync() {
|
|
|
649
752
|
// hands control back to the renderer while the kernel paginates the file.
|
|
650
753
|
async function readQueue() {
|
|
651
754
|
try {
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
return
|
|
655
|
-
|
|
656
|
-
jobs: Array.isArray(data.jobs) ? data.jobs : [],
|
|
657
|
-
scheduledFor: data.scheduledFor ?? null,
|
|
658
|
-
lastRunAt: data.lastRunAt ?? null,
|
|
659
|
-
paused: data.paused ?? null,
|
|
660
|
-
};
|
|
661
|
-
} catch {
|
|
662
|
-
return { config: { ...DEFAULT_CONFIG }, jobs: [], scheduledFor: null, lastRunAt: null, paused: null };
|
|
755
|
+
return shapeQueue(await fsp.readFile(QUEUE_PATH, 'utf8'));
|
|
756
|
+
} catch (e) {
|
|
757
|
+
if (e?.code === 'ENOENT') return EMPTY_QUEUE();
|
|
758
|
+
return unreadableQueue(e);
|
|
663
759
|
}
|
|
664
760
|
}
|
|
665
761
|
|
|
666
762
|
async function writeQueue(state) {
|
|
763
|
+
// Last line of defence: never persist a state derived from a failed read.
|
|
764
|
+
if (state && state.unreadable) {
|
|
765
|
+
throw new Error(`refusing to write queue.json from an unreadable read (${state.unreadable})`);
|
|
766
|
+
}
|
|
667
767
|
ensureDirs();
|
|
668
768
|
await config.writeJson(QUEUE_PATH, state);
|
|
669
769
|
}
|
|
@@ -679,6 +779,12 @@ let mutateTail = Promise.resolve();
|
|
|
679
779
|
function mutate(fn) {
|
|
680
780
|
const next = mutateTail.then(async () => {
|
|
681
781
|
const state = await readQueue();
|
|
782
|
+
// Bail BEFORE fn runs: a mutator handed an unreadable (therefore empty)
|
|
783
|
+
// state would compute its result from a queue that isn't there, and
|
|
784
|
+
// writeQueue would then persist that fiction over the real file.
|
|
785
|
+
if (state.unreadable) {
|
|
786
|
+
throw new Error(`queue mutation skipped: queue.json unreadable (${state.unreadable})`);
|
|
787
|
+
}
|
|
682
788
|
const ret = await fn(state);
|
|
683
789
|
await writeQueue(state);
|
|
684
790
|
return ret;
|
|
@@ -811,6 +917,12 @@ function validatePromptForSpawn(body, srcLabel) {
|
|
|
811
917
|
* Newly-discovered PRDs land as `pending`.
|
|
812
918
|
*/
|
|
813
919
|
async function reconcile(state) {
|
|
920
|
+
// Defence in depth — tickQueue already gates on this, but reconcile is the
|
|
921
|
+
// function that would do the damage (every unmatched PRD .md becomes a
|
|
922
|
+
// fresh 'pending' row), so it refuses the poison state itself.
|
|
923
|
+
if (state && state.unreadable) {
|
|
924
|
+
throw new Error(`reconcile skipped: queue.json unreadable (${state.unreadable})`);
|
|
925
|
+
}
|
|
814
926
|
const files = await listPrdFiles();
|
|
815
927
|
const onDisk = new Map();
|
|
816
928
|
for (const f of files) {
|
|
@@ -828,7 +940,22 @@ async function reconcile(state) {
|
|
|
828
940
|
const seen = new Set();
|
|
829
941
|
for (const job of state.jobs) {
|
|
830
942
|
const p = onDisk.get(job.slug);
|
|
831
|
-
if (!p)
|
|
943
|
+
if (!p) {
|
|
944
|
+
// A terminal job whose .md is gone was archived on purpose — dropping
|
|
945
|
+
// its row is the intended end of the auto-archive flow.
|
|
946
|
+
//
|
|
947
|
+
// A PENDING or RUNNING job whose .md is merely not VISIBLE is a
|
|
948
|
+
// different thing entirely, and dropping it destroys queued work: the
|
|
949
|
+
// file may be unreadable, on a project whose dir failed to enumerate,
|
|
950
|
+
// or mid-move. "I can't see it" is not "the user deleted it", so the
|
|
951
|
+
// row survives — worst case it re-resolves on the next pass.
|
|
952
|
+
if (job.status === 'pending' || job.status === 'running') {
|
|
953
|
+
seen.add(job.slug);
|
|
954
|
+
next.push({ ...job });
|
|
955
|
+
console.warn(`[scheduler] reconcile: keeping ${job.status} job ${job.slug} — PRD source not visible in any candidate dir`);
|
|
956
|
+
}
|
|
957
|
+
continue;
|
|
958
|
+
}
|
|
832
959
|
seen.add(job.slug);
|
|
833
960
|
next.push({
|
|
834
961
|
...job,
|
|
@@ -876,6 +1003,18 @@ async function reconcile(state) {
|
|
|
876
1003
|
// review and failed PRD files are NEVER auto-archived."
|
|
877
1004
|
continue;
|
|
878
1005
|
}
|
|
1006
|
+
// history.jsonl may not exist yet (nothing has crossed HISTORY_RETENTION_MS
|
|
1007
|
+
// since the feature shipped), which leaves historyBySlug empty and the
|
|
1008
|
+
// guard above inert. Fall back to reading the slug's own newest run
|
|
1009
|
+
// sidecars straight off disk — same "don't resurrect an already-terminal
|
|
1010
|
+
// slug" intent, independent of history.jsonl's existence.
|
|
1011
|
+
const fallback = latestTerminalOutcomeForSlug(slug, { runsDir: RUNS_DIR });
|
|
1012
|
+
if (fallback) {
|
|
1013
|
+
if (fallback.status === 'completed') {
|
|
1014
|
+
historyArchiveCandidates.push({ slug, status: fallback.status, finishedAt: fallback.finishedAt });
|
|
1015
|
+
}
|
|
1016
|
+
continue;
|
|
1017
|
+
}
|
|
879
1018
|
const entry = {
|
|
880
1019
|
slug,
|
|
881
1020
|
title: p.title,
|
|
@@ -1223,8 +1362,21 @@ async function clearPause(source) {
|
|
|
1223
1362
|
if (wasPaused) await broadcast({ flush: true });
|
|
1224
1363
|
}
|
|
1225
1364
|
|
|
1226
|
-
/**
|
|
1227
|
-
|
|
1365
|
+
/**
|
|
1366
|
+
* Mutate a job in place to "pending" with cleared run metadata.
|
|
1367
|
+
*
|
|
1368
|
+
* Refuses (no-ops, returns false) on a job already in a terminal success
|
|
1369
|
+
* state ('completed') unless opts.force is true — resetting a completed job
|
|
1370
|
+
* re-fires the PRD and re-executes already-shipped work (the false-failure
|
|
1371
|
+
* class PRD 812-workbench-review-nits-cleanup demonstrated: a completed job
|
|
1372
|
+
* was reset to pending and re-ran a correct no-op that then got flagged
|
|
1373
|
+
* needs_review). All internal call sites operate on jobs that are still
|
|
1374
|
+
* 'running'/'failed' at the point they call this, so the guard is a no-op
|
|
1375
|
+
* for them; only an external reset request (IPC/admin API) can target an
|
|
1376
|
+
* already-'completed' job, and that path is exactly what this guards.
|
|
1377
|
+
*/
|
|
1378
|
+
function resetJobFields(job, errorMsg, opts = {}) {
|
|
1379
|
+
if (job.status === 'completed' && opts.force !== true) return false;
|
|
1228
1380
|
job.status = 'pending';
|
|
1229
1381
|
job.runId = null;
|
|
1230
1382
|
job.startedAt = null;
|
|
@@ -1233,6 +1385,10 @@ function resetJobFields(job, errorMsg) {
|
|
|
1233
1385
|
job.error = errorMsg ?? null;
|
|
1234
1386
|
delete job.runtime;
|
|
1235
1387
|
delete job.verifierVerdict;
|
|
1388
|
+
// Deliberately NOT deleting job.landedCommit: it must outlive a reset so a
|
|
1389
|
+
// re-fired run of this same slug can pass it to verifyRun as
|
|
1390
|
+
// priorLandedCommit (pass_no_commit_prior_run_verified exemption).
|
|
1391
|
+
return true;
|
|
1236
1392
|
}
|
|
1237
1393
|
|
|
1238
1394
|
// Grace period between a boot orphan's SIGTERM and reading its log to
|
|
@@ -1325,24 +1481,41 @@ function isNotifiableTerminalStatus(effectiveStatus) {
|
|
|
1325
1481
|
* notifyOriginatingTab(job) → void
|
|
1326
1482
|
*
|
|
1327
1483
|
* On a true terminal transition (completed/failed — never the benign
|
|
1328
|
-
* rateLimited auto-pause, which resets the job to pending instead),
|
|
1329
|
-
* short status
|
|
1330
|
-
*
|
|
1331
|
-
*
|
|
1332
|
-
*
|
|
1333
|
-
*
|
|
1334
|
-
*
|
|
1335
|
-
*
|
|
1336
|
-
*
|
|
1484
|
+
* rateLimited auto-pause, which resets the job to pending instead), publish
|
|
1485
|
+
* a short status notification for the PRD that queued this job.
|
|
1486
|
+
*
|
|
1487
|
+
* Resolution order (PRD 814): (1) if the PRD's `sourcePromptId` resolves to
|
|
1488
|
+
* a known, still-active PromptSession (minted for a dev-work dispatch, PRD
|
|
1489
|
+
* 813) under the job's cwd, append a 'response' PromptSessionEvent to THAT
|
|
1490
|
+
* session's own event chain — its own scoped PromptSessionConversation, not
|
|
1491
|
+
* whatever tab happens to be active — and stop; (2) otherwise, fall back to
|
|
1492
|
+
* today's behavior: push a short status prompt into the chat tab that queued
|
|
1493
|
+
* this PRD via enqueueExternalPrompt (PRD 753), resolved via the PRD's own
|
|
1494
|
+
* `sourceTabId` frontmatter, then the first open tab (per sessionsStore's
|
|
1495
|
+
* persisted tabs.json) whose cwd matches the job's cwd — first match only,
|
|
1496
|
+
* no fan-out to multiple matching tabs; (3) no-op. Never throws to the
|
|
1497
|
+
* caller (fire-and-forget from spawnJob). Deps are injectable (mirrors
|
|
1498
|
+
* partitionBootOrphans's isAlive param) so unit tests can exercise the
|
|
1499
|
+
* resolution logic without touching disk/electron.
|
|
1337
1500
|
*/
|
|
1338
1501
|
async function notifyOriginatingTab(job, {
|
|
1339
1502
|
parsePrdRaw = prdParser.parsePrdRaw,
|
|
1340
1503
|
loadSessions = sessionsStore.load,
|
|
1341
1504
|
sendPrompt = enqueueExternalPrompt,
|
|
1505
|
+
appendResponseEvent = appendResponseEventIfKnown,
|
|
1342
1506
|
} = {}) {
|
|
1343
1507
|
try {
|
|
1344
1508
|
const prdPath = prdPathForJob(job);
|
|
1345
1509
|
const prd = await parsePrdRaw(prdPath).catch(() => null);
|
|
1510
|
+
const message = `PRD ${job.slug} finished: ${job.status}. Check Scheduler for details.`;
|
|
1511
|
+
|
|
1512
|
+
if (prd?.sourcePromptId) {
|
|
1513
|
+
const routed = await appendResponseEvent(job.cwd || null, prd.sourcePromptId, message).catch((e) => {
|
|
1514
|
+
console.error('[scheduler] notifyOriginatingTab appendResponseEvent error', job?.slug, e);
|
|
1515
|
+
return false;
|
|
1516
|
+
});
|
|
1517
|
+
if (routed) return;
|
|
1518
|
+
}
|
|
1346
1519
|
|
|
1347
1520
|
let targetTabId = prd?.sourceTabId || null;
|
|
1348
1521
|
if (!targetTabId) {
|
|
@@ -1358,7 +1531,7 @@ async function notifyOriginatingTab(job, {
|
|
|
1358
1531
|
return;
|
|
1359
1532
|
}
|
|
1360
1533
|
|
|
1361
|
-
sendPrompt(targetTabId,
|
|
1534
|
+
sendPrompt(targetTabId, message);
|
|
1362
1535
|
} catch (e) {
|
|
1363
1536
|
console.error('[scheduler] notifyOriginatingTab error', job?.slug, e);
|
|
1364
1537
|
}
|
|
@@ -1442,6 +1615,36 @@ function classifyFailureOutcome({ exitCode, networkError, durationMs, transientR
|
|
|
1442
1615
|
return { action: 'retry', transientKind, retries };
|
|
1443
1616
|
}
|
|
1444
1617
|
|
|
1618
|
+
/**
|
|
1619
|
+
* Commit-guard verdict decision. Pure/no I/O so the false-positive defenses
|
|
1620
|
+
* can be unit-tested directly rather than only through a live spawnJob run.
|
|
1621
|
+
* Returns the flagged verifyResult replacement, or null if the guard should
|
|
1622
|
+
* not fire (any of the four defenses applies).
|
|
1623
|
+
*
|
|
1624
|
+
* The fourth defense (legitimateNoOp) exists because runVerify.cjs's own
|
|
1625
|
+
* pass_no_commit exemptions (COMPLETED_EQUIVALENT_VERDICTS members like
|
|
1626
|
+
* pass_no_commit_already_shipped) already independently proved a truthful
|
|
1627
|
+
* PASS-with-no-commit is correct; without this check the commit-guard
|
|
1628
|
+
* double-punishes that same honest no-op for dirt a concurrent interactive
|
|
1629
|
+
* session left behind (incidents: 655-needs-review-rca-feedback-hook,
|
|
1630
|
+
* 672-fix-feedback-session-manager, 2026-07-31).
|
|
1631
|
+
*/
|
|
1632
|
+
function commitGuardVerdict({ newlyDirty, siblingRunning, jobSelfCommitted, legitimateNoOp, verifyResult }) {
|
|
1633
|
+
if (!newlyDirty || newlyDirty.length === 0) return null;
|
|
1634
|
+
if (siblingRunning || jobSelfCommitted || legitimateNoOp) return null;
|
|
1635
|
+
const sample = newlyDirty.slice(0, 3).join(', ');
|
|
1636
|
+
const carried = [...(verifyResult?.annotations ?? [])];
|
|
1637
|
+
if (verifyResult && verifyResult.verdict !== 'clean') {
|
|
1638
|
+
carried.push({ verdict: verifyResult.verdict, reason: verifyResult.reason });
|
|
1639
|
+
}
|
|
1640
|
+
return {
|
|
1641
|
+
verdict: 'uncommitted_changes',
|
|
1642
|
+
reason: `finish protocol incomplete: ${newlyDirty.length} uncommitted file(s) left in working tree (e.g. ${sample})`,
|
|
1643
|
+
downgradeTo: 'needs_review',
|
|
1644
|
+
annotations: carried.length ? carried : undefined,
|
|
1645
|
+
};
|
|
1646
|
+
}
|
|
1647
|
+
|
|
1445
1648
|
// ---------- execution ----------
|
|
1446
1649
|
|
|
1447
1650
|
function pickRunDir() {
|
|
@@ -1490,16 +1693,35 @@ async function executeJob(job, runDir, defaultCwd, onPid) {
|
|
|
1490
1693
|
|
|
1491
1694
|
// Read full PRD body fresh from disk (queue stored only the preview).
|
|
1492
1695
|
let prompt;
|
|
1493
|
-
|
|
1696
|
+
let prdPath = prdPathForJob(job);
|
|
1494
1697
|
try {
|
|
1495
1698
|
const parsed = await parsePrd(prdPath);
|
|
1496
1699
|
// Centrally enforce the review → security-review → verify → commit finish
|
|
1497
1700
|
// sequence on every job, regardless of what the PRD body says.
|
|
1498
1701
|
prompt = parsed.body + FINISH_PROTOCOL;
|
|
1499
1702
|
} catch (e) {
|
|
1500
|
-
|
|
1501
|
-
|
|
1502
|
-
|
|
1703
|
+
// The project-scoped dir isn't the only place a PRD source can live — a
|
|
1704
|
+
// writer that hasn't migrated to prdLocations.cjs yet (or a not-yet-run
|
|
1705
|
+
// boot migration) can leave it in the legacy global dir. Fall back to
|
|
1706
|
+
// findPrdDir's full candidate search before failing the job outright.
|
|
1707
|
+
const fallbackDir = await findPrdDir(job.slug);
|
|
1708
|
+
if (fallbackDir) {
|
|
1709
|
+
const fallbackPath = path.join(fallbackDir, `${job.slug}.md`);
|
|
1710
|
+
safeLog(`[scheduler] PRD not in project dir; found ${job.slug}.md in ${fallbackDir}\n`);
|
|
1711
|
+
try {
|
|
1712
|
+
const parsed = await parsePrd(fallbackPath);
|
|
1713
|
+
prompt = parsed.body + FINISH_PROTOCOL;
|
|
1714
|
+
prdPath = fallbackPath;
|
|
1715
|
+
} catch (e2) {
|
|
1716
|
+
safeLog(`[scheduler] failed to read PRD: ${e2?.message}\n`);
|
|
1717
|
+
closeFd();
|
|
1718
|
+
return { exitCode: -1, durationMs: 0, error: e2?.message };
|
|
1719
|
+
}
|
|
1720
|
+
} else {
|
|
1721
|
+
safeLog(`[scheduler] failed to read PRD: ${e?.message}\n`);
|
|
1722
|
+
closeFd();
|
|
1723
|
+
return { exitCode: -1, durationMs: 0, error: e?.message };
|
|
1724
|
+
}
|
|
1503
1725
|
}
|
|
1504
1726
|
|
|
1505
1727
|
const promptCheck = validatePromptForSpawn(prompt, prdPath);
|
|
@@ -1655,7 +1877,7 @@ async function executeJob(job, runDir, defaultCwd, onPid) {
|
|
|
1655
1877
|
sl(`\n[scheduler] ${errMsg}\n`);
|
|
1656
1878
|
// Sync write: inside a Promise executor callback; must flush meta
|
|
1657
1879
|
// before resolve() so the spawnJob mutate() that follows sees it.
|
|
1658
|
-
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs });
|
|
1880
|
+
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA });
|
|
1659
1881
|
resolve({ exitCode: -1, durationMs, error: errMsg, sessionId });
|
|
1660
1882
|
return;
|
|
1661
1883
|
}
|
|
@@ -1685,6 +1907,7 @@ async function executeJob(job, runDir, defaultCwd, onPid) {
|
|
|
1685
1907
|
slug: job.slug, cwd, sessionId, exitCode: effectiveCode, rateLimited, networkError,
|
|
1686
1908
|
startedAt, finishedAt: Date.now(), durationMs,
|
|
1687
1909
|
agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
|
|
1910
|
+
schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
|
|
1688
1911
|
});
|
|
1689
1912
|
resolve({ exitCode: effectiveCode, durationMs, rateLimited, networkError, sessionId });
|
|
1690
1913
|
},
|
|
@@ -1813,6 +2036,36 @@ ${logTail}
|
|
|
1813
2036
|
DO NOT attempt the fix. ONLY write the file. When the file exists, exit immediately.`;
|
|
1814
2037
|
}
|
|
1815
2038
|
|
|
2039
|
+
/**
|
|
2040
|
+
* Pure predicate: a run whose meta.json shows a clean exit (exitCode 0) and
|
|
2041
|
+
* whose verdicts.json verdict is completed-equivalent (clean / the
|
|
2042
|
+
* pass_no_commit exemptions) has nothing left to diagnose — spawning an
|
|
2043
|
+
* Opus investigation for it just manufactures a depth+1 fix-of-a-fix PRD for
|
|
2044
|
+
* already-shipped work. Any missing/malformed input is treated as "don't
|
|
2045
|
+
* skip" (fail-open: never let a missing artifact suppress a real
|
|
2046
|
+
* investigation). Exported for tests.
|
|
2047
|
+
*/
|
|
2048
|
+
function shouldSkipInvestigationForCleanRun({ meta, verdicts }) {
|
|
2049
|
+
if (!meta || meta.exitCode !== 0) return false;
|
|
2050
|
+
if (!verdicts || !COMPLETED_EQUIVALENT_VERDICTS.has(verdicts.verdict)) return false;
|
|
2051
|
+
return true;
|
|
2052
|
+
}
|
|
2053
|
+
|
|
2054
|
+
/**
|
|
2055
|
+
* Reads <runDir>/<slug>.meta.json + <slug>.verdicts.json off disk for the
|
|
2056
|
+
* shouldSkipInvestigationForCleanRun guard. Fails safe to {} on any read/parse
|
|
2057
|
+
* error (never suppresses an investigation on a missing artifact).
|
|
2058
|
+
*/
|
|
2059
|
+
function readRunOutcomeSidecars(runDir, slug) {
|
|
2060
|
+
const readJson = (p) => {
|
|
2061
|
+
try { return JSON.parse(fs.readFileSync(p, 'utf8')); } catch { return null; }
|
|
2062
|
+
};
|
|
2063
|
+
return {
|
|
2064
|
+
meta: readJson(path.join(runDir, `${slug}.meta.json`)),
|
|
2065
|
+
verdicts: readJson(path.join(runDir, `${slug}.verdicts.json`)),
|
|
2066
|
+
};
|
|
2067
|
+
}
|
|
2068
|
+
|
|
1816
2069
|
/**
|
|
1817
2070
|
* Spawn an Opus investigation session for a failed job. The investigator's job
|
|
1818
2071
|
* is to read the failure log + original PRD, identify the root cause, and write
|
|
@@ -1821,13 +2074,22 @@ DO NOT attempt the fix. ONLY write the file. When the file exists, exit immediat
|
|
|
1821
2074
|
* run out-of-band, so they don't consume the concurrency cap. They DO consume
|
|
1822
2075
|
* tokens, which the when-available throttle will reflect on the next poll.
|
|
1823
2076
|
*
|
|
1824
|
-
* Skipped if the failed job is itself a fix-plan (avoids infinite recursion)
|
|
2077
|
+
* Skipped if the failed job is itself a fix-plan (avoids infinite recursion),
|
|
2078
|
+
* or if the run being investigated actually verified clean (nothing to fix —
|
|
2079
|
+
* see shouldSkipInvestigationForCleanRun).
|
|
1825
2080
|
*/
|
|
1826
2081
|
async function spawnInvestigation(failedJob, runDir) {
|
|
1827
2082
|
if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth)) {
|
|
1828
2083
|
console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
|
|
1829
2084
|
return { deferred: false };
|
|
1830
2085
|
}
|
|
2086
|
+
{
|
|
2087
|
+
const { meta, verdicts } = readRunOutcomeSidecars(runDir, failedJob.slug);
|
|
2088
|
+
if (shouldSkipInvestigationForCleanRun({ meta, verdicts })) {
|
|
2089
|
+
console.log(`[scheduler] skip investigation: ${failedJob.slug} last run verified ${verdicts.verdict} (exit 0) — nothing to diagnose`);
|
|
2090
|
+
return { deferred: false };
|
|
2091
|
+
}
|
|
2092
|
+
}
|
|
1831
2093
|
if (investigationsInFlight >= MAX_CONCURRENT_INVESTIGATIONS) {
|
|
1832
2094
|
// Queue for retry when a slot frees rather than dropping — otherwise a failed
|
|
1833
2095
|
// job (never 'needs_review', so reverifyNeedsReview won't retry it) would
|
|
@@ -2030,6 +2292,11 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2030
2292
|
// in its tool output (see incidents: PRD 39, 44, 56 on 2026-05-23→24).
|
|
2031
2293
|
// Called outside mutate() so the queue lock is not held during I/O.
|
|
2032
2294
|
let verifyResult = null;
|
|
2295
|
+
// Persisted onto the job row (see the mutate() block below) whenever
|
|
2296
|
+
// this run's own HEAD advances, so a LATER re-fire of the same slug can
|
|
2297
|
+
// pass it back into verifyRun as priorLandedCommit (see the
|
|
2298
|
+
// pass_no_commit_prior_run_verified exemption in runVerify.cjs).
|
|
2299
|
+
let jobLandedCommitThisRun = null;
|
|
2033
2300
|
if (res.exitCode === 0 && !res.rateLimited) {
|
|
2034
2301
|
// Detect whether the job self-committed by comparing HEAD before/after.
|
|
2035
2302
|
// Used by the sentinel override: SCHEDULER_VERDICT: PASS + a landed
|
|
@@ -2042,15 +2309,30 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2042
2309
|
job.startedAt,
|
|
2043
2310
|
new Date().toISOString(),
|
|
2044
2311
|
);
|
|
2312
|
+
if (guardHeadBefore && headAtExit && headAtExit !== guardHeadBefore) {
|
|
2313
|
+
jobLandedCommitThisRun = headAtExit;
|
|
2314
|
+
}
|
|
2045
2315
|
|
|
2046
2316
|
const prdPath = prdPathForJob(job);
|
|
2047
2317
|
const stateForDeps = await readQueue();
|
|
2318
|
+
// priorLandedCommit: the commit a PREVIOUS run of this same slug landed,
|
|
2319
|
+
// if any — prefer the live jobs[] row (survives a resetJob, see
|
|
2320
|
+
// resetJobFields), fall back to history.jsonl for a slug that already
|
|
2321
|
+
// left jobs[]. Never the commit THIS run just made (committedDuringRun
|
|
2322
|
+
// already covers that case).
|
|
2323
|
+
const liveRow = stateForDeps.jobs.find((j) => j.slug === job.slug);
|
|
2324
|
+
let priorLandedCommit = liveRow?.landedCommit ?? null;
|
|
2325
|
+
if (!priorLandedCommit) {
|
|
2326
|
+
const hist = await queueHistory.historyTerminalBySlug().catch(() => null);
|
|
2327
|
+
priorLandedCommit = hist?.get(job.slug)?.landedCommit ?? null;
|
|
2328
|
+
}
|
|
2048
2329
|
verifyResult = await verifyRun({
|
|
2049
2330
|
runDir,
|
|
2050
2331
|
prdPath,
|
|
2051
2332
|
queueEntry: job,
|
|
2052
2333
|
allJobs: stateForDeps.jobs,
|
|
2053
2334
|
committedDuringRun,
|
|
2335
|
+
priorLandedCommit,
|
|
2054
2336
|
}).catch((e) => ({
|
|
2055
2337
|
verdict: 'verify_unavailable',
|
|
2056
2338
|
reason: `verifier threw: ${e?.message ?? String(e)}`,
|
|
@@ -2076,6 +2358,17 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2076
2358
|
// deliverable; leftover dirt is presumptively a concurrent external edit
|
|
2077
2359
|
// (e.g. an interactive session editing the same repo), not the job's
|
|
2078
2360
|
// unsaved work — so skip rather than false-flag a completed job.
|
|
2361
|
+
// - legitimate-no-op skip: if the verifier already independently proved
|
|
2362
|
+
// this run's PASS-with-no-commit is truthful (verdict is one of
|
|
2363
|
+
// COMPLETED_EQUIVALENT_VERDICTS — pass_no_commit_target_verified,
|
|
2364
|
+
// _prior_run_verified, _already_shipped), the run itself did nothing
|
|
2365
|
+
// wrong; dirt left by a concurrent interactive session (e.g.
|
|
2366
|
+
// /process-feedback writing new PRD .md files into the same repo
|
|
2367
|
+
// while this job's own AC turned out to already be satisfied) is not
|
|
2368
|
+
// this job's unfinished work. Without this skip, runVerify.cjs's
|
|
2369
|
+
// exemption and this guard double-punish the same honest no-op from
|
|
2370
|
+
// two different code paths (incidents: 655-needs-review-rca-feedback-hook,
|
|
2371
|
+
// 672-fix-feedback-session-manager, 2026-07-31).
|
|
2079
2372
|
// Non-git cwds resolve to null and are skipped (the guard is best-effort).
|
|
2080
2373
|
//
|
|
2081
2374
|
// Runs even when a transcript-pattern verdict already fired: the commit-guard
|
|
@@ -2086,7 +2379,8 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2086
2379
|
// annotation, so a real "finish protocol incomplete" is distinguishable from
|
|
2087
2380
|
// transcript noise in the queue (feedback 2026-06-10 addendum).
|
|
2088
2381
|
const guardWillRefire = verifyResult && verifyResult.downgradeTo === 'pending';
|
|
2089
|
-
|
|
2382
|
+
const guardIsLegitimateNoOp = verifyResult && COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict);
|
|
2383
|
+
if (res.exitCode === 0 && !res.rateLimited && !guardWillRefire && !guardIsLegitimateNoOp) {
|
|
2090
2384
|
const after = await uncommittedChanges(guardCwd);
|
|
2091
2385
|
if (after && after.length > 0) {
|
|
2092
2386
|
const baseSet = new Set(guardBaseline || []);
|
|
@@ -2097,19 +2391,15 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2097
2391
|
);
|
|
2098
2392
|
const guardHeadAfter = await gitHead(guardCwd);
|
|
2099
2393
|
const jobSelfCommitted = guardHeadBefore && guardHeadAfter && guardHeadAfter !== guardHeadBefore;
|
|
2100
|
-
|
|
2101
|
-
|
|
2102
|
-
|
|
2103
|
-
|
|
2104
|
-
|
|
2105
|
-
|
|
2106
|
-
|
|
2107
|
-
|
|
2108
|
-
|
|
2109
|
-
reason: `finish protocol incomplete: ${newlyDirty.length} uncommitted file(s) left in working tree (e.g. ${sample})`,
|
|
2110
|
-
downgradeTo: 'needs_review',
|
|
2111
|
-
annotations: carried.length ? carried : undefined,
|
|
2112
|
-
};
|
|
2394
|
+
const guardVerdict = commitGuardVerdict({
|
|
2395
|
+
newlyDirty,
|
|
2396
|
+
siblingRunning,
|
|
2397
|
+
jobSelfCommitted,
|
|
2398
|
+
legitimateNoOp: guardIsLegitimateNoOp,
|
|
2399
|
+
verifyResult,
|
|
2400
|
+
});
|
|
2401
|
+
if (guardVerdict) {
|
|
2402
|
+
verifyResult = guardVerdict;
|
|
2113
2403
|
console.log(`[scheduler] commit-guard: ${job.slug} left ${newlyDirty.length} files uncommitted → needs_review`);
|
|
2114
2404
|
}
|
|
2115
2405
|
}
|
|
@@ -2138,6 +2428,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2138
2428
|
let investigationJobSnapshot = null;
|
|
2139
2429
|
let needsReviewRcaSnapshot = null;
|
|
2140
2430
|
let terminalNotifySnapshot = null;
|
|
2431
|
+
const newlyCompletedPrds = [];
|
|
2141
2432
|
await mutate((s) => {
|
|
2142
2433
|
const i2 = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
2143
2434
|
if (i2 >= 0) {
|
|
@@ -2158,11 +2449,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2158
2449
|
effectiveStatus = 'failed';
|
|
2159
2450
|
} else if (
|
|
2160
2451
|
!verifyResult
|
|
2161
|
-
|| verifyResult.verdict
|
|
2162
|
-
// pass_no_commit_target_verified: -merge-main postcondition exemption
|
|
2163
|
-
// (runVerify.cjs) — an independently gh-confirmed clean merge target,
|
|
2164
|
-
// not a plain unsubstantiated PASS. Completed, same as 'clean'.
|
|
2165
|
-
|| verifyResult.verdict === 'pass_no_commit_target_verified'
|
|
2452
|
+
|| COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict)
|
|
2166
2453
|
) {
|
|
2167
2454
|
effectiveStatus = 'completed';
|
|
2168
2455
|
} else if (verifyResult.downgradeTo === 'pending') {
|
|
@@ -2180,6 +2467,13 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2180
2467
|
s.jobs[i2].error = effectiveStatus === 'needs_review'
|
|
2181
2468
|
? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
|
|
2182
2469
|
: (res.error || null);
|
|
2470
|
+
// Persist the commit THIS run landed (if HEAD advanced) so a later
|
|
2471
|
+
// re-fire of the same slug can prove its own no-op re-run is
|
|
2472
|
+
// truthful via the pass_no_commit_prior_run_verified exemption.
|
|
2473
|
+
// Survives resetJobFields — see that function's comment.
|
|
2474
|
+
if (jobLandedCommitThisRun) {
|
|
2475
|
+
s.jobs[i2].landedCommit = jobLandedCommitThisRun;
|
|
2476
|
+
}
|
|
2183
2477
|
// Persist the verifier's verdict string so the renderer can show it.
|
|
2184
2478
|
if (verifyResult?.verdict && verifyResult.verdict !== 'clean') {
|
|
2185
2479
|
s.jobs[i2].verifierVerdict = verifyResult.verdict;
|
|
@@ -2201,6 +2495,9 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2201
2495
|
if (isNotifiableTerminalStatus(effectiveStatus)) {
|
|
2202
2496
|
terminalNotifySnapshot = { ...s.jobs[i2] };
|
|
2203
2497
|
}
|
|
2498
|
+
if (effectiveStatus === 'completed') {
|
|
2499
|
+
newlyCompletedPrds.push({ slug: s.jobs[i2].slug, cwd: s.jobs[i2].cwd });
|
|
2500
|
+
}
|
|
2204
2501
|
if (effectiveStatus === 'failed') {
|
|
2205
2502
|
actuallyFailed = true;
|
|
2206
2503
|
failedJobSnapshot = { ...s.jobs[i2] };
|
|
@@ -2252,11 +2549,15 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2252
2549
|
if (priorStatus === 'needs_review') {
|
|
2253
2550
|
delete orig.verifierVerdict;
|
|
2254
2551
|
}
|
|
2552
|
+
newlyCompletedPrds.push({ slug: orig.slug, cwd: orig.cwd });
|
|
2255
2553
|
}
|
|
2256
2554
|
}
|
|
2257
2555
|
}
|
|
2258
2556
|
}
|
|
2259
2557
|
});
|
|
2558
|
+
for (const { slug, cwd } of newlyCompletedPrds) {
|
|
2559
|
+
await archiveCompletedPrd(slug, cwd);
|
|
2560
|
+
}
|
|
2260
2561
|
await broadcast({ flush: true });
|
|
2261
2562
|
|
|
2262
2563
|
if (terminalNotifySnapshot) {
|
|
@@ -2377,6 +2678,12 @@ let tickTail = Promise.resolve();
|
|
|
2377
2678
|
function tickQueue() {
|
|
2378
2679
|
const next = tickTail.then(async () => {
|
|
2379
2680
|
const state = await readQueue();
|
|
2681
|
+
// Never reconcile against an unreadable queue: reconcile() would see zero
|
|
2682
|
+
// job rows for every PRD on disk and resurrect the lot as 'pending'.
|
|
2683
|
+
if (state.unreadable) {
|
|
2684
|
+
console.error('[scheduler] tickQueue skipped: queue.json unreadable');
|
|
2685
|
+
return { fired: false, reason: 'unreadable' };
|
|
2686
|
+
}
|
|
2380
2687
|
if (state.paused) {
|
|
2381
2688
|
console.log('[scheduler] tickQueue skipped: paused');
|
|
2382
2689
|
return { fired: false, reason: 'paused' };
|
|
@@ -2462,6 +2769,8 @@ function forceTickOutcome(result) {
|
|
|
2462
2769
|
return { ok: true, kind: 'info', message: `Already running — ${result.runningCount} job(s) in flight` };
|
|
2463
2770
|
case 'paused':
|
|
2464
2771
|
return { ok: true, kind: 'warn', message: 'Scheduler is paused' };
|
|
2772
|
+
case 'unreadable':
|
|
2773
|
+
return { ok: false, kind: 'error', message: 'queue.json is unreadable — scheduling halted; a .corrupt-<ts> copy was saved next to it' };
|
|
2465
2774
|
case 'cancelled':
|
|
2466
2775
|
return { ok: true, kind: 'warn', message: 'Batch cancelled — try again' };
|
|
2467
2776
|
case 'memory-deferred':
|
|
@@ -2477,6 +2786,10 @@ function forceTickOutcome(result) {
|
|
|
2477
2786
|
|
|
2478
2787
|
async function runDueJobs() {
|
|
2479
2788
|
const state = await readQueue();
|
|
2789
|
+
if (state.unreadable) {
|
|
2790
|
+
console.error('[scheduler] runDueJobs skipped: queue.json unreadable');
|
|
2791
|
+
return { fired: false, reason: 'unreadable' };
|
|
2792
|
+
}
|
|
2480
2793
|
if (state.paused) {
|
|
2481
2794
|
console.log('[scheduler] runDueJobs skipped: paused');
|
|
2482
2795
|
return { fired: false, reason: 'paused' };
|
|
@@ -2730,7 +3043,7 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
|
|
|
2730
3043
|
// investigation jobs correctly found "nothing to fix" but were flagged
|
|
2731
3044
|
// anyway). For non-fix-plan jobs the exemption never applies, so rescanning
|
|
2732
3045
|
// their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
|
|
2733
|
-
const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'pass_no_commit']);
|
|
3046
|
+
const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'pass_no_commit', 'pass_no_commit_already_shipped']);
|
|
2734
3047
|
|
|
2735
3048
|
// Bounds fix-plan recursion: depth 1 = the original job, depth 2 = its fix
|
|
2736
3049
|
// (gets exactly one follow-up investigation if it also lands in
|
|
@@ -2882,6 +3195,13 @@ async function reverifyNeedsReview() {
|
|
|
2882
3195
|
// commit-guard uses gitHead() (before/after HEAD diff); here the run is
|
|
2883
3196
|
// already over so we query git log filtered to [startedAt, finishedAt+60s].
|
|
2884
3197
|
const committedDuringRun = await committedInWindow(job.cwd, job.startedAt, job.finishedAt);
|
|
3198
|
+
// priorLandedCommit: same lookup as spawnJob's post-run verify — the live
|
|
3199
|
+
// jobs[] row first (survives a resetJob), else history.jsonl.
|
|
3200
|
+
let priorLandedCommit = job.landedCommit ?? null;
|
|
3201
|
+
if (!priorLandedCommit) {
|
|
3202
|
+
const hist = await queueHistory.historyTerminalBySlug().catch(() => null);
|
|
3203
|
+
priorLandedCommit = hist?.get(job.slug)?.landedCommit ?? null;
|
|
3204
|
+
}
|
|
2885
3205
|
let v = null;
|
|
2886
3206
|
try {
|
|
2887
3207
|
v = await verifyRun({
|
|
@@ -2891,12 +3211,10 @@ async function reverifyNeedsReview() {
|
|
|
2891
3211
|
allJobs: snap.jobs,
|
|
2892
3212
|
committedDuringRun,
|
|
2893
3213
|
allowPreSentinelHeal: true,
|
|
3214
|
+
priorLandedCommit,
|
|
2894
3215
|
});
|
|
2895
3216
|
} catch { leftForReview.push({ slug: job.slug, reason: 'verifyRun threw' }); continue; }
|
|
2896
|
-
|
|
2897
|
-
// (runVerify.cjs) — same "heal it" treatment as 'clean', see spawnJob's
|
|
2898
|
-
// effectiveStatus branch above for the primary-path equivalent.
|
|
2899
|
-
if (v && (v.verdict === 'clean' || v.verdict === 'pass_no_commit_target_verified')) {
|
|
3217
|
+
if (v && COMPLETED_EQUIVALENT_VERDICTS.has(v.verdict)) {
|
|
2900
3218
|
healed.push(job.slug);
|
|
2901
3219
|
} else {
|
|
2902
3220
|
leftForReview.push({ slug: job.slug, reason: v ? `${v.verdict}: ${v.reason}` : 'null verdict' });
|
|
@@ -2904,15 +3222,20 @@ async function reverifyNeedsReview() {
|
|
|
2904
3222
|
}
|
|
2905
3223
|
if (healed.length) {
|
|
2906
3224
|
const healSet = new Set(healed);
|
|
3225
|
+
const healedPrds = [];
|
|
2907
3226
|
await mutate((s) => {
|
|
2908
3227
|
for (const j of s.jobs) {
|
|
2909
3228
|
if (j.status === 'needs_review' && healSet.has(j.slug)) {
|
|
2910
3229
|
j.status = 'completed';
|
|
2911
3230
|
j.error = null;
|
|
2912
3231
|
delete j.verifierVerdict;
|
|
3232
|
+
healedPrds.push({ slug: j.slug, cwd: j.cwd });
|
|
2913
3233
|
}
|
|
2914
3234
|
}
|
|
2915
3235
|
});
|
|
3236
|
+
for (const { slug, cwd } of healedPrds) {
|
|
3237
|
+
await archiveCompletedPrd(slug, cwd);
|
|
3238
|
+
}
|
|
2916
3239
|
console.log(`[scheduler] boot reverify: healed ${healed.length} stale needs_review → completed (${healed.join(', ')})`);
|
|
2917
3240
|
await broadcast();
|
|
2918
3241
|
}
|
|
@@ -2934,6 +3257,7 @@ async function reverifyNeedsReview() {
|
|
|
2934
3257
|
// boot/tick, because the promotion only ran inside `if (healed.length)`
|
|
2935
3258
|
// scoped to that single pass's fresh heals.)
|
|
2936
3259
|
const promoted = [];
|
|
3260
|
+
const promotedPrds = [];
|
|
2937
3261
|
await mutate((s) => {
|
|
2938
3262
|
for (const job of s.jobs) {
|
|
2939
3263
|
if (job.status !== 'completed' || !isFixPlanSlug(job.slug)) continue;
|
|
@@ -2946,8 +3270,12 @@ async function reverifyNeedsReview() {
|
|
|
2946
3270
|
orig.completedBy = job.slug;
|
|
2947
3271
|
if (priorStatus === 'needs_review') delete orig.verifierVerdict;
|
|
2948
3272
|
promoted.push(`${orig.slug} (was ${priorStatus}, via ${job.slug})`);
|
|
3273
|
+
promotedPrds.push({ slug: orig.slug, cwd: orig.cwd });
|
|
2949
3274
|
}
|
|
2950
3275
|
});
|
|
3276
|
+
for (const { slug, cwd } of promotedPrds) {
|
|
3277
|
+
await archiveCompletedPrd(slug, cwd);
|
|
3278
|
+
}
|
|
2951
3279
|
if (promoted.length) {
|
|
2952
3280
|
console.log(`[scheduler] boot reverify: auto-promoted ${promoted.length} original(s): ${promoted.join(', ')}`);
|
|
2953
3281
|
await broadcast();
|
|
@@ -3094,13 +3422,21 @@ function registerScheduleHandlers() {
|
|
|
3094
3422
|
|
|
3095
3423
|
ipcMain.handle('schedule:reset-job', validated(schemas.scheduleSlug, async ({ slug }) => {
|
|
3096
3424
|
if (!(await safeSlugPath(slug))) return { ok: false, error: 'invalid slug' };
|
|
3097
|
-
const
|
|
3425
|
+
const outcome = await mutate((state) => {
|
|
3098
3426
|
const idx = state.jobs.findIndex((j) => j.slug === slug);
|
|
3099
|
-
if (idx < 0) return
|
|
3100
|
-
resetJobFields
|
|
3101
|
-
|
|
3427
|
+
if (idx < 0) return 'not-found';
|
|
3428
|
+
// Guard is in resetJobFields: refuses to reset an already-'completed'
|
|
3429
|
+
// job, which would otherwise re-fire a PRD whose deliverable already
|
|
3430
|
+
// landed (see resetJobFields' doc comment for the incident).
|
|
3431
|
+
return resetJobFields(state.jobs[idx]) ? 'ok' : 'refused';
|
|
3102
3432
|
});
|
|
3103
|
-
if (
|
|
3433
|
+
if (outcome === 'not-found') return { ok: false, error: 'not found' };
|
|
3434
|
+
if (outcome === 'refused') {
|
|
3435
|
+
return {
|
|
3436
|
+
ok: false,
|
|
3437
|
+
error: 'job already completed — resetting it would re-execute shipped work; archive the PRD instead',
|
|
3438
|
+
};
|
|
3439
|
+
}
|
|
3104
3440
|
await broadcast({ flush: true });
|
|
3105
3441
|
return { ok: true };
|
|
3106
3442
|
}));
|
|
@@ -3315,6 +3651,7 @@ async function init() {
|
|
|
3315
3651
|
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
3316
3652
|
bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
|
|
3317
3653
|
}
|
|
3654
|
+
const bootReconciledCompletions = [];
|
|
3318
3655
|
await mutate((state) => {
|
|
3319
3656
|
for (const j of state.jobs) {
|
|
3320
3657
|
if (j.status !== 'running' || !immediateSlugs.includes(j.slug)) continue;
|
|
@@ -3322,9 +3659,13 @@ async function init() {
|
|
|
3322
3659
|
const pid = j.runtime?.pid;
|
|
3323
3660
|
const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
|
|
3324
3661
|
applyOrphanOutcome(j, outcome, killNote);
|
|
3662
|
+
if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
|
|
3325
3663
|
console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
|
|
3326
3664
|
}
|
|
3327
3665
|
});
|
|
3666
|
+
for (const { slug, cwd } of bootReconciledCompletions) {
|
|
3667
|
+
await archiveCompletedPrd(slug, cwd);
|
|
3668
|
+
}
|
|
3328
3669
|
|
|
3329
3670
|
// Still-alive orphans: SIGTERM (+ killOrphanClaudePid's own deferred SIGKILL
|
|
3330
3671
|
// follow-up) now, but classification waits until BOOT_ORPHAN_KILL_GRACE_MS
|
|
@@ -3344,6 +3685,7 @@ async function init() {
|
|
|
3344
3685
|
setTimeout(() => {
|
|
3345
3686
|
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
3346
3687
|
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
3688
|
+
let deferredCompletedCwd;
|
|
3347
3689
|
mutate((state) => {
|
|
3348
3690
|
const cur = state.jobs.find((x) => x.slug === slug);
|
|
3349
3691
|
// Race guard: bail if the job already resolved, OR if it's already been
|
|
@@ -3353,6 +3695,9 @@ async function init() {
|
|
|
3353
3695
|
if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
|
|
3354
3696
|
applyOrphanOutcome(cur, outcome, killNote);
|
|
3355
3697
|
console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
|
|
3698
|
+
deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
|
|
3699
|
+
}).then(() => {
|
|
3700
|
+
if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
|
|
3356
3701
|
}).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
|
|
3357
3702
|
}, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
|
|
3358
3703
|
}
|
|
@@ -3551,15 +3896,25 @@ const remote = {
|
|
|
3551
3896
|
}
|
|
3552
3897
|
},
|
|
3553
3898
|
|
|
3554
|
-
async resetJob(slug) {
|
|
3899
|
+
async resetJob(slug, opts = {}) {
|
|
3555
3900
|
if (!(await safeSlugPath(slug))) return { ok: false, error: 'invalid slug' };
|
|
3556
|
-
const
|
|
3901
|
+
const outcome = await mutate((state) => {
|
|
3557
3902
|
const idx = state.jobs.findIndex((j) => j.slug === slug);
|
|
3558
|
-
if (idx < 0) return
|
|
3559
|
-
resetJobFields
|
|
3560
|
-
|
|
3903
|
+
if (idx < 0) return { kind: 'not-found' };
|
|
3904
|
+
// Terminal-status guard lives in resetJobFields itself; force:true
|
|
3905
|
+
// threads through to override it.
|
|
3906
|
+
if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true })) {
|
|
3907
|
+
return { kind: 'refused' };
|
|
3908
|
+
}
|
|
3909
|
+
return { kind: 'ok' };
|
|
3561
3910
|
});
|
|
3562
|
-
if (
|
|
3911
|
+
if (outcome.kind === 'not-found') return { ok: false, error: 'not found' };
|
|
3912
|
+
if (outcome.kind === 'refused') {
|
|
3913
|
+
return {
|
|
3914
|
+
ok: false,
|
|
3915
|
+
error: 'job already completed — resetting it would re-execute shipped work; archive the PRD instead, or pass force:true',
|
|
3916
|
+
};
|
|
3917
|
+
}
|
|
3563
3918
|
await broadcast({ flush: true });
|
|
3564
3919
|
return { ok: true, slug, status: 'pending' };
|
|
3565
3920
|
},
|
|
@@ -3625,9 +3980,10 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
3625
3980
|
sendJson(res, 400, { ok: false, error: 'missing slug' });
|
|
3626
3981
|
return;
|
|
3627
3982
|
}
|
|
3628
|
-
const
|
|
3983
|
+
const force = parsed.force === true;
|
|
3984
|
+
const result = await remoteObj.resetJob(slug, { force });
|
|
3629
3985
|
sendJson(res, 200, result);
|
|
3630
3986
|
});
|
|
3631
3987
|
}
|
|
3632
3988
|
|
|
3633
|
-
module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, feedbackSweepDue, FEEDBACK_SWEEP_TICK_INTERVAL, sweepFeedback, registerAdminRoutes, notifyOriginatingTab, isNotifiableTerminalStatus, candidatePrdsDirs, prdDirForCwd, prdPathForJob, findPrdDir, runPrdMigration };
|
|
3989
|
+
module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, writeQueue, reconcile, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, feedbackSweepDue, FEEDBACK_SWEEP_TICK_INTERVAL, sweepFeedback, registerAdminRoutes, notifyOriginatingTab, isNotifiableTerminalStatus, candidatePrdsDirs, prdDirForCwd, prdPathForJob, findPrdDir, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields };
|