claude-code-session-manager 0.64.0 → 0.66.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/AgentLibrary-Bzg89D5Y.js +3 -0
- package/dist/assets/{History-Dp3-_bjQ.js → History-CNH9vA0A.js} +2 -2
- package/dist/assets/{Hooks-DMQFgGDd.js → Hooks-CyTksPza.js} +3 -3
- package/dist/assets/{HostBilko-zNU1NtBh.js → HostBilko-DXQVDNHn.js} +1 -1
- package/dist/assets/{Library-BrODBbeL.js → Library-9E2UOIsm.js} +1 -1
- package/dist/assets/{ListDetail-C5vbA1gC.js → ListDetail-CQWU_Yn5.js} +1 -1
- package/dist/assets/{MarkdownEditor-CpzAEiiG.js → MarkdownEditor-DQgfpSef.js} +1 -1
- package/dist/assets/{McpServers-DWuw-flj.js → McpServers-r3qwDIj2.js} +2 -2
- package/dist/assets/{Memory-B4TC9SsJ.js → Memory-BaxOpj-3.js} +6 -6
- package/dist/assets/{Panel-Bap7G-pI.js → Panel-BhFD8Lqo.js} +1 -1
- package/dist/assets/{Permissions-BIpNWItH.js → Permissions-Cj-mODQQ.js} +3 -3
- package/dist/assets/{Plugins-DrwklmW4.js → Plugins-DEL3Fqng.js} +2 -2
- package/dist/assets/{ProvenanceBadge-CCp_EhjV.js → ProvenanceBadge-CIAg6-JQ.js} +1 -1
- package/dist/assets/Scheduler-YOuZKkES.js +14 -0
- package/dist/assets/{ScopeSwitcher-BlQvpyLE.js → ScopeSwitcher-DZ_3gEus.js} +1 -1
- package/dist/assets/{Settings-BzvmZvGa.js → Settings-B0x4oflz.js} +3 -3
- package/dist/assets/{SkillReferenceGraph-CGtwmCjm.js → SkillReferenceGraph-CoIwsol8.js} +1 -1
- package/dist/assets/{Skills-Cauvbtk7.js → Skills-CN8R6AWn.js} +2 -2
- package/dist/assets/{SystemPrompt-BqzM76PV.js → SystemPrompt-DceChpAi.js} +1 -1
- package/dist/assets/{TagLibrary-DVKPjeRu.js → TagLibrary-BKmz2W7B.js} +1 -1
- package/dist/assets/{TiptapBody-CUWdBAYf.js → TiptapBody-W7n5SwPM.js} +1 -1
- package/dist/assets/{Toggle-Bos67yDJ.js → Toggle-QuxlVHGI.js} +1 -1
- package/dist/assets/{index-LlWpj2VJ.css → index-14dBLqE_.css} +1 -1
- package/dist/assets/{index-BfPVknuV.js → index-TejhHSzN.js} +526 -525
- package/dist/assets/{settingsSchema-DKuyaEK5.js → settingsSchema-DNqx6BKJ.js} +1 -1
- package/dist/index.html +2 -2
- package/package.json +5 -1
- package/plugins/session-manager-dev/skills/builder/4-manual/SKILL.md +120 -0
- package/plugins/session-manager-dev/skills/builder/SKILL.md +13 -3
- package/plugins/session-manager-dev/skills/develop/SKILL.md +41 -13
- package/plugins/session-manager-dev/skills/ops-sweep/SKILL.md +10 -0
- package/src/main/__tests__/agentLibrary.test.cjs +40 -0
- package/src/main/__tests__/flatPrdTickSweep.test.cjs +110 -0
- package/src/main/__tests__/prdAdminRouteParity.test.cjs +68 -0
- package/src/main/__tests__/prdAdminRoutes.test.cjs +311 -0
- package/src/main/__tests__/prdCreate.test.cjs +7 -2
- package/src/main/__tests__/prdMigration.test.cjs +17 -0
- package/src/main/__tests__/prdMigrationLegacyAdopt.test.cjs +91 -0
- package/src/main/__tests__/reconcileFlatPrdSweep.test.cjs +109 -0
- package/src/main/__tests__/scheduleJobSchema.test.cjs +127 -0
- package/src/main/__tests__/scheduleJobStatusDrift.test.cjs +65 -0
- package/src/main/__tests__/scheduleJobTransitions.test.cjs +152 -0
- package/src/main/__tests__/scheduleJobTransitionsGrep.test.cjs +59 -0
- package/src/main/__tests__/scheduler-reconcile-invalid-repair.test.cjs +203 -0
- package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +196 -0
- package/src/main/agentLibrary.cjs +40 -2
- package/src/main/index.cjs +2 -0
- package/src/main/ipcSchemas.cjs +60 -0
- package/src/main/lib/localAdminHttp.cjs +10 -3
- package/src/main/lib/prdAdminRoutes.cjs +175 -0
- package/src/main/lib/prdCreate.cjs +37 -2
- package/src/main/lib/prdFrontmatter.cjs +179 -1
- package/src/main/lib/prdMigration.cjs +82 -5
- package/src/main/lib/queueStore.cjs +41 -7
- package/src/main/lib/scheduleJobSchema.cjs +114 -0
- package/src/main/lib/scheduleJobTransitions.cjs +164 -0
- package/src/main/scheduler/prdParser.cjs +7 -0
- package/src/main/scheduler.cjs +649 -137
- package/src/preload/api.d.ts +54 -2
- package/src/preload/index.cjs +12 -0
- package/dist/assets/AgentLibrary-9KJ4fedH.js +0 -1
- package/dist/assets/Scheduler-DfoTegGG.js +0 -14
- /package/plugins/session-manager-dev/skills/builder/{4-report → 5-report}/SKILL.md +0 -0
package/src/main/scheduler.cjs
CHANGED
|
@@ -88,7 +88,10 @@ const queueOps = require('./queueOps.cjs');
|
|
|
88
88
|
// home-dir layout.
|
|
89
89
|
const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
|
|
90
90
|
const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
|
|
91
|
+
const { transitionJob } = require('./lib/scheduleJobTransitions.cjs');
|
|
91
92
|
const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
|
|
93
|
+
const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
|
|
94
|
+
const { appendAuditEvent } = require('./lib/auditLog.cjs');
|
|
92
95
|
|
|
93
96
|
// ---------- origin session resolution (PRD 832) ----------
|
|
94
97
|
// An Epic IS a tagged claude session — job rows carry the originating
|
|
@@ -110,8 +113,8 @@ function resolveOriginSessionId(cwd, epicId) {
|
|
|
110
113
|
const sessionSlots = require('./lib/sessionSlots.cjs');
|
|
111
114
|
const jobWorktree = require('./lib/jobWorktree.cjs');
|
|
112
115
|
const queueStore = require('./lib/queueStore.cjs');
|
|
113
|
-
const { splitFrontmatter } = require('./lib/prdFrontmatter.cjs');
|
|
114
|
-
const { migratePrds, consolidateFlatPrds } = require('./lib/prdMigration.cjs');
|
|
116
|
+
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
117
|
+
const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
|
|
115
118
|
const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
|
|
116
119
|
|
|
117
120
|
// Captured once at module load so every run's meta sidecar can record how
|
|
@@ -720,7 +723,7 @@ async function retireCompletedSlugs(slugs) {
|
|
|
720
723
|
for (const j of s.jobs) {
|
|
721
724
|
if (!j || !slugSet.has(j.slug)) continue;
|
|
722
725
|
if (j.status !== 'pending' && j.status !== 'running') continue;
|
|
723
|
-
j
|
|
726
|
+
if (!transitionJob(j, 'completed', { reason: 'manual archive of an already-shipped PRD', source: 'retireCompletedSlugs' })) continue;
|
|
724
727
|
j.finishedAt = new Date().toISOString();
|
|
725
728
|
j.exitCode = 0;
|
|
726
729
|
j.error = null;
|
|
@@ -756,6 +759,50 @@ function ensureDirs() {
|
|
|
756
759
|
* unparseable cwd, cwd not on disk) are left in place and logged as a
|
|
757
760
|
* warning — never silently dropped — so a human can fix the frontmatter.
|
|
758
761
|
*/
|
|
762
|
+
/**
|
|
763
|
+
* consolidateAllFlatPrds(cwds) — run consolidateFlatPrds() over every given
|
|
764
|
+
* project cwd, logging outcomes. Called from TWO places: once at boot (over
|
|
765
|
+
* every historical project, via runPrdMigration below) AND at the top of
|
|
766
|
+
* every reconcile() call (over every project reconcile itself would
|
|
767
|
+
* otherwise scan), BEFORE reconcile scans the flat dir for PRD sources. The
|
|
768
|
+
* reconcile()-level call is what makes "anything written to the retired flat
|
|
769
|
+
* prds/ dir is swept into prds-archived/ without being executed" actually
|
|
770
|
+
* true regardless of which of reconcile's several callers (tickQueue's poll,
|
|
771
|
+
* job completion, the schedule:state/schedule:rescan IPC handlers,
|
|
772
|
+
* rescheduleTimer) triggers the pass: a PRD dropped in the flat dir has no
|
|
773
|
+
* queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
|
|
774
|
+
* sweep archives it before reconcile can ever turn it into a pending job.
|
|
775
|
+
*/
|
|
776
|
+
async function consolidateAllFlatPrds(cwds) {
|
|
777
|
+
for (const cwd of cwds) {
|
|
778
|
+
try {
|
|
779
|
+
const c = await consolidateFlatPrds(cwd);
|
|
780
|
+
if (c.moved > 0) {
|
|
781
|
+
console.log(`[scheduler] flat-PRD consolidation: archived ${c.moved} file(s) in ${cwd}`);
|
|
782
|
+
}
|
|
783
|
+
for (const f of c.failed) {
|
|
784
|
+
logs.writeLine({
|
|
785
|
+
level: 'warn', scope: 'scheduler',
|
|
786
|
+
message: `flat-PRD consolidation: could not archive ${f.file}`,
|
|
787
|
+
meta: { cwd, reason: f.reason },
|
|
788
|
+
});
|
|
789
|
+
}
|
|
790
|
+
// Deliberately left behind because a live job still points at them
|
|
791
|
+
// (PRD 992). Logged so a permanently-stuck flat PRD is visible rather
|
|
792
|
+
// than looking like a clean consolidation.
|
|
793
|
+
for (const s of c.skipped ?? []) {
|
|
794
|
+
logs.writeLine({
|
|
795
|
+
level: 'info', scope: 'scheduler',
|
|
796
|
+
message: `flat-PRD consolidation: left ${s.file} in place`,
|
|
797
|
+
meta: { cwd, reason: s.reason },
|
|
798
|
+
});
|
|
799
|
+
}
|
|
800
|
+
} catch (e) {
|
|
801
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'flat-PRD consolidation failed', meta: { cwd, error: e?.message } });
|
|
802
|
+
}
|
|
803
|
+
}
|
|
804
|
+
}
|
|
805
|
+
|
|
759
806
|
async function runPrdMigration() {
|
|
760
807
|
let result;
|
|
761
808
|
try {
|
|
@@ -782,33 +829,30 @@ async function runPrdMigration() {
|
|
|
782
829
|
// still sitting flat consolidates into `prds-archived/` for later special
|
|
783
830
|
// processing. Queue rows for moved files are reaped by the archived-twin
|
|
784
831
|
// retirement. Idempotent per project; failures are logged, never fatal.
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
message: `flat-PRD consolidation: left ${s.file} in place`,
|
|
805
|
-
meta: { cwd, reason: s.reason },
|
|
806
|
-
});
|
|
807
|
-
}
|
|
808
|
-
} catch (e) {
|
|
809
|
-
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'flat-PRD consolidation failed', meta: { cwd, error: e?.message } });
|
|
832
|
+
// (This boot-time pass is redundant with the one reconcile() now also runs
|
|
833
|
+
// on every pass, but stays here so a fresh boot's very first log line
|
|
834
|
+
// still reports the initial sweep — see consolidateAllFlatPrds's own
|
|
835
|
+
// comment for why reconcile() is the load-bearing call site.)
|
|
836
|
+
await consolidateAllFlatPrds(allProjectCwds());
|
|
837
|
+
|
|
838
|
+
// Rollout migration for the PRD-authoring-lockdown feature: stamp every
|
|
839
|
+
// pre-existing PRD as legacy-adopted BEFORE reconcile() ever runs its
|
|
840
|
+
// provenance gate against it. Must run every boot (idempotent, cheap
|
|
841
|
+
// scan-and-skip) rather than once — a project opened for the first time
|
|
842
|
+
// after this shipped still has pre-existing unstamped PRDs the very first
|
|
843
|
+
// time reconcile() sees them.
|
|
844
|
+
try {
|
|
845
|
+
const adopted = await legacyAdoptExistingPrds();
|
|
846
|
+
if (adopted.stamped > 0) {
|
|
847
|
+
console.log(`[scheduler] legacy-adopt migration: stamped ${adopted.stamped} pre-existing PRD(s) as createdVia=legacy-adopted`);
|
|
848
|
+
}
|
|
849
|
+
for (const f of adopted.failed) {
|
|
850
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'legacy-adopt migration: could not stamp PRD', meta: f });
|
|
810
851
|
}
|
|
852
|
+
} catch (e) {
|
|
853
|
+
logs.writeLine({ level: 'error', scope: 'scheduler', message: 'legacy-adopt migration failed', meta: { error: e?.message } });
|
|
811
854
|
}
|
|
855
|
+
|
|
812
856
|
return result;
|
|
813
857
|
}
|
|
814
858
|
|
|
@@ -913,6 +957,41 @@ function appendHeartbeat(entry) {
|
|
|
913
957
|
}
|
|
914
958
|
}
|
|
915
959
|
|
|
960
|
+
/**
|
|
961
|
+
* computeStallSummary(state) → { stalled, total, running, pending, byProject }
|
|
962
|
+
*
|
|
963
|
+
* Pure, no IO. `state` is a merged queue-store read ({ jobs, invalidJobs,
|
|
964
|
+
* paused }). "Stalled" = the queue holds work — valid rows OR rows
|
|
965
|
+
* quarantined for an invalid status — but nothing is running or pending and
|
|
966
|
+
* the scheduler isn't paused. The 2026-08-07 incident sat exactly in this
|
|
967
|
+
* state for 4+ hours: 2 jobs, 0 running, 0 pending, and the only visible
|
|
968
|
+
* symptom was a heartbeat `counts` object that had silently minted a
|
|
969
|
+
* `queued` bucket instead of reporting anything actionable. `byProject`
|
|
970
|
+
* breaks the stalled rows down by cwd for the log line / toast.
|
|
971
|
+
*/
|
|
972
|
+
function computeStallSummary(state) {
|
|
973
|
+
const jobs = Array.isArray(state?.jobs) ? state.jobs : [];
|
|
974
|
+
const invalidJobs = Array.isArray(state?.invalidJobs) ? state.invalidJobs : [];
|
|
975
|
+
let running = 0;
|
|
976
|
+
let pending = 0;
|
|
977
|
+
const byProject = {};
|
|
978
|
+
for (const j of jobs) {
|
|
979
|
+
if (j.status === 'running') running += 1;
|
|
980
|
+
if (j.status === 'pending') pending += 1;
|
|
981
|
+
const key = j.cwd || '(unknown)';
|
|
982
|
+
byProject[key] = byProject[key] || {};
|
|
983
|
+
byProject[key][j.status] = (byProject[key][j.status] || 0) + 1;
|
|
984
|
+
}
|
|
985
|
+
for (const inv of invalidJobs) {
|
|
986
|
+
const key = inv.row?.cwd || '(unknown)';
|
|
987
|
+
byProject[key] = byProject[key] || {};
|
|
988
|
+
byProject[key].invalid = (byProject[key].invalid || 0) + 1;
|
|
989
|
+
}
|
|
990
|
+
const total = jobs.length + invalidJobs.length;
|
|
991
|
+
const stalled = total > 0 && running === 0 && pending === 0 && !state?.paused;
|
|
992
|
+
return { stalled, total, running, pending, byProject };
|
|
993
|
+
}
|
|
994
|
+
|
|
916
995
|
// An empty queue and an unreadable queue are NOT the same thing, and
|
|
917
996
|
// conflating them is destructive: reconcile() treats every PRD .md with no
|
|
918
997
|
// matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
|
|
@@ -1172,6 +1251,15 @@ async function reconcile(state) {
|
|
|
1172
1251
|
if (state && state.unreadable) {
|
|
1173
1252
|
throw new Error(`reconcile skipped: queue.json unreadable (${state.unreadable})`);
|
|
1174
1253
|
}
|
|
1254
|
+
// Sweep the retired flat prds/ dir BEFORE scanning it below. reconcile()
|
|
1255
|
+
// has several callers besides tickQueue's ~60s poll (broadcast,
|
|
1256
|
+
// rescheduleTimer, the schedule:state IPC handler, schedule:rescan) — this
|
|
1257
|
+
// lives here, not in any one caller, so the "a hand-written PRD in the flat
|
|
1258
|
+
// dir is swept before it can become a job" guarantee holds regardless of
|
|
1259
|
+
// which caller triggers this reconcile pass. A freshly hand-written file
|
|
1260
|
+
// has no queue row yet, so it is never "live" and gets archived here
|
|
1261
|
+
// instead of ever reaching the onDisk scan below.
|
|
1262
|
+
await consolidateAllFlatPrds(allProjectCwds());
|
|
1175
1263
|
const files = await listPrdFiles();
|
|
1176
1264
|
const onDisk = new Map();
|
|
1177
1265
|
for (const f of files) {
|
|
@@ -1213,13 +1301,16 @@ async function reconcile(state) {
|
|
|
1213
1301
|
// file may be unreadable, on a project whose dir failed to enumerate,
|
|
1214
1302
|
// or mid-move. "I can't see it" is not "the user deleted it", so the
|
|
1215
1303
|
// row survives — worst case it re-resolves on the next pass.
|
|
1216
|
-
if (job.status === 'pending' || job.status === 'running') {
|
|
1304
|
+
if (job.status === 'pending' || job.status === 'running' || job.status === 'quarantined') {
|
|
1217
1305
|
// Exception: a PENDING row whose PRD has an archived twin was
|
|
1218
1306
|
// retired on purpose (work landed by other means — e.g. implemented
|
|
1219
1307
|
// inline — and the source .md moved to prds-archived/). Keeping it
|
|
1220
1308
|
// would show a phantom "scheduled" job forever; firing it would just
|
|
1221
1309
|
// hit executeJob's archived-twin skip anyway. Running rows are left
|
|
1222
|
-
// alone — the reaper owns their lifecycle.
|
|
1310
|
+
// alone — the reaper owns their lifecycle. A quarantined row's file
|
|
1311
|
+
// going merely-not-visible must survive too — quarantine is meant to
|
|
1312
|
+
// be loud and reversible, never a silent drop (see this function's
|
|
1313
|
+
// header comment on the 2026-08-01 outage a silent skip caused).
|
|
1223
1314
|
if (job.status === 'pending' && (await archivedTwinExists(job))) {
|
|
1224
1315
|
console.log(`[scheduler] reconcile: retiring pending job ${job.slug} — PRD already archived (work landed elsewhere)`);
|
|
1225
1316
|
continue;
|
|
@@ -1233,7 +1324,7 @@ async function reconcile(state) {
|
|
|
1233
1324
|
continue;
|
|
1234
1325
|
}
|
|
1235
1326
|
seen.add(job.slug);
|
|
1236
|
-
|
|
1327
|
+
const updatedJob = {
|
|
1237
1328
|
...job,
|
|
1238
1329
|
title: p.title,
|
|
1239
1330
|
cwd: p.cwd,
|
|
@@ -1249,7 +1340,23 @@ async function reconcile(state) {
|
|
|
1249
1340
|
originSessionId: job.originSessionId
|
|
1250
1341
|
?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
|
|
1251
1342
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1252
|
-
}
|
|
1343
|
+
};
|
|
1344
|
+
// Adopt path: a row parked 'quarantined' (no createdVia provenance when
|
|
1345
|
+
// discovered) whose PRD file now carries a stamp — written via the
|
|
1346
|
+
// update-prd API's adopt patch, either the Scheduler tab's one-click
|
|
1347
|
+
// "adopt PRD" action or a manual scheduler_update_prd call — promotes to
|
|
1348
|
+
// 'pending' the very next reconcile pass. This is the ONLY way a
|
|
1349
|
+
// quarantined row becomes runnable; nothing else in reconcile() clears
|
|
1350
|
+
// that status.
|
|
1351
|
+
if (updatedJob.status === 'quarantined' && p.createdVia) {
|
|
1352
|
+
transitionJob(updatedJob, 'pending', {
|
|
1353
|
+
reason: `adopted via API (createdVia=${p.createdVia})`,
|
|
1354
|
+
source: 'reconcile-adopt',
|
|
1355
|
+
});
|
|
1356
|
+
console.log(`[scheduler] reconcile: adopted quarantined PRD ${job.slug} — createdVia=${p.createdVia}`);
|
|
1357
|
+
appendAuditEvent('scheduler_prd_adopted', { slug: job.slug, cwd: p.cwd, createdVia: p.createdVia, source: 'reconcile' });
|
|
1358
|
+
}
|
|
1359
|
+
next.push(updatedJob);
|
|
1253
1360
|
}
|
|
1254
1361
|
// Slugs on disk with no matching state.jobs row are normally brand-new
|
|
1255
1362
|
// PRDs — but once queueHistory.partitionJobs (above, later this same
|
|
@@ -1264,7 +1371,11 @@ async function reconcile(state) {
|
|
|
1264
1371
|
for (const [slug] of onDisk) {
|
|
1265
1372
|
if (!seen.has(slug)) unmatchedSlugs.push(slug);
|
|
1266
1373
|
}
|
|
1267
|
-
|
|
1374
|
+
// Rows quarantined by queueStore.shapeJobs because their `status` failed
|
|
1375
|
+
// ScheduleJobSchema (e.g. the 1021/1022 incident's `"status": "queued"`) —
|
|
1376
|
+
// see the repair pass below, right after historyBySlug is available.
|
|
1377
|
+
const invalidJobs = Array.isArray(state.invalidJobs) ? state.invalidJobs : [];
|
|
1378
|
+
const historyBySlug = (unmatchedSlugs.length > 0 || terminalDroppedNeedingHistoryCheck.length > 0 || invalidJobs.length > 0)
|
|
1268
1379
|
? await queueHistory.historyTerminalBySlug()
|
|
1269
1380
|
: new Map();
|
|
1270
1381
|
|
|
@@ -1282,6 +1393,79 @@ async function reconcile(state) {
|
|
|
1282
1393
|
}
|
|
1283
1394
|
}
|
|
1284
1395
|
|
|
1396
|
+
// Repair pass: an invalid row must self-heal within this one tick, not
|
|
1397
|
+
// wait for its slug to also drop out of `seen` via some unrelated code
|
|
1398
|
+
// path. Before this pass, reconcile was add-only (`if (seen.has(slug))
|
|
1399
|
+
// continue` below) — a quarantined row simply vanished from state.jobs
|
|
1400
|
+
// with no log of what its bad status actually was and no repair, which is
|
|
1401
|
+
// how the 1021/1022 rows sat invisible for 4+ hours (2026-08-07).
|
|
1402
|
+
let repairedInvalidCount = 0;
|
|
1403
|
+
for (const inv of invalidJobs) {
|
|
1404
|
+
if (seen.has(inv.slug)) continue; // a valid row for this slug already exists
|
|
1405
|
+
const oldStatus = inv.row?.status;
|
|
1406
|
+
const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir: RUNS_DIR });
|
|
1407
|
+
if (hist) {
|
|
1408
|
+
// Never resurrect: this slug already has a durable terminal record
|
|
1409
|
+
// elsewhere (history.jsonl or a run sidecar) — repairing its corrupted
|
|
1410
|
+
// row back to 'pending' would re-execute already-shipped work. Drop
|
|
1411
|
+
// the row (its real outcome is recorded elsewhere), loudly.
|
|
1412
|
+
console.warn(`[scheduler] reconcile: dropping invalid queue row ${inv.slug} (status was ${JSON.stringify(oldStatus)}) — already terminal (${hist.status}) in history/run sidecar, not resurrecting`);
|
|
1413
|
+
appendAuditEvent('scheduler_row_repaired', {
|
|
1414
|
+
slug: inv.slug, cwd: inv.row?.cwd ?? null, oldStatus: oldStatus ?? null,
|
|
1415
|
+
action: 'dropped-already-terminal', terminalStatus: hist.status, issues: inv.issues,
|
|
1416
|
+
});
|
|
1417
|
+
continue;
|
|
1418
|
+
}
|
|
1419
|
+
const p = onDisk.get(inv.slug);
|
|
1420
|
+
if (!p) {
|
|
1421
|
+
// PRD file also gone with no terminal record anywhere — nothing to
|
|
1422
|
+
// repair against. queueStore already logged the quarantine once.
|
|
1423
|
+
continue;
|
|
1424
|
+
}
|
|
1425
|
+
const job = {
|
|
1426
|
+
...inv.row,
|
|
1427
|
+
slug: inv.slug,
|
|
1428
|
+
title: p.title,
|
|
1429
|
+
cwd: p.cwd,
|
|
1430
|
+
parallelGroup: p.parallelGroup,
|
|
1431
|
+
estimateMinutes: p.estimateMinutes,
|
|
1432
|
+
sourcePromptId: p.sourcePromptId ?? inv.row?.sourcePromptId ?? null,
|
|
1433
|
+
sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
|
|
1434
|
+
epicId: p.epicId ?? inv.row?.epicId ?? null,
|
|
1435
|
+
dependsOn: p.dependsOn,
|
|
1436
|
+
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
1437
|
+
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1438
|
+
};
|
|
1439
|
+
const reason = `reconcile: repaired invalid status ${JSON.stringify(oldStatus)}`;
|
|
1440
|
+
// A repair is not a lifecycle transition — the corrupted `status` was
|
|
1441
|
+
// never a legal predecessor to check against LEGAL_TRANSITIONS, so this
|
|
1442
|
+
// goes through transitionJob's allowAnyFrom escape hatch (still gets the
|
|
1443
|
+
// normal mutation/statusHistory/audit trail, just skips the legality
|
|
1444
|
+
// gate on `from`) rather than a bare field assignment.
|
|
1445
|
+
transitionJob(job, 'pending', { reason, source: 'reconcile-repair', allowAnyFrom: true });
|
|
1446
|
+
if (job.runId || job.startedAt || job.runtime) {
|
|
1447
|
+
// This row had actually begun executing before its status got
|
|
1448
|
+
// corrupted.
|
|
1449
|
+
job.runId = null;
|
|
1450
|
+
job.startedAt = null;
|
|
1451
|
+
job.finishedAt = null;
|
|
1452
|
+
job.exitCode = null;
|
|
1453
|
+
delete job.runtime;
|
|
1454
|
+
delete job.verifierVerdict;
|
|
1455
|
+
}
|
|
1456
|
+
job.error = null;
|
|
1457
|
+
seen.add(inv.slug);
|
|
1458
|
+
next.push(job);
|
|
1459
|
+
repairedInvalidCount += 1;
|
|
1460
|
+
console.warn(`[scheduler] reconcile: repaired invalid queue row ${inv.slug} — status was ${JSON.stringify(oldStatus)}, reset to 'pending' (${inv.issues})`);
|
|
1461
|
+
appendAuditEvent('scheduler_row_repaired', {
|
|
1462
|
+
slug: inv.slug, cwd: p.cwd, oldStatus: oldStatus ?? null, newStatus: 'pending', issues: inv.issues,
|
|
1463
|
+
});
|
|
1464
|
+
}
|
|
1465
|
+
if (repairedInvalidCount > 0) {
|
|
1466
|
+
console.warn(`[scheduler] reconcile: repaired ${repairedInvalidCount} invalid queue row(s) this pass`);
|
|
1467
|
+
}
|
|
1468
|
+
|
|
1285
1469
|
// Terminal-in-history slugs whose .md file is still on disk: fed into the
|
|
1286
1470
|
// auto-archive selection pass below (as synthetic completed entries) so
|
|
1287
1471
|
// their file can still be swept, without ever creating a live job row
|
|
@@ -1302,6 +1486,7 @@ async function reconcile(state) {
|
|
|
1302
1486
|
return idx.sessions[epicId]?.status ?? null;
|
|
1303
1487
|
}
|
|
1304
1488
|
|
|
1489
|
+
let staleNewDiscoveryCount = 0;
|
|
1305
1490
|
for (const [slug, p] of onDisk) {
|
|
1306
1491
|
if (seen.has(slug)) continue;
|
|
1307
1492
|
// Security gate: a PRD's file location IS its Epic membership
|
|
@@ -1385,8 +1570,47 @@ async function reconcile(state) {
|
|
|
1385
1570
|
const parent = healTargetForFix(slug, state.jobs);
|
|
1386
1571
|
entry.investigationDepth = parent ? (parent.investigationDepth ?? 1) + 1 : 2;
|
|
1387
1572
|
}
|
|
1573
|
+
// Provenance gate (PRD-authoring lockdown): a PRD discovered with no
|
|
1574
|
+
// `createdVia` stamp was never written through scheduler_create_prd/
|
|
1575
|
+
// chat:create-prd (prdCreate.cjs always stamps 'scheduler-api') or the
|
|
1576
|
+
// legacy-adopt boot migration ('legacy-adopted') — it bypassed the
|
|
1577
|
+
// sanctioned API, most likely via a raw Write/Edit tool call the
|
|
1578
|
+
// guard-prd-writes.cjs PreToolUse hook should have denied. Fix-plan PRDs
|
|
1579
|
+
// are exempt: spawnInvestigation's own probe writes them directly by
|
|
1580
|
+
// design (a trusted, scheduler-spawned internal loop, not an
|
|
1581
|
+
// agent/human authoring a PRD), matching the isFixPlanSlug convention
|
|
1582
|
+
// used everywhere else this distinction matters.
|
|
1583
|
+
//
|
|
1584
|
+
// Quarantine is loud and reversible, never a silent skip (see the
|
|
1585
|
+
// 2026-08-01 23-PRD outage this file's header references for what a
|
|
1586
|
+
// SILENT skip costs): logged at warn, audited, and surfaced in the
|
|
1587
|
+
// Scheduler tab's Quarantined filter with a one-click adopt action
|
|
1588
|
+
// (schedule:adopt-prd) that stamps the file via the same update-prd API
|
|
1589
|
+
// route the MCP tool uses — reconcile()'s adopt path above promotes it
|
|
1590
|
+
// to 'pending' on the very next pass, within one tick of being stamped.
|
|
1591
|
+
if (!p.createdVia && !isFixPlanSlug(slug)) {
|
|
1592
|
+
entry.status = 'quarantined';
|
|
1593
|
+
console.warn(`[scheduler] reconcile: quarantining unstamped PRD ${slug} (${p.path}) — no createdVia provenance; adopt it from the Scheduler tab's Quarantined filter or via scheduler_update_prd to make it runnable`);
|
|
1594
|
+
appendAuditEvent('prd_quarantined', { slug, cwd: p.cwd, path: p.path, reason: 'missing createdVia provenance frontmatter' });
|
|
1595
|
+
}
|
|
1596
|
+
// A PRD with no queue row and no terminal record is normally a
|
|
1597
|
+
// brand-new file — but one whose mtime already predates a full poll
|
|
1598
|
+
// interval means it sat unpicked (a prior reconcile pass should have
|
|
1599
|
+
// caught it, or it's arriving from a source that bypassed the app's
|
|
1600
|
+
// normal write path). Report it rather than silently treating "first
|
|
1601
|
+
// seen this pass" as "just created".
|
|
1602
|
+
try {
|
|
1603
|
+
const ageMs = Date.now() - fs.statSync(p.path).mtimeMs;
|
|
1604
|
+
if (ageMs > POLL_INTERVAL_MS) {
|
|
1605
|
+
staleNewDiscoveryCount += 1;
|
|
1606
|
+
console.warn(`[scheduler] reconcile: discovered PRD ${slug} with no queue row and no terminal record — file is ${Math.round(ageMs / 1000)}s old, only first seen this pass`);
|
|
1607
|
+
}
|
|
1608
|
+
} catch { /* stat is best-effort reporting only */ }
|
|
1388
1609
|
next.push(entry);
|
|
1389
1610
|
}
|
|
1611
|
+
if (staleNewDiscoveryCount > 0) {
|
|
1612
|
+
console.warn(`[scheduler] reconcile: ${staleNewDiscoveryCount} PRD(s) discovered this pass were already older than one poll interval with no prior queue row`);
|
|
1613
|
+
}
|
|
1390
1614
|
const sorted = next.sort((a, b) => b.slug.localeCompare(a.slug));
|
|
1391
1615
|
|
|
1392
1616
|
// Move terminal jobs past the retention window out to history.jsonl so
|
|
@@ -1467,6 +1691,13 @@ let resumeTimer = null;
|
|
|
1467
1691
|
let pollLoopTimer = null;
|
|
1468
1692
|
let rescheduleInterval = null;
|
|
1469
1693
|
let heartbeatInterval = null;
|
|
1694
|
+
// Stall-detector state (computeStallSummary), read/written only inside the
|
|
1695
|
+
// heartbeat interval below. stallSince: wall-clock ms the stalled condition
|
|
1696
|
+
// was first observed, null when clear. stallToasted: rate-limits the
|
|
1697
|
+
// error-log + toast to once per stall episode (cleared the moment the queue
|
|
1698
|
+
// stops being stalled) rather than every 60s heartbeat tick.
|
|
1699
|
+
let stallSince = null;
|
|
1700
|
+
let stallToasted = false;
|
|
1470
1701
|
// (The 5-minute feedback sweep that used to piggyback on this heartbeat is
|
|
1471
1702
|
// gone: it scanned each active project's session-manager-operations/feedback/
|
|
1472
1703
|
// and auto-queued a /process-feedback PRD. Both the folder and that skill are
|
|
@@ -1738,7 +1969,7 @@ async function clearPause(source) {
|
|
|
1738
1969
|
*/
|
|
1739
1970
|
function resetJobFields(job, errorMsg, opts = {}) {
|
|
1740
1971
|
if (job.status === 'completed' && opts.force !== true) return false;
|
|
1741
|
-
job
|
|
1972
|
+
if (!transitionJob(job, 'pending', { reason: errorMsg ?? 'reset to pending', source: opts.source ?? 'resetJobFields' })) return false;
|
|
1742
1973
|
job.runId = null;
|
|
1743
1974
|
job.startedAt = null;
|
|
1744
1975
|
job.finishedAt = null;
|
|
@@ -1799,13 +2030,13 @@ function partitionBootOrphans(jobs, isAlive = claudePidAlive) {
|
|
|
1799
2030
|
function applyOrphanOutcome(job, outcome, killNote = '') {
|
|
1800
2031
|
const now = new Date().toISOString();
|
|
1801
2032
|
if (outcome === 'success') {
|
|
1802
|
-
job
|
|
2033
|
+
transitionJob(job, 'completed', { reason: 'boot orphan reconciliation: run succeeded', source: 'applyOrphanOutcome' });
|
|
1803
2034
|
job.exitCode = 0;
|
|
1804
2035
|
job.error = null;
|
|
1805
2036
|
job.finishedAt = now;
|
|
1806
2037
|
delete job.runtime;
|
|
1807
2038
|
} else if (outcome === 'failed') {
|
|
1808
|
-
job
|
|
2039
|
+
transitionJob(job, 'failed', { reason: `orphaned: app restarted while running${killNote}`, source: 'applyOrphanOutcome' });
|
|
1809
2040
|
job.exitCode = job.exitCode ?? 1;
|
|
1810
2041
|
job.error = `orphaned: app restarted while running${killNote}`;
|
|
1811
2042
|
job.finishedAt = now;
|
|
@@ -1813,10 +2044,10 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
|
|
|
1813
2044
|
} else {
|
|
1814
2045
|
const tries = job.orphanRetries ?? 0;
|
|
1815
2046
|
if (tries < ORPHAN_REQUEUE_CAP) {
|
|
1816
|
-
resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}
|
|
2047
|
+
resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}`, { source: 'applyOrphanOutcome' });
|
|
1817
2048
|
job.orphanRetries = tries + 1;
|
|
1818
2049
|
} else {
|
|
1819
|
-
job
|
|
2050
|
+
transitionJob(job, 'failed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`, source: 'applyOrphanOutcome' });
|
|
1820
2051
|
job.exitCode = job.exitCode ?? 1;
|
|
1821
2052
|
job.error = `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`;
|
|
1822
2053
|
job.finishedAt = now;
|
|
@@ -2860,7 +3091,7 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
2860
3091
|
// "nothing is happening" even though an Opus process was actively running.
|
|
2861
3092
|
await mutate((s) => {
|
|
2862
3093
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
2863
|
-
if (j) j
|
|
3094
|
+
if (j) transitionJob(j, 'investigating', { reason: 'spawning investigation probe', source: 'spawnInvestigation:start' });
|
|
2864
3095
|
});
|
|
2865
3096
|
await broadcast({ flush: true });
|
|
2866
3097
|
|
|
@@ -2905,7 +3136,7 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
2905
3136
|
// 'investigating' must never be the job's resting state.
|
|
2906
3137
|
mutate((s) => {
|
|
2907
3138
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
2908
|
-
if (j && j.status === 'investigating') j
|
|
3139
|
+
if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation probe exited — restoring prior status', source: 'spawnInvestigation:onExit' });
|
|
2909
3140
|
})
|
|
2910
3141
|
.then(() => broadcast({ flush: true }))
|
|
2911
3142
|
.catch(() => {});
|
|
@@ -2960,7 +3191,7 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
2960
3191
|
releaseSlot();
|
|
2961
3192
|
mutate((s) => {
|
|
2962
3193
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
2963
|
-
if (j && j.status === 'investigating') j
|
|
3194
|
+
if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation spawn threw before exiting — restoring prior status', source: 'spawnInvestigation:catch' });
|
|
2964
3195
|
})
|
|
2965
3196
|
.then(() => broadcast({ flush: true }))
|
|
2966
3197
|
.catch(() => {});
|
|
@@ -2982,7 +3213,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2982
3213
|
await mutate((s) => {
|
|
2983
3214
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
2984
3215
|
if (idx >= 0) {
|
|
2985
|
-
s.jobs[idx]
|
|
3216
|
+
transitionJob(s.jobs[idx], 'running', { reason: 'dispatched for execution', source: 'spawnJob:dispatch' });
|
|
2986
3217
|
s.jobs[idx].runId = runId;
|
|
2987
3218
|
s.jobs[idx].startedAt = new Date().toISOString();
|
|
2988
3219
|
}
|
|
@@ -3069,7 +3300,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3069
3300
|
await mutate((s) => {
|
|
3070
3301
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3071
3302
|
if (idx >= 0) {
|
|
3072
|
-
s.jobs[idx].
|
|
3303
|
+
transitionJob(s.jobs[idx], 'completed', { reason: res.note ?? 'PRD archived or missing — treated as already-shipped', source: 'spawnJob:skip-archived' });
|
|
3073
3304
|
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
3074
3305
|
s.jobs[idx].exitCode = 0;
|
|
3075
3306
|
s.jobs[idx].error = null;
|
|
@@ -3239,10 +3470,19 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3239
3470
|
const newlyCompletedPrds = [];
|
|
3240
3471
|
await mutate((s) => {
|
|
3241
3472
|
const i2 = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3473
|
+
// A job already moved off 'running' by someone else (namely
|
|
3474
|
+
// remote.cancelJob, PRD 1024 — it SIGTERMs the process then finalizes
|
|
3475
|
+
// the row to 'failed' before this exit handler necessarily runs) is
|
|
3476
|
+
// not this run's to finalize: doing so anyway could re-legalize the
|
|
3477
|
+
// row via a legal failed->completed/needs_review edge (see
|
|
3478
|
+
// scheduleJobTransitions.cjs's LEGAL_TRANSITIONS) and silently
|
|
3479
|
+
// undo the cancellation. Skip — the row already reflects its real
|
|
3480
|
+
// terminal state.
|
|
3481
|
+
if (i2 >= 0 && s.jobs[i2].status !== 'running') return;
|
|
3242
3482
|
if (i2 >= 0) {
|
|
3243
3483
|
const treatAsPending = res.rateLimited || (s.paused && s.paused.reason === 'rate_limit');
|
|
3244
3484
|
if (treatAsPending) {
|
|
3245
|
-
resetJobFields(s.jobs[i2], res.rateLimited ? 'paused: rate limit' : 'paused: queue halted');
|
|
3485
|
+
resetJobFields(s.jobs[i2], res.rateLimited ? 'paused: rate limit' : 'paused: queue halted', { source: 'spawnJob:halt-reset' });
|
|
3246
3486
|
} else {
|
|
3247
3487
|
// Determine effective status, applying the verifier verdict for exit=0 runs.
|
|
3248
3488
|
let effectiveStatus;
|
|
@@ -3262,14 +3502,14 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3262
3502
|
effectiveStatus = 'completed';
|
|
3263
3503
|
} else if (verifyResult.downgradeTo === 'pending') {
|
|
3264
3504
|
// HALT or deps_unmet: reset to pending so the job re-fires.
|
|
3265
|
-
resetJobFields(s.jobs[i2], verifyResult.reason);
|
|
3505
|
+
resetJobFields(s.jobs[i2], verifyResult.reason, { source: 'spawnJob:verify-downgrade' });
|
|
3266
3506
|
return; // job already mutated by resetJobFields; skip the rest
|
|
3267
3507
|
} else {
|
|
3268
3508
|
// transcript_errors or verify_unavailable: escalate to needs_review.
|
|
3269
3509
|
effectiveStatus = 'needs_review';
|
|
3270
3510
|
}
|
|
3271
3511
|
|
|
3272
|
-
s.jobs[i2].
|
|
3512
|
+
transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
|
|
3273
3513
|
s.jobs[i2].finishedAt = new Date().toISOString();
|
|
3274
3514
|
s.jobs[i2].exitCode = res.exitCode;
|
|
3275
3515
|
s.jobs[i2].error = effectiveStatus === 'needs_review'
|
|
@@ -3356,7 +3596,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3356
3596
|
if (orig) {
|
|
3357
3597
|
const priorStatus = orig.status;
|
|
3358
3598
|
console.log(`[scheduler] auto-promote: ${orig.slug} (${priorStatus}) → completed because ${job.slug} succeeded`);
|
|
3359
|
-
orig.
|
|
3599
|
+
transitionJob(orig, 'completed', { reason: `auto-promoted: fix plan ${job.slug} succeeded`, source: 'spawnJob:auto-promote' });
|
|
3360
3600
|
orig.exitCode = 0;
|
|
3361
3601
|
orig.error = null;
|
|
3362
3602
|
orig.completedBy = job.slug;
|
|
@@ -3444,7 +3684,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3444
3684
|
await mutate((s) => {
|
|
3445
3685
|
const i = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3446
3686
|
if (i >= 0) {
|
|
3447
|
-
resetJobFields(s.jobs[i], null);
|
|
3687
|
+
resetJobFields(s.jobs[i], null, { source: 'spawnJob:transient-retry' });
|
|
3448
3688
|
s.jobs[i].transientRetries = decision.retries + 1;
|
|
3449
3689
|
}
|
|
3450
3690
|
});
|
|
@@ -3454,7 +3694,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3454
3694
|
await mutate((s) => {
|
|
3455
3695
|
const i = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3456
3696
|
if (i >= 0) {
|
|
3457
|
-
s.jobs[i].
|
|
3697
|
+
transitionJob(s.jobs[i], 'failed', { reason: `transient failure (${decision.transientKind}) left uncommitted work — not auto-requeued`, source: 'spawnJob:fail-dirty' });
|
|
3458
3698
|
s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample}) — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
|
|
3459
3699
|
}
|
|
3460
3700
|
});
|
|
@@ -3515,6 +3755,9 @@ function tickQueue() {
|
|
|
3515
3755
|
}
|
|
3516
3756
|
if (cancelToken.cancelled) return { fired: false, reason: 'cancelled' };
|
|
3517
3757
|
|
|
3758
|
+
// The retired-flat-dir sweep now lives inside reconcile() itself (see its
|
|
3759
|
+
// own comment) so every caller of reconcile — not just this tick — gets
|
|
3760
|
+
// the guarantee.
|
|
3518
3761
|
await reconcile(state);
|
|
3519
3762
|
// Session-Manager's machine-wide slot pool is the ONLY concurrency limit
|
|
3520
3763
|
// the picker answers to (plus the memory gate below). The scheduler used
|
|
@@ -3702,7 +3945,7 @@ async function reapDeadRunningJobs() {
|
|
|
3702
3945
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
3703
3946
|
if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
|
|
3704
3947
|
const success = outcome === 'success';
|
|
3705
|
-
s.jobs[idx]
|
|
3948
|
+
transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: `reaped: process gone (outcome=${outcome})`, source: 'reapDeadRunningJobs' });
|
|
3706
3949
|
s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
|
|
3707
3950
|
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
3708
3951
|
s.jobs[idx].error = success ? null : `reaped: process gone, no success result in log (${outcome})`;
|
|
@@ -4125,7 +4368,7 @@ async function reverifyNeedsReview() {
|
|
|
4125
4368
|
await mutate((s) => {
|
|
4126
4369
|
for (const j of s.jobs) {
|
|
4127
4370
|
if (j.status === 'needs_review' && healSet.has(j.slug)) {
|
|
4128
|
-
j
|
|
4371
|
+
transitionJob(j, 'completed', { reason: 'boot reverify: stale needs_review healed', source: 'reverifyNeedsReview:heal' });
|
|
4129
4372
|
j.error = null;
|
|
4130
4373
|
delete j.verifierVerdict;
|
|
4131
4374
|
healedPrds.push({ slug: j.slug, cwd: j.cwd });
|
|
@@ -4163,7 +4406,7 @@ async function reverifyNeedsReview() {
|
|
|
4163
4406
|
const orig = healTargetForFix(job.slug, s.jobs);
|
|
4164
4407
|
if (!orig) continue;
|
|
4165
4408
|
const priorStatus = orig.status;
|
|
4166
|
-
orig.
|
|
4409
|
+
transitionJob(orig, 'completed', { reason: `auto-promoted: fix plan ${job.slug} already completed`, source: 'reverifyNeedsReview:auto-promote' });
|
|
4167
4410
|
orig.exitCode = 0;
|
|
4168
4411
|
orig.error = null;
|
|
4169
4412
|
orig.completedBy = job.slug;
|
|
@@ -4341,7 +4584,7 @@ function registerScheduleHandlers() {
|
|
|
4341
4584
|
// Guard is in resetJobFields: refuses to reset an already-'completed'
|
|
4342
4585
|
// job, which would otherwise re-fire a PRD whose deliverable already
|
|
4343
4586
|
// landed (see resetJobFields' doc comment for the incident).
|
|
4344
|
-
return resetJobFields(state.jobs[idx]) ? 'ok' : 'refused';
|
|
4587
|
+
return resetJobFields(state.jobs[idx], null, { source: 'ipc:schedule:reset-job' }) ? 'ok' : 'refused';
|
|
4345
4588
|
});
|
|
4346
4589
|
if (outcome === 'not-found') return { ok: false, error: 'not found' };
|
|
4347
4590
|
if (outcome === 'refused') {
|
|
@@ -4354,6 +4597,36 @@ function registerScheduleHandlers() {
|
|
|
4354
4597
|
return { ok: true };
|
|
4355
4598
|
}));
|
|
4356
4599
|
|
|
4600
|
+
// Renderer-facing counterpart to prdCreate.cjs's chat:create-prd handler
|
|
4601
|
+
// (index.cjs): calls the SAME remote.updatePrd the admin HTTP route/MCP
|
|
4602
|
+
// tool use, so "stamps it through the API" holds for the Scheduler tab's
|
|
4603
|
+
// one-click adopt action too, not just a direct fs write. Only a
|
|
4604
|
+
// 'quarantined' row is eligible — see reconcile()'s provenance gate.
|
|
4605
|
+
ipcMain.handle('schedule:adopt-prd', validated(schemas.scheduleSlug, async ({ slug }) => {
|
|
4606
|
+
if (!(await safeSlugPath(slug))) return { ok: false, kind: 'error', message: 'invalid slug' };
|
|
4607
|
+
const state = await readQueue();
|
|
4608
|
+
const job = state.jobs.find((j) => j.slug === slug);
|
|
4609
|
+
if (!job) return { ok: false, kind: 'error', message: 'not found' };
|
|
4610
|
+
if (job.status !== 'quarantined') {
|
|
4611
|
+
return { ok: false, kind: 'error', message: `job status is "${job.status}" — only a quarantined PRD may be adopted` };
|
|
4612
|
+
}
|
|
4613
|
+
const result = await remote.updatePrd({
|
|
4614
|
+
slug,
|
|
4615
|
+
cwd: job.cwd,
|
|
4616
|
+
frontmatter: { createdVia: 'legacy-adopted', issuedAt: new Date().toISOString() },
|
|
4617
|
+
});
|
|
4618
|
+
if (!result.ok) return { ok: false, kind: 'error', message: result.error ?? 'adopt failed' };
|
|
4619
|
+
appendAuditEvent('scheduler_prd_adopted', { slug, cwd: job.cwd ?? null, source: 'ipc:schedule:adopt-prd' });
|
|
4620
|
+
// Promote the row to 'pending' immediately rather than waiting for the
|
|
4621
|
+
// next poll tick — the Scheduler tab's "adopt PRD" click should be
|
|
4622
|
+
// visibly effective within this one round-trip.
|
|
4623
|
+
const freshState = await readQueue();
|
|
4624
|
+
await reconcile(freshState);
|
|
4625
|
+
await writeQueue(freshState);
|
|
4626
|
+
await broadcast({ flush: true });
|
|
4627
|
+
return { ok: true, kind: 'info', message: `Adopted ${slug} — it will run as a normal pending job` };
|
|
4628
|
+
}));
|
|
4629
|
+
|
|
4357
4630
|
ipcMain.handle('schedule:run-now', async () => {
|
|
4358
4631
|
// Manual run-now overrides any auto-pause. Clear it first.
|
|
4359
4632
|
await clearPause('run-now');
|
|
@@ -4483,85 +4756,7 @@ function registerScheduleHandlers() {
|
|
|
4483
4756
|
}
|
|
4484
4757
|
}));
|
|
4485
4758
|
|
|
4486
|
-
ipcMain.handle('schedule:list-prds', async () =>
|
|
4487
|
-
ensureDirs();
|
|
4488
|
-
const out = [];
|
|
4489
|
-
const seenSlugs = new Set();
|
|
4490
|
-
|
|
4491
|
-
async function readDirInto(dir, { archived }) {
|
|
4492
|
-
let entries;
|
|
4493
|
-
try {
|
|
4494
|
-
entries = await fsp.readdir(dir);
|
|
4495
|
-
} catch (e) {
|
|
4496
|
-
if (e?.code !== 'ENOENT') {
|
|
4497
|
-
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: readdir failed', meta: { dir, error: e?.message } });
|
|
4498
|
-
}
|
|
4499
|
-
return;
|
|
4500
|
-
}
|
|
4501
|
-
for (const name of entries) {
|
|
4502
|
-
if (!name.endsWith('.md') || name.startsWith('.')) continue;
|
|
4503
|
-
const filePath = path.join(dir, name);
|
|
4504
|
-
try {
|
|
4505
|
-
const parsed = await parsePrd(filePath);
|
|
4506
|
-
// A slug can't be both live and archived at once, but a duplicate
|
|
4507
|
-
// slug found in two archive dirs (shouldn't happen — archiving is
|
|
4508
|
-
// a single rename — but is cheap to guard) is skipped rather than
|
|
4509
|
-
// double-counted.
|
|
4510
|
-
if (seenSlugs.has(parsed.slug)) continue;
|
|
4511
|
-
seenSlugs.add(parsed.slug);
|
|
4512
|
-
const stat = await fsp.stat(filePath);
|
|
4513
|
-
const entry = {
|
|
4514
|
-
slug: parsed.slug,
|
|
4515
|
-
parallelGroup: parsed.parallelGroup,
|
|
4516
|
-
title: parsed.title,
|
|
4517
|
-
cwd: parsed.cwd || '',
|
|
4518
|
-
estimateMinutes: parsed.estimateMinutes,
|
|
4519
|
-
sourcePromptId: parsed.sourcePromptId,
|
|
4520
|
-
epicId: parsed.epicId ?? null,
|
|
4521
|
-
mtimeMs: stat.mtimeMs,
|
|
4522
|
-
archived,
|
|
4523
|
-
};
|
|
4524
|
-
out.push(entry);
|
|
4525
|
-
} catch (e) {
|
|
4526
|
-
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: skipping unparseable file', meta: { name, error: e?.message } });
|
|
4527
|
-
}
|
|
4528
|
-
}
|
|
4529
|
-
}
|
|
4530
|
-
|
|
4531
|
-
// Live PRDs first, so an archived duplicate (shouldn't exist, but a
|
|
4532
|
-
// stale rename copy is possible) never shadows the still-runnable live
|
|
4533
|
-
// entry.
|
|
4534
|
-
for (const dir of candidatePrdsDirs()) {
|
|
4535
|
-
await readDirInto(dir, { archived: false });
|
|
4536
|
-
}
|
|
4537
|
-
|
|
4538
|
-
const archivedStart = out.length;
|
|
4539
|
-
for (const dir of candidateArchivedPrdsDirs()) {
|
|
4540
|
-
await readDirInto(dir, { archived: true });
|
|
4541
|
-
}
|
|
4542
|
-
|
|
4543
|
-
// Archived PRDs need a status: archiveCompletedPrd (scheduler.cjs) only
|
|
4544
|
-
// ever archives a job whose effective status is 'completed' — a 'failed'
|
|
4545
|
-
// job's PRD source stays in the live prds/ dir (still visible/countable
|
|
4546
|
-
// there already). Still resolve the real job status defensively (live
|
|
4547
|
-
// queue row, falling back to history.jsonl) rather than hard-coding
|
|
4548
|
-
// 'completed', so this stays correct if that archiving invariant ever
|
|
4549
|
-
// changes.
|
|
4550
|
-
if (out.length > archivedStart) {
|
|
4551
|
-
const [state, histBySlug] = await Promise.all([
|
|
4552
|
-
readQueue(),
|
|
4553
|
-
queueHistory.historyTerminalBySlug().catch(() => new Map()),
|
|
4554
|
-
]);
|
|
4555
|
-
const liveStatusBySlug = new Map(state.jobs.map((j) => [j.slug, j.status]));
|
|
4556
|
-
for (let i = archivedStart; i < out.length; i++) {
|
|
4557
|
-
const entry = out[i];
|
|
4558
|
-
entry.archivedStatus = resolveArchivedPrdStatus(entry.slug, liveStatusBySlug, histBySlug);
|
|
4559
|
-
}
|
|
4560
|
-
}
|
|
4561
|
-
|
|
4562
|
-
out.sort((a, b) => a.slug.localeCompare(b.slug, undefined, { numeric: true }));
|
|
4563
|
-
return out;
|
|
4564
|
-
});
|
|
4759
|
+
ipcMain.handle('schedule:list-prds', async () => listPrdsInternal());
|
|
4565
4760
|
|
|
4566
4761
|
// Return last N completed/failed jobs from queue.json, newest first.
|
|
4567
4762
|
// Purely additive: no schema change, no archive-folder read needed.
|
|
@@ -4766,12 +4961,54 @@ async function init() {
|
|
|
4766
4961
|
if (heartbeatInterval) clearInterval(heartbeatInterval);
|
|
4767
4962
|
heartbeatInterval = setInterval(() => {
|
|
4768
4963
|
const s = readQueueSync();
|
|
4769
|
-
|
|
4770
|
-
|
|
4964
|
+
// Initialise from the real status union (scheduleJobSchema.cjs) rather
|
|
4965
|
+
// than a hand-maintained subset — the old `{ pending, running, completed,
|
|
4966
|
+
// failed }` literal silently minted a NEW key for any other value
|
|
4967
|
+
// (`counts[j.status] = (counts[j.status]||0)+1`), which is exactly how a
|
|
4968
|
+
// heartbeat with a `queued: 2` bucket looked like "normal" 24h
|
|
4969
|
+
// visibility instead of the alarm it should have been. Any row whose
|
|
4970
|
+
// status isn't in JOB_STATUSES (shouldn't happen post-quarantine, but
|
|
4971
|
+
// this is the last line of defence) routes into `unknown`, never a
|
|
4972
|
+
// freshly-minted key.
|
|
4973
|
+
const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
|
|
4974
|
+
counts.unknown = 0;
|
|
4975
|
+
for (const j of s.jobs) {
|
|
4976
|
+
if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
|
|
4977
|
+
counts[j.status] += 1;
|
|
4978
|
+
} else {
|
|
4979
|
+
counts.unknown += 1;
|
|
4980
|
+
}
|
|
4981
|
+
}
|
|
4982
|
+
|
|
4983
|
+
const stall = computeStallSummary(s);
|
|
4984
|
+
if (stall.stalled) {
|
|
4985
|
+
if (stallSince === null) stallSince = Date.now();
|
|
4986
|
+
if (!stallToasted && Date.now() - stallSince >= POLL_INTERVAL_MS) {
|
|
4987
|
+
stallToasted = true;
|
|
4988
|
+
console.error(
|
|
4989
|
+
`[scheduler] STALL DETECTED: ${stall.total} job(s) queued, 0 running, 0 pending, not paused, `
|
|
4990
|
+
+ `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
|
|
4991
|
+
stall.byProject,
|
|
4992
|
+
);
|
|
4993
|
+
appendAuditEvent('scheduler_stall_detected', { total: stall.total, byProject: stall.byProject });
|
|
4994
|
+
if (mainWindow && !mainWindow.isDestroyed()) {
|
|
4995
|
+
sendIfAlive(mainWindow, 'schedule:stall', {
|
|
4996
|
+
message: `Scheduler stall: ${stall.total} job(s) queued but none running or pending. Check the Scheduler tab.`,
|
|
4997
|
+
total: stall.total,
|
|
4998
|
+
byProject: stall.byProject,
|
|
4999
|
+
});
|
|
5000
|
+
}
|
|
5001
|
+
}
|
|
5002
|
+
} else {
|
|
5003
|
+
stallSince = null;
|
|
5004
|
+
stallToasted = false;
|
|
5005
|
+
}
|
|
5006
|
+
|
|
4771
5007
|
appendHeartbeat({
|
|
4772
5008
|
ts: Date.now(),
|
|
4773
5009
|
pid: process.pid,
|
|
4774
5010
|
counts,
|
|
5011
|
+
stall: { stalled: stall.stalled, total: stall.total },
|
|
4775
5012
|
paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
|
|
4776
5013
|
nextReset: cachedNextReset,
|
|
4777
5014
|
utilization: cachedUtilization,
|
|
@@ -4802,6 +5039,102 @@ async function init() {
|
|
|
4802
5039
|
}
|
|
4803
5040
|
}
|
|
4804
5041
|
|
|
5042
|
+
/**
|
|
5043
|
+
* listPrdsInternal() → every live + archived PRD across every project,
|
|
5044
|
+
* with each entry's real job status folded in (`status`: the live queue
|
|
5045
|
+
* row's status, or the resolved terminal status for an archived entry, or
|
|
5046
|
+
* null when no queue row exists yet — e.g. a PRD just written and not yet
|
|
5047
|
+
* picked up by reconcile()). Single source of truth for both the renderer's
|
|
5048
|
+
* `schedule:list-prds` IPC handler and the admin HTTP `GET
|
|
5049
|
+
* /admin/scheduler/prds` route (PRD 1024) — neither re-implements this scan.
|
|
5050
|
+
*/
|
|
5051
|
+
async function listPrdsInternal() {
|
|
5052
|
+
ensureDirs();
|
|
5053
|
+
const out = [];
|
|
5054
|
+
const seenSlugs = new Set();
|
|
5055
|
+
|
|
5056
|
+
async function readDirInto(dir, { archived }) {
|
|
5057
|
+
let entries;
|
|
5058
|
+
try {
|
|
5059
|
+
entries = await fsp.readdir(dir);
|
|
5060
|
+
} catch (e) {
|
|
5061
|
+
if (e?.code !== 'ENOENT') {
|
|
5062
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: readdir failed', meta: { dir, error: e?.message } });
|
|
5063
|
+
}
|
|
5064
|
+
return;
|
|
5065
|
+
}
|
|
5066
|
+
for (const name of entries) {
|
|
5067
|
+
if (!name.endsWith('.md') || name.startsWith('.')) continue;
|
|
5068
|
+
const filePath = path.join(dir, name);
|
|
5069
|
+
try {
|
|
5070
|
+
const parsed = await parsePrd(filePath);
|
|
5071
|
+
// A slug can't be both live and archived at once, but a duplicate
|
|
5072
|
+
// slug found in two archive dirs (shouldn't happen — archiving is
|
|
5073
|
+
// a single rename — but is cheap to guard) is skipped rather than
|
|
5074
|
+
// double-counted.
|
|
5075
|
+
if (seenSlugs.has(parsed.slug)) continue;
|
|
5076
|
+
seenSlugs.add(parsed.slug);
|
|
5077
|
+
const stat = await fsp.stat(filePath);
|
|
5078
|
+
const entry = {
|
|
5079
|
+
slug: parsed.slug,
|
|
5080
|
+
parallelGroup: parsed.parallelGroup,
|
|
5081
|
+
title: parsed.title,
|
|
5082
|
+
cwd: parsed.cwd || '',
|
|
5083
|
+
estimateMinutes: parsed.estimateMinutes,
|
|
5084
|
+
sourcePromptId: parsed.sourcePromptId,
|
|
5085
|
+
epicId: parsed.epicId ?? null,
|
|
5086
|
+
mtimeMs: stat.mtimeMs,
|
|
5087
|
+
archived,
|
|
5088
|
+
};
|
|
5089
|
+
out.push(entry);
|
|
5090
|
+
} catch (e) {
|
|
5091
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: skipping unparseable file', meta: { name, error: e?.message } });
|
|
5092
|
+
}
|
|
5093
|
+
}
|
|
5094
|
+
}
|
|
5095
|
+
|
|
5096
|
+
// Live PRDs first, so an archived duplicate (shouldn't exist, but a
|
|
5097
|
+
// stale rename copy is possible) never shadows the still-runnable live
|
|
5098
|
+
// entry.
|
|
5099
|
+
for (const dir of candidatePrdsDirs()) {
|
|
5100
|
+
await readDirInto(dir, { archived: false });
|
|
5101
|
+
}
|
|
5102
|
+
|
|
5103
|
+
const archivedStart = out.length;
|
|
5104
|
+
for (const dir of candidateArchivedPrdsDirs()) {
|
|
5105
|
+
await readDirInto(dir, { archived: true });
|
|
5106
|
+
}
|
|
5107
|
+
|
|
5108
|
+
// Every entry (live and archived) gets a real job status folded in.
|
|
5109
|
+
// Archived PRDs need one resolved defensively (live queue row, falling
|
|
5110
|
+
// back to history.jsonl) rather than hard-coded 'completed', so this
|
|
5111
|
+
// stays correct if the archive-only-completed invariant ever changes; a
|
|
5112
|
+
// live entry with no queue row yet (just written, not yet reconciled)
|
|
5113
|
+
// gets `status: null`.
|
|
5114
|
+
const [state, histBySlug] = await Promise.all([
|
|
5115
|
+
readQueue(),
|
|
5116
|
+
queueHistory.historyTerminalBySlug().catch(() => new Map()),
|
|
5117
|
+
]);
|
|
5118
|
+
const liveStatusBySlug = new Map(state.jobs.map((j) => [j.slug, j.status]));
|
|
5119
|
+
for (let i = 0; i < out.length; i++) {
|
|
5120
|
+
const entry = out[i];
|
|
5121
|
+
// `entry` is a freshly-synthesized PRD-listing row, not a persisted
|
|
5122
|
+
// ScheduleJob — assigning its `status` here is not a queue-job status
|
|
5123
|
+
// transition (no queue.json row is mutated, no statusHistory/audit
|
|
5124
|
+
// trail applies), so it is intentionally exempt from the
|
|
5125
|
+
// transitionJob-only rule enforced by scheduleJobTransitionsGrep.test.cjs.
|
|
5126
|
+
if (i < archivedStart) {
|
|
5127
|
+
entry.status = liveStatusBySlug.get(entry.slug) ?? null;
|
|
5128
|
+
} else {
|
|
5129
|
+
entry.archivedStatus = resolveArchivedPrdStatus(entry.slug, liveStatusBySlug, histBySlug);
|
|
5130
|
+
entry.status = entry.archivedStatus;
|
|
5131
|
+
}
|
|
5132
|
+
}
|
|
5133
|
+
|
|
5134
|
+
out.sort((a, b) => a.slug.localeCompare(b.slug, undefined, { numeric: true }));
|
|
5135
|
+
return out;
|
|
5136
|
+
}
|
|
5137
|
+
|
|
4805
5138
|
// remote — in-process (non-IPC) scheduler accessors, used by prdCreate.cjs
|
|
4806
5139
|
// and other main-process callers. (Named for the retired web-remote relay,
|
|
4807
5140
|
// its original consumer; kept because it still has in-process callers.)
|
|
@@ -4921,7 +5254,7 @@ const remote = {
|
|
|
4921
5254
|
// Best-effort: record the dispatch on the Epic's event chain.
|
|
4922
5255
|
try { await appendPrdCreatedEvent(cwd, epicTrace, slug); } catch { /* trace only */ }
|
|
4923
5256
|
}
|
|
4924
|
-
return { ok: true, bytesWritten: stat.size };
|
|
5257
|
+
return { ok: true, bytesWritten: stat.size, path: resolved, epicId: epicTrace };
|
|
4925
5258
|
} catch (e) {
|
|
4926
5259
|
return { ok: false, error: e?.message ?? 'write failed' };
|
|
4927
5260
|
}
|
|
@@ -4934,7 +5267,7 @@ const remote = {
|
|
|
4934
5267
|
if (idx < 0) return { kind: 'not-found' };
|
|
4935
5268
|
// Terminal-status guard lives in resetJobFields itself; force:true
|
|
4936
5269
|
// threads through to override it.
|
|
4937
|
-
if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true })) {
|
|
5270
|
+
if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true, source: 'remote:resetJob' })) {
|
|
4938
5271
|
return { kind: 'refused' };
|
|
4939
5272
|
}
|
|
4940
5273
|
return { kind: 'ok' };
|
|
@@ -4955,6 +5288,185 @@ const remote = {
|
|
|
4955
5288
|
return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd }));
|
|
4956
5289
|
},
|
|
4957
5290
|
|
|
5291
|
+
// Single queue row lookup, used by cancelJob/updatePrd's status guards and
|
|
5292
|
+
// the admin GET /admin/scheduler/prds?slug= route (PRD 1024).
|
|
5293
|
+
async getJob(slug) {
|
|
5294
|
+
const state = await readQueue();
|
|
5295
|
+
const job = state.jobs.find((j) => j.slug === slug);
|
|
5296
|
+
return job ? { slug: job.slug, title: job.title, status: job.status, cwd: job.cwd, error: job.error ?? null } : null;
|
|
5297
|
+
},
|
|
5298
|
+
|
|
5299
|
+
// Every live+archived PRD across every project (listPrdsInternal, shared
|
|
5300
|
+
// with the renderer's schedule:list-prds IPC handler), filtered by the
|
|
5301
|
+
// admin route's cwd/epicId/status query params.
|
|
5302
|
+
async listPrds(filter = {}) {
|
|
5303
|
+
const all = await listPrdsInternal();
|
|
5304
|
+
return all.filter((entry) => {
|
|
5305
|
+
if (filter.cwd && entry.cwd !== filter.cwd) return false;
|
|
5306
|
+
if (filter.epicId && entry.epicId !== filter.epicId) return false;
|
|
5307
|
+
if (filter.status && entry.status !== filter.status) return false;
|
|
5308
|
+
return true;
|
|
5309
|
+
});
|
|
5310
|
+
},
|
|
5311
|
+
|
|
5312
|
+
// Full body + parsed frontmatter for one PRD, live or archived. Mirrors
|
|
5313
|
+
// readPrd's dir-search + symlink-defense pattern (see that method's
|
|
5314
|
+
// comment) rather than sharing code with it, since readPrd intentionally
|
|
5315
|
+
// returns raw text only and is a much narrower/hotter path (executeJob's
|
|
5316
|
+
// PRD re-reads) that shouldn't grow a second return shape.
|
|
5317
|
+
async getPrdParsed(slug, cwd) {
|
|
5318
|
+
let dir = null;
|
|
5319
|
+
let filePath = null;
|
|
5320
|
+
if (cwd) {
|
|
5321
|
+
for (const d of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
|
|
5322
|
+
const p = safeSlugPathIn(d, slug);
|
|
5323
|
+
if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
|
|
5324
|
+
}
|
|
5325
|
+
if (!filePath) {
|
|
5326
|
+
for (const d of listArchivedPrdDirs(cwd)) {
|
|
5327
|
+
const p = safeSlugPathIn(d, slug);
|
|
5328
|
+
if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
|
|
5329
|
+
}
|
|
5330
|
+
}
|
|
5331
|
+
} else {
|
|
5332
|
+
dir = await findPrdDir(slug);
|
|
5333
|
+
filePath = dir ? safeSlugPathIn(dir, slug) : null;
|
|
5334
|
+
if (!filePath) {
|
|
5335
|
+
for (const d of candidateArchivedPrdsDirs()) {
|
|
5336
|
+
const p = safeSlugPathIn(d, slug);
|
|
5337
|
+
if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
|
|
5338
|
+
}
|
|
5339
|
+
}
|
|
5340
|
+
}
|
|
5341
|
+
if (!filePath) return { ok: false, error: 'invalid slug' };
|
|
5342
|
+
try {
|
|
5343
|
+
// Symlink defense, matching readPrd/writePrd's comment: safeSlugPathIn
|
|
5344
|
+
// is lexical and does not resolve symlinks.
|
|
5345
|
+
const real = await fsp.realpath(filePath);
|
|
5346
|
+
if (!real.startsWith(dir + path.sep)) return { ok: false, error: 'invalid slug' };
|
|
5347
|
+
const [raw, parsed] = await Promise.all([fsp.readFile(real, 'utf8'), prdParser.parsePrdRaw(real)]);
|
|
5348
|
+
return {
|
|
5349
|
+
ok: true,
|
|
5350
|
+
slug: parsed.slug,
|
|
5351
|
+
frontmatter: {
|
|
5352
|
+
title: parsed.title,
|
|
5353
|
+
cwd: parsed.cwd,
|
|
5354
|
+
estimateMinutes: parsed.estimateMinutes,
|
|
5355
|
+
parallelGroup: parsed.parallelGroup,
|
|
5356
|
+
sourcePromptId: parsed.sourcePromptId,
|
|
5357
|
+
sourceTabId: parsed.sourceTabId,
|
|
5358
|
+
epicId: parsed.epicId,
|
|
5359
|
+
dependsOn: parsed.dependsOn,
|
|
5360
|
+
createdVia: parsed.createdVia,
|
|
5361
|
+
issuedAt: parsed.issuedAt,
|
|
5362
|
+
},
|
|
5363
|
+
body: parsed.body,
|
|
5364
|
+
raw,
|
|
5365
|
+
};
|
|
5366
|
+
} catch (e) {
|
|
5367
|
+
return { ok: false, error: e?.message ?? 'read failed' };
|
|
5368
|
+
}
|
|
5369
|
+
},
|
|
5370
|
+
|
|
5371
|
+
// Edits a NOT-yet-running PRD's frontmatter and/or body in place, refusing
|
|
5372
|
+
// once a queue row exists for it and that row is anything but 'pending'
|
|
5373
|
+
// (running/completed/failed/needs_review — editing the spec under a live
|
|
5374
|
+
// or already-finished executor would silently rewrite history). Reuses
|
|
5375
|
+
// prdFrontmatter.cjs's parsePrdFile/serializePrdFile round-trip pair (PRD
|
|
5376
|
+
// 1024) so unrecognized keys (e.g. dependsOn) and untouched recognized
|
|
5377
|
+
// keys' original line formatting survive unchanged.
|
|
5378
|
+
async updatePrd({ slug, cwd, frontmatter, body }) {
|
|
5379
|
+
const job = await this.getJob(slug);
|
|
5380
|
+
// 'quarantined' is also editable: it's the ONLY way a quarantined PRD's
|
|
5381
|
+
// createdVia stamp gets written (the adopt action below), so refusing it
|
|
5382
|
+
// here would make quarantine irreversible through the API.
|
|
5383
|
+
if (job && job.status !== 'pending' && job.status !== 'quarantined') {
|
|
5384
|
+
return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
|
|
5385
|
+
}
|
|
5386
|
+
|
|
5387
|
+
let dir = null;
|
|
5388
|
+
let filePath = null;
|
|
5389
|
+
if (cwd) {
|
|
5390
|
+
for (const d of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
|
|
5391
|
+
const p = safeSlugPathIn(d, slug);
|
|
5392
|
+
if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
|
|
5393
|
+
}
|
|
5394
|
+
} else {
|
|
5395
|
+
dir = await findPrdDir(slug);
|
|
5396
|
+
filePath = dir ? safeSlugPathIn(dir, slug) : null;
|
|
5397
|
+
}
|
|
5398
|
+
if (!filePath) return { ok: false, error: 'PRD not found' };
|
|
5399
|
+
|
|
5400
|
+
let raw;
|
|
5401
|
+
try {
|
|
5402
|
+
// Symlink defense, matching writePrd's comment: safeSlugPathIn is
|
|
5403
|
+
// lexical and does not resolve symlinks. updatePrd is a WRITE path
|
|
5404
|
+
// (unlike getPrdParsed's read-only realpath check), so also reject a
|
|
5405
|
+
// target that is itself already a symlink — a rogue job could plant
|
|
5406
|
+
// one inside the PRDs dir pointing outside the safe root.
|
|
5407
|
+
const real = await fsp.realpath(filePath);
|
|
5408
|
+
if (!real.startsWith(dir + path.sep)) return { ok: false, error: 'invalid slug' };
|
|
5409
|
+
const existing = await fsp.lstat(filePath).catch(() => null);
|
|
5410
|
+
if (existing && existing.isSymbolicLink()) return { ok: false, error: 'invalid slug' };
|
|
5411
|
+
raw = await fsp.readFile(real, 'utf8');
|
|
5412
|
+
} catch (e) {
|
|
5413
|
+
return { ok: false, error: e?.message ?? 'read failed' };
|
|
5414
|
+
}
|
|
5415
|
+
|
|
5416
|
+
const { frontmatter: fm, body: origBody } = parsePrdFile(raw);
|
|
5417
|
+
if (frontmatter) {
|
|
5418
|
+
for (const key of Object.keys(frontmatter)) {
|
|
5419
|
+
if (frontmatter[key] === undefined) continue;
|
|
5420
|
+
fm[key] = frontmatter[key];
|
|
5421
|
+
}
|
|
5422
|
+
}
|
|
5423
|
+
const newBody = body !== undefined ? body : origBody;
|
|
5424
|
+
const newRaw = serializePrdFile(fm, newBody);
|
|
5425
|
+
|
|
5426
|
+
try {
|
|
5427
|
+
await config.writeTextAtomic(filePath, newRaw, { writer: 'scheduler' });
|
|
5428
|
+
const stat = await fsp.stat(filePath);
|
|
5429
|
+
return { ok: true, slug, bytesWritten: stat.size };
|
|
5430
|
+
} catch (e) {
|
|
5431
|
+
return { ok: false, error: e?.message ?? 'write failed' };
|
|
5432
|
+
}
|
|
5433
|
+
},
|
|
5434
|
+
|
|
5435
|
+
// Cancels a job that hasn't finished yet. A 'running' job's process group
|
|
5436
|
+
// is SIGTERM'd (reusing killOrphanClaudePid — the same kill path boot
|
|
5437
|
+
// reconciliation uses for an orphaned running job) before its queue row is
|
|
5438
|
+
// finalized; a 'pending' job has no process to kill. There is no
|
|
5439
|
+
// 'cancelled' status in the closed job-status set (pending/running/
|
|
5440
|
+
// completed/failed/needs_review — see CLAUDE.md's domain model), so a
|
|
5441
|
+
// cancelled job lands in 'failed' with an error naming the cause,
|
|
5442
|
+
// consistent with every other non-success terminal outcome. Refuses a
|
|
5443
|
+
// slug that's already terminal — nothing left to cancel.
|
|
5444
|
+
async cancelJob(slug) {
|
|
5445
|
+
const state = await readQueue();
|
|
5446
|
+
const job = state.jobs.find((j) => j.slug === slug);
|
|
5447
|
+
if (!job) return { ok: false, error: 'not found' };
|
|
5448
|
+
if (job.status === 'completed' || job.status === 'failed' || job.status === 'needs_review') {
|
|
5449
|
+
return { ok: false, error: `job already terminal (status: "${job.status}") — nothing to cancel` };
|
|
5450
|
+
}
|
|
5451
|
+
const wasRunning = job.status === 'running';
|
|
5452
|
+
const pid = job.runtime?.pid;
|
|
5453
|
+
if (wasRunning && pid) {
|
|
5454
|
+
killOrphanClaudePid(pid);
|
|
5455
|
+
}
|
|
5456
|
+
await mutate((s) => {
|
|
5457
|
+
const idx = s.jobs.findIndex((j) => j.slug === slug);
|
|
5458
|
+
if (idx < 0) return;
|
|
5459
|
+
const j = s.jobs[idx];
|
|
5460
|
+
transitionJob(j, 'failed', { reason: 'cancelled via admin API', source: 'remote:cancelJob' });
|
|
5461
|
+
j.error = 'cancelled via admin API';
|
|
5462
|
+
j.finishedAt = new Date().toISOString();
|
|
5463
|
+
j.exitCode = j.exitCode ?? null;
|
|
5464
|
+
delete j.runtime;
|
|
5465
|
+
});
|
|
5466
|
+
await broadcast({ flush: true });
|
|
5467
|
+
return { ok: true, slug, status: 'failed', wasRunning, cwd: job.cwd ?? null };
|
|
5468
|
+
},
|
|
5469
|
+
|
|
4958
5470
|
// Exposes the module-level allocateParallelGroup (PRD 548) to callers that
|
|
4959
5471
|
// only hold the `remote` object (lib/prdCreate.cjs's create-prd route) —
|
|
4960
5472
|
// reuses the same allocator the file-based /develop authoring path relies
|
|
@@ -4995,4 +5507,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
4995
5507
|
});
|
|
4996
5508
|
}
|
|
4997
5509
|
|
|
4998
|
-
module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveNotifyPrd, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob };
|
|
5510
|
+
module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary };
|