claude-code-session-manager 0.65.0 → 0.67.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/AgentLibrary-CiqimwWl.js +3 -0
- package/dist/assets/{History-DB-9-zwc.js → History-DbzjII2Z.js} +2 -2
- package/dist/assets/{Hooks-DnxqMRrR.js → Hooks-C0PTplI_.js} +3 -3
- package/dist/assets/{HostBilko-Cw6JocV7.js → HostBilko-vGc1FYLD.js} +1 -1
- package/dist/assets/{Library-BYSB0dmY.js → Library-4tU4rg2h.js} +1 -1
- package/dist/assets/{ListDetail-BfKqnL0r.js → ListDetail-BGIvkCwZ.js} +1 -1
- package/dist/assets/{MarkdownEditor-Cs-l5JOv.js → MarkdownEditor-ChFdpCam.js} +1 -1
- package/dist/assets/{McpServers-bBqj3hGg.js → McpServers-DsRWWpU3.js} +2 -2
- package/dist/assets/{Memory-DzPGXT5J.js → Memory-DmxoxxxY.js} +6 -6
- package/dist/assets/{Panel-BX6UZ-W0.js → Panel-KC-rv3jS.js} +1 -1
- package/dist/assets/{Permissions-Db7S4T3Z.js → Permissions-DrDjCRFl.js} +3 -3
- package/dist/assets/{Plugins-CIuey9R3.js → Plugins-KexSQwk0.js} +2 -2
- package/dist/assets/{ProvenanceBadge-FPWNWc_-.js → ProvenanceBadge-Rg-jL94y.js} +1 -1
- package/dist/assets/SaveBar-BbEo8U2e.js +1 -0
- package/dist/assets/Scheduler-C4Ti9TCc.js +14 -0
- package/dist/assets/ScopeSwitcher-Di9BmYze.js +1 -0
- package/dist/assets/Settings-CxyhPuYQ.js +3 -0
- package/dist/assets/{SkillReferenceGraph-CprLcOCe.js → SkillReferenceGraph-0q26RHrn.js} +1 -1
- package/dist/assets/Skills-BXkVyCo5.js +3 -0
- package/dist/assets/SystemPrompt-Bwe3NTbi.js +1 -0
- package/dist/assets/{TagLibrary-q66s4N3i.js → TagLibrary-CnwEf6EI.js} +1 -1
- package/dist/assets/{TiptapBody-I22aArc2.js → TiptapBody-faCtLj9L.js} +1 -1
- package/dist/assets/{Toggle-D9sapSAv.js → Toggle-Bw7G_-RR.js} +1 -1
- package/dist/assets/{index-BRwaw_1W.js → index-B6dJ2CsU.js} +666 -665
- package/dist/assets/{index-LlWpj2VJ.css → index-CEnMgeQU.css} +1 -1
- package/dist/assets/{settingsSchema-CTc4qelV.js → settingsSchema-B5C9hZoS.js} +1 -1
- package/dist/index.html +2 -2
- package/package.json +1 -1
- package/plugins/session-manager-dev/skills/develop/SKILL.md +73 -15
- package/plugins/session-manager-dev/skills/ops-sweep/SKILL.md +10 -0
- package/src/main/__tests__/agentLibrary.test.cjs +40 -0
- package/src/main/__tests__/develop-skill-failure-modes.test.cjs +70 -0
- package/src/main/__tests__/flatPrdTickSweep.test.cjs +110 -0
- package/src/main/__tests__/health-per-project-stall.test.cjs +82 -0
- package/src/main/__tests__/prdAdminRouteParity.test.cjs +68 -0
- package/src/main/__tests__/prdAdminRoutes.test.cjs +311 -0
- package/src/main/__tests__/prdCreate.test.cjs +7 -2
- package/src/main/__tests__/prdMigration.test.cjs +17 -0
- package/src/main/__tests__/prdMigrationLegacyAdopt.test.cjs +91 -0
- package/src/main/__tests__/reconcileFlatPrdSweep.test.cjs +109 -0
- package/src/main/__tests__/scheduleJobSchema.test.cjs +127 -0
- package/src/main/__tests__/scheduleJobStatusDrift.test.cjs +65 -0
- package/src/main/__tests__/scheduleJobTransitions.test.cjs +152 -0
- package/src/main/__tests__/scheduleJobTransitionsGrep.test.cjs +59 -0
- package/src/main/__tests__/scheduler-reconcile-invalid-repair.test.cjs +203 -0
- package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +196 -0
- package/src/main/__tests__/scheduler-stall-per-project.test.cjs +108 -0
- package/src/main/agentLibrary.cjs +40 -2
- package/src/main/health.cjs +97 -2
- package/src/main/index.cjs +2 -0
- package/src/main/ipcSchemas.cjs +60 -0
- package/src/main/lib/localAdminHttp.cjs +10 -3
- package/src/main/lib/prdAdminRoutes.cjs +175 -0
- package/src/main/lib/prdCreate.cjs +37 -2
- package/src/main/lib/prdFrontmatter.cjs +179 -1
- package/src/main/lib/prdMigration.cjs +82 -5
- package/src/main/lib/queueStore.cjs +41 -7
- package/src/main/lib/scheduleJobSchema.cjs +114 -0
- package/src/main/lib/scheduleJobTransitions.cjs +164 -0
- package/src/main/lib/schedulerConfig.cjs +7 -0
- package/src/main/scheduler/prdParser.cjs +7 -0
- package/src/main/scheduler.cjs +769 -138
- package/src/preload/api.d.ts +54 -2
- package/src/preload/index.cjs +12 -0
- package/dist/assets/AgentLibrary-DYriNDGf.js +0 -1
- package/dist/assets/Scheduler-X5y252Qw.js +0 -14
- package/dist/assets/ScopeSwitcher-DU7M_q5-.js +0 -1
- package/dist/assets/Settings-DEptXcEY.js +0 -3
- package/dist/assets/Skills-CGN56X1i.js +0 -3
- package/dist/assets/SystemPrompt-DIEK7FpJ.js +0 -1
package/src/main/scheduler.cjs
CHANGED
|
@@ -75,7 +75,11 @@ const {
|
|
|
75
75
|
USAGE_REFRESH_INTERVAL_MS,
|
|
76
76
|
MAX_JOB_DURATION_MS,
|
|
77
77
|
BROADCAST_COALESCE_MS,
|
|
78
|
+
QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
|
|
78
79
|
} = require('./lib/schedulerConfig.cjs');
|
|
80
|
+
const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
|
|
81
|
+
? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
|
|
82
|
+
: QUARANTINE_ESCALATE_MS_DEFAULT;
|
|
79
83
|
const { pickForProject, pickNextBatch, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
|
|
80
84
|
const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
|
|
81
85
|
const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
|
|
@@ -88,7 +92,10 @@ const queueOps = require('./queueOps.cjs');
|
|
|
88
92
|
// home-dir layout.
|
|
89
93
|
const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
|
|
90
94
|
const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
|
|
95
|
+
const { transitionJob } = require('./lib/scheduleJobTransitions.cjs');
|
|
91
96
|
const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
|
|
97
|
+
const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
|
|
98
|
+
const { appendAuditEvent } = require('./lib/auditLog.cjs');
|
|
92
99
|
|
|
93
100
|
// ---------- origin session resolution (PRD 832) ----------
|
|
94
101
|
// An Epic IS a tagged claude session — job rows carry the originating
|
|
@@ -110,8 +117,8 @@ function resolveOriginSessionId(cwd, epicId) {
|
|
|
110
117
|
const sessionSlots = require('./lib/sessionSlots.cjs');
|
|
111
118
|
const jobWorktree = require('./lib/jobWorktree.cjs');
|
|
112
119
|
const queueStore = require('./lib/queueStore.cjs');
|
|
113
|
-
const { splitFrontmatter } = require('./lib/prdFrontmatter.cjs');
|
|
114
|
-
const { migratePrds, consolidateFlatPrds } = require('./lib/prdMigration.cjs');
|
|
120
|
+
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
121
|
+
const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
|
|
115
122
|
const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
|
|
116
123
|
|
|
117
124
|
// Captured once at module load so every run's meta sidecar can record how
|
|
@@ -720,7 +727,7 @@ async function retireCompletedSlugs(slugs) {
|
|
|
720
727
|
for (const j of s.jobs) {
|
|
721
728
|
if (!j || !slugSet.has(j.slug)) continue;
|
|
722
729
|
if (j.status !== 'pending' && j.status !== 'running') continue;
|
|
723
|
-
j
|
|
730
|
+
if (!transitionJob(j, 'completed', { reason: 'manual archive of an already-shipped PRD', source: 'retireCompletedSlugs' })) continue;
|
|
724
731
|
j.finishedAt = new Date().toISOString();
|
|
725
732
|
j.exitCode = 0;
|
|
726
733
|
j.error = null;
|
|
@@ -756,6 +763,50 @@ function ensureDirs() {
|
|
|
756
763
|
* unparseable cwd, cwd not on disk) are left in place and logged as a
|
|
757
764
|
* warning — never silently dropped — so a human can fix the frontmatter.
|
|
758
765
|
*/
|
|
766
|
+
/**
|
|
767
|
+
* consolidateAllFlatPrds(cwds) — run consolidateFlatPrds() over every given
|
|
768
|
+
* project cwd, logging outcomes. Called from TWO places: once at boot (over
|
|
769
|
+
* every historical project, via runPrdMigration below) AND at the top of
|
|
770
|
+
* every reconcile() call (over every project reconcile itself would
|
|
771
|
+
* otherwise scan), BEFORE reconcile scans the flat dir for PRD sources. The
|
|
772
|
+
* reconcile()-level call is what makes "anything written to the retired flat
|
|
773
|
+
* prds/ dir is swept into prds-archived/ without being executed" actually
|
|
774
|
+
* true regardless of which of reconcile's several callers (tickQueue's poll,
|
|
775
|
+
* job completion, the schedule:state/schedule:rescan IPC handlers,
|
|
776
|
+
* rescheduleTimer) triggers the pass: a PRD dropped in the flat dir has no
|
|
777
|
+
* queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
|
|
778
|
+
* sweep archives it before reconcile can ever turn it into a pending job.
|
|
779
|
+
*/
|
|
780
|
+
async function consolidateAllFlatPrds(cwds) {
|
|
781
|
+
for (const cwd of cwds) {
|
|
782
|
+
try {
|
|
783
|
+
const c = await consolidateFlatPrds(cwd);
|
|
784
|
+
if (c.moved > 0) {
|
|
785
|
+
console.log(`[scheduler] flat-PRD consolidation: archived ${c.moved} file(s) in ${cwd}`);
|
|
786
|
+
}
|
|
787
|
+
for (const f of c.failed) {
|
|
788
|
+
logs.writeLine({
|
|
789
|
+
level: 'warn', scope: 'scheduler',
|
|
790
|
+
message: `flat-PRD consolidation: could not archive ${f.file}`,
|
|
791
|
+
meta: { cwd, reason: f.reason },
|
|
792
|
+
});
|
|
793
|
+
}
|
|
794
|
+
// Deliberately left behind because a live job still points at them
|
|
795
|
+
// (PRD 992). Logged so a permanently-stuck flat PRD is visible rather
|
|
796
|
+
// than looking like a clean consolidation.
|
|
797
|
+
for (const s of c.skipped ?? []) {
|
|
798
|
+
logs.writeLine({
|
|
799
|
+
level: 'info', scope: 'scheduler',
|
|
800
|
+
message: `flat-PRD consolidation: left ${s.file} in place`,
|
|
801
|
+
meta: { cwd, reason: s.reason },
|
|
802
|
+
});
|
|
803
|
+
}
|
|
804
|
+
} catch (e) {
|
|
805
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'flat-PRD consolidation failed', meta: { cwd, error: e?.message } });
|
|
806
|
+
}
|
|
807
|
+
}
|
|
808
|
+
}
|
|
809
|
+
|
|
759
810
|
async function runPrdMigration() {
|
|
760
811
|
let result;
|
|
761
812
|
try {
|
|
@@ -782,33 +833,30 @@ async function runPrdMigration() {
|
|
|
782
833
|
// still sitting flat consolidates into `prds-archived/` for later special
|
|
783
834
|
// processing. Queue rows for moved files are reaped by the archived-twin
|
|
784
835
|
// retirement. Idempotent per project; failures are logged, never fatal.
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
message: `flat-PRD consolidation: left ${s.file} in place`,
|
|
805
|
-
meta: { cwd, reason: s.reason },
|
|
806
|
-
});
|
|
807
|
-
}
|
|
808
|
-
} catch (e) {
|
|
809
|
-
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'flat-PRD consolidation failed', meta: { cwd, error: e?.message } });
|
|
836
|
+
// (This boot-time pass is redundant with the one reconcile() now also runs
|
|
837
|
+
// on every pass, but stays here so a fresh boot's very first log line
|
|
838
|
+
// still reports the initial sweep — see consolidateAllFlatPrds's own
|
|
839
|
+
// comment for why reconcile() is the load-bearing call site.)
|
|
840
|
+
await consolidateAllFlatPrds(allProjectCwds());
|
|
841
|
+
|
|
842
|
+
// Rollout migration for the PRD-authoring-lockdown feature: stamp every
|
|
843
|
+
// pre-existing PRD as legacy-adopted BEFORE reconcile() ever runs its
|
|
844
|
+
// provenance gate against it. Must run every boot (idempotent, cheap
|
|
845
|
+
// scan-and-skip) rather than once — a project opened for the first time
|
|
846
|
+
// after this shipped still has pre-existing unstamped PRDs the very first
|
|
847
|
+
// time reconcile() sees them.
|
|
848
|
+
try {
|
|
849
|
+
const adopted = await legacyAdoptExistingPrds();
|
|
850
|
+
if (adopted.stamped > 0) {
|
|
851
|
+
console.log(`[scheduler] legacy-adopt migration: stamped ${adopted.stamped} pre-existing PRD(s) as createdVia=legacy-adopted`);
|
|
852
|
+
}
|
|
853
|
+
for (const f of adopted.failed) {
|
|
854
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'legacy-adopt migration: could not stamp PRD', meta: f });
|
|
810
855
|
}
|
|
856
|
+
} catch (e) {
|
|
857
|
+
logs.writeLine({ level: 'error', scope: 'scheduler', message: 'legacy-adopt migration failed', meta: { error: e?.message } });
|
|
811
858
|
}
|
|
859
|
+
|
|
812
860
|
return result;
|
|
813
861
|
}
|
|
814
862
|
|
|
@@ -913,6 +961,88 @@ function appendHeartbeat(entry) {
|
|
|
913
961
|
}
|
|
914
962
|
}
|
|
915
963
|
|
|
964
|
+
/**
|
|
965
|
+
* computeStallSummary(state) → { stalled, total, running, pending, byProject }
|
|
966
|
+
*
|
|
967
|
+
* Pure, no IO. `state` is a merged queue-store read ({ jobs, invalidJobs,
|
|
968
|
+
* paused }). The engine (reconcile/reaper/auto-fix/reverify) already
|
|
969
|
+
* operates machine-wide via queueStore's stateCwds() — this function is
|
|
970
|
+
* MONITORING, and monitoring must not collapse per-project reality into one
|
|
971
|
+
* boolean. `stalled` (top-level) is the pre-existing machine-wide roll-up:
|
|
972
|
+
* the queue holds work — valid rows OR rows quarantined for an invalid
|
|
973
|
+
* status — but nothing anywhere is running or pending and the scheduler
|
|
974
|
+
* isn't paused. The 2026-08-07 incident sat exactly in this state for 4+
|
|
975
|
+
* hours: 2 jobs, 0 running, 0 pending, and the only visible symptom was a
|
|
976
|
+
* heartbeat `counts` object that had silently minted a `queued` bucket
|
|
977
|
+
* instead of reporting anything actionable.
|
|
978
|
+
*
|
|
979
|
+
* `byProject[cwd].stalled` is the PER-PROJECT verdict added for the
|
|
980
|
+
* "burrow went dark while other projects were busy" gap: a project can hold
|
|
981
|
+
* jobs (including ones parked `quarantined`) with 0 running and 0 pending
|
|
982
|
+
* while the machine-wide `stalled` above reads false because a different
|
|
983
|
+
* project has running/pending work. Each project's own status counts
|
|
984
|
+
* (`byProject[cwd][status]`) already summed to a total before this — the
|
|
985
|
+
* fix is only the boolean, not the counting.
|
|
986
|
+
*/
|
|
987
|
+
function computeStallSummary(state) {
|
|
988
|
+
const jobs = Array.isArray(state?.jobs) ? state.jobs : [];
|
|
989
|
+
const invalidJobs = Array.isArray(state?.invalidJobs) ? state.invalidJobs : [];
|
|
990
|
+
let running = 0;
|
|
991
|
+
let pending = 0;
|
|
992
|
+
const byProject = {};
|
|
993
|
+
for (const j of jobs) {
|
|
994
|
+
if (j.status === 'running') running += 1;
|
|
995
|
+
if (j.status === 'pending') pending += 1;
|
|
996
|
+
const key = j.cwd || '(unknown)';
|
|
997
|
+
byProject[key] = byProject[key] || {};
|
|
998
|
+
byProject[key][j.status] = (byProject[key][j.status] || 0) + 1;
|
|
999
|
+
}
|
|
1000
|
+
for (const inv of invalidJobs) {
|
|
1001
|
+
const key = inv.row?.cwd || '(unknown)';
|
|
1002
|
+
byProject[key] = byProject[key] || {};
|
|
1003
|
+
byProject[key].invalid = (byProject[key].invalid || 0) + 1;
|
|
1004
|
+
}
|
|
1005
|
+
const total = jobs.length + invalidJobs.length;
|
|
1006
|
+
const stalled = total > 0 && running === 0 && pending === 0 && !state?.paused;
|
|
1007
|
+
for (const key of Object.keys(byProject)) {
|
|
1008
|
+
const counts = byProject[key];
|
|
1009
|
+
const projRunning = counts.running || 0;
|
|
1010
|
+
const projPending = counts.pending || 0;
|
|
1011
|
+
const projTotal = Object.keys(counts)
|
|
1012
|
+
.filter((k) => k !== 'stalled')
|
|
1013
|
+
.reduce((sum, k) => sum + counts[k], 0);
|
|
1014
|
+
counts.stalled = projTotal > 0 && projRunning === 0 && projPending === 0 && !state?.paused;
|
|
1015
|
+
}
|
|
1016
|
+
return { stalled, total, running, pending, byProject };
|
|
1017
|
+
}
|
|
1018
|
+
|
|
1019
|
+
/**
|
|
1020
|
+
* findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
|
|
1021
|
+
*
|
|
1022
|
+
* Pure, no IO. A 'quarantined' row (no createdVia provenance) can otherwise
|
|
1023
|
+
* sit forever with nothing looking at it — quarantine only ever clears via a
|
|
1024
|
+
* human adopting or archiving it. This is the escalation half of that gate:
|
|
1025
|
+
* any quarantined row whose recorded quarantine timestamp (statusHistory's
|
|
1026
|
+
* `to === 'quarantined'` entry — stamped at creation, or backfilled from the
|
|
1027
|
+
* PRD file's mtime by reconcile() for rows quarantined before that stamp
|
|
1028
|
+
* existed) is older than `thresholdMs` is reported so the caller can
|
|
1029
|
+
* warn-log and surface it distinctly. A row with no recoverable timestamp is
|
|
1030
|
+
* skipped rather than guessed at.
|
|
1031
|
+
*/
|
|
1032
|
+
function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
|
|
1033
|
+
const stale = [];
|
|
1034
|
+
for (const j of jobs ?? []) {
|
|
1035
|
+
if (j.status !== 'quarantined') continue;
|
|
1036
|
+
const entry = (j.statusHistory || []).find((h) => h.to === 'quarantined');
|
|
1037
|
+
if (!entry) continue;
|
|
1038
|
+
const since = Date.parse(entry.at);
|
|
1039
|
+
if (Number.isNaN(since)) continue;
|
|
1040
|
+
const ageMs = now - since;
|
|
1041
|
+
if (ageMs >= thresholdMs) stale.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs });
|
|
1042
|
+
}
|
|
1043
|
+
return stale;
|
|
1044
|
+
}
|
|
1045
|
+
|
|
916
1046
|
// An empty queue and an unreadable queue are NOT the same thing, and
|
|
917
1047
|
// conflating them is destructive: reconcile() treats every PRD .md with no
|
|
918
1048
|
// matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
|
|
@@ -1172,6 +1302,15 @@ async function reconcile(state) {
|
|
|
1172
1302
|
if (state && state.unreadable) {
|
|
1173
1303
|
throw new Error(`reconcile skipped: queue.json unreadable (${state.unreadable})`);
|
|
1174
1304
|
}
|
|
1305
|
+
// Sweep the retired flat prds/ dir BEFORE scanning it below. reconcile()
|
|
1306
|
+
// has several callers besides tickQueue's ~60s poll (broadcast,
|
|
1307
|
+
// rescheduleTimer, the schedule:state IPC handler, schedule:rescan) — this
|
|
1308
|
+
// lives here, not in any one caller, so the "a hand-written PRD in the flat
|
|
1309
|
+
// dir is swept before it can become a job" guarantee holds regardless of
|
|
1310
|
+
// which caller triggers this reconcile pass. A freshly hand-written file
|
|
1311
|
+
// has no queue row yet, so it is never "live" and gets archived here
|
|
1312
|
+
// instead of ever reaching the onDisk scan below.
|
|
1313
|
+
await consolidateAllFlatPrds(allProjectCwds());
|
|
1175
1314
|
const files = await listPrdFiles();
|
|
1176
1315
|
const onDisk = new Map();
|
|
1177
1316
|
for (const f of files) {
|
|
@@ -1213,13 +1352,16 @@ async function reconcile(state) {
|
|
|
1213
1352
|
// file may be unreadable, on a project whose dir failed to enumerate,
|
|
1214
1353
|
// or mid-move. "I can't see it" is not "the user deleted it", so the
|
|
1215
1354
|
// row survives — worst case it re-resolves on the next pass.
|
|
1216
|
-
if (job.status === 'pending' || job.status === 'running') {
|
|
1355
|
+
if (job.status === 'pending' || job.status === 'running' || job.status === 'quarantined') {
|
|
1217
1356
|
// Exception: a PENDING row whose PRD has an archived twin was
|
|
1218
1357
|
// retired on purpose (work landed by other means — e.g. implemented
|
|
1219
1358
|
// inline — and the source .md moved to prds-archived/). Keeping it
|
|
1220
1359
|
// would show a phantom "scheduled" job forever; firing it would just
|
|
1221
1360
|
// hit executeJob's archived-twin skip anyway. Running rows are left
|
|
1222
|
-
// alone — the reaper owns their lifecycle.
|
|
1361
|
+
// alone — the reaper owns their lifecycle. A quarantined row's file
|
|
1362
|
+
// going merely-not-visible must survive too — quarantine is meant to
|
|
1363
|
+
// be loud and reversible, never a silent drop (see this function's
|
|
1364
|
+
// header comment on the 2026-08-01 outage a silent skip caused).
|
|
1223
1365
|
if (job.status === 'pending' && (await archivedTwinExists(job))) {
|
|
1224
1366
|
console.log(`[scheduler] reconcile: retiring pending job ${job.slug} — PRD already archived (work landed elsewhere)`);
|
|
1225
1367
|
continue;
|
|
@@ -1233,7 +1375,7 @@ async function reconcile(state) {
|
|
|
1233
1375
|
continue;
|
|
1234
1376
|
}
|
|
1235
1377
|
seen.add(job.slug);
|
|
1236
|
-
|
|
1378
|
+
const updatedJob = {
|
|
1237
1379
|
...job,
|
|
1238
1380
|
title: p.title,
|
|
1239
1381
|
cwd: p.cwd,
|
|
@@ -1249,7 +1391,38 @@ async function reconcile(state) {
|
|
|
1249
1391
|
originSessionId: job.originSessionId
|
|
1250
1392
|
?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
|
|
1251
1393
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1252
|
-
}
|
|
1394
|
+
};
|
|
1395
|
+
// Adopt path: a row parked 'quarantined' (no createdVia provenance when
|
|
1396
|
+
// discovered) whose PRD file now carries a stamp — written via the
|
|
1397
|
+
// update-prd API's adopt patch, either the Scheduler tab's one-click
|
|
1398
|
+
// "adopt PRD" action or a manual scheduler_update_prd call — promotes to
|
|
1399
|
+
// 'pending' the very next reconcile pass. This is the ONLY way a
|
|
1400
|
+
// quarantined row becomes runnable; nothing else in reconcile() clears
|
|
1401
|
+
// that status.
|
|
1402
|
+
if (updatedJob.status === 'quarantined' && p.createdVia) {
|
|
1403
|
+
transitionJob(updatedJob, 'pending', {
|
|
1404
|
+
reason: `adopted via API (createdVia=${p.createdVia})`,
|
|
1405
|
+
source: 'reconcile-adopt',
|
|
1406
|
+
});
|
|
1407
|
+
console.log(`[scheduler] reconcile: adopted quarantined PRD ${job.slug} — createdVia=${p.createdVia}`);
|
|
1408
|
+
appendAuditEvent('scheduler_prd_adopted', { slug: job.slug, cwd: p.cwd, createdVia: p.createdVia, source: 'reconcile' });
|
|
1409
|
+
}
|
|
1410
|
+
// Backfill a quarantine timestamp for rows quarantined before the
|
|
1411
|
+
// statusHistory stamp below existed (e.g. the burrow-project rows
|
|
1412
|
+
// quarantined under the PRD-authoring lockdown) — findStaleQuarantinedJobs
|
|
1413
|
+
// needs SOME timestamp to escalate an un-adopted row past its age
|
|
1414
|
+
// threshold, and the PRD file's own mtime is the best available proxy
|
|
1415
|
+
// for "when this file first showed up unstamped" for a row that has
|
|
1416
|
+
// never been touched since.
|
|
1417
|
+
if (updatedJob.status === 'quarantined' && !(updatedJob.statusHistory || []).some((h) => h.to === 'quarantined')) {
|
|
1418
|
+
try {
|
|
1419
|
+
const at = new Date(fs.statSync(p.path).mtimeMs).toISOString();
|
|
1420
|
+
const history = Array.isArray(updatedJob.statusHistory) ? [...updatedJob.statusHistory] : [];
|
|
1421
|
+
history.push({ from: null, to: 'quarantined', reason: 'backfilled from PRD file mtime', source: 'reconcile-backfill', at });
|
|
1422
|
+
updatedJob.statusHistory = history;
|
|
1423
|
+
} catch { /* best-effort only — a missing/unreadable file just skips the backfill */ }
|
|
1424
|
+
}
|
|
1425
|
+
next.push(updatedJob);
|
|
1253
1426
|
}
|
|
1254
1427
|
// Slugs on disk with no matching state.jobs row are normally brand-new
|
|
1255
1428
|
// PRDs — but once queueHistory.partitionJobs (above, later this same
|
|
@@ -1264,7 +1437,11 @@ async function reconcile(state) {
|
|
|
1264
1437
|
for (const [slug] of onDisk) {
|
|
1265
1438
|
if (!seen.has(slug)) unmatchedSlugs.push(slug);
|
|
1266
1439
|
}
|
|
1267
|
-
|
|
1440
|
+
// Rows quarantined by queueStore.shapeJobs because their `status` failed
|
|
1441
|
+
// ScheduleJobSchema (e.g. the 1021/1022 incident's `"status": "queued"`) —
|
|
1442
|
+
// see the repair pass below, right after historyBySlug is available.
|
|
1443
|
+
const invalidJobs = Array.isArray(state.invalidJobs) ? state.invalidJobs : [];
|
|
1444
|
+
const historyBySlug = (unmatchedSlugs.length > 0 || terminalDroppedNeedingHistoryCheck.length > 0 || invalidJobs.length > 0)
|
|
1268
1445
|
? await queueHistory.historyTerminalBySlug()
|
|
1269
1446
|
: new Map();
|
|
1270
1447
|
|
|
@@ -1282,6 +1459,79 @@ async function reconcile(state) {
|
|
|
1282
1459
|
}
|
|
1283
1460
|
}
|
|
1284
1461
|
|
|
1462
|
+
// Repair pass: an invalid row must self-heal within this one tick, not
|
|
1463
|
+
// wait for its slug to also drop out of `seen` via some unrelated code
|
|
1464
|
+
// path. Before this pass, reconcile was add-only (`if (seen.has(slug))
|
|
1465
|
+
// continue` below) — a quarantined row simply vanished from state.jobs
|
|
1466
|
+
// with no log of what its bad status actually was and no repair, which is
|
|
1467
|
+
// how the 1021/1022 rows sat invisible for 4+ hours (2026-08-07).
|
|
1468
|
+
let repairedInvalidCount = 0;
|
|
1469
|
+
for (const inv of invalidJobs) {
|
|
1470
|
+
if (seen.has(inv.slug)) continue; // a valid row for this slug already exists
|
|
1471
|
+
const oldStatus = inv.row?.status;
|
|
1472
|
+
const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir: RUNS_DIR });
|
|
1473
|
+
if (hist) {
|
|
1474
|
+
// Never resurrect: this slug already has a durable terminal record
|
|
1475
|
+
// elsewhere (history.jsonl or a run sidecar) — repairing its corrupted
|
|
1476
|
+
// row back to 'pending' would re-execute already-shipped work. Drop
|
|
1477
|
+
// the row (its real outcome is recorded elsewhere), loudly.
|
|
1478
|
+
console.warn(`[scheduler] reconcile: dropping invalid queue row ${inv.slug} (status was ${JSON.stringify(oldStatus)}) — already terminal (${hist.status}) in history/run sidecar, not resurrecting`);
|
|
1479
|
+
appendAuditEvent('scheduler_row_repaired', {
|
|
1480
|
+
slug: inv.slug, cwd: inv.row?.cwd ?? null, oldStatus: oldStatus ?? null,
|
|
1481
|
+
action: 'dropped-already-terminal', terminalStatus: hist.status, issues: inv.issues,
|
|
1482
|
+
});
|
|
1483
|
+
continue;
|
|
1484
|
+
}
|
|
1485
|
+
const p = onDisk.get(inv.slug);
|
|
1486
|
+
if (!p) {
|
|
1487
|
+
// PRD file also gone with no terminal record anywhere — nothing to
|
|
1488
|
+
// repair against. queueStore already logged the quarantine once.
|
|
1489
|
+
continue;
|
|
1490
|
+
}
|
|
1491
|
+
const job = {
|
|
1492
|
+
...inv.row,
|
|
1493
|
+
slug: inv.slug,
|
|
1494
|
+
title: p.title,
|
|
1495
|
+
cwd: p.cwd,
|
|
1496
|
+
parallelGroup: p.parallelGroup,
|
|
1497
|
+
estimateMinutes: p.estimateMinutes,
|
|
1498
|
+
sourcePromptId: p.sourcePromptId ?? inv.row?.sourcePromptId ?? null,
|
|
1499
|
+
sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
|
|
1500
|
+
epicId: p.epicId ?? inv.row?.epicId ?? null,
|
|
1501
|
+
dependsOn: p.dependsOn,
|
|
1502
|
+
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
1503
|
+
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
1504
|
+
};
|
|
1505
|
+
const reason = `reconcile: repaired invalid status ${JSON.stringify(oldStatus)}`;
|
|
1506
|
+
// A repair is not a lifecycle transition — the corrupted `status` was
|
|
1507
|
+
// never a legal predecessor to check against LEGAL_TRANSITIONS, so this
|
|
1508
|
+
// goes through transitionJob's allowAnyFrom escape hatch (still gets the
|
|
1509
|
+
// normal mutation/statusHistory/audit trail, just skips the legality
|
|
1510
|
+
// gate on `from`) rather than a bare field assignment.
|
|
1511
|
+
transitionJob(job, 'pending', { reason, source: 'reconcile-repair', allowAnyFrom: true });
|
|
1512
|
+
if (job.runId || job.startedAt || job.runtime) {
|
|
1513
|
+
// This row had actually begun executing before its status got
|
|
1514
|
+
// corrupted.
|
|
1515
|
+
job.runId = null;
|
|
1516
|
+
job.startedAt = null;
|
|
1517
|
+
job.finishedAt = null;
|
|
1518
|
+
job.exitCode = null;
|
|
1519
|
+
delete job.runtime;
|
|
1520
|
+
delete job.verifierVerdict;
|
|
1521
|
+
}
|
|
1522
|
+
job.error = null;
|
|
1523
|
+
seen.add(inv.slug);
|
|
1524
|
+
next.push(job);
|
|
1525
|
+
repairedInvalidCount += 1;
|
|
1526
|
+
console.warn(`[scheduler] reconcile: repaired invalid queue row ${inv.slug} — status was ${JSON.stringify(oldStatus)}, reset to 'pending' (${inv.issues})`);
|
|
1527
|
+
appendAuditEvent('scheduler_row_repaired', {
|
|
1528
|
+
slug: inv.slug, cwd: p.cwd, oldStatus: oldStatus ?? null, newStatus: 'pending', issues: inv.issues,
|
|
1529
|
+
});
|
|
1530
|
+
}
|
|
1531
|
+
if (repairedInvalidCount > 0) {
|
|
1532
|
+
console.warn(`[scheduler] reconcile: repaired ${repairedInvalidCount} invalid queue row(s) this pass`);
|
|
1533
|
+
}
|
|
1534
|
+
|
|
1285
1535
|
// Terminal-in-history slugs whose .md file is still on disk: fed into the
|
|
1286
1536
|
// auto-archive selection pass below (as synthetic completed entries) so
|
|
1287
1537
|
// their file can still be swept, without ever creating a live job row
|
|
@@ -1302,6 +1552,7 @@ async function reconcile(state) {
|
|
|
1302
1552
|
return idx.sessions[epicId]?.status ?? null;
|
|
1303
1553
|
}
|
|
1304
1554
|
|
|
1555
|
+
let staleNewDiscoveryCount = 0;
|
|
1305
1556
|
for (const [slug, p] of onDisk) {
|
|
1306
1557
|
if (seen.has(slug)) continue;
|
|
1307
1558
|
// Security gate: a PRD's file location IS its Epic membership
|
|
@@ -1385,8 +1636,59 @@ async function reconcile(state) {
|
|
|
1385
1636
|
const parent = healTargetForFix(slug, state.jobs);
|
|
1386
1637
|
entry.investigationDepth = parent ? (parent.investigationDepth ?? 1) + 1 : 2;
|
|
1387
1638
|
}
|
|
1639
|
+
// Provenance gate (PRD-authoring lockdown): a PRD discovered with no
|
|
1640
|
+
// `createdVia` stamp was never written through scheduler_create_prd/
|
|
1641
|
+
// chat:create-prd (prdCreate.cjs always stamps 'scheduler-api') or the
|
|
1642
|
+
// legacy-adopt boot migration ('legacy-adopted') — it bypassed the
|
|
1643
|
+
// sanctioned API, most likely via a raw Write/Edit tool call the
|
|
1644
|
+
// guard-prd-writes.cjs PreToolUse hook should have denied. Fix-plan PRDs
|
|
1645
|
+
// are exempt: spawnInvestigation's own probe writes them directly by
|
|
1646
|
+
// design (a trusted, scheduler-spawned internal loop, not an
|
|
1647
|
+
// agent/human authoring a PRD), matching the isFixPlanSlug convention
|
|
1648
|
+
// used everywhere else this distinction matters.
|
|
1649
|
+
//
|
|
1650
|
+
// Quarantine is loud and reversible, never a silent skip (see the
|
|
1651
|
+
// 2026-08-01 23-PRD outage this file's header references for what a
|
|
1652
|
+
// SILENT skip costs): logged at warn, audited, and surfaced in the
|
|
1653
|
+
// Scheduler tab's Quarantined filter with a one-click adopt action
|
|
1654
|
+
// (schedule:adopt-prd) that stamps the file via the same update-prd API
|
|
1655
|
+
// route the MCP tool uses — reconcile()'s adopt path above promotes it
|
|
1656
|
+
// to 'pending' on the very next pass, within one tick of being stamped.
|
|
1657
|
+
if (!p.createdVia && !isFixPlanSlug(slug)) {
|
|
1658
|
+
entry.status = 'quarantined';
|
|
1659
|
+
// Stamped at creation (not via transitionJob, since this is a
|
|
1660
|
+
// brand-new row minted directly at 'quarantined' rather than
|
|
1661
|
+
// transitioning through 'pending') so findStaleQuarantinedJobs has a
|
|
1662
|
+
// real quarantine timestamp to escalate against, instead of only the
|
|
1663
|
+
// reconcile-backfill fallback above.
|
|
1664
|
+
entry.statusHistory = [{
|
|
1665
|
+
from: null,
|
|
1666
|
+
to: 'quarantined',
|
|
1667
|
+
reason: 'missing createdVia provenance frontmatter',
|
|
1668
|
+
source: 'reconcile',
|
|
1669
|
+
at: new Date().toISOString(),
|
|
1670
|
+
}];
|
|
1671
|
+
console.warn(`[scheduler] reconcile: quarantining unstamped PRD ${slug} (${p.path}) — no createdVia provenance; adopt it from the Scheduler tab's Quarantined filter or via scheduler_update_prd to make it runnable`);
|
|
1672
|
+
appendAuditEvent('prd_quarantined', { slug, cwd: p.cwd, path: p.path, reason: 'missing createdVia provenance frontmatter' });
|
|
1673
|
+
}
|
|
1674
|
+
// A PRD with no queue row and no terminal record is normally a
|
|
1675
|
+
// brand-new file — but one whose mtime already predates a full poll
|
|
1676
|
+
// interval means it sat unpicked (a prior reconcile pass should have
|
|
1677
|
+
// caught it, or it's arriving from a source that bypassed the app's
|
|
1678
|
+
// normal write path). Report it rather than silently treating "first
|
|
1679
|
+
// seen this pass" as "just created".
|
|
1680
|
+
try {
|
|
1681
|
+
const ageMs = Date.now() - fs.statSync(p.path).mtimeMs;
|
|
1682
|
+
if (ageMs > POLL_INTERVAL_MS) {
|
|
1683
|
+
staleNewDiscoveryCount += 1;
|
|
1684
|
+
console.warn(`[scheduler] reconcile: discovered PRD ${slug} with no queue row and no terminal record — file is ${Math.round(ageMs / 1000)}s old, only first seen this pass`);
|
|
1685
|
+
}
|
|
1686
|
+
} catch { /* stat is best-effort reporting only */ }
|
|
1388
1687
|
next.push(entry);
|
|
1389
1688
|
}
|
|
1689
|
+
if (staleNewDiscoveryCount > 0) {
|
|
1690
|
+
console.warn(`[scheduler] reconcile: ${staleNewDiscoveryCount} PRD(s) discovered this pass were already older than one poll interval with no prior queue row`);
|
|
1691
|
+
}
|
|
1390
1692
|
const sorted = next.sort((a, b) => b.slug.localeCompare(a.slug));
|
|
1391
1693
|
|
|
1392
1694
|
// Move terminal jobs past the retention window out to history.jsonl so
|
|
@@ -1467,6 +1769,16 @@ let resumeTimer = null;
|
|
|
1467
1769
|
let pollLoopTimer = null;
|
|
1468
1770
|
let rescheduleInterval = null;
|
|
1469
1771
|
let heartbeatInterval = null;
|
|
1772
|
+
// Stall-detector state (computeStallSummary), read/written only inside the
|
|
1773
|
+
// heartbeat interval below. Keyed per-project cwd (never a single value) —
|
|
1774
|
+
// a single module-level flag would let one busy project's activity clear or
|
|
1775
|
+
// suppress another stalled project's alert. stallSince.get(cwd): wall-clock
|
|
1776
|
+
// ms that project's stalled condition was first observed, absent when clear.
|
|
1777
|
+
// stallToasted.get(cwd): rate-limits that project's error-log + toast to
|
|
1778
|
+
// once per stall episode (cleared the moment that project stops being
|
|
1779
|
+
// stalled) rather than every 60s heartbeat tick.
|
|
1780
|
+
let stallSince = new Map();
|
|
1781
|
+
let stallToasted = new Map();
|
|
1470
1782
|
// (The 5-minute feedback sweep that used to piggyback on this heartbeat is
|
|
1471
1783
|
// gone: it scanned each active project's session-manager-operations/feedback/
|
|
1472
1784
|
// and auto-queued a /process-feedback PRD. Both the folder and that skill are
|
|
@@ -1738,7 +2050,7 @@ async function clearPause(source) {
|
|
|
1738
2050
|
*/
|
|
1739
2051
|
function resetJobFields(job, errorMsg, opts = {}) {
|
|
1740
2052
|
if (job.status === 'completed' && opts.force !== true) return false;
|
|
1741
|
-
job
|
|
2053
|
+
if (!transitionJob(job, 'pending', { reason: errorMsg ?? 'reset to pending', source: opts.source ?? 'resetJobFields' })) return false;
|
|
1742
2054
|
job.runId = null;
|
|
1743
2055
|
job.startedAt = null;
|
|
1744
2056
|
job.finishedAt = null;
|
|
@@ -1799,13 +2111,13 @@ function partitionBootOrphans(jobs, isAlive = claudePidAlive) {
|
|
|
1799
2111
|
function applyOrphanOutcome(job, outcome, killNote = '') {
|
|
1800
2112
|
const now = new Date().toISOString();
|
|
1801
2113
|
if (outcome === 'success') {
|
|
1802
|
-
job
|
|
2114
|
+
transitionJob(job, 'completed', { reason: 'boot orphan reconciliation: run succeeded', source: 'applyOrphanOutcome' });
|
|
1803
2115
|
job.exitCode = 0;
|
|
1804
2116
|
job.error = null;
|
|
1805
2117
|
job.finishedAt = now;
|
|
1806
2118
|
delete job.runtime;
|
|
1807
2119
|
} else if (outcome === 'failed') {
|
|
1808
|
-
job
|
|
2120
|
+
transitionJob(job, 'failed', { reason: `orphaned: app restarted while running${killNote}`, source: 'applyOrphanOutcome' });
|
|
1809
2121
|
job.exitCode = job.exitCode ?? 1;
|
|
1810
2122
|
job.error = `orphaned: app restarted while running${killNote}`;
|
|
1811
2123
|
job.finishedAt = now;
|
|
@@ -1813,10 +2125,10 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
|
|
|
1813
2125
|
} else {
|
|
1814
2126
|
const tries = job.orphanRetries ?? 0;
|
|
1815
2127
|
if (tries < ORPHAN_REQUEUE_CAP) {
|
|
1816
|
-
resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}
|
|
2128
|
+
resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}`, { source: 'applyOrphanOutcome' });
|
|
1817
2129
|
job.orphanRetries = tries + 1;
|
|
1818
2130
|
} else {
|
|
1819
|
-
job
|
|
2131
|
+
transitionJob(job, 'failed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`, source: 'applyOrphanOutcome' });
|
|
1820
2132
|
job.exitCode = job.exitCode ?? 1;
|
|
1821
2133
|
job.error = `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`;
|
|
1822
2134
|
job.finishedAt = now;
|
|
@@ -2860,7 +3172,7 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
2860
3172
|
// "nothing is happening" even though an Opus process was actively running.
|
|
2861
3173
|
await mutate((s) => {
|
|
2862
3174
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
2863
|
-
if (j) j
|
|
3175
|
+
if (j) transitionJob(j, 'investigating', { reason: 'spawning investigation probe', source: 'spawnInvestigation:start' });
|
|
2864
3176
|
});
|
|
2865
3177
|
await broadcast({ flush: true });
|
|
2866
3178
|
|
|
@@ -2905,7 +3217,7 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
2905
3217
|
// 'investigating' must never be the job's resting state.
|
|
2906
3218
|
mutate((s) => {
|
|
2907
3219
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
2908
|
-
if (j && j.status === 'investigating') j
|
|
3220
|
+
if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation probe exited — restoring prior status', source: 'spawnInvestigation:onExit' });
|
|
2909
3221
|
})
|
|
2910
3222
|
.then(() => broadcast({ flush: true }))
|
|
2911
3223
|
.catch(() => {});
|
|
@@ -2960,7 +3272,7 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
2960
3272
|
releaseSlot();
|
|
2961
3273
|
mutate((s) => {
|
|
2962
3274
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
2963
|
-
if (j && j.status === 'investigating') j
|
|
3275
|
+
if (j && j.status === 'investigating') transitionJob(j, failedJob.status || 'failed', { reason: 'investigation spawn threw before exiting — restoring prior status', source: 'spawnInvestigation:catch' });
|
|
2964
3276
|
})
|
|
2965
3277
|
.then(() => broadcast({ flush: true }))
|
|
2966
3278
|
.catch(() => {});
|
|
@@ -2982,7 +3294,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
2982
3294
|
await mutate((s) => {
|
|
2983
3295
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
2984
3296
|
if (idx >= 0) {
|
|
2985
|
-
s.jobs[idx]
|
|
3297
|
+
transitionJob(s.jobs[idx], 'running', { reason: 'dispatched for execution', source: 'spawnJob:dispatch' });
|
|
2986
3298
|
s.jobs[idx].runId = runId;
|
|
2987
3299
|
s.jobs[idx].startedAt = new Date().toISOString();
|
|
2988
3300
|
}
|
|
@@ -3069,7 +3381,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3069
3381
|
await mutate((s) => {
|
|
3070
3382
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3071
3383
|
if (idx >= 0) {
|
|
3072
|
-
s.jobs[idx].
|
|
3384
|
+
transitionJob(s.jobs[idx], 'completed', { reason: res.note ?? 'PRD archived or missing — treated as already-shipped', source: 'spawnJob:skip-archived' });
|
|
3073
3385
|
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
3074
3386
|
s.jobs[idx].exitCode = 0;
|
|
3075
3387
|
s.jobs[idx].error = null;
|
|
@@ -3239,10 +3551,19 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3239
3551
|
const newlyCompletedPrds = [];
|
|
3240
3552
|
await mutate((s) => {
|
|
3241
3553
|
const i2 = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3554
|
+
// A job already moved off 'running' by someone else (namely
|
|
3555
|
+
// remote.cancelJob, PRD 1024 — it SIGTERMs the process then finalizes
|
|
3556
|
+
// the row to 'failed' before this exit handler necessarily runs) is
|
|
3557
|
+
// not this run's to finalize: doing so anyway could re-legalize the
|
|
3558
|
+
// row via a legal failed->completed/needs_review edge (see
|
|
3559
|
+
// scheduleJobTransitions.cjs's LEGAL_TRANSITIONS) and silently
|
|
3560
|
+
// undo the cancellation. Skip — the row already reflects its real
|
|
3561
|
+
// terminal state.
|
|
3562
|
+
if (i2 >= 0 && s.jobs[i2].status !== 'running') return;
|
|
3242
3563
|
if (i2 >= 0) {
|
|
3243
3564
|
const treatAsPending = res.rateLimited || (s.paused && s.paused.reason === 'rate_limit');
|
|
3244
3565
|
if (treatAsPending) {
|
|
3245
|
-
resetJobFields(s.jobs[i2], res.rateLimited ? 'paused: rate limit' : 'paused: queue halted');
|
|
3566
|
+
resetJobFields(s.jobs[i2], res.rateLimited ? 'paused: rate limit' : 'paused: queue halted', { source: 'spawnJob:halt-reset' });
|
|
3246
3567
|
} else {
|
|
3247
3568
|
// Determine effective status, applying the verifier verdict for exit=0 runs.
|
|
3248
3569
|
let effectiveStatus;
|
|
@@ -3262,14 +3583,14 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3262
3583
|
effectiveStatus = 'completed';
|
|
3263
3584
|
} else if (verifyResult.downgradeTo === 'pending') {
|
|
3264
3585
|
// HALT or deps_unmet: reset to pending so the job re-fires.
|
|
3265
|
-
resetJobFields(s.jobs[i2], verifyResult.reason);
|
|
3586
|
+
resetJobFields(s.jobs[i2], verifyResult.reason, { source: 'spawnJob:verify-downgrade' });
|
|
3266
3587
|
return; // job already mutated by resetJobFields; skip the rest
|
|
3267
3588
|
} else {
|
|
3268
3589
|
// transcript_errors or verify_unavailable: escalate to needs_review.
|
|
3269
3590
|
effectiveStatus = 'needs_review';
|
|
3270
3591
|
}
|
|
3271
3592
|
|
|
3272
|
-
s.jobs[i2].
|
|
3593
|
+
transitionJob(s.jobs[i2], effectiveStatus, { reason: sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`, source: 'spawnJob:finalize' });
|
|
3273
3594
|
s.jobs[i2].finishedAt = new Date().toISOString();
|
|
3274
3595
|
s.jobs[i2].exitCode = res.exitCode;
|
|
3275
3596
|
s.jobs[i2].error = effectiveStatus === 'needs_review'
|
|
@@ -3356,7 +3677,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3356
3677
|
if (orig) {
|
|
3357
3678
|
const priorStatus = orig.status;
|
|
3358
3679
|
console.log(`[scheduler] auto-promote: ${orig.slug} (${priorStatus}) → completed because ${job.slug} succeeded`);
|
|
3359
|
-
orig.
|
|
3680
|
+
transitionJob(orig, 'completed', { reason: `auto-promoted: fix plan ${job.slug} succeeded`, source: 'spawnJob:auto-promote' });
|
|
3360
3681
|
orig.exitCode = 0;
|
|
3361
3682
|
orig.error = null;
|
|
3362
3683
|
orig.completedBy = job.slug;
|
|
@@ -3444,7 +3765,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3444
3765
|
await mutate((s) => {
|
|
3445
3766
|
const i = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3446
3767
|
if (i >= 0) {
|
|
3447
|
-
resetJobFields(s.jobs[i], null);
|
|
3768
|
+
resetJobFields(s.jobs[i], null, { source: 'spawnJob:transient-retry' });
|
|
3448
3769
|
s.jobs[i].transientRetries = decision.retries + 1;
|
|
3449
3770
|
}
|
|
3450
3771
|
});
|
|
@@ -3454,7 +3775,7 @@ async function spawnJob(job, runId, runDir, defaultCwd) {
|
|
|
3454
3775
|
await mutate((s) => {
|
|
3455
3776
|
const i = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
3456
3777
|
if (i >= 0) {
|
|
3457
|
-
s.jobs[i].
|
|
3778
|
+
transitionJob(s.jobs[i], 'failed', { reason: `transient failure (${decision.transientKind}) left uncommitted work — not auto-requeued`, source: 'spawnJob:fail-dirty' });
|
|
3458
3779
|
s.jobs[i].error = `transient failure (${decision.transientKind}) left ${newlyDirtyCount} uncommitted file(s) in working tree (e.g. ${dirtySample}) — not auto-requeued to avoid overwriting partial work; review and commit or discard manually`;
|
|
3459
3780
|
}
|
|
3460
3781
|
});
|
|
@@ -3515,6 +3836,9 @@ function tickQueue() {
|
|
|
3515
3836
|
}
|
|
3516
3837
|
if (cancelToken.cancelled) return { fired: false, reason: 'cancelled' };
|
|
3517
3838
|
|
|
3839
|
+
// The retired-flat-dir sweep now lives inside reconcile() itself (see its
|
|
3840
|
+
// own comment) so every caller of reconcile — not just this tick — gets
|
|
3841
|
+
// the guarantee.
|
|
3518
3842
|
await reconcile(state);
|
|
3519
3843
|
// Session-Manager's machine-wide slot pool is the ONLY concurrency limit
|
|
3520
3844
|
// the picker answers to (plus the memory gate below). The scheduler used
|
|
@@ -3702,7 +4026,7 @@ async function reapDeadRunningJobs() {
|
|
|
3702
4026
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
3703
4027
|
if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
|
|
3704
4028
|
const success = outcome === 'success';
|
|
3705
|
-
s.jobs[idx]
|
|
4029
|
+
transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: `reaped: process gone (outcome=${outcome})`, source: 'reapDeadRunningJobs' });
|
|
3706
4030
|
s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
|
|
3707
4031
|
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
3708
4032
|
s.jobs[idx].error = success ? null : `reaped: process gone, no success result in log (${outcome})`;
|
|
@@ -4125,7 +4449,7 @@ async function reverifyNeedsReview() {
|
|
|
4125
4449
|
await mutate((s) => {
|
|
4126
4450
|
for (const j of s.jobs) {
|
|
4127
4451
|
if (j.status === 'needs_review' && healSet.has(j.slug)) {
|
|
4128
|
-
j
|
|
4452
|
+
transitionJob(j, 'completed', { reason: 'boot reverify: stale needs_review healed', source: 'reverifyNeedsReview:heal' });
|
|
4129
4453
|
j.error = null;
|
|
4130
4454
|
delete j.verifierVerdict;
|
|
4131
4455
|
healedPrds.push({ slug: j.slug, cwd: j.cwd });
|
|
@@ -4163,7 +4487,7 @@ async function reverifyNeedsReview() {
|
|
|
4163
4487
|
const orig = healTargetForFix(job.slug, s.jobs);
|
|
4164
4488
|
if (!orig) continue;
|
|
4165
4489
|
const priorStatus = orig.status;
|
|
4166
|
-
orig.
|
|
4490
|
+
transitionJob(orig, 'completed', { reason: `auto-promoted: fix plan ${job.slug} already completed`, source: 'reverifyNeedsReview:auto-promote' });
|
|
4167
4491
|
orig.exitCode = 0;
|
|
4168
4492
|
orig.error = null;
|
|
4169
4493
|
orig.completedBy = job.slug;
|
|
@@ -4341,7 +4665,7 @@ function registerScheduleHandlers() {
|
|
|
4341
4665
|
// Guard is in resetJobFields: refuses to reset an already-'completed'
|
|
4342
4666
|
// job, which would otherwise re-fire a PRD whose deliverable already
|
|
4343
4667
|
// landed (see resetJobFields' doc comment for the incident).
|
|
4344
|
-
return resetJobFields(state.jobs[idx]) ? 'ok' : 'refused';
|
|
4668
|
+
return resetJobFields(state.jobs[idx], null, { source: 'ipc:schedule:reset-job' }) ? 'ok' : 'refused';
|
|
4345
4669
|
});
|
|
4346
4670
|
if (outcome === 'not-found') return { ok: false, error: 'not found' };
|
|
4347
4671
|
if (outcome === 'refused') {
|
|
@@ -4354,6 +4678,36 @@ function registerScheduleHandlers() {
|
|
|
4354
4678
|
return { ok: true };
|
|
4355
4679
|
}));
|
|
4356
4680
|
|
|
4681
|
+
// Renderer-facing counterpart to prdCreate.cjs's chat:create-prd handler
|
|
4682
|
+
// (index.cjs): calls the SAME remote.updatePrd the admin HTTP route/MCP
|
|
4683
|
+
// tool use, so "stamps it through the API" holds for the Scheduler tab's
|
|
4684
|
+
// one-click adopt action too, not just a direct fs write. Only a
|
|
4685
|
+
// 'quarantined' row is eligible — see reconcile()'s provenance gate.
|
|
4686
|
+
ipcMain.handle('schedule:adopt-prd', validated(schemas.scheduleSlug, async ({ slug }) => {
|
|
4687
|
+
if (!(await safeSlugPath(slug))) return { ok: false, kind: 'error', message: 'invalid slug' };
|
|
4688
|
+
const state = await readQueue();
|
|
4689
|
+
const job = state.jobs.find((j) => j.slug === slug);
|
|
4690
|
+
if (!job) return { ok: false, kind: 'error', message: 'not found' };
|
|
4691
|
+
if (job.status !== 'quarantined') {
|
|
4692
|
+
return { ok: false, kind: 'error', message: `job status is "${job.status}" — only a quarantined PRD may be adopted` };
|
|
4693
|
+
}
|
|
4694
|
+
const result = await remote.updatePrd({
|
|
4695
|
+
slug,
|
|
4696
|
+
cwd: job.cwd,
|
|
4697
|
+
frontmatter: { createdVia: 'legacy-adopted', issuedAt: new Date().toISOString() },
|
|
4698
|
+
});
|
|
4699
|
+
if (!result.ok) return { ok: false, kind: 'error', message: result.error ?? 'adopt failed' };
|
|
4700
|
+
appendAuditEvent('scheduler_prd_adopted', { slug, cwd: job.cwd ?? null, source: 'ipc:schedule:adopt-prd' });
|
|
4701
|
+
// Promote the row to 'pending' immediately rather than waiting for the
|
|
4702
|
+
// next poll tick — the Scheduler tab's "adopt PRD" click should be
|
|
4703
|
+
// visibly effective within this one round-trip.
|
|
4704
|
+
const freshState = await readQueue();
|
|
4705
|
+
await reconcile(freshState);
|
|
4706
|
+
await writeQueue(freshState);
|
|
4707
|
+
await broadcast({ flush: true });
|
|
4708
|
+
return { ok: true, kind: 'info', message: `Adopted ${slug} — it will run as a normal pending job` };
|
|
4709
|
+
}));
|
|
4710
|
+
|
|
4357
4711
|
ipcMain.handle('schedule:run-now', async () => {
|
|
4358
4712
|
// Manual run-now overrides any auto-pause. Clear it first.
|
|
4359
4713
|
await clearPause('run-now');
|
|
@@ -4483,85 +4837,7 @@ function registerScheduleHandlers() {
|
|
|
4483
4837
|
}
|
|
4484
4838
|
}));
|
|
4485
4839
|
|
|
4486
|
-
ipcMain.handle('schedule:list-prds', async () =>
|
|
4487
|
-
ensureDirs();
|
|
4488
|
-
const out = [];
|
|
4489
|
-
const seenSlugs = new Set();
|
|
4490
|
-
|
|
4491
|
-
async function readDirInto(dir, { archived }) {
|
|
4492
|
-
let entries;
|
|
4493
|
-
try {
|
|
4494
|
-
entries = await fsp.readdir(dir);
|
|
4495
|
-
} catch (e) {
|
|
4496
|
-
if (e?.code !== 'ENOENT') {
|
|
4497
|
-
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: readdir failed', meta: { dir, error: e?.message } });
|
|
4498
|
-
}
|
|
4499
|
-
return;
|
|
4500
|
-
}
|
|
4501
|
-
for (const name of entries) {
|
|
4502
|
-
if (!name.endsWith('.md') || name.startsWith('.')) continue;
|
|
4503
|
-
const filePath = path.join(dir, name);
|
|
4504
|
-
try {
|
|
4505
|
-
const parsed = await parsePrd(filePath);
|
|
4506
|
-
// A slug can't be both live and archived at once, but a duplicate
|
|
4507
|
-
// slug found in two archive dirs (shouldn't happen — archiving is
|
|
4508
|
-
// a single rename — but is cheap to guard) is skipped rather than
|
|
4509
|
-
// double-counted.
|
|
4510
|
-
if (seenSlugs.has(parsed.slug)) continue;
|
|
4511
|
-
seenSlugs.add(parsed.slug);
|
|
4512
|
-
const stat = await fsp.stat(filePath);
|
|
4513
|
-
const entry = {
|
|
4514
|
-
slug: parsed.slug,
|
|
4515
|
-
parallelGroup: parsed.parallelGroup,
|
|
4516
|
-
title: parsed.title,
|
|
4517
|
-
cwd: parsed.cwd || '',
|
|
4518
|
-
estimateMinutes: parsed.estimateMinutes,
|
|
4519
|
-
sourcePromptId: parsed.sourcePromptId,
|
|
4520
|
-
epicId: parsed.epicId ?? null,
|
|
4521
|
-
mtimeMs: stat.mtimeMs,
|
|
4522
|
-
archived,
|
|
4523
|
-
};
|
|
4524
|
-
out.push(entry);
|
|
4525
|
-
} catch (e) {
|
|
4526
|
-
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: skipping unparseable file', meta: { name, error: e?.message } });
|
|
4527
|
-
}
|
|
4528
|
-
}
|
|
4529
|
-
}
|
|
4530
|
-
|
|
4531
|
-
// Live PRDs first, so an archived duplicate (shouldn't exist, but a
|
|
4532
|
-
// stale rename copy is possible) never shadows the still-runnable live
|
|
4533
|
-
// entry.
|
|
4534
|
-
for (const dir of candidatePrdsDirs()) {
|
|
4535
|
-
await readDirInto(dir, { archived: false });
|
|
4536
|
-
}
|
|
4537
|
-
|
|
4538
|
-
const archivedStart = out.length;
|
|
4539
|
-
for (const dir of candidateArchivedPrdsDirs()) {
|
|
4540
|
-
await readDirInto(dir, { archived: true });
|
|
4541
|
-
}
|
|
4542
|
-
|
|
4543
|
-
// Archived PRDs need a status: archiveCompletedPrd (scheduler.cjs) only
|
|
4544
|
-
// ever archives a job whose effective status is 'completed' — a 'failed'
|
|
4545
|
-
// job's PRD source stays in the live prds/ dir (still visible/countable
|
|
4546
|
-
// there already). Still resolve the real job status defensively (live
|
|
4547
|
-
// queue row, falling back to history.jsonl) rather than hard-coding
|
|
4548
|
-
// 'completed', so this stays correct if that archiving invariant ever
|
|
4549
|
-
// changes.
|
|
4550
|
-
if (out.length > archivedStart) {
|
|
4551
|
-
const [state, histBySlug] = await Promise.all([
|
|
4552
|
-
readQueue(),
|
|
4553
|
-
queueHistory.historyTerminalBySlug().catch(() => new Map()),
|
|
4554
|
-
]);
|
|
4555
|
-
const liveStatusBySlug = new Map(state.jobs.map((j) => [j.slug, j.status]));
|
|
4556
|
-
for (let i = archivedStart; i < out.length; i++) {
|
|
4557
|
-
const entry = out[i];
|
|
4558
|
-
entry.archivedStatus = resolveArchivedPrdStatus(entry.slug, liveStatusBySlug, histBySlug);
|
|
4559
|
-
}
|
|
4560
|
-
}
|
|
4561
|
-
|
|
4562
|
-
out.sort((a, b) => a.slug.localeCompare(b.slug, undefined, { numeric: true }));
|
|
4563
|
-
return out;
|
|
4564
|
-
});
|
|
4840
|
+
ipcMain.handle('schedule:list-prds', async () => listPrdsInternal());
|
|
4565
4841
|
|
|
4566
4842
|
// Return last N completed/failed jobs from queue.json, newest first.
|
|
4567
4843
|
// Purely additive: no schema change, no archive-folder read needed.
|
|
@@ -4729,6 +5005,7 @@ async function init() {
|
|
|
4729
5005
|
if (rescheduleInterval) clearInterval(rescheduleInterval);
|
|
4730
5006
|
rescheduleInterval = setInterval(() => {
|
|
4731
5007
|
rescheduleTimer().catch(() => {});
|
|
5008
|
+
const s = readQueueSync();
|
|
4732
5009
|
// Periodic self-heal: re-run the verifier over stale needs_review jobs so a
|
|
4733
5010
|
// job whose work actually landed (committed in-window, no FAIL sentinel)
|
|
4734
5011
|
// auto-clears WITHOUT waiting for the next app restart. Cheap-guarded — the
|
|
@@ -4738,10 +5015,32 @@ async function init() {
|
|
|
4738
5015
|
// MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
|
|
4739
5016
|
// past it), so this interval firing cannot fan out investigations.
|
|
4740
5017
|
if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
|
|
4741
|
-
const s = readQueueSync();
|
|
4742
5018
|
if (s.jobs.some((j) => j.status === 'needs_review')) {
|
|
4743
5019
|
reverifyNeedsReview().catch(() => {});
|
|
4744
5020
|
}
|
|
5021
|
+
// A quarantined row only ever promotes to 'pending' through
|
|
5022
|
+
// reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
|
|
5023
|
+
// it re-checks the PRD file's createdVia stamp every pass. broadcast()
|
|
5024
|
+
// already runs reconcile+writeQueue on every normal poll tick, but an
|
|
5025
|
+
// idle queue (nothing pending/running to fire) can back off that
|
|
5026
|
+
// cadence for a long time; this guarantees an adopted-but-still-
|
|
5027
|
+
// quarantined row is re-checked within 10 minutes regardless.
|
|
5028
|
+
if (s.jobs.some((j) => j.status === 'quarantined')) {
|
|
5029
|
+
broadcast().catch(() => {});
|
|
5030
|
+
}
|
|
5031
|
+
}
|
|
5032
|
+
// Age-based escalation (independent of the self-heal kill-switch above —
|
|
5033
|
+
// this is a monitoring signal, not an auto-fix action): a quarantined
|
|
5034
|
+
// row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
|
|
5035
|
+
// warn-logged by project + slug + age so it cannot sit stranded and
|
|
5036
|
+
// silent (the four burrow-project rows this PRD was written against).
|
|
5037
|
+
for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
|
|
5038
|
+
console.warn(
|
|
5039
|
+
`[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
|
|
5040
|
+
+ `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
5041
|
+
+ `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
|
|
5042
|
+
);
|
|
5043
|
+
appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
|
|
4745
5044
|
}
|
|
4746
5045
|
}, 10 * 60_000);
|
|
4747
5046
|
|
|
@@ -4766,12 +5065,69 @@ async function init() {
|
|
|
4766
5065
|
if (heartbeatInterval) clearInterval(heartbeatInterval);
|
|
4767
5066
|
heartbeatInterval = setInterval(() => {
|
|
4768
5067
|
const s = readQueueSync();
|
|
4769
|
-
|
|
4770
|
-
|
|
5068
|
+
// Initialise from the real status union (scheduleJobSchema.cjs) rather
|
|
5069
|
+
// than a hand-maintained subset — the old `{ pending, running, completed,
|
|
5070
|
+
// failed }` literal silently minted a NEW key for any other value
|
|
5071
|
+
// (`counts[j.status] = (counts[j.status]||0)+1`), which is exactly how a
|
|
5072
|
+
// heartbeat with a `queued: 2` bucket looked like "normal" 24h
|
|
5073
|
+
// visibility instead of the alarm it should have been. Any row whose
|
|
5074
|
+
// status isn't in JOB_STATUSES (shouldn't happen post-quarantine, but
|
|
5075
|
+
// this is the last line of defence) routes into `unknown`, never a
|
|
5076
|
+
// freshly-minted key.
|
|
5077
|
+
const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
|
|
5078
|
+
counts.unknown = 0;
|
|
5079
|
+
for (const j of s.jobs) {
|
|
5080
|
+
if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
|
|
5081
|
+
counts[j.status] += 1;
|
|
5082
|
+
} else {
|
|
5083
|
+
counts.unknown += 1;
|
|
5084
|
+
}
|
|
5085
|
+
}
|
|
5086
|
+
|
|
5087
|
+
const stall = computeStallSummary(s);
|
|
5088
|
+
// Per-project alerting (see computeStallSummary's header): a project
|
|
5089
|
+
// stalled while others are busy must still fire, and one project
|
|
5090
|
+
// recovering must not clear or suppress another's still-open episode —
|
|
5091
|
+
// that is exactly what a single module-level stallSince/stallToasted
|
|
5092
|
+
// flag masked before (the burrow-vs-others incident this PRD fixes).
|
|
5093
|
+
const now = Date.now();
|
|
5094
|
+
const stalledCwds = Object.keys(stall.byProject).filter((cwd) => stall.byProject[cwd].stalled);
|
|
5095
|
+
for (const cwd of [...stallSince.keys()]) {
|
|
5096
|
+
if (!stalledCwds.includes(cwd)) {
|
|
5097
|
+
stallSince.delete(cwd);
|
|
5098
|
+
stallToasted.delete(cwd);
|
|
5099
|
+
}
|
|
5100
|
+
}
|
|
5101
|
+
const toAlert = [];
|
|
5102
|
+
for (const cwd of stalledCwds) {
|
|
5103
|
+
if (!stallSince.has(cwd)) stallSince.set(cwd, now);
|
|
5104
|
+
if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
|
|
5105
|
+
stallToasted.set(cwd, true);
|
|
5106
|
+
toAlert.push(cwd);
|
|
5107
|
+
}
|
|
5108
|
+
}
|
|
5109
|
+
if (toAlert.length > 0) {
|
|
5110
|
+
console.error(
|
|
5111
|
+
`[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
|
|
5112
|
+
+ `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
|
|
5113
|
+
stall.byProject,
|
|
5114
|
+
);
|
|
5115
|
+
appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: stall.total, byProject: stall.byProject });
|
|
5116
|
+
if (mainWindow && !mainWindow.isDestroyed()) {
|
|
5117
|
+
sendIfAlive(mainWindow, 'schedule:stall', {
|
|
5118
|
+
message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
|
|
5119
|
+
projects: toAlert,
|
|
5120
|
+
total: stall.total,
|
|
5121
|
+
byProject: stall.byProject,
|
|
5122
|
+
});
|
|
5123
|
+
}
|
|
5124
|
+
}
|
|
5125
|
+
|
|
4771
5126
|
appendHeartbeat({
|
|
4772
5127
|
ts: Date.now(),
|
|
4773
5128
|
pid: process.pid,
|
|
4774
5129
|
counts,
|
|
5130
|
+
stall: { stalled: stall.stalled, total: stall.total },
|
|
4775
5131
|
paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
|
|
4776
5132
|
nextReset: cachedNextReset,
|
|
4777
5133
|
utilization: cachedUtilization,
|
|
@@ -4802,6 +5158,102 @@ async function init() {
|
|
|
4802
5158
|
}
|
|
4803
5159
|
}
|
|
4804
5160
|
|
|
5161
|
+
/**
|
|
5162
|
+
* listPrdsInternal() → every live + archived PRD across every project,
|
|
5163
|
+
* with each entry's real job status folded in (`status`: the live queue
|
|
5164
|
+
* row's status, or the resolved terminal status for an archived entry, or
|
|
5165
|
+
* null when no queue row exists yet — e.g. a PRD just written and not yet
|
|
5166
|
+
* picked up by reconcile()). Single source of truth for both the renderer's
|
|
5167
|
+
* `schedule:list-prds` IPC handler and the admin HTTP `GET
|
|
5168
|
+
* /admin/scheduler/prds` route (PRD 1024) — neither re-implements this scan.
|
|
5169
|
+
*/
|
|
5170
|
+
async function listPrdsInternal() {
|
|
5171
|
+
ensureDirs();
|
|
5172
|
+
const out = [];
|
|
5173
|
+
const seenSlugs = new Set();
|
|
5174
|
+
|
|
5175
|
+
async function readDirInto(dir, { archived }) {
|
|
5176
|
+
let entries;
|
|
5177
|
+
try {
|
|
5178
|
+
entries = await fsp.readdir(dir);
|
|
5179
|
+
} catch (e) {
|
|
5180
|
+
if (e?.code !== 'ENOENT') {
|
|
5181
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: readdir failed', meta: { dir, error: e?.message } });
|
|
5182
|
+
}
|
|
5183
|
+
return;
|
|
5184
|
+
}
|
|
5185
|
+
for (const name of entries) {
|
|
5186
|
+
if (!name.endsWith('.md') || name.startsWith('.')) continue;
|
|
5187
|
+
const filePath = path.join(dir, name);
|
|
5188
|
+
try {
|
|
5189
|
+
const parsed = await parsePrd(filePath);
|
|
5190
|
+
// A slug can't be both live and archived at once, but a duplicate
|
|
5191
|
+
// slug found in two archive dirs (shouldn't happen — archiving is
|
|
5192
|
+
// a single rename — but is cheap to guard) is skipped rather than
|
|
5193
|
+
// double-counted.
|
|
5194
|
+
if (seenSlugs.has(parsed.slug)) continue;
|
|
5195
|
+
seenSlugs.add(parsed.slug);
|
|
5196
|
+
const stat = await fsp.stat(filePath);
|
|
5197
|
+
const entry = {
|
|
5198
|
+
slug: parsed.slug,
|
|
5199
|
+
parallelGroup: parsed.parallelGroup,
|
|
5200
|
+
title: parsed.title,
|
|
5201
|
+
cwd: parsed.cwd || '',
|
|
5202
|
+
estimateMinutes: parsed.estimateMinutes,
|
|
5203
|
+
sourcePromptId: parsed.sourcePromptId,
|
|
5204
|
+
epicId: parsed.epicId ?? null,
|
|
5205
|
+
mtimeMs: stat.mtimeMs,
|
|
5206
|
+
archived,
|
|
5207
|
+
};
|
|
5208
|
+
out.push(entry);
|
|
5209
|
+
} catch (e) {
|
|
5210
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'list-prds: skipping unparseable file', meta: { name, error: e?.message } });
|
|
5211
|
+
}
|
|
5212
|
+
}
|
|
5213
|
+
}
|
|
5214
|
+
|
|
5215
|
+
// Live PRDs first, so an archived duplicate (shouldn't exist, but a
|
|
5216
|
+
// stale rename copy is possible) never shadows the still-runnable live
|
|
5217
|
+
// entry.
|
|
5218
|
+
for (const dir of candidatePrdsDirs()) {
|
|
5219
|
+
await readDirInto(dir, { archived: false });
|
|
5220
|
+
}
|
|
5221
|
+
|
|
5222
|
+
const archivedStart = out.length;
|
|
5223
|
+
for (const dir of candidateArchivedPrdsDirs()) {
|
|
5224
|
+
await readDirInto(dir, { archived: true });
|
|
5225
|
+
}
|
|
5226
|
+
|
|
5227
|
+
// Every entry (live and archived) gets a real job status folded in.
|
|
5228
|
+
// Archived PRDs need one resolved defensively (live queue row, falling
|
|
5229
|
+
// back to history.jsonl) rather than hard-coded 'completed', so this
|
|
5230
|
+
// stays correct if the archive-only-completed invariant ever changes; a
|
|
5231
|
+
// live entry with no queue row yet (just written, not yet reconciled)
|
|
5232
|
+
// gets `status: null`.
|
|
5233
|
+
const [state, histBySlug] = await Promise.all([
|
|
5234
|
+
readQueue(),
|
|
5235
|
+
queueHistory.historyTerminalBySlug().catch(() => new Map()),
|
|
5236
|
+
]);
|
|
5237
|
+
const liveStatusBySlug = new Map(state.jobs.map((j) => [j.slug, j.status]));
|
|
5238
|
+
for (let i = 0; i < out.length; i++) {
|
|
5239
|
+
const entry = out[i];
|
|
5240
|
+
// `entry` is a freshly-synthesized PRD-listing row, not a persisted
|
|
5241
|
+
// ScheduleJob — assigning its `status` here is not a queue-job status
|
|
5242
|
+
// transition (no queue.json row is mutated, no statusHistory/audit
|
|
5243
|
+
// trail applies), so it is intentionally exempt from the
|
|
5244
|
+
// transitionJob-only rule enforced by scheduleJobTransitionsGrep.test.cjs.
|
|
5245
|
+
if (i < archivedStart) {
|
|
5246
|
+
entry.status = liveStatusBySlug.get(entry.slug) ?? null;
|
|
5247
|
+
} else {
|
|
5248
|
+
entry.archivedStatus = resolveArchivedPrdStatus(entry.slug, liveStatusBySlug, histBySlug);
|
|
5249
|
+
entry.status = entry.archivedStatus;
|
|
5250
|
+
}
|
|
5251
|
+
}
|
|
5252
|
+
|
|
5253
|
+
out.sort((a, b) => a.slug.localeCompare(b.slug, undefined, { numeric: true }));
|
|
5254
|
+
return out;
|
|
5255
|
+
}
|
|
5256
|
+
|
|
4805
5257
|
// remote — in-process (non-IPC) scheduler accessors, used by prdCreate.cjs
|
|
4806
5258
|
// and other main-process callers. (Named for the retired web-remote relay,
|
|
4807
5259
|
// its original consumer; kept because it still has in-process callers.)
|
|
@@ -4921,7 +5373,7 @@ const remote = {
|
|
|
4921
5373
|
// Best-effort: record the dispatch on the Epic's event chain.
|
|
4922
5374
|
try { await appendPrdCreatedEvent(cwd, epicTrace, slug); } catch { /* trace only */ }
|
|
4923
5375
|
}
|
|
4924
|
-
return { ok: true, bytesWritten: stat.size };
|
|
5376
|
+
return { ok: true, bytesWritten: stat.size, path: resolved, epicId: epicTrace };
|
|
4925
5377
|
} catch (e) {
|
|
4926
5378
|
return { ok: false, error: e?.message ?? 'write failed' };
|
|
4927
5379
|
}
|
|
@@ -4934,7 +5386,7 @@ const remote = {
|
|
|
4934
5386
|
if (idx < 0) return { kind: 'not-found' };
|
|
4935
5387
|
// Terminal-status guard lives in resetJobFields itself; force:true
|
|
4936
5388
|
// threads through to override it.
|
|
4937
|
-
if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true })) {
|
|
5389
|
+
if (!resetJobFields(state.jobs[idx], null, { force: opts.force === true, source: 'remote:resetJob' })) {
|
|
4938
5390
|
return { kind: 'refused' };
|
|
4939
5391
|
}
|
|
4940
5392
|
return { kind: 'ok' };
|
|
@@ -4955,6 +5407,185 @@ const remote = {
|
|
|
4955
5407
|
return state.jobs.map((j) => ({ slug: j.slug, title: j.title, status: j.status, cwd: j.cwd }));
|
|
4956
5408
|
},
|
|
4957
5409
|
|
|
5410
|
+
// Single queue row lookup, used by cancelJob/updatePrd's status guards and
|
|
5411
|
+
// the admin GET /admin/scheduler/prds?slug= route (PRD 1024).
|
|
5412
|
+
async getJob(slug) {
|
|
5413
|
+
const state = await readQueue();
|
|
5414
|
+
const job = state.jobs.find((j) => j.slug === slug);
|
|
5415
|
+
return job ? { slug: job.slug, title: job.title, status: job.status, cwd: job.cwd, error: job.error ?? null } : null;
|
|
5416
|
+
},
|
|
5417
|
+
|
|
5418
|
+
// Every live+archived PRD across every project (listPrdsInternal, shared
|
|
5419
|
+
// with the renderer's schedule:list-prds IPC handler), filtered by the
|
|
5420
|
+
// admin route's cwd/epicId/status query params.
|
|
5421
|
+
async listPrds(filter = {}) {
|
|
5422
|
+
const all = await listPrdsInternal();
|
|
5423
|
+
return all.filter((entry) => {
|
|
5424
|
+
if (filter.cwd && entry.cwd !== filter.cwd) return false;
|
|
5425
|
+
if (filter.epicId && entry.epicId !== filter.epicId) return false;
|
|
5426
|
+
if (filter.status && entry.status !== filter.status) return false;
|
|
5427
|
+
return true;
|
|
5428
|
+
});
|
|
5429
|
+
},
|
|
5430
|
+
|
|
5431
|
+
// Full body + parsed frontmatter for one PRD, live or archived. Mirrors
|
|
5432
|
+
// readPrd's dir-search + symlink-defense pattern (see that method's
|
|
5433
|
+
// comment) rather than sharing code with it, since readPrd intentionally
|
|
5434
|
+
// returns raw text only and is a much narrower/hotter path (executeJob's
|
|
5435
|
+
// PRD re-reads) that shouldn't grow a second return shape.
|
|
5436
|
+
async getPrdParsed(slug, cwd) {
|
|
5437
|
+
let dir = null;
|
|
5438
|
+
let filePath = null;
|
|
5439
|
+
if (cwd) {
|
|
5440
|
+
for (const d of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
|
|
5441
|
+
const p = safeSlugPathIn(d, slug);
|
|
5442
|
+
if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
|
|
5443
|
+
}
|
|
5444
|
+
if (!filePath) {
|
|
5445
|
+
for (const d of listArchivedPrdDirs(cwd)) {
|
|
5446
|
+
const p = safeSlugPathIn(d, slug);
|
|
5447
|
+
if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
|
|
5448
|
+
}
|
|
5449
|
+
}
|
|
5450
|
+
} else {
|
|
5451
|
+
dir = await findPrdDir(slug);
|
|
5452
|
+
filePath = dir ? safeSlugPathIn(dir, slug) : null;
|
|
5453
|
+
if (!filePath) {
|
|
5454
|
+
for (const d of candidateArchivedPrdsDirs()) {
|
|
5455
|
+
const p = safeSlugPathIn(d, slug);
|
|
5456
|
+
if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
|
|
5457
|
+
}
|
|
5458
|
+
}
|
|
5459
|
+
}
|
|
5460
|
+
if (!filePath) return { ok: false, error: 'invalid slug' };
|
|
5461
|
+
try {
|
|
5462
|
+
// Symlink defense, matching readPrd/writePrd's comment: safeSlugPathIn
|
|
5463
|
+
// is lexical and does not resolve symlinks.
|
|
5464
|
+
const real = await fsp.realpath(filePath);
|
|
5465
|
+
if (!real.startsWith(dir + path.sep)) return { ok: false, error: 'invalid slug' };
|
|
5466
|
+
const [raw, parsed] = await Promise.all([fsp.readFile(real, 'utf8'), prdParser.parsePrdRaw(real)]);
|
|
5467
|
+
return {
|
|
5468
|
+
ok: true,
|
|
5469
|
+
slug: parsed.slug,
|
|
5470
|
+
frontmatter: {
|
|
5471
|
+
title: parsed.title,
|
|
5472
|
+
cwd: parsed.cwd,
|
|
5473
|
+
estimateMinutes: parsed.estimateMinutes,
|
|
5474
|
+
parallelGroup: parsed.parallelGroup,
|
|
5475
|
+
sourcePromptId: parsed.sourcePromptId,
|
|
5476
|
+
sourceTabId: parsed.sourceTabId,
|
|
5477
|
+
epicId: parsed.epicId,
|
|
5478
|
+
dependsOn: parsed.dependsOn,
|
|
5479
|
+
createdVia: parsed.createdVia,
|
|
5480
|
+
issuedAt: parsed.issuedAt,
|
|
5481
|
+
},
|
|
5482
|
+
body: parsed.body,
|
|
5483
|
+
raw,
|
|
5484
|
+
};
|
|
5485
|
+
} catch (e) {
|
|
5486
|
+
return { ok: false, error: e?.message ?? 'read failed' };
|
|
5487
|
+
}
|
|
5488
|
+
},
|
|
5489
|
+
|
|
5490
|
+
// Edits a NOT-yet-running PRD's frontmatter and/or body in place, refusing
|
|
5491
|
+
// once a queue row exists for it and that row is anything but 'pending'
|
|
5492
|
+
// (running/completed/failed/needs_review — editing the spec under a live
|
|
5493
|
+
// or already-finished executor would silently rewrite history). Reuses
|
|
5494
|
+
// prdFrontmatter.cjs's parsePrdFile/serializePrdFile round-trip pair (PRD
|
|
5495
|
+
// 1024) so unrecognized keys (e.g. dependsOn) and untouched recognized
|
|
5496
|
+
// keys' original line formatting survive unchanged.
|
|
5497
|
+
async updatePrd({ slug, cwd, frontmatter, body }) {
|
|
5498
|
+
const job = await this.getJob(slug);
|
|
5499
|
+
// 'quarantined' is also editable: it's the ONLY way a quarantined PRD's
|
|
5500
|
+
// createdVia stamp gets written (the adopt action below), so refusing it
|
|
5501
|
+
// here would make quarantine irreversible through the API.
|
|
5502
|
+
if (job && job.status !== 'pending' && job.status !== 'quarantined') {
|
|
5503
|
+
return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
|
|
5504
|
+
}
|
|
5505
|
+
|
|
5506
|
+
let dir = null;
|
|
5507
|
+
let filePath = null;
|
|
5508
|
+
if (cwd) {
|
|
5509
|
+
for (const d of [prdDirForCwd(cwd), ...listEpicPrdDirs(cwd)]) {
|
|
5510
|
+
const p = safeSlugPathIn(d, slug);
|
|
5511
|
+
if (p && fs.existsSync(p)) { dir = d; filePath = p; break; }
|
|
5512
|
+
}
|
|
5513
|
+
} else {
|
|
5514
|
+
dir = await findPrdDir(slug);
|
|
5515
|
+
filePath = dir ? safeSlugPathIn(dir, slug) : null;
|
|
5516
|
+
}
|
|
5517
|
+
if (!filePath) return { ok: false, error: 'PRD not found' };
|
|
5518
|
+
|
|
5519
|
+
let raw;
|
|
5520
|
+
try {
|
|
5521
|
+
// Symlink defense, matching writePrd's comment: safeSlugPathIn is
|
|
5522
|
+
// lexical and does not resolve symlinks. updatePrd is a WRITE path
|
|
5523
|
+
// (unlike getPrdParsed's read-only realpath check), so also reject a
|
|
5524
|
+
// target that is itself already a symlink — a rogue job could plant
|
|
5525
|
+
// one inside the PRDs dir pointing outside the safe root.
|
|
5526
|
+
const real = await fsp.realpath(filePath);
|
|
5527
|
+
if (!real.startsWith(dir + path.sep)) return { ok: false, error: 'invalid slug' };
|
|
5528
|
+
const existing = await fsp.lstat(filePath).catch(() => null);
|
|
5529
|
+
if (existing && existing.isSymbolicLink()) return { ok: false, error: 'invalid slug' };
|
|
5530
|
+
raw = await fsp.readFile(real, 'utf8');
|
|
5531
|
+
} catch (e) {
|
|
5532
|
+
return { ok: false, error: e?.message ?? 'read failed' };
|
|
5533
|
+
}
|
|
5534
|
+
|
|
5535
|
+
const { frontmatter: fm, body: origBody } = parsePrdFile(raw);
|
|
5536
|
+
if (frontmatter) {
|
|
5537
|
+
for (const key of Object.keys(frontmatter)) {
|
|
5538
|
+
if (frontmatter[key] === undefined) continue;
|
|
5539
|
+
fm[key] = frontmatter[key];
|
|
5540
|
+
}
|
|
5541
|
+
}
|
|
5542
|
+
const newBody = body !== undefined ? body : origBody;
|
|
5543
|
+
const newRaw = serializePrdFile(fm, newBody);
|
|
5544
|
+
|
|
5545
|
+
try {
|
|
5546
|
+
await config.writeTextAtomic(filePath, newRaw, { writer: 'scheduler' });
|
|
5547
|
+
const stat = await fsp.stat(filePath);
|
|
5548
|
+
return { ok: true, slug, bytesWritten: stat.size };
|
|
5549
|
+
} catch (e) {
|
|
5550
|
+
return { ok: false, error: e?.message ?? 'write failed' };
|
|
5551
|
+
}
|
|
5552
|
+
},
|
|
5553
|
+
|
|
5554
|
+
// Cancels a job that hasn't finished yet. A 'running' job's process group
|
|
5555
|
+
// is SIGTERM'd (reusing killOrphanClaudePid — the same kill path boot
|
|
5556
|
+
// reconciliation uses for an orphaned running job) before its queue row is
|
|
5557
|
+
// finalized; a 'pending' job has no process to kill. There is no
|
|
5558
|
+
// 'cancelled' status in the closed job-status set (pending/running/
|
|
5559
|
+
// completed/failed/needs_review — see CLAUDE.md's domain model), so a
|
|
5560
|
+
// cancelled job lands in 'failed' with an error naming the cause,
|
|
5561
|
+
// consistent with every other non-success terminal outcome. Refuses a
|
|
5562
|
+
// slug that's already terminal — nothing left to cancel.
|
|
5563
|
+
async cancelJob(slug) {
|
|
5564
|
+
const state = await readQueue();
|
|
5565
|
+
const job = state.jobs.find((j) => j.slug === slug);
|
|
5566
|
+
if (!job) return { ok: false, error: 'not found' };
|
|
5567
|
+
if (job.status === 'completed' || job.status === 'failed' || job.status === 'needs_review') {
|
|
5568
|
+
return { ok: false, error: `job already terminal (status: "${job.status}") — nothing to cancel` };
|
|
5569
|
+
}
|
|
5570
|
+
const wasRunning = job.status === 'running';
|
|
5571
|
+
const pid = job.runtime?.pid;
|
|
5572
|
+
if (wasRunning && pid) {
|
|
5573
|
+
killOrphanClaudePid(pid);
|
|
5574
|
+
}
|
|
5575
|
+
await mutate((s) => {
|
|
5576
|
+
const idx = s.jobs.findIndex((j) => j.slug === slug);
|
|
5577
|
+
if (idx < 0) return;
|
|
5578
|
+
const j = s.jobs[idx];
|
|
5579
|
+
transitionJob(j, 'failed', { reason: 'cancelled via admin API', source: 'remote:cancelJob' });
|
|
5580
|
+
j.error = 'cancelled via admin API';
|
|
5581
|
+
j.finishedAt = new Date().toISOString();
|
|
5582
|
+
j.exitCode = j.exitCode ?? null;
|
|
5583
|
+
delete j.runtime;
|
|
5584
|
+
});
|
|
5585
|
+
await broadcast({ flush: true });
|
|
5586
|
+
return { ok: true, slug, status: 'failed', wasRunning, cwd: job.cwd ?? null };
|
|
5587
|
+
},
|
|
5588
|
+
|
|
4958
5589
|
// Exposes the module-level allocateParallelGroup (PRD 548) to callers that
|
|
4959
5590
|
// only hold the `remote` object (lib/prdCreate.cjs's create-prd route) —
|
|
4960
5591
|
// reuses the same allocator the file-based /develop authoring path relies
|
|
@@ -4995,4 +5626,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
4995
5626
|
});
|
|
4996
5627
|
}
|
|
4997
5628
|
|
|
4998
|
-
module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveNotifyPrd, runPrdMigration, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob };
|
|
5629
|
+
module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS };
|