claude-code-session-manager 0.39.1 → 0.39.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{TiptapBody-CtuLATFR.js → TiptapBody-MCzz_6Zm.js} +1 -1
- package/dist/assets/{index-CefVSjmm.js → index-BPbLVT6E.js} +490 -488
- package/dist/assets/{index-B6JNrbpD.css → index-DVlD8N1X.css} +2 -2
- package/dist/index.html +2 -2
- package/package.json +3 -3
- package/scripts/lib/activeSessions.cjs +26 -3
- package/scripts/lib/watchdogHelpers.cjs +13 -6
- package/src/main/__tests__/browserView-destroyed-handler.test.cjs +84 -0
- package/src/main/__tests__/prdCreate.test.cjs +6 -0
- package/src/main/__tests__/prdLocations.test.cjs +41 -0
- package/src/main/__tests__/prdMigration.test.cjs +41 -0
- package/src/main/__tests__/promptSessionEvents.test.cjs +106 -0
- package/src/main/__tests__/queueHistory.test.cjs +2 -2
- package/src/main/__tests__/rcaFeedbackHook.test.cjs +108 -0
- package/src/main/__tests__/runVerify.test.cjs +459 -4
- package/src/main/__tests__/scheduler-admin-routes.test.cjs +52 -3
- package/src/main/__tests__/scheduler-archive-completed-prd.test.cjs +102 -0
- package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +126 -0
- package/src/main/__tests__/scheduler-committed-in-window.test.cjs +58 -3
- package/src/main/__tests__/scheduler-find-prd-dir.test.cjs +42 -0
- package/src/main/__tests__/scheduler-investigation-clean-skip.test.cjs +63 -0
- package/src/main/__tests__/scheduler-meta-code-sha.test.cjs +23 -0
- package/src/main/__tests__/scheduler-notify-originating-tab.test.cjs +68 -0
- package/src/main/__tests__/scheduler-reset-job-fields-guard.test.cjs +77 -0
- package/src/main/__tests__/scheduler-unreadable-queue-guard.test.cjs +62 -0
- package/src/main/browserView.cjs +5 -4
- package/src/main/chatRunner.cjs +90 -7
- package/src/main/config.cjs +9 -0
- package/src/main/ipcSchemas.cjs +1 -1
- package/src/main/lib/__tests__/terminalRunOutcome.test.cjs +118 -0
- package/src/main/lib/prdLocations.cjs +43 -6
- package/src/main/lib/prdMigration.cjs +5 -3
- package/src/main/lib/queueHistory.cjs +1 -1
- package/src/main/lib/rcaFeedbackHook.cjs +55 -5
- package/src/main/lib/terminalRunOutcome.cjs +100 -0
- package/src/main/promptSessionEvents.cjs +87 -0
- package/src/main/runVerify.cjs +195 -3
- package/src/main/scheduler.cjs +436 -80
- package/src/preload/api.d.ts +5 -1
package/src/main/chatRunner.cjs
CHANGED
|
@@ -36,7 +36,12 @@
|
|
|
36
36
|
* chat:run:complete { tabId, sessionId, finalMessage }
|
|
37
37
|
* chat:run:needs-input { tabId, sessionId, questions, raw }
|
|
38
38
|
* chat:run:error { tabId, sessionId, message }
|
|
39
|
+
* — the kill-ceiling variant additionally carries
|
|
40
|
+
* { elapsedMs, ceilingMs, lastToolUses } so a resumed
|
|
41
|
+
* turn can tell whether an external side effect
|
|
42
|
+
* landed before verifying/retrying
|
|
39
43
|
* chat:run:notice { tabId, sessionId, message } — informational, not terminal
|
|
44
|
+
* (also fires at 80% of the kill ceiling as a wrap-up nudge)
|
|
40
45
|
* chat:context-usage { tabId, sessionId, usedTokens, totalTokens, usedPct, categories }
|
|
41
46
|
* — result of a silent `/context` probe
|
|
42
47
|
*/
|
|
@@ -239,6 +244,34 @@ function hasMcpConsentDenial(text) {
|
|
|
239
244
|
return MCP_CONSENT_DENIAL_MARKERS.some((marker) => lower.includes(marker));
|
|
240
245
|
}
|
|
241
246
|
|
|
247
|
+
// Number of recent tool uses kept for the kill-message context (Ask 3).
|
|
248
|
+
const RECENT_TOOL_USE_LIMIT = 3;
|
|
249
|
+
const TOOL_USE_DETAIL_MAX_LEN = 60;
|
|
250
|
+
|
|
251
|
+
// Renders a single classified tool_use (from classifyToolUse) plus a short
|
|
252
|
+
// input-derived detail string into a human-readable descriptor, e.g.
|
|
253
|
+
// "Bash(eas submit --platform ios …)". Pure formatting — no new stream
|
|
254
|
+
// parsing; `detail` is lifted from the same already-parsed block.
|
|
255
|
+
function renderToolUseDescriptor({ label, detail }) {
|
|
256
|
+
if (!detail) return label;
|
|
257
|
+
const truncated = detail.length > TOOL_USE_DETAIL_MAX_LEN
|
|
258
|
+
? `${detail.slice(0, TOOL_USE_DETAIL_MAX_LEN)}…`
|
|
259
|
+
: detail;
|
|
260
|
+
return `${label}(${truncated})`;
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
// Pulls a short descriptive string out of a tool_use block's input, reusing
|
|
264
|
+
// fields already present on the already-parsed block — not a new parser.
|
|
265
|
+
function describeToolUseInput(block) {
|
|
266
|
+
const input = block?.input;
|
|
267
|
+
if (!input || typeof input !== 'object') return '';
|
|
268
|
+
if (typeof input.command === 'string') return input.command;
|
|
269
|
+
if (typeof input.description === 'string') return input.description;
|
|
270
|
+
if (typeof input.pattern === 'string') return input.pattern;
|
|
271
|
+
if (typeof input.file_path === 'string') return input.file_path;
|
|
272
|
+
return '';
|
|
273
|
+
}
|
|
274
|
+
|
|
242
275
|
// Instruction prepended to every prompt. Tells the agent how to signal that
|
|
243
276
|
// it needs clarification vs. having completed the task.
|
|
244
277
|
const STOP_SIGNAL_INSTRUCTION =
|
|
@@ -252,6 +285,17 @@ const STOP_SIGNAL_INSTRUCTION =
|
|
|
252
285
|
`guess on what's genuinely blocked, but always answer what you can first. ` +
|
|
253
286
|
`Otherwise complete the task and end with a concise summary of what you did.\n\n`;
|
|
254
287
|
|
|
288
|
+
// ─── Hard wall-clock kill ceiling ────────────────────────────────────────
|
|
289
|
+
// Defined here (ahead of CHAT_MODE_TRUTH_INSTRUCTION) so the prompt's stated
|
|
290
|
+
// budget is always derived from this single constant — never a hand-written
|
|
291
|
+
// duplicate that could drift from the real timer below.
|
|
292
|
+
const KILL_CEILING_MS = 30 * 60 * 1000; // 30 minutes
|
|
293
|
+
const KILL_CEILING_MIN = KILL_CEILING_MS / 60_000;
|
|
294
|
+
// 80% warning point — gives the model a turn-visible nudge to wrap up before
|
|
295
|
+
// the hard kill fires at 100%. Not configurable; see PRD out-of-scope note
|
|
296
|
+
// re: making KILL_CEILING_MS itself configurable.
|
|
297
|
+
const WARN_CEILING_MS = Math.floor(KILL_CEILING_MS * 0.8);
|
|
298
|
+
|
|
255
299
|
// Instruction prepended to every prompt. Tells the agent the truth about this
|
|
256
300
|
// execution mode: this Chat tab is a one-shot headless `claude -p` run — no
|
|
257
301
|
// process survives after this turn ends, so background shells, scheduled
|
|
@@ -265,9 +309,12 @@ const CHAT_MODE_TRUTH_INSTRUCTION =
|
|
|
265
309
|
`there is no later turn in which that could happen, so that promise would ` +
|
|
266
310
|
`go unfulfilled and leave the user waiting with no explanation. If you need ` +
|
|
267
311
|
`to poll something, do it synchronously within this turn with a bounded ` +
|
|
268
|
-
`timeout, then report the actual result.
|
|
269
|
-
|
|
270
|
-
`
|
|
312
|
+
`timeout, then report the actual result. This turn is hard-killed after ` +
|
|
313
|
+
`${KILL_CEILING_MIN} minutes of wall-clock. Size every synchronous poll to ` +
|
|
314
|
+
`finish inside that budget; if the work cannot fit, do the part that fits, ` +
|
|
315
|
+
`report exactly what landed, and say what remains. End this turn with ` +
|
|
316
|
+
`either a real result or an explicit statement that the user needs to ` +
|
|
317
|
+
`reply for the work to continue.\n\n`;
|
|
271
318
|
|
|
272
319
|
// ─── Serial run queue (v0.34) ───────────────────────────────────────────────
|
|
273
320
|
// CONCURRENCY_CAP=2 (default) governs ALL runs — silent probes and manual
|
|
@@ -299,9 +346,6 @@ let activeCount = 0;
|
|
|
299
346
|
let executor = executeRun;
|
|
300
347
|
function __setExecutor(fn) { executor = fn || executeRun; }
|
|
301
348
|
|
|
302
|
-
// ─── Hard wall-clock kill ceiling ────────────────────────────────────────
|
|
303
|
-
const KILL_CEILING_MS = 30 * 60 * 1000; // 30 minutes
|
|
304
|
-
|
|
305
349
|
// ─── Window reference (set by attachWindow) ────────────────────────────────
|
|
306
350
|
|
|
307
351
|
let mainWindow = null;
|
|
@@ -378,6 +422,10 @@ function pump() {
|
|
|
378
422
|
*/
|
|
379
423
|
function executeRun({ tabId, sessionId, prompt, cwd, resume, silent, onSilentResult, promptId }) {
|
|
380
424
|
return new Promise((resolve) => {
|
|
425
|
+
const startedAt = Date.now();
|
|
426
|
+
// Last few tool_use blocks seen on the stream, oldest first — surfaced in
|
|
427
|
+
// the kill message (Ask 3) so a resumed turn knows what might have landed.
|
|
428
|
+
const recentToolUses = [];
|
|
381
429
|
let settled = false;
|
|
382
430
|
// Frees the lane exactly once: drops the cancel fn and resolves the promise
|
|
383
431
|
// the pump is awaiting. Both exit and error paths funnel through here.
|
|
@@ -487,12 +535,40 @@ function executeRun({ tabId, sessionId, prompt, cwd, resume, silent, onSilentRes
|
|
|
487
535
|
// executor invocation returns — see comment above the inFlight Map decl.
|
|
488
536
|
inFlight.set(tabId, { cancelFn, donePromise: null });
|
|
489
537
|
|
|
538
|
+
// 80%-of-ceiling warning — a turn-visible nudge to wrap up before the hard
|
|
539
|
+
// kill fires at 100%. Does not extend or otherwise affect the kill timer
|
|
540
|
+
// below; it is purely informational (Ask 2).
|
|
541
|
+
const warnTimer = setTimeout(() => {
|
|
542
|
+
if (silent) return; // silent probes are short-lived; nothing to warn
|
|
543
|
+
broadcast('chat:run:notice', {
|
|
544
|
+
tabId,
|
|
545
|
+
sessionId,
|
|
546
|
+
message:
|
|
547
|
+
`Heads up: this turn has been running for ${Math.round(WARN_CEILING_MS / 60_000)} ` +
|
|
548
|
+
`minutes and will be force-killed at the ${KILL_CEILING_MIN}-minute ceiling if it's ` +
|
|
549
|
+
`still going. Wrap up now — report exactly what has landed so far and what remains, ` +
|
|
550
|
+
`before the hard kill fires.`,
|
|
551
|
+
});
|
|
552
|
+
}, WARN_CEILING_MS);
|
|
553
|
+
if (warnTimer.unref) warnTimer.unref();
|
|
554
|
+
|
|
490
555
|
// Hard wall-clock ceiling — SIGTERM + SIGKILL on expiry
|
|
491
556
|
const killTimer = setTimeout(() => {
|
|
557
|
+
const elapsedMs = Date.now() - startedAt;
|
|
558
|
+
const lastToolUses = recentToolUses.slice();
|
|
559
|
+
const lastActionsText = lastToolUses.length > 0
|
|
560
|
+
? lastToolUses.map(renderToolUseDescriptor).join(', ')
|
|
561
|
+
: 'none observed';
|
|
492
562
|
emitTerminal('chat:run:error', {
|
|
493
563
|
tabId,
|
|
494
564
|
sessionId,
|
|
495
|
-
|
|
565
|
+
elapsedMs,
|
|
566
|
+
ceilingMs: KILL_CEILING_MS,
|
|
567
|
+
lastToolUses,
|
|
568
|
+
message:
|
|
569
|
+
`Killed after ${Math.round(elapsedMs / 60_000)}m (ceiling ${KILL_CEILING_MIN}m). ` +
|
|
570
|
+
`Last actions: ${lastActionsText}. External side effects may have completed — verify ` +
|
|
571
|
+
`before retrying.`,
|
|
496
572
|
});
|
|
497
573
|
cancelFn();
|
|
498
574
|
}, KILL_CEILING_MS);
|
|
@@ -520,6 +596,8 @@ function executeRun({ tabId, sessionId, prompt, cwd, resume, silent, onSilentRes
|
|
|
520
596
|
if (!silent) broadcast('chat:run:output', { tabId, delta: block.text });
|
|
521
597
|
} else if (block.type === 'tool_use' && typeof block.name === 'string') {
|
|
522
598
|
const classified = classifyToolUse(block);
|
|
599
|
+
recentToolUses.push({ ...classified, detail: describeToolUseInput(block) });
|
|
600
|
+
if (recentToolUses.length > RECENT_TOOL_USE_LIMIT) recentToolUses.shift();
|
|
523
601
|
if (!silent) broadcast('chat:run:tool-use', { tabId, id: block.id, ...classified });
|
|
524
602
|
}
|
|
525
603
|
}
|
|
@@ -602,6 +680,7 @@ function executeRun({ tabId, sessionId, prompt, cwd, resume, silent, onSilentRes
|
|
|
602
680
|
|
|
603
681
|
child.on('error', (err) => {
|
|
604
682
|
clearTimeout(killTimer);
|
|
683
|
+
clearTimeout(warnTimer);
|
|
605
684
|
emitTerminal('chat:run:error', {
|
|
606
685
|
tabId,
|
|
607
686
|
sessionId,
|
|
@@ -618,6 +697,7 @@ function executeRun({ tabId, sessionId, prompt, cwd, resume, silent, onSilentRes
|
|
|
618
697
|
// terminalSent latch.
|
|
619
698
|
child.on('close', (code, signal) => {
|
|
620
699
|
clearTimeout(killTimer);
|
|
700
|
+
clearTimeout(warnTimer);
|
|
621
701
|
// Flush any partial line that didn't end with \n
|
|
622
702
|
if (lineBuffer.trim()) processLine(lineBuffer.trim());
|
|
623
703
|
|
|
@@ -759,6 +839,9 @@ module.exports = {
|
|
|
759
839
|
probeContextUsage,
|
|
760
840
|
STOP_SENTINEL,
|
|
761
841
|
CHAT_MODE_TRUTH_INSTRUCTION,
|
|
842
|
+
KILL_CEILING_MS,
|
|
843
|
+
KILL_CEILING_MIN,
|
|
844
|
+
WARN_CEILING_MS,
|
|
762
845
|
__setExecutor,
|
|
763
846
|
enqueueExternalPrompt,
|
|
764
847
|
registerAdminRoute,
|
package/src/main/config.cjs
CHANGED
|
@@ -143,6 +143,15 @@ function validateWrite(realAbs) {
|
|
|
143
143
|
if (realAbs === feedbackSub || realAbs.startsWith(feedbackSub + path.sep)) {
|
|
144
144
|
return;
|
|
145
145
|
}
|
|
146
|
+
// PromptSession persistence (active-index.json + per-session archives,
|
|
147
|
+
// promptSessions.ts) and the scheduler's own read-modify-write of that
|
|
148
|
+
// same active index (promptSessionEvents.cjs, PRD 814) — narrowly
|
|
149
|
+
// scoped to session-manager-operations/prompt-sessions/, this repo's
|
|
150
|
+
// existing per-project artifact-store convention.
|
|
151
|
+
const promptSessionsSub = path.join(realRoot, 'session-manager-operations', 'prompt-sessions');
|
|
152
|
+
if (realAbs === promptSessionsSub || realAbs.startsWith(promptSessionsSub + path.sep)) {
|
|
153
|
+
return;
|
|
154
|
+
}
|
|
146
155
|
}
|
|
147
156
|
}
|
|
148
157
|
throw new Error(`Write outside allowed write boundaries: ${realAbs}`);
|
package/src/main/ipcSchemas.cjs
CHANGED
|
@@ -278,7 +278,7 @@ const schedulerCreatePrd = z.object({
|
|
|
278
278
|
sourceTabId: z.string().min(1).max(128).regex(NO_NEWLINE_RE, 'must not contain newlines').optional(),
|
|
279
279
|
// User-selected Feature/Bug tag (PRD 774) carried from the originating
|
|
280
280
|
// PromptTicket — deterministic, never LLM-classified.
|
|
281
|
-
tag: z.enum(['feature', 'bug']).optional(),
|
|
281
|
+
tag: z.enum(['feature', 'bug', 'discussion']).optional(),
|
|
282
282
|
});
|
|
283
283
|
|
|
284
284
|
// Bulk archive: slug list, capped to limit unbounded retag/archive payloads.
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* terminalRunOutcome.test.cjs — unit tests for the history-independent
|
|
3
|
+
* terminal-run-outcome probe (PRD 812-689-fix-fix-distribute-adminserver-routes).
|
|
4
|
+
*
|
|
5
|
+
* Run: timeout 120 npx vitest run src/main/lib/__tests__/terminalRunOutcome.test.cjs
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
'use strict';
|
|
9
|
+
|
|
10
|
+
import { test, expect } from 'vitest';
|
|
11
|
+
const fs = require('node:fs');
|
|
12
|
+
const os = require('node:os');
|
|
13
|
+
const path = require('node:path');
|
|
14
|
+
const {
|
|
15
|
+
latestTerminalOutcomeForSlug,
|
|
16
|
+
MAX_DIRS_SCANNED,
|
|
17
|
+
} = require('../terminalRunOutcome.cjs');
|
|
18
|
+
|
|
19
|
+
function mkTmpRunsDir() {
|
|
20
|
+
return fs.mkdtempSync(path.join(os.tmpdir(), 'terminal-run-outcome-'));
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
function writeRun(runsDir, runId, slug, meta, verdicts) {
|
|
24
|
+
const dir = path.join(runsDir, runId);
|
|
25
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
26
|
+
if (meta !== undefined) {
|
|
27
|
+
fs.writeFileSync(path.join(dir, `${slug}.meta.json`), typeof meta === 'string' ? meta : JSON.stringify(meta));
|
|
28
|
+
}
|
|
29
|
+
if (verdicts !== undefined) {
|
|
30
|
+
fs.writeFileSync(path.join(dir, `${slug}.verdicts.json`), typeof verdicts === 'string' ? verdicts : JSON.stringify(verdicts));
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
test('returns completed for newest run with exitCode 0 + clean verdict', () => {
|
|
35
|
+
const runsDir = mkTmpRunsDir();
|
|
36
|
+
writeRun(runsDir, '2026-07-31T07-38-29-081Z', 'my-slug', { exitCode: 0, finishedAt: 1785483574748 }, { verdict: 'clean' });
|
|
37
|
+
const result = latestTerminalOutcomeForSlug('my-slug', { runsDir });
|
|
38
|
+
expect(result).toEqual({ status: 'completed', runId: '2026-07-31T07-38-29-081Z', finishedAt: new Date(1785483574748).toISOString() });
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
test('returns completed for pass_no_commit_already_shipped verdict', () => {
|
|
42
|
+
const runsDir = mkTmpRunsDir();
|
|
43
|
+
writeRun(runsDir, '2026-07-31T07-38-29-081Z', 'my-slug', { exitCode: 0, finishedAt: 1785483574749 }, { verdict: 'pass_no_commit_already_shipped' });
|
|
44
|
+
const result = latestTerminalOutcomeForSlug('my-slug', { runsDir });
|
|
45
|
+
expect(result).toEqual({ status: 'completed', runId: '2026-07-31T07-38-29-081Z', finishedAt: new Date(1785483574749).toISOString() });
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
test('returns failed for non-zero exitCode', () => {
|
|
49
|
+
const runsDir = mkTmpRunsDir();
|
|
50
|
+
writeRun(runsDir, '2026-07-31T07-38-29-081Z', 'my-slug', { exitCode: 1, finishedAt: 1785483574750 });
|
|
51
|
+
const result = latestTerminalOutcomeForSlug('my-slug', { runsDir });
|
|
52
|
+
expect(result).toEqual({ status: 'failed', runId: '2026-07-31T07-38-29-081Z', finishedAt: new Date(1785483574750).toISOString() });
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
test('returns failed for exitCode 0 with a non-completed-equivalent verdict', () => {
|
|
56
|
+
const runsDir = mkTmpRunsDir();
|
|
57
|
+
writeRun(runsDir, '2026-07-31T07-38-29-081Z', 'my-slug', { exitCode: 0, finishedAt: 1785483574751 }, { verdict: 'transcript_errors' });
|
|
58
|
+
const result = latestTerminalOutcomeForSlug('my-slug', { runsDir });
|
|
59
|
+
expect(result).toEqual({ status: 'failed', runId: '2026-07-31T07-38-29-081Z', finishedAt: new Date(1785483574751).toISOString() });
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
test('returns null when there is no run dir for the slug', () => {
|
|
63
|
+
const runsDir = mkTmpRunsDir();
|
|
64
|
+
writeRun(runsDir, '2026-07-31T07-38-29-081Z', 'other-slug', { exitCode: 0 }, { verdict: 'clean' });
|
|
65
|
+
expect(latestTerminalOutcomeForSlug('my-slug', { runsDir })).toBeNull();
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
test('returns null on malformed meta.json', () => {
|
|
69
|
+
const runsDir = mkTmpRunsDir();
|
|
70
|
+
writeRun(runsDir, '2026-07-31T07-38-29-081Z', 'my-slug', '{not json', { verdict: 'clean' });
|
|
71
|
+
expect(latestTerminalOutcomeForSlug('my-slug', { runsDir })).toBeNull();
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
test('returns null on malformed verdicts.json', () => {
|
|
75
|
+
const runsDir = mkTmpRunsDir();
|
|
76
|
+
writeRun(runsDir, '2026-07-31T07-38-29-081Z', 'my-slug', { exitCode: 0 }, '{not json');
|
|
77
|
+
expect(latestTerminalOutcomeForSlug('my-slug', { runsDir })).toBeNull();
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
test('returns null when runsDir does not exist', () => {
|
|
81
|
+
expect(latestTerminalOutcomeForSlug('my-slug', { runsDir: '/nonexistent/path/xyz' })).toBeNull();
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
test('picks the newest of several run dirs for the same slug', () => {
|
|
85
|
+
const runsDir = mkTmpRunsDir();
|
|
86
|
+
writeRun(runsDir, '2026-07-01T00-00-00-000Z', 'my-slug', { exitCode: 1, finishedAt: 10 });
|
|
87
|
+
writeRun(runsDir, '2026-07-31T07-38-29-081Z', 'my-slug', { exitCode: 0, finishedAt: 1785483574752 }, { verdict: 'clean' });
|
|
88
|
+
writeRun(runsDir, '2026-06-01T00-00-00-000Z', 'my-slug', { exitCode: 1, finishedAt: 5 });
|
|
89
|
+
const result = latestTerminalOutcomeForSlug('my-slug', { runsDir });
|
|
90
|
+
expect(result).toEqual({ status: 'completed', runId: '2026-07-31T07-38-29-081Z', finishedAt: new Date(1785483574752).toISOString() });
|
|
91
|
+
});
|
|
92
|
+
|
|
93
|
+
test('stats at most the newest few run dirs (bound enforced)', () => {
|
|
94
|
+
const runsDir = mkTmpRunsDir();
|
|
95
|
+
// Create many more matching run dirs than MAX_DIRS_SCANNED, all with a
|
|
96
|
+
// non-terminal-equivalent verdict so the loop never early-returns before
|
|
97
|
+
// exhausting the candidate slice — proves the bound is actually applied.
|
|
98
|
+
const total = MAX_DIRS_SCANNED + 10;
|
|
99
|
+
for (let i = 0; i < total; i++) {
|
|
100
|
+
const ts = `2026-07-${String(i + 1).padStart(2, '0')}T00-00-00-000Z`;
|
|
101
|
+
writeRun(runsDir, ts, 'my-slug', { exitCode: 0, finishedAt: i }, { verdict: 'transcript_errors' });
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
let readFileCalls = 0;
|
|
105
|
+
const fsImpl = {
|
|
106
|
+
readdirSync: (...a) => fs.readdirSync(...a),
|
|
107
|
+
existsSync: (...a) => fs.existsSync(...a),
|
|
108
|
+
readFileSync: (...a) => {
|
|
109
|
+
readFileCalls++;
|
|
110
|
+
return fs.readFileSync(...a);
|
|
111
|
+
},
|
|
112
|
+
};
|
|
113
|
+
|
|
114
|
+
latestTerminalOutcomeForSlug('my-slug', { runsDir, fsImpl });
|
|
115
|
+
// Each scanned dir reads meta.json, then (since exitCode===0) verdicts.json
|
|
116
|
+
// too, so at most MAX_DIRS_SCANNED * 2 readFileSync calls.
|
|
117
|
+
expect(readFileCalls).toBeLessThanOrEqual(MAX_DIRS_SCANNED * 2);
|
|
118
|
+
});
|
|
@@ -15,8 +15,9 @@
|
|
|
15
15
|
*/
|
|
16
16
|
'use strict';
|
|
17
17
|
|
|
18
|
+
const fs = require('node:fs');
|
|
18
19
|
const path = require('node:path');
|
|
19
|
-
const { activeProjectCwds } = require('../../../scripts/lib/activeSessions.cjs');
|
|
20
|
+
const { activeProjectCwds, allProjectCwds } = require('../../../scripts/lib/activeSessions.cjs');
|
|
20
21
|
|
|
21
22
|
const PRD_SUBPATH = ['session-manager-operations', 'scheduler', 'prds'];
|
|
22
23
|
|
|
@@ -34,13 +35,49 @@ function resolvePrdWriteDir(cwd) {
|
|
|
34
35
|
|
|
35
36
|
/**
|
|
36
37
|
* resolvePrdsDirs(maxAgeMin?, opts?) → string[]
|
|
37
|
-
*
|
|
38
|
-
*
|
|
39
|
-
*
|
|
38
|
+
*
|
|
39
|
+
* Every `<cwd>/session-manager-operations/scheduler/prds` dir that actually
|
|
40
|
+
* EXISTS on disk, across every project this machine has ever opened — plus
|
|
41
|
+
* the currently-active projects' dirs even if they haven't been created yet
|
|
42
|
+
* (write paths need a destination before the first PRD lands there).
|
|
43
|
+
*
|
|
44
|
+
* Deliberately NOT filtered by recency. This function answers "where do PRD
|
|
45
|
+
* source files live", and a project being quiet says nothing about whether it
|
|
46
|
+
* owns queued work. It used to return only activeProjectCwds' 90-minute
|
|
47
|
+
* window, which made a quiet project's PRDs unscannable — and reconcile()
|
|
48
|
+
* reads an unscannable PRD as a deleted one, silently dropping its queue row
|
|
49
|
+
* (2026-07-31: 142 PRDs across 6 quiet projects). Recency stays where it
|
|
50
|
+
* belongs: the feedback sweep, which genuinely only cares about live work.
|
|
51
|
+
*
|
|
52
|
+
* `maxAgeMin` is still honoured for the active-project half so existing
|
|
53
|
+
* callers and tests keep their semantics; `opts` is forwarded to the
|
|
54
|
+
* underlying scan (e.g. `projectsDir` override for tests).
|
|
40
55
|
*/
|
|
41
56
|
function resolvePrdsDirs(maxAgeMin, opts) {
|
|
42
|
-
const
|
|
43
|
-
|
|
57
|
+
const dirs = [];
|
|
58
|
+
const seen = new Set();
|
|
59
|
+
const add = (dir) => {
|
|
60
|
+
if (seen.has(dir)) return;
|
|
61
|
+
seen.add(dir);
|
|
62
|
+
dirs.push(dir);
|
|
63
|
+
};
|
|
64
|
+
|
|
65
|
+
// Every historical project that has a PRD dir on disk — the set that
|
|
66
|
+
// matters for discovery, regardless of when it was last touched.
|
|
67
|
+
for (const cwd of allProjectCwds(opts)) {
|
|
68
|
+
let dir;
|
|
69
|
+
try { dir = resolvePrdWriteDir(cwd); } catch { continue; }
|
|
70
|
+
if (fs.existsSync(dir)) add(dir);
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// Active projects are added unconditionally: a brand-new project has no
|
|
74
|
+
// prds/ dir yet, and callers that resolve a write destination must still
|
|
75
|
+
// find it. Scans over a non-existent dir are a harmless ENOENT no-op.
|
|
76
|
+
for (const cwd of activeProjectCwds(maxAgeMin, opts)) {
|
|
77
|
+
try { add(resolvePrdWriteDir(cwd)); } catch { /* unusable cwd */ }
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
return dirs;
|
|
44
81
|
}
|
|
45
82
|
|
|
46
83
|
module.exports = { resolvePrdWriteDir, resolvePrdsDirs, PRD_SUBPATH };
|
|
@@ -18,6 +18,7 @@ const fsp = require('node:fs/promises');
|
|
|
18
18
|
const path = require('node:path');
|
|
19
19
|
const { splitFrontmatter } = require('./prdFrontmatter.cjs');
|
|
20
20
|
const { resolvePrdWriteDir } = require('./prdLocations.cjs');
|
|
21
|
+
const { expandHome } = require('./expandHome.cjs');
|
|
21
22
|
|
|
22
23
|
/**
|
|
23
24
|
* Move every `.md` file in legacyPrdsDir whose frontmatter `cwd` resolves to
|
|
@@ -55,13 +56,14 @@ async function migratePrds(legacyPrdsDir) {
|
|
|
55
56
|
}
|
|
56
57
|
|
|
57
58
|
const { fm } = splitFrontmatter(raw);
|
|
58
|
-
const
|
|
59
|
-
if (!
|
|
59
|
+
const rawCwd = fm.cwd && fm.cwd.trim();
|
|
60
|
+
if (!rawCwd) {
|
|
60
61
|
unresolved.push({ file: name, reason: 'no cwd in frontmatter' });
|
|
61
62
|
continue;
|
|
62
63
|
}
|
|
64
|
+
const cwd = expandHome(rawCwd);
|
|
63
65
|
if (!fs.existsSync(cwd)) {
|
|
64
|
-
unresolved.push({ file: name, reason: `cwd does not exist on disk: ${
|
|
66
|
+
unresolved.push({ file: name, reason: `cwd does not exist on disk: ${rawCwd}` });
|
|
65
67
|
continue;
|
|
66
68
|
}
|
|
67
69
|
|
|
@@ -196,7 +196,7 @@ async function historyTerminalBySlug() {
|
|
|
196
196
|
if (!line.trim()) continue;
|
|
197
197
|
try {
|
|
198
198
|
const j = JSON.parse(line);
|
|
199
|
-
if (j?.slug) map.set(j.slug, { status: j.status, finishedAt: j.finishedAt });
|
|
199
|
+
if (j?.slug) map.set(j.slug, { status: j.status, finishedAt: j.finishedAt, landedCommit: j.landedCommit ?? null });
|
|
200
200
|
} catch {
|
|
201
201
|
// corrupt/partial line — ignore
|
|
202
202
|
}
|
|
@@ -57,6 +57,8 @@ const VERDICT_LABELS = {
|
|
|
57
57
|
uncommitted_changes: 'uncommitted changes',
|
|
58
58
|
no_verdict_sentinel: 'no commit or verdict sentinel',
|
|
59
59
|
pass_no_commit: 'PASS sentinel but no commit landed',
|
|
60
|
+
pass_no_commit_already_shipped: 'PASS with no commit — deliverables already shipped',
|
|
61
|
+
pass_no_commit_prior_run_verified: 'PASS with no commit — prior run of this slug already landed the work',
|
|
60
62
|
};
|
|
61
63
|
|
|
62
64
|
function humanVerdict(verdict) {
|
|
@@ -66,6 +68,8 @@ function humanVerdict(verdict) {
|
|
|
66
68
|
// ─── Failure-class matching (deterministic, no LLM) ─────────────────────────
|
|
67
69
|
|
|
68
70
|
const FAILURE_CLASSES = {
|
|
71
|
+
ALREADY_SHIPPED: 'already-shipped',
|
|
72
|
+
SELF_QUEUE: 'self-queue',
|
|
69
73
|
STUCK_LOOP: 'stuck-loop',
|
|
70
74
|
POST_AC_OVERRUN: 'post-ac-overrun',
|
|
71
75
|
NO_SENTINEL: 'no-sentinel',
|
|
@@ -75,6 +79,10 @@ const FAILURE_CLASSES = {
|
|
|
75
79
|
};
|
|
76
80
|
|
|
77
81
|
const PREVENTION_HINTS = {
|
|
82
|
+
[FAILURE_CLASSES.ALREADY_SHIPPED]:
|
|
83
|
+
"This run found its acceptance criteria already satisfied by a prior commit and correctly made no change, so no commit landed and the verifier returned `pass_no_commit`. This is a stale re-run, not an execution failure — the PRD's `.md` was never moved out of `session-manager-operations/scheduler/prds/` after the work shipped. Archive the PRD into `session-manager-operations/scheduler/prds-archived/` instead of re-queuing or re-running it.",
|
|
84
|
+
[FAILURE_CLASSES.SELF_QUEUE]:
|
|
85
|
+
"This run either invoked /develop or /process-feedback from inside its own headless execution, or backgrounded a long-running command and called ScheduleWakeup to check back later — both are the 'you ARE the executor — never re-queue or self-schedule' anti-pattern. A headless PRD run must perform its own acceptance criteria directly and has no next turn to resume it (standards.md → Execution discipline).",
|
|
78
86
|
[FAILURE_CLASSES.STUCK_LOOP]:
|
|
79
87
|
'Bound every command with `timeout <N> <cmd>` — never leave an unbounded `until`/`while true`/`sleep` poll in a PRD body (PRD_AUTHORING.md loop-hang guidance).',
|
|
80
88
|
[FAILURE_CLASSES.POST_AC_OVERRUN]:
|
|
@@ -104,6 +112,9 @@ function extractRcaBlock(text) {
|
|
|
104
112
|
return m ? m[1].trim() : null;
|
|
105
113
|
}
|
|
106
114
|
|
|
115
|
+
const ALREADY_SHIPPED_RE = /already (fully )?(satisfied|implemented|committed|done|shipped)|was (already )?(implemented|committed) in|nothing (new )?to commit|no (code )?changes were needed/i;
|
|
116
|
+
const SELF_QUEUE_SKILL_RE = /Launching skill: session-manager-dev:(develop|process-feedback)/;
|
|
117
|
+
const SELF_QUEUE_WAKEUP_RE = /ScheduleWakeup/;
|
|
107
118
|
const STUCK_LOOP_RE = /\b(until\s|while\s+true|sleep\s)/i;
|
|
108
119
|
const AC_CHECKBOX_RE = /^\s*[-*]\s*\[[xX]\]/;
|
|
109
120
|
const SENTINEL_PASS_RE = /SCHEDULER_VERDICT:\s*PASS/;
|
|
@@ -120,6 +131,29 @@ const POST_AC_OVERRUN_MIN_TAIL_FRACTION = 0.3;
|
|
|
120
131
|
function classifyFailure({ verdict, logTail }) {
|
|
121
132
|
const lines = (logTail || '').split('\n');
|
|
122
133
|
|
|
134
|
+
// Checked first, before SELF_QUEUE/STUCK_LOOP: a correct executor that finds
|
|
135
|
+
// its acceptance criteria already satisfied by a prior commit makes no
|
|
136
|
+
// change and truthfully prints a PASS sentinel, so the run lands
|
|
137
|
+
// `pass_no_commit` (or `no_verdict_sentinel`, e.g. when it exits before the
|
|
138
|
+
// finish-protocol sentinel). Gated tightly on verdict so a tail that merely
|
|
139
|
+
// *mentions* "already implemented" (e.g. quoting a PRD body) while genuinely
|
|
140
|
+
// failing for another reason doesn't get misclassified as this benign case
|
|
141
|
+
// (incident: PRD 812-rca-self-delegation-failure-class re-ran 27 minutes
|
|
142
|
+
// after its own fix landed in 9cf0384 and was misclassified NO_SENTINEL).
|
|
143
|
+
if (
|
|
144
|
+
(verdict === 'pass_no_commit' || verdict === 'no_verdict_sentinel') &&
|
|
145
|
+
ALREADY_SHIPPED_RE.test(logTail || '')
|
|
146
|
+
) {
|
|
147
|
+
return FAILURE_CLASSES.ALREADY_SHIPPED;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// Checked before STUCK_LOOP: a backgrounded command + ScheduleWakeup
|
|
151
|
+
// (variant b) can land words like "sleep"/"poll" in the tail via the PRD's
|
|
152
|
+
// own AC text, which would otherwise false-match STUCK_LOOP_RE (see PRD 771).
|
|
153
|
+
if (SELF_QUEUE_SKILL_RE.test(logTail || '') || SELF_QUEUE_WAKEUP_RE.test(logTail || '')) {
|
|
154
|
+
return FAILURE_CLASSES.SELF_QUEUE;
|
|
155
|
+
}
|
|
156
|
+
|
|
123
157
|
const tailWindow = lines.slice(-STUCK_LOOP_WINDOW);
|
|
124
158
|
if (tailWindow.some((l) => STUCK_LOOP_RE.test(l))) {
|
|
125
159
|
return FAILURE_CLASSES.STUCK_LOOP;
|
|
@@ -301,12 +335,28 @@ async function fileRcaFeedback({ job, runDir, verdict, annotations, investigatio
|
|
|
301
335
|
console.log(`[rca] skip: destPath escaped dest dir (${destPath})`);
|
|
302
336
|
return { filed: false, reason: 'unsafe-path' };
|
|
303
337
|
}
|
|
338
|
+
// /process-feedback archives dispositioned items to <dest.dir>/processed/
|
|
339
|
+
// immediately at disposition time (its own README convention) — long
|
|
340
|
+
// before this hook's own run-verify/self-heal passes might touch the same
|
|
341
|
+
// (slug, runId) again. Once archived, the live-dir check alone goes false
|
|
342
|
+
// and a re-trigger would refile a duplicate straight into the live inbox.
|
|
343
|
+
const processedPath = path.resolve(path.join(dest.dir, 'processed', fileName));
|
|
344
|
+
if (!processedPath.startsWith(path.resolve(path.join(dest.dir, 'processed')) + path.sep)) {
|
|
345
|
+
console.log(`[rca] skip: processedPath escaped processed dir (${processedPath})`);
|
|
346
|
+
return { filed: false, reason: 'unsafe-path' };
|
|
347
|
+
}
|
|
304
348
|
|
|
305
|
-
const
|
|
349
|
+
const liveExists = fs.existsSync(destPath);
|
|
350
|
+
const processedExists = !liveExists && fs.existsSync(processedPath);
|
|
351
|
+
const alreadyFiled = liveExists || processedExists;
|
|
306
352
|
if (alreadyFiled && !investigationText) {
|
|
353
|
+
const existingPath = liveExists ? destPath : processedPath;
|
|
307
354
|
console.log(`[rca] skip: already filed for ${job.slug}@${job.runId}`);
|
|
308
|
-
return { filed: false, reason: 'duplicate', path:
|
|
355
|
+
return { filed: false, reason: 'duplicate', path: existingPath };
|
|
309
356
|
}
|
|
357
|
+
// An investigationText update must target wherever the file actually
|
|
358
|
+
// lives (live dir or processed/), not assume the live dir.
|
|
359
|
+
const targetPath = liveExists || !processedExists ? destPath : processedPath;
|
|
310
360
|
|
|
311
361
|
const meta = readRunMeta(runDir, job.slug);
|
|
312
362
|
const logPath = runDir ? path.join(runDir, `${job.slug}.log`) : null;
|
|
@@ -320,10 +370,10 @@ async function fileRcaFeedback({ job, runDir, verdict, annotations, investigatio
|
|
|
320
370
|
});
|
|
321
371
|
|
|
322
372
|
config.addAllowedRoot(dest.allowlistRoot);
|
|
323
|
-
await config.writeTextAtomic(
|
|
373
|
+
await config.writeTextAtomic(targetPath, markdown);
|
|
324
374
|
|
|
325
|
-
console.log(`[rca] ${alreadyFiled ? 'updated (investigation)' : 'filed'} ${
|
|
326
|
-
return { filed: true, path:
|
|
375
|
+
console.log(`[rca] ${alreadyFiled ? 'updated (investigation)' : 'filed'} ${targetPath}`);
|
|
376
|
+
return { filed: true, path: targetPath, updated: alreadyFiled };
|
|
327
377
|
} catch (e) {
|
|
328
378
|
console.error('[rca] error filing RCA feedback', e?.message ?? String(e));
|
|
329
379
|
return { filed: false, reason: 'error', error: e?.message ?? String(e) };
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* terminalRunOutcome.cjs — history-independent fallback for detecting an
|
|
3
|
+
* already-terminal (completed/failed) run for a PRD slug, by reading the
|
|
4
|
+
* newest run directory's <slug>.meta.json + <slug>.verdicts.json sidecars
|
|
5
|
+
* directly off disk instead of relying on history.jsonl (which may not
|
|
6
|
+
* exist yet — see PRD 812-689-fix-fix-distribute-adminserver-routes).
|
|
7
|
+
*
|
|
8
|
+
* Pure-ish: takes an injectable runsDir + fsImpl so tests can point it at a
|
|
9
|
+
* temp directory. Fails safe to null on any fs/JSON error so callers treat
|
|
10
|
+
* "unknown" the same as "no history match" (resurrect as pending).
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
'use strict';
|
|
14
|
+
|
|
15
|
+
const fs = require('node:fs');
|
|
16
|
+
const path = require('node:path');
|
|
17
|
+
|
|
18
|
+
/** Completed-equivalent verdicts — a clean, exit-0 run with nothing left to fix. */
|
|
19
|
+
const COMPLETED_EQUIVALENT_VERDICTS = new Set([
|
|
20
|
+
'clean',
|
|
21
|
+
'pass_no_commit_target_verified',
|
|
22
|
+
'pass_no_commit_already_shipped',
|
|
23
|
+
'pass_no_commit_prior_run_verified',
|
|
24
|
+
]);
|
|
25
|
+
|
|
26
|
+
// Bounds the scan: only the newest few run dirs are stat'd per slug, never
|
|
27
|
+
// the full runs/ directory (which can hold thousands of entries).
|
|
28
|
+
const MAX_DIRS_SCANNED = 5;
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Returns { status: 'completed'|'failed', runId, finishedAt } for the newest
|
|
32
|
+
* run directory containing a `<slug>.meta.json`, or null if none is found or
|
|
33
|
+
* any fs/JSON error occurs (fail-safe: caller falls back to resurrecting the
|
|
34
|
+
* slug, unchanged current behavior).
|
|
35
|
+
*/
|
|
36
|
+
function latestTerminalOutcomeForSlug(slug, { runsDir, fsImpl = fs } = {}) {
|
|
37
|
+
if (!slug || !runsDir) return null;
|
|
38
|
+
let dirs;
|
|
39
|
+
try {
|
|
40
|
+
dirs = fsImpl.readdirSync(runsDir);
|
|
41
|
+
} catch {
|
|
42
|
+
return null;
|
|
43
|
+
}
|
|
44
|
+
if (!Array.isArray(dirs) || dirs.length === 0) return null;
|
|
45
|
+
|
|
46
|
+
// Filter (cheap existence check) to dirs that actually have a run for this
|
|
47
|
+
// slug, mirroring resolveRunId's existing pattern in scheduler.cjs — this
|
|
48
|
+
// is a stat, not a read+parse.
|
|
49
|
+
const matches = dirs.filter((d) => {
|
|
50
|
+
try {
|
|
51
|
+
return fsImpl.existsSync(path.join(runsDir, d, `${slug}.meta.json`));
|
|
52
|
+
} catch {
|
|
53
|
+
return false;
|
|
54
|
+
}
|
|
55
|
+
});
|
|
56
|
+
if (!matches.length) return null;
|
|
57
|
+
|
|
58
|
+
// Dir names are ISO timestamps with `:`/`.` replaced by `-` — lexical
|
|
59
|
+
// descending sort is chronological descending. Only read+parse (the
|
|
60
|
+
// expensive part) the newest few.
|
|
61
|
+
matches.sort().reverse();
|
|
62
|
+
const candidates = matches.slice(0, MAX_DIRS_SCANNED);
|
|
63
|
+
|
|
64
|
+
for (const dir of candidates) {
|
|
65
|
+
const metaPath = path.join(runsDir, dir, `${slug}.meta.json`);
|
|
66
|
+
let meta;
|
|
67
|
+
try {
|
|
68
|
+
meta = JSON.parse(fsImpl.readFileSync(metaPath, 'utf8'));
|
|
69
|
+
} catch {
|
|
70
|
+
return null;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// meta.json's finishedAt is an epoch-ms number; callers (e.g. the
|
|
74
|
+
// history-archive-candidate path, which feeds Date.parse) expect an ISO
|
|
75
|
+
// string like queueHistory's finishedAt — normalize here.
|
|
76
|
+
const finishedAt = typeof meta.finishedAt === 'number'
|
|
77
|
+
? new Date(meta.finishedAt).toISOString()
|
|
78
|
+
: (meta.finishedAt ?? null);
|
|
79
|
+
|
|
80
|
+
if (meta.exitCode !== 0) {
|
|
81
|
+
return { status: 'failed', runId: dir, finishedAt };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
const verdictsPath = path.join(runsDir, dir, `${slug}.verdicts.json`);
|
|
85
|
+
let verdicts;
|
|
86
|
+
try {
|
|
87
|
+
verdicts = JSON.parse(fsImpl.readFileSync(verdictsPath, 'utf8'));
|
|
88
|
+
} catch {
|
|
89
|
+
return null;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
if (COMPLETED_EQUIVALENT_VERDICTS.has(verdicts.verdict)) {
|
|
93
|
+
return { status: 'completed', runId: dir, finishedAt };
|
|
94
|
+
}
|
|
95
|
+
return { status: 'failed', runId: dir, finishedAt };
|
|
96
|
+
}
|
|
97
|
+
return null;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
module.exports = { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS, MAX_DIRS_SCANNED };
|