claude-code-session-manager 0.78.0 → 0.79.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/dist/assets/{AgentLibrary-13pfo8uY.js → AgentLibrary-COtVRqBR.js} +1 -1
  2. package/dist/assets/{DataModel-SUyQbFlg.js → DataModel-CSEKw_OR.js} +1 -1
  3. package/dist/assets/{History-2GJMS703.js → History-CHHovrAO.js} +1 -1
  4. package/dist/assets/{Hooks-DM2nS3RT.js → Hooks-BZU6C3x6.js} +1 -1
  5. package/dist/assets/{HostBilko-BLeC-lpp.js → HostBilko-CqTUoq37.js} +1 -1
  6. package/dist/assets/{Library-BaRkU9m0.js → Library-BtxdyTLz.js} +1 -1
  7. package/dist/assets/{ListDetail-D5scjSKq.js → ListDetail-qZc7Zm-6.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-B1lgAo9T.js → MarkdownEditor-BHe_4fJR.js} +1 -1
  9. package/dist/assets/{McpServers-BzyThQSM.js → McpServers-7Z98HLNo.js} +1 -1
  10. package/dist/assets/{Memory-7UdaOTtl.js → Memory-CR72KoyP.js} +1 -1
  11. package/dist/assets/{Panel-JbTMaOPq.js → Panel-pL6H3dpQ.js} +1 -1
  12. package/dist/assets/{Permissions-UBam0bJG.js → Permissions-CWSWjyXM.js} +1 -1
  13. package/dist/assets/{Plugins-B3gUDkeb.js → Plugins-CN6lX2lt.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-CeHOub7m.js → ProvenanceBadge-BXSXwIsk.js} +1 -1
  15. package/dist/assets/{SaveBar-BcvQEq6h.js → SaveBar-BlB5TGpR.js} +1 -1
  16. package/dist/assets/{Scheduler-Dc5qiP24.js → Scheduler-DRciWUmR.js} +1 -1
  17. package/dist/assets/{ScopeSwitcher-BvGQmw4Y.js → ScopeSwitcher-kFrXtjpr.js} +1 -1
  18. package/dist/assets/{Settings-C2dEFb-v.js → Settings-BXuyf4lJ.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-CIwlosBc.js → SkillReferenceGraph-Dfacb0PE.js} +1 -1
  20. package/dist/assets/{Skills-C0GzzVrQ.js → Skills-CHqcpiyt.js} +1 -1
  21. package/dist/assets/{SystemPrompt-mtGPK8zo.js → SystemPrompt-fxXm0BZr.js} +1 -1
  22. package/dist/assets/{TagLibrary-DX54-mpd.js → TagLibrary-DOz65ZTz.js} +1 -1
  23. package/dist/assets/{TiptapBody-yADC2RWE.js → TiptapBody-D0bWx_9o.js} +1 -1
  24. package/dist/assets/{Toggle-CRxaCYLI.js → Toggle-C9jBwGSx.js} +1 -1
  25. package/dist/assets/{index-D6ymGESc.js → index-DPYa6jbM.js} +3 -3
  26. package/dist/assets/{settingsSchema-TtMvT5Sx.js → settingsSchema-BTPw1bR3.js} +1 -1
  27. package/dist/index.html +1 -1
  28. package/package.json +1 -1
  29. package/src/main/__tests__/runLogRetention.test.cjs +59 -0
  30. package/src/main/__tests__/scheduler-never-stop.test.cjs +157 -0
  31. package/src/main/__tests__/scheduler-no-orphan-run-dir.test.cjs +81 -0
  32. package/src/main/__tests__/scheduler-rate-limit-cooldown-freshness.test.cjs +123 -0
  33. package/src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs +158 -0
  34. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +30 -0
  35. package/src/main/lib/__tests__/reaperHelpers.test.cjs +90 -1
  36. package/src/main/lib/jobDirtFilter.cjs +54 -0
  37. package/src/main/lib/rateLimitDetect.cjs +35 -0
  38. package/src/main/lib/reaperHelpers.cjs +31 -6
  39. package/src/main/lib/runLogRetention.cjs +82 -4
  40. package/src/main/scheduler.cjs +353 -31
@@ -0,0 +1,123 @@
1
+ /**
2
+ * scheduler-rate-limit-cooldown-freshness.test.cjs — PRD 1119.
3
+ *
4
+ * setPaused()'s manual-override cooldown used to suppress WRITING any pause
5
+ * for 5 minutes after a human Resume/Run now, unconditionally — even when
6
+ * the triggering rate-limit observation came from a run that started AFTER
7
+ * the manual clear (i.e. brand-new evidence, not the stale auto-detection
8
+ * the cooldown exists to ignore). Combined with the halt-reset that resets a
9
+ * rate-limited job straight back to 'pending', this produced a hot dispatch
10
+ * loop: dispatch -> 429 in seconds -> reset to pending -> pause suppressed
11
+ * -> dispatch again, on a ~12s cadence, until the cooldown itself expired.
12
+ *
13
+ * isCooldownSuppressed() is the pure predicate this PRD extracts so both
14
+ * directions of the fix (fresh observation bypasses the cooldown; stale one
15
+ * stays suppressed) are directly unit-testable without touching queue.json,
16
+ * mutate(), or any fs state. nextRapidRateLimitCount() is the pure reducer
17
+ * behind the independent rapid-repeat hard-pause cap (a second circuit
18
+ * breaker for when the computed resumeAt is itself stale/wrong and would
19
+ * otherwise let the resume timer keep re-clearing the pause every ~30s,
20
+ * with every subsequent dispatch genuinely "fresh").
21
+ *
22
+ * Run: timeout 60 npx vitest run src/main/__tests__/scheduler-rate-limit-cooldown-freshness.test.cjs
23
+ */
24
+
25
+ 'use strict';
26
+
27
+ import { test, expect } from 'vitest';
28
+ const {
29
+ isCooldownSuppressed,
30
+ nextRapidRateLimitCount,
31
+ CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
32
+ RAPID_RATE_LIMIT_WINDOW_MS,
33
+ MANUAL_PAUSE_COOLDOWN_MS,
34
+ } = require('../scheduler.cjs');
35
+
36
+ const clearedAt = 1_000_000;
37
+
38
+ test('a rate_limit pause triggered by a run that STARTED AFTER the manual clear engages regardless of the cooldown', () => {
39
+ const now = clearedAt + 60_000; // well inside the 5-minute cooldown window
40
+ const observedAt = clearedAt + 5_000; // the triggering run started after the clear
41
+ const suppressed = isCooldownSuppressed({ pauseClearedManuallyAt: clearedAt, now, observedAt, force: false });
42
+ expect(suppressed).toBe(false);
43
+ });
44
+
45
+ test('a rate_limit pause triggered by an observation OLDER than the manual clear stays suppressed', () => {
46
+ const now = clearedAt + 60_000;
47
+ const observedAt = clearedAt - 5_000; // stale: this run started before the human cleared the pause
48
+ const suppressed = isCooldownSuppressed({ pauseClearedManuallyAt: clearedAt, now, observedAt, force: false });
49
+ expect(suppressed).toBe(true);
50
+ });
51
+
52
+ test('no observedAt at all (unrelated pause reasons, e.g. auth/network) is treated as stale — unchanged prior behavior', () => {
53
+ const now = clearedAt + 60_000;
54
+ const suppressed = isCooldownSuppressed({ pauseClearedManuallyAt: clearedAt, now, observedAt: null, force: false });
55
+ expect(suppressed).toBe(true);
56
+ });
57
+
58
+ test('once the cooldown window itself has elapsed, suppression lifts regardless of freshness', () => {
59
+ const now = clearedAt + MANUAL_PAUSE_COOLDOWN_MS + 1;
60
+ const suppressed = isCooldownSuppressed({ pauseClearedManuallyAt: clearedAt, now, observedAt: null, force: false });
61
+ expect(suppressed).toBe(false);
62
+ });
63
+
64
+ test('no manual clear on record never suppresses', () => {
65
+ const suppressed = isCooldownSuppressed({ pauseClearedManuallyAt: null, now: Date.now(), observedAt: null, force: false });
66
+ expect(suppressed).toBe(false);
67
+ });
68
+
69
+ test('force always bypasses the cooldown, fresh or not', () => {
70
+ const now = clearedAt + 1_000;
71
+ const staleObservedAt = clearedAt - 5_000;
72
+ const suppressed = isCooldownSuppressed({ pauseClearedManuallyAt: clearedAt, now, observedAt: staleObservedAt, force: true });
73
+ expect(suppressed).toBe(false);
74
+ });
75
+
76
+ // ---------- rapid-repeat consecutive counter ----------
77
+
78
+ test('a rate-limited run under the rapid window increments the counter', () => {
79
+ expect(nextRapidRateLimitCount(0, { rateLimited: true, durationMs: 4_000 })).toBe(1);
80
+ expect(nextRapidRateLimitCount(1, { rateLimited: true, durationMs: 4_000 })).toBe(2);
81
+ });
82
+
83
+ test('a rate-limited run that ran for >= the rapid window does not increment the counter', () => {
84
+ expect(nextRapidRateLimitCount(1, { rateLimited: true, durationMs: RAPID_RATE_LIMIT_WINDOW_MS })).toBe(1);
85
+ });
86
+
87
+ test('any non-rate-limited outcome resets the counter to 0', () => {
88
+ expect(nextRapidRateLimitCount(2, { rateLimited: false, durationMs: 999_999 })).toBe(0);
89
+ });
90
+
91
+ test('driving the counter to the documented threshold engages the hard-pause condition', () => {
92
+ let count = 0;
93
+ for (let i = 0; i < CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD; i += 1) {
94
+ count = nextRapidRateLimitCount(count, { rateLimited: true, durationMs: 3_000 });
95
+ }
96
+ expect(count).toBe(CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD);
97
+ expect(count >= CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD).toBe(true);
98
+ });
99
+
100
+ test('reaching the threshold forces the pause through even against a STALE observation still inside the cooldown window', () => {
101
+ // This is the scenario the hard cap exists for: freshness alone is not
102
+ // enough when the computed resumeAt is itself stale/wrong and the resume
103
+ // timer keeps re-clearing the pause, so a LATER dispatch's observation can
104
+ // legitimately look "stale" relative to the most recent clear. The
105
+ // rapid-repeat counter is an independent, unconditional circuit breaker.
106
+ let count = 0;
107
+ for (let i = 0; i < CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD; i += 1) {
108
+ count = nextRapidRateLimitCount(count, { rateLimited: true, durationMs: 3_000 });
109
+ }
110
+ const force = count >= CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD;
111
+ const staleObservedAt = clearedAt - 999_999;
112
+ const suppressed = isCooldownSuppressed({ pauseClearedManuallyAt: clearedAt, now: clearedAt + 1_000, observedAt: staleObservedAt, force });
113
+ expect(suppressed).toBe(false);
114
+ });
115
+
116
+ test('one non-rate-limited run in between resets the streak so the hard pause never engages', () => {
117
+ let count = 0;
118
+ count = nextRapidRateLimitCount(count, { rateLimited: true, durationMs: 3_000 });
119
+ count = nextRapidRateLimitCount(count, { rateLimited: true, durationMs: 3_000 });
120
+ count = nextRapidRateLimitCount(count, { rateLimited: false, durationMs: 60_000 }); // job actually ran fine
121
+ count = nextRapidRateLimitCount(count, { rateLimited: true, durationMs: 3_000 });
122
+ expect(count).toBeLessThan(CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD);
123
+ });
@@ -0,0 +1,158 @@
1
+ /**
2
+ * scheduler-rate-limit-spin-guard.test.cjs — PRD 1119 end-to-end repro of the
3
+ * 2026-09-05 incident: a human pressed Resume/Run now on job 204, and for the
4
+ * next hour the scheduler dispatched it, watched it 429 in seconds, reset it
5
+ * to pending (halt-reset), and dispatched it again — 291 times, one every
6
+ * ~12s — because setPaused()'s manual-override cooldown suppressed WRITING
7
+ * the rate_limit pause for the whole 5-minute window, even though every one
8
+ * of those 429s was a brand-new observation from a run that started AFTER
9
+ * the human's clear.
10
+ *
11
+ * This drives the REAL dispatch path (spawnJob, via a stub `claude` binary
12
+ * that always 429s in well under a second) repeatedly, immediately after a
13
+ * simulated manual Resume, and asserts the scheduler engages a genuine pause
14
+ * almost immediately — never anywhere near the 291-dispatch runaway, and
15
+ * never past the documented CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD hard cap
16
+ * even in the worst case. Compressed in wall-clock time (a stub exits in
17
+ * milliseconds, not the ~12s a real `claude -p` round trip took) — the
18
+ * assertion is on DISPATCH COUNT, which is what the fix actually bounds,
19
+ * not on wall-clock duration.
20
+ *
21
+ * Run: timeout 60 npx vitest run src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs
22
+ */
23
+
24
+ 'use strict';
25
+
26
+ import { test, expect, beforeAll, afterAll, afterEach } from 'vitest';
27
+ const fs = require('node:fs');
28
+ const os = require('node:os');
29
+ const path = require('node:path');
30
+ const { execFileSync } = require('node:child_process');
31
+
32
+ let tmpHome;
33
+ let originalHome;
34
+ let scheduler;
35
+ let queueStore;
36
+
37
+ beforeAll(() => {
38
+ originalHome = process.env.HOME;
39
+ tmpHome = fs.mkdtempSync(path.join(os.tmpdir(), 'sm-rate-limit-spin-guard-'));
40
+ process.env.HOME = tmpHome;
41
+ // Worktree isolation is irrelevant to this repro (it's about pause
42
+ // suppression, not tree isolation) and each real `git worktree add` adds
43
+ // real wall-clock overhead this test does not need to pay 40 times over.
44
+ process.env.SM_JOB_WORKTREE_DISABLE = '1';
45
+ scheduler = require('../scheduler.cjs');
46
+ queueStore = require('../lib/queueStore.cjs');
47
+ });
48
+
49
+ afterAll(() => {
50
+ process.env.HOME = originalHome;
51
+ delete process.env.SM_JOB_WORKTREE_DISABLE;
52
+ fs.rmSync(tmpHome, { recursive: true, force: true });
53
+ });
54
+
55
+ afterEach(() => {
56
+ delete process.env.SM_CLAUDE_BIN;
57
+ });
58
+
59
+ function git(args, cwd) {
60
+ return execFileSync('git', args, { cwd, encoding: 'utf8' });
61
+ }
62
+
63
+ function initRepo(dir) {
64
+ fs.mkdirSync(dir, { recursive: true });
65
+ git(['init', '-q'], dir);
66
+ git(['config', 'user.email', 'test@example.com'], dir);
67
+ git(['config', 'user.name', 'Test'], dir);
68
+ fs.writeFileSync(path.join(dir, 'README.md'), 'hello\n', 'utf8');
69
+ git(['add', '-A'], dir);
70
+ git(['commit', '-q', '-m', 'initial'], dir);
71
+ }
72
+
73
+ function registerActiveProject(cwd) {
74
+ const projectsDir = path.join(tmpHome, '.claude', 'projects');
75
+ const slugDir = path.join(projectsDir, 'spin-guard-project-slug');
76
+ fs.mkdirSync(slugDir, { recursive: true });
77
+ fs.writeFileSync(path.join(slugDir, 'transcript.jsonl'), JSON.stringify({ cwd }) + '\n');
78
+ queueStore.bustCwdCache();
79
+ }
80
+
81
+ function writeProjectQueue(cwd, jobs) {
82
+ const stateDir = path.join(cwd, 'session-manager-operations', 'scheduler', 'state');
83
+ fs.mkdirSync(stateDir, { recursive: true });
84
+ const queuePath = path.join(stateDir, 'queue.json');
85
+ fs.writeFileSync(queuePath, JSON.stringify({ jobs }, null, 2));
86
+ return queuePath;
87
+ }
88
+
89
+ // Stub `claude` binary: always dies as a rate limit, in well under 30s (same
90
+ // technique as scheduler-prd-persona-spawn.test.cjs's argv-marker stub) —
91
+ // emits the canonical 429 signal detectRateLimitInLog looks for, then exits
92
+ // non-zero almost instantly, and bumps a counter file each invocation so the
93
+ // test can assert exactly how many times it was actually spawned.
94
+ function writeAlways429ClaudeStub(counterPath) {
95
+ const stubPath = path.join(os.tmpdir(), `sm-claude-stub-429-${process.pid}-${Math.floor(Math.random() * 1e9)}.cjs`);
96
+ const body = `
97
+ const fs = require('fs');
98
+ fs.appendFileSync(${JSON.stringify(counterPath)}, 'x');
99
+ process.stdout.write(JSON.stringify({ type: 'result', subtype: 'error', is_error: true, api_error_status: 429 }) + '\\n');
100
+ process.exit(1);
101
+ `;
102
+ fs.writeFileSync(stubPath, `#!${process.execPath}\n${body}\n`, { mode: 0o755 });
103
+ return stubPath;
104
+ }
105
+
106
+ test('a manual Resume immediately followed by a permanently-429ing job never produces a runaway dispatch loop', async () => {
107
+ const projectCwd = fs.mkdtempSync(path.join(tmpHome, 'spin-guard-project-'));
108
+ initRepo(projectCwd);
109
+ registerActiveProject(projectCwd);
110
+ const slug = `1119-spin-guard-${process.pid}`;
111
+ const prdsDir = path.join(projectCwd, 'session-manager-operations', 'scheduler', 'prds');
112
+ fs.mkdirSync(prdsDir, { recursive: true });
113
+ fs.writeFileSync(path.join(prdsDir, `${slug}.md`), 'Do the thing.', 'utf8');
114
+
115
+ const counterPath = path.join(tmpHome, 'dispatch-count.txt');
116
+ fs.writeFileSync(counterPath, '');
117
+ process.env.SM_CLAUDE_BIN = writeAlways429ClaudeStub(counterPath);
118
+
119
+ writeProjectQueue(projectCwd, [
120
+ { slug, status: 'pending', cwd: projectCwd },
121
+ ]);
122
+
123
+ // The human just pressed Resume/Run now — this is the exact call clearPause
124
+ // makes from the IPC handlers, and it starts the 5-minute cooldown window.
125
+ await scheduler.clearPause('manual');
126
+
127
+ const job = { slug, cwd: projectCwd };
128
+
129
+ // Simulate the tick loop's dispatch decision being driven repeatedly "over
130
+ // ten minutes" — compressed here to however many cycles a permanently-429
131
+ // job can actually complete, since each cycle is a real (fast) subprocess
132
+ // spawn rather than a real 12s `claude -p` round trip. tickQueue()'s own
133
+ // paused early-return (unchanged by this PRD) is exactly what this loop
134
+ // reproduces by hand: stop dispatching the moment the queue is paused.
135
+ const MAX_TICKS = 40;
136
+ for (let i = 0; i < MAX_TICKS; i += 1) {
137
+ // eslint-disable-next-line no-await-in-loop
138
+ const state = await queueStore.readMerged();
139
+ if (state.paused) break;
140
+ const row = state.jobs.find((j) => j.slug === slug);
141
+ if (!row || row.status !== 'pending') break;
142
+ const runId = `run-${i}`;
143
+ const runDir = path.join(scheduler.RUNS_DIR, runId);
144
+ fs.mkdirSync(runDir, { recursive: true });
145
+ // eslint-disable-next-line no-await-in-loop
146
+ await scheduler.spawnJob(job, runId, runDir, projectCwd);
147
+ }
148
+
149
+ const dispatchCount = fs.readFileSync(counterPath, 'utf8').length;
150
+ expect(dispatchCount).toBeLessThanOrEqual(scheduler.CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD);
151
+ expect(dispatchCount).toBeGreaterThan(0);
152
+ expect(dispatchCount).toBeLessThan(10); // nowhere near the 291-dispatch incident
153
+
154
+ const finalState = await queueStore.readMerged();
155
+ expect(finalState.paused).toBeTruthy();
156
+ expect(finalState.paused.reason).toBe('rate_limit');
157
+ expect(finalState.jobs.find((j) => j.slug === slug).status).toBe('pending');
158
+ });
@@ -271,6 +271,36 @@ test('reapDeadRunningJobs: a failed reap with a persisted guardBaseline names th
271
271
  assert.match(lastTransition.reason, /left 1 files uncommitted/);
272
272
  });
273
273
 
274
+ test('reapDeadRunningJobs: a rate-limited death (api_error_status:429) is reset to pending, not stamped failed, and the reason names the rate limit (PRD 1117)', async () => {
275
+ const projectCwd = path.join(tmpHome, 'h-project');
276
+ fs.mkdirSync(projectCwd, { recursive: true });
277
+ registerActiveProject(projectCwd);
278
+
279
+ const runId = 'run-rate-limited';
280
+ const queuePath = writeProjectQueue(projectCwd, [
281
+ {
282
+ slug: 'rate-limited-job',
283
+ status: 'running',
284
+ cwd: projectCwd,
285
+ runId,
286
+ runtime: { pid: 999999 }, // guaranteed-dead pid — the reaper won the race
287
+ },
288
+ ]);
289
+ // Real-shape tail: a 429 result event, the exact signal the reaper
290
+ // previously had no branch for and stamped terminal 'failed'.
291
+ writeRunLog(runId, 'rate-limited-job', [
292
+ '{"type":"result","subtype":"success","is_error":true,"api_error_status":429,"rateLimitType":"seven_day","result":"You\'ve reached your Fable limit.","uuid":"x"}',
293
+ ]);
294
+
295
+ await reapDeadRunningJobs();
296
+
297
+ const jobs = JSON.parse(fs.readFileSync(queuePath, 'utf8')).jobs;
298
+ const row = jobs.find((j) => j.slug === 'rate-limited-job');
299
+ assert.equal(row.status, 'pending', 'a rate-limited reap must be retryable, never terminal failed');
300
+ assert.match(row.error, /rate limit/i, 'the reset reason must name the rate limit, not read as a generic reap failure');
301
+ assert.equal(row.runtime, undefined);
302
+ });
303
+
274
304
  test('reapDeadRunningJobs skips in-place salvage (no whole-tree dump) when the row has no persisted guardBaseline', async () => {
275
305
  const projectCwd = path.join(tmpHome, 'f-project');
276
306
  initRepo(projectCwd);
@@ -16,7 +16,11 @@
16
16
  // vitest, NOT node:test — this repo's suite is vitest-only (CLAUDE.md).
17
17
  import { test } from 'vitest';
18
18
  const assert = require('node:assert/strict');
19
- const { selectReapableJobs, mapOutcomeToGateOutcome } = require('../reaperHelpers.cjs');
19
+ const fs = require('node:fs');
20
+ const os = require('node:os');
21
+ const path = require('node:path');
22
+ const { selectReapableJobs, mapOutcomeToGateOutcome, classifyRunOutcome } = require('../reaperHelpers.cjs');
23
+ const { detectRateLimitInLog } = require('../rateLimitDetect.cjs');
20
24
 
21
25
  const NOW = Date.parse('2026-09-01T12:00:00.000Z');
22
26
  const agoMin = (m) => new Date(NOW - m * 60_000).toISOString();
@@ -131,3 +135,88 @@ test('mapOutcomeToGateOutcome: no_result -> never_ran', () => {
131
135
  test('mapOutcomeToGateOutcome: unknown -> unknown', () => {
132
136
  assert.strictEqual(mapOutcomeToGateOutcome('unknown'), 'unknown');
133
137
  });
138
+
139
+ // detectRateLimitInLog / classifyRunOutcome — PRD 1117: the reaper's
140
+ // rate-limit detection must be the SAME single source of truth spawnJob
141
+ // uses (detectRateLimitInLog, shared via lib/rateLimitDetect.cjs), and a
142
+ // rate-limited death must classify as a NEW, distinct 'rate_limited'
143
+ // outcome — never collapsed into 'failed', and never re-labelling
144
+ // 'success'/'failed'/'no_result' (which keep their prior meanings, per the
145
+ // mapOutcomeToGateOutcome tests above, unmodified).
146
+
147
+ function writeTmpLog(contents) {
148
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'sm-reaper-rate-limit-test-'));
149
+ const p = path.join(dir, 'run.log');
150
+ fs.writeFileSync(p, contents);
151
+ return p;
152
+ }
153
+
154
+ test('detectRateLimitInLog: rateLimitType":"five_hour" variant', () => {
155
+ const p = writeTmpLog('{"type":"result","rate_limit_info":{"rateLimitType":"five_hour"}}\n');
156
+ assert.strictEqual(detectRateLimitInLog(p), true);
157
+ });
158
+
159
+ test('detectRateLimitInLog: rateLimitType":"seven_day" variant (missed pre-PRD-1117)', () => {
160
+ const p = writeTmpLog('{"type":"rate_limit_event","rate_limit_info":{"rateLimitType":"seven_day","unifiedWindows":{"five_hour":{"utilization":0}}}}\n');
161
+ assert.strictEqual(detectRateLimitInLog(p), true);
162
+ });
163
+
164
+ test('detectRateLimitInLog: api_error_status":429 variant', () => {
165
+ const p = writeTmpLog('{"type":"result","is_error":true,"api_error_status":429,"result":"nope"}\n');
166
+ assert.strictEqual(detectRateLimitInLog(p), true);
167
+ });
168
+
169
+ test('detectRateLimitInLog: "You\'ve reached your <model> limit" variant (missed pre-PRD-1117)', () => {
170
+ const p = writeTmpLog('{"type":"result","result":"You\'ve reached your Fable limit. Switch to another model."}\n');
171
+ assert.strictEqual(detectRateLimitInLog(p), true);
172
+ });
173
+
174
+ test('detectRateLimitInLog: real evidence log (2026-09-05 204-mercury-steam-horse 429)', () => {
175
+ const evidencePath = '/home/bilko/.claude/session-manager/scheduled-plans/runs/2026-09-05T16-32-18-381Z/204-mercury-steam-horse.log';
176
+ if (!fs.existsSync(evidencePath)) return; // machine-local fixture; skip elsewhere
177
+ assert.strictEqual(detectRateLimitInLog(evidencePath), true);
178
+ });
179
+
180
+ test('classifyRunOutcome: a 429 log tail classifies as the new distinct rate_limited outcome, not failed', () => {
181
+ const p = writeTmpLog('{"type":"result","subtype":"success","is_error":true,"api_error_status":429,"rateLimitType":"seven_day","result":"You\'ve reached your Fable limit."}\n');
182
+ assert.strictEqual(classifyRunOutcome(p), 'rate_limited');
183
+ });
184
+
185
+ test('classifyRunOutcome: a genuine error with no rate-limit signal still classifies as failed', () => {
186
+ const p = writeTmpLog('{"type":"result","subtype":"error","is_error":true,"result":"Error: expected 2 but got 3"}\n');
187
+ assert.strictEqual(classifyRunOutcome(p), 'failed');
188
+ });
189
+
190
+ // REGRESSION GUARD (2026-09-05). The rate-limit check originally ran as the
191
+ // FIRST statement of classifyRunOutcome, before the result event was parsed.
192
+ // Because the CLI emits an informational rate_limit_event with
193
+ // status:"allowed_warning" on nearly every run once utilization is non-zero,
194
+ // that misclassified genuinely successful runs as 'rate_limited' — which in
195
+ // reapDeadRunningJobs re-queues finished work and spuriously pauses the whole
196
+ // machine. The fixture below is the shape of a REAL successful run log, not a
197
+ // synthetic one-liner: the allowed_warning event that every run carries, then
198
+ // a clean result. If someone moves the check back to the top, this goes red.
199
+ test('classifyRunOutcome: a SUCCESSFUL run whose tail carries an allowed_warning rate_limit_event is still success', () => {
200
+ const p = writeTmpLog([
201
+ '{"type":"rate_limit_event","rate_limit_info":{"status":"allowed_warning","resetsAt":1788643800,"rateLimitType":"five_hour","utilization":0.42,"isUsingOverage":false}}',
202
+ '{"type":"rate_limit_event","rate_limit_info":{"status":"allowed_warning","resetsAt":1789059600,"rateLimitType":"seven_day","utilization":0.59,"isUsingOverage":false}}',
203
+ '{"type":"result","subtype":"success","is_error":false,"num_turns":55,"result":"SCHEDULER_VERDICT: PASS"}',
204
+ ].join('\n') + '\n');
205
+ assert.strictEqual(classifyRunOutcome(p), 'success');
206
+ assert.strictEqual(mapOutcomeToGateOutcome(classifyRunOutcome(p)), 'passed');
207
+ });
208
+
209
+ // The same tail, but the run actually errored: NOW the rate-limit signal is
210
+ // what it claims to be, and the outcome must be the retryable one.
211
+ test('classifyRunOutcome: the same allowed_warning tail on an ERRORED run classifies as rate_limited', () => {
212
+ const p = writeTmpLog([
213
+ '{"type":"rate_limit_event","rate_limit_info":{"status":"allowed_warning","rateLimitType":"seven_day","utilization":0.59}}',
214
+ '{"type":"result","subtype":"success","is_error":true,"api_error_status":429,"result":"You\'ve reached your Fable limit."}',
215
+ ].join('\n') + '\n');
216
+ assert.strictEqual(classifyRunOutcome(p), 'rate_limited');
217
+ });
218
+
219
+ test('classifyRunOutcome: a clean success log still classifies as success', () => {
220
+ const p = writeTmpLog('{"type":"result","subtype":"success","is_error":false,"result":"done"}\n');
221
+ assert.strictEqual(classifyRunOutcome(p), 'success');
222
+ });
@@ -0,0 +1,54 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * jobDirtFilter.cjs — strip paths a scheduled job can NEVER be responsible
5
+ * for out of its post-run dirty delta.
6
+ *
7
+ * The commit guard parks a job on `needs_review` ("finish protocol
8
+ * incomplete: N uncommitted file(s)") whenever its run leaves tracked files
9
+ * dirty. That is the right rule for source files. It is wrong for
10
+ * `session-manager-operations/`: the SINGLE-WRITER LAW (lib/opsOwnership.cjs)
11
+ * says the app itself owns every namespace under that root — `scheduler`
12
+ * writes queue.json/history.jsonl on every dispatch and every finalize,
13
+ * `prompt-sessions` writes active-index.json and transcripts continuously.
14
+ * Those writes land DURING the job's own run, inside its own guard window,
15
+ * and get attributed to the job.
16
+ *
17
+ * Result before this filter: a job that did everything right and committed
18
+ * cleanly still parked, because the scheduler's own bookkeeping was dirty
19
+ * underneath it. Observed 2026-09-05 on social-signals-trader PRDs 4055 and
20
+ * 4056 (leftover lists led with active-index.json / queue.json /
21
+ * history.jsonl / .max-allocated-group) and on starry-night-ships PRD 206.
22
+ * Each parked row then blocked every job that dependsOn it, which is how a
23
+ * queue with 40 ready PRDs stops dead.
24
+ *
25
+ * Deliberately narrow: ONLY the app-owned operations root. A job that dirties
26
+ * real source and walks away is still a genuine finish-protocol violation and
27
+ * must still park.
28
+ */
29
+
30
+ /** The one app-owned root. Matches at any depth, and only as a path segment. */
31
+ const OPS_ROOT_SEGMENT = 'session-manager-operations';
32
+
33
+ /** True when `p` lives under an app-owned operations root. Pure. */
34
+ function isAppOwnedChurn(p) {
35
+ if (typeof p !== 'string' || p === '') return false;
36
+ // Normalise Windows-style separators defensively; git porcelain emits '/'.
37
+ const parts = p.replace(/\\/g, '/').split('/');
38
+ return parts.includes(OPS_ROOT_SEGMENT);
39
+ }
40
+
41
+ /**
42
+ * Remove app-owned churn from a dirty-path list.
43
+ *
44
+ * Preserves the caller's null contract: `null` means "git status itself was
45
+ * unavailable", which is NOT the same fact as "the job left nothing", and
46
+ * every caller already branches on it.
47
+ */
48
+ function stripAppOwnedChurn(paths) {
49
+ if (paths === null || paths === undefined) return paths;
50
+ if (!Array.isArray(paths)) return paths;
51
+ return paths.filter((p) => !isAppOwnedChurn(p));
52
+ }
53
+
54
+ module.exports = { stripAppOwnedChurn, isAppOwnedChurn, OPS_ROOT_SEGMENT };
@@ -0,0 +1,35 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * rateLimitDetect.cjs — the single source of truth for "did this claude -p
5
+ * log tail show a rate-limit death". Shared by spawnJob() (a still-running
6
+ * process that just exited) and reapDeadRunningJobs()/classifyRunOutcome()
7
+ * (a process the reaper found already dead) — see PRD 1117. Before this,
8
+ * only spawnJob checked this; the reaper had no rate-limit branch at all,
9
+ * so a rate-limited exit that the reaper won the race to finalize got
10
+ * stamped terminal 'failed' instead of retryable 'pending'. There must
11
+ * never be a second, independently-drifting regex set at either call site.
12
+ */
13
+
14
+ const { readTail } = require('./fileTail.cjs');
15
+
16
+ /** Scan the tail of a job's log for the canonical rate-limit signal. We look
17
+ * at the last 16 KB — final result event always lands at the end. Covers
18
+ * both unified-window rate-limit types (five_hour, seven_day), the raw
19
+ * 429 status, and the two human-readable limit-message phrasings the CLI
20
+ * emits ("You've hit your limit" / "You've reached your <model> limit"). */
21
+ function detectRateLimitInLog(logPath) {
22
+ try {
23
+ const text = readTail(logPath, 16384);
24
+ if (!text) return false;
25
+ return /"rateLimitType":"five_hour"/.test(text)
26
+ || /"rateLimitType":"seven_day"/.test(text)
27
+ || /"api_error_status":429/.test(text)
28
+ || /You'?ve hit your limit/.test(text)
29
+ || /You'?ve reached your .* limit/.test(text);
30
+ } catch {
31
+ return false;
32
+ }
33
+ }
34
+
35
+ module.exports = { detectRateLimitInLog };
@@ -9,6 +9,7 @@
9
9
 
10
10
  const fs = require('node:fs');
11
11
  const { readTail } = require('./fileTail.cjs');
12
+ const { detectRateLimitInLog } = require('./rateLimitDetect.cjs');
12
13
 
13
14
  /**
14
15
  * Return true if pid is alive AND its cmdline looks like a claude process.
@@ -38,11 +39,16 @@ function claudePidAlive(pid) {
38
39
  * of its log file and scanning for the LAST `{"type":"result"}` JSONL event.
39
40
  *
40
41
  * Returns:
41
- * 'success' — last result event has subtype=success and is_error !== true
42
- * 'failed' — last result event exists but indicates an error
43
- * 'no_result' — no result event found in the tail (process may have been killed
44
- * before emitting one, or the log is absent/empty)
45
- * 'unknown' — unexpected error reading/parsing (outer catch)
42
+ * 'success' — last result event has subtype=success and is_error !== true
43
+ * 'rate_limited' — the log tail shows the same rate-limit signal spawnJob's own
44
+ * live-process check uses (detectRateLimitInLog, the shared
45
+ * single source of truth) — a NEW, distinct outcome from
46
+ * 'failed' (PRD 1117): a rate-limited death is retryable, not
47
+ * a genuine gate failure, and must never collapse into 'failed'
48
+ * 'failed' — last result event exists but indicates a genuine error
49
+ * 'no_result' — no result event found in the tail (process may have been killed
50
+ * before emitting one, or the log is absent/empty)
51
+ * 'unknown' — unexpected error reading/parsing (outer catch)
46
52
  */
47
53
  function classifyRunOutcome(logPath) {
48
54
  try {
@@ -56,8 +62,27 @@ function classifyRunOutcome(logPath) {
56
62
  if (obj && obj.type === 'result') lastResult = obj;
57
63
  } catch { /* partial line at tail boundary or non-JSON scheduler log line */ }
58
64
  }
65
+ // ORDER IS LOAD-BEARING. The rate-limit check must come AFTER the success
66
+ // determination, never before it. The CLI emits an informational
67
+ // `rate_limit_event` with status:"allowed_warning" on essentially every
68
+ // run once utilization is non-zero, and detectRateLimitInLog matches its
69
+ // "rateLimitType" field — so checking first classified genuinely
70
+ // SUCCESSFUL runs as rate_limited. Measured on 2026-09-05 against the
71
+ // eight most recent runs whose own meta.json recorded exitCode:0 and
72
+ // rateLimited:false, four came back 'rate_limited' (e.g.
73
+ // 200-campaign-toolkit-weak-points-and-stuns: 55 turns, is_error:false,
74
+ // terminalReasonFromHarness:'completed', landed commit 7fd05f7 — matched
75
+ // purely on an allowed_warning five_hour event). In reapDeadRunningJobs
76
+ // that resets a finished job to 'pending' to re-run shipped work AND
77
+ // engages setPaused('rate_limit') with no rate limit in effect — strictly
78
+ // worse than the terminal-'failed' bug the rate_limit branch was added to
79
+ // fix. See the guard test in this file's __tests__ sibling.
80
+ if (lastResult && lastResult.subtype === 'success' && lastResult.is_error !== true) return 'success';
81
+ // Still checked ahead of 'no_result': a run killed mid-flight by a rate
82
+ // limit may never emit a result event at all, and that is a rate-limited
83
+ // death, not silence.
84
+ if (detectRateLimitInLog(logPath)) return 'rate_limited';
59
85
  if (!lastResult) return 'no_result';
60
- if (lastResult.subtype === 'success' && lastResult.is_error !== true) return 'success';
61
86
  return 'failed';
62
87
  } catch {
63
88
  return 'unknown';