claude-code-session-manager 0.66.0 → 0.67.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/dist/assets/{AgentLibrary-Bzg89D5Y.js → AgentLibrary-CiqimwWl.js} +1 -1
  2. package/dist/assets/{History-CNH9vA0A.js → History-DbzjII2Z.js} +1 -1
  3. package/dist/assets/{Hooks-CyTksPza.js → Hooks-C0PTplI_.js} +2 -2
  4. package/dist/assets/{HostBilko-DXQVDNHn.js → HostBilko-vGc1FYLD.js} +1 -1
  5. package/dist/assets/{Library-9E2UOIsm.js → Library-4tU4rg2h.js} +1 -1
  6. package/dist/assets/{ListDetail-CQWU_Yn5.js → ListDetail-BGIvkCwZ.js} +1 -1
  7. package/dist/assets/{MarkdownEditor-DQgfpSef.js → MarkdownEditor-ChFdpCam.js} +1 -1
  8. package/dist/assets/{McpServers-r3qwDIj2.js → McpServers-DsRWWpU3.js} +2 -2
  9. package/dist/assets/{Memory-BaxOpj-3.js → Memory-DmxoxxxY.js} +1 -1
  10. package/dist/assets/{Panel-BhFD8Lqo.js → Panel-KC-rv3jS.js} +1 -1
  11. package/dist/assets/{Permissions-Cj-mODQQ.js → Permissions-DrDjCRFl.js} +2 -2
  12. package/dist/assets/{Plugins-DEL3Fqng.js → Plugins-KexSQwk0.js} +2 -2
  13. package/dist/assets/{ProvenanceBadge-CIAg6-JQ.js → ProvenanceBadge-Rg-jL94y.js} +1 -1
  14. package/dist/assets/SaveBar-BbEo8U2e.js +1 -0
  15. package/dist/assets/{Scheduler-YOuZKkES.js → Scheduler-C4Ti9TCc.js} +1 -1
  16. package/dist/assets/ScopeSwitcher-Di9BmYze.js +1 -0
  17. package/dist/assets/Settings-CxyhPuYQ.js +3 -0
  18. package/dist/assets/{SkillReferenceGraph-CoIwsol8.js → SkillReferenceGraph-0q26RHrn.js} +1 -1
  19. package/dist/assets/{Skills-CN8R6AWn.js → Skills-BXkVyCo5.js} +2 -2
  20. package/dist/assets/SystemPrompt-Bwe3NTbi.js +1 -0
  21. package/dist/assets/{TagLibrary-BKmz2W7B.js → TagLibrary-CnwEf6EI.js} +1 -1
  22. package/dist/assets/{TiptapBody-W7n5SwPM.js → TiptapBody-faCtLj9L.js} +1 -1
  23. package/dist/assets/{Toggle-QuxlVHGI.js → Toggle-Bw7G_-RR.js} +1 -1
  24. package/dist/assets/{index-TejhHSzN.js → index-B6dJ2CsU.js} +323 -323
  25. package/dist/assets/{index-14dBLqE_.css → index-CEnMgeQU.css} +1 -1
  26. package/dist/assets/{settingsSchema-DNqx6BKJ.js → settingsSchema-B5C9hZoS.js} +1 -1
  27. package/dist/index.html +2 -2
  28. package/package.json +1 -1
  29. package/plugins/session-manager-dev/skills/develop/SKILL.md +41 -11
  30. package/src/main/__tests__/develop-skill-failure-modes.test.cjs +70 -0
  31. package/src/main/__tests__/health-per-project-stall.test.cjs +82 -0
  32. package/src/main/__tests__/scheduler-stall-per-project.test.cjs +108 -0
  33. package/src/main/health.cjs +97 -2
  34. package/src/main/lib/schedulerConfig.cjs +7 -0
  35. package/src/main/scheduler.cjs +154 -35
  36. package/dist/assets/ScopeSwitcher-DZ_3gEus.js +0 -1
  37. package/dist/assets/Settings-B0x4oflz.js +0 -3
  38. package/dist/assets/SystemPrompt-DceChpAi.js +0 -1
@@ -210,11 +210,20 @@ can't load skills.
210
210
  separate PRDs into one oversized one; if the sub-task groups would each take real time on
211
211
  their own, that's a signal to split into a chain link instead of one bloated PRD.
212
212
 
213
+ **Preflight — confirm the tool is even in your tool list before you start composing PRDs.**
214
+ Check for `mcp__session-manager-scheduler__scheduler_create_prd` in your available tools as
215
+ the very first thing you do in this step, before any drafting — catching a missing tool here
216
+ costs nothing; catching it after you've already composed and written PRD bodies means
217
+ discarding that work. If it's absent, see "Two failure modes" immediately below — case (b),
218
+ not the reachable-but-erroring fallback.
219
+
213
220
  **`scheduler_create_prd` is the ONLY sanctioned way to author a PRD — not a preference, a
214
221
  rule.** Every PRD reaches disk through the MCP tool
215
222
  (`mcp__session-manager-scheduler__scheduler_create_prd`). Hand-writing the file directly is a
216
- degraded, LAST-RESORT fallback (below) reserved for the single case where the tool itself is
217
- unreachable — never a co-equal alternative to reach for out of habit or convenience. Its input
223
+ degraded, LAST-RESORT fallback (below) reserved for the single case where the tool is present
224
+ but errors as unreachable — never a co-equal alternative to reach for out of habit or
225
+ convenience, and never applicable when the tool isn't in your list at all (see "Two failure
226
+ modes" below). Its input
218
227
  (`title`, `cwd`, `estimateMinutes`, `goal`, `acceptanceCriteria[]`, `implementationNotes`,
219
228
  `outOfScope[]`) maps directly onto the sections below — pass them straight through. **Always
220
229
  pass `sourcePromptId` explicitly, set to the `<epic-id>` resolved in the Epic-gated step
@@ -227,18 +236,37 @@ can't load skills.
227
236
  `parallelGroup` is DEPRECATED and ignored — express ordering with the `dependsOn` input
228
237
  (slugs that must complete first); independent PRDs simply omit it and may run in parallel.
229
238
 
230
- **Fallback — only when the tool errors with "app not running" / admin API unreachable**
231
- (the session-manager Electron app must be running for this MCP tool to work; if it isn't,
232
- don't block on it). This is a deliberate bypass of the service boundary, not a shortcut:
233
- using it means the frontmatter validation, atomic `NN` allocation, standards-pointer
239
+ **Two failure modes — do not conflate them. They have opposite correct responses.**
240
+
241
+ - **(a) Tool PRESENT but ERRORS as "app not running" / admin API unreachable.** The tool
242
+ shows up in your tool list (`mcp__session-manager-scheduler__scheduler_create_prd` is
243
+ callable), but calling it fails because the session-manager Electron app that hosts the
244
+ admin API isn't running right now. This is the ONLY case the manual-write fallback below
245
+ covers. Do not use this path when the tool is reachable but merely returned a validation
246
+ error (bad frontmatter, unresolvable Epic, etc.) — fix the input and retry the tool; a
247
+ validation error is not "the app is not running."
248
+ - **(b) Tool ABSENT from your tool list entirely.** You never see
249
+ `mcp__session-manager-scheduler__scheduler_create_prd` offered at all — there is no error to
250
+ catch, because the tool call is never attempted. This means the `session-manager-scheduler`
251
+ MCP server is not registered for the project you're running against — a **misconfiguration**,
252
+ not "the app is offline." **STOP. Do not write any PRD file, hand-authored or otherwise.**
253
+ Report to the human, by name: "the `session-manager-scheduler` MCP tool is not available in
254
+ this session — the server isn't registered for this project." Point them at the fix: it
255
+ should be registered once at USER scope (`claude mcp add session-manager-scheduler --scope
256
+ user -- node <path-to-session-manager-repo>/scripts/scheduler-mcp-server.cjs`, or run
257
+ `scripts/install-scheduler-mcp-user-scope.sh` from the session-manager repo) so every
258
+ project gets the tool without a per-repo `.mcp.json` edit — do not work around a missing
259
+ tool by hand-writing the file, and do not add a project-local `.mcp.json` entry yourself as
260
+ a substitute; that's the human's call and re-introduces the per-repo drift this fix removes.
261
+
262
+ **Fallback for case (a) only.** This is a deliberate bypass of the service boundary, not a
263
+ shortcut: using it means the frontmatter validation, atomic `NN` allocation, standards-pointer
234
264
  insertion, and Epic-existence check that `scheduler_create_prd` normally performs did not
235
265
  run. **You MUST call this out, visibly, in your report** — state plainly that the app wasn't
236
266
  running, that you hand-authored the PRD file directly instead of using the tool, name the
237
267
  exact file, and flag it for human verification (this bypass is also what
238
268
  `scripts/audit-ops-hygiene.cjs` and the `ops-sweep` skill look for and report as a hygiene
239
- finding, independent of your own report). Do not use this path when the tool is reachable but
240
- merely returned a validation error (bad frontmatter, unresolvable Epic, etc.) — fix the input
241
- and retry the tool; a validation error is not "the app is not running."
269
+ finding, independent of your own report).
242
270
  When you do use it: compute the highest in-use number deterministically yourself — never
243
271
  eyeball or narrow-grep the `ls` (a narrowed pattern like `'^10[0-9]'` silently misses `110+`
244
272
  and collides). PRDs are stored per-project, so `NN` allocation for a given PRD only needs
@@ -448,8 +476,10 @@ single definition of "tracked to done" for both entry paths.
448
476
  ## Notes
449
477
 
450
478
  - Submit each PRD through `scheduler_create_prd`, then confirm — don't draft them inline in chat
451
- for review first, and don't hand-write the file yourself unless the tool is unreachable (app
452
- not running); see the fallback note above, including its mandatory bypass warning.
479
+ for review first. Only hand-write the file when the tool is PRESENT but ERRORS as unreachable
480
+ (app not running) — see the "Two failure modes" note above, including its mandatory bypass
481
+ warning. If the tool is ABSENT from your tool list, that's a misconfiguration, not an offline
482
+ app: stop and tell the human, never hand-write the file.
453
483
  - Don't combine unrelated features into one PRD. One focused, completable unit each.
454
484
  - Don't add a `parallelGroup` frontmatter key — the filename `NN-` prefix drives grouping.
455
485
  - Don't write a PRD to `data/prds/`, `docs/prds/`, the project's own folder, or anywhere outside
@@ -0,0 +1,70 @@
1
+ /**
2
+ * develop-skill-failure-modes.test.cjs — drift guard for the two
3
+ * scheduler_create_prd failure-mode branches in the /develop skill.
4
+ *
5
+ * PRD 1024-1030 incident (2026-08-08): social-signals-trader's session
6
+ * hand-wrote 7 PRDs because scheduler_create_prd was ABSENT from its tool
7
+ * list (the project's .mcp.json didn't register the server yet) — not
8
+ * because the tool errored as unreachable. The old wording only named the
9
+ * "tool errors ... unreachable" case, which an absent tool never matches,
10
+ * so an agent in that situation had no documented path except hand-writing.
11
+ * This test asserts the fix (two named failure modes, with the absent-tool
12
+ * case explicitly forbidding hand-writing) is present and doesn't silently
13
+ * regress back to the single-case wording.
14
+ *
15
+ * Run: timeout 120 npx vitest run src/main/__tests__/develop-skill-failure-modes.test.cjs
16
+ */
17
+
18
+ 'use strict';
19
+
20
+ const fs = require('fs');
21
+ const path = require('path');
22
+
23
+ import { test, expect } from 'vitest';
24
+
25
+ const SKILL_MD = path.resolve(
26
+ __dirname,
27
+ '..',
28
+ '..',
29
+ '..',
30
+ 'plugins',
31
+ 'session-manager-dev',
32
+ 'skills',
33
+ 'develop',
34
+ 'SKILL.md'
35
+ );
36
+
37
+ function readSkill() {
38
+ return fs.readFileSync(SKILL_MD, 'utf8');
39
+ }
40
+
41
+ test('SKILL.md names both failure modes explicitly', () => {
42
+ const text = readSkill();
43
+ expect(text).toMatch(/tool PRESENT but ERRORS/i);
44
+ expect(text).toMatch(/tool ABSENT from your tool list/i);
45
+ });
46
+
47
+ test('the absent-tool branch explicitly forbids hand-writing and says STOP', () => {
48
+ const text = readSkill();
49
+ const absentIdx = text.search(/tool ABSENT from your tool list/i);
50
+ expect(absentIdx).toBeGreaterThan(-1);
51
+ const nearby = text.slice(absentIdx, absentIdx + 1200);
52
+ expect(nearby).toMatch(/STOP/);
53
+ expect(nearby).toMatch(/[Dd]o not write any PRD file/);
54
+ expect(nearby).toMatch(/misconfiguration/i);
55
+ });
56
+
57
+ test('the present-but-erroring branch is the only one pointed at the manual-write fallback', () => {
58
+ const text = readSkill();
59
+ expect(text).toMatch(/Fallback for case \(a\) only/);
60
+ });
61
+
62
+ test('the standalone preflight check appears before PRD composition begins', () => {
63
+ const text = readSkill();
64
+ expect(text).toMatch(/Preflight — confirm the tool is even in your tool list/);
65
+ });
66
+
67
+ test('a preflight-tool-registration fix is named for the human (user-scope claude mcp add)', () => {
68
+ const text = readSkill();
69
+ expect(text).toMatch(/claude mcp add session-manager-scheduler --scope\s*\n?\s*user/);
70
+ });
@@ -0,0 +1,82 @@
1
+ /**
2
+ * health-per-project-stall.test.cjs — per-project problem-count rollup and
3
+ * per-project stall-past-threshold escalation for the scheduler_queue health
4
+ * check. Covers the gap where a project holding ONLY failed/needs_review/
5
+ * quarantined rows (0 running, 0 pending) never tripped evaluateTickLiveness
6
+ * (which requires actual pending work) — the exact way the burrow project's
7
+ * four quarantined PRDs went dark.
8
+ *
9
+ * Run: timeout 120 node --test src/main/__tests__/health-per-project-stall.test.cjs
10
+ */
11
+
12
+ 'use strict';
13
+
14
+ const { test } = require('node:test');
15
+ const assert = require('node:assert/strict');
16
+ const {
17
+ computeProjectProblemCounts,
18
+ evaluatePerProjectStall,
19
+ TICK_STALL_THRESHOLD_MS,
20
+ } = require('../health.cjs');
21
+ const { computeStallSummary } = require('../scheduler.cjs');
22
+
23
+ test('computeProjectProblemCounts: breaks down failed/needs_review/quarantined by project, ignores healthy statuses', () => {
24
+ const jobs = [
25
+ { slug: 'a', cwd: '/burrow', status: 'quarantined' },
26
+ { slug: 'b', cwd: '/burrow', status: 'quarantined' },
27
+ { slug: 'c', cwd: '/burrow', status: 'quarantined' },
28
+ { slug: 'd', cwd: '/burrow', status: 'quarantined' },
29
+ { slug: 'e', cwd: '/other', status: 'failed' },
30
+ { slug: 'f', cwd: '/other', status: 'needs_review' },
31
+ { slug: 'g', cwd: '/other', status: 'running' },
32
+ { slug: 'h', cwd: '/other', status: 'pending' },
33
+ { slug: 'i', cwd: '/other', status: 'completed' },
34
+ ];
35
+ const counts = computeProjectProblemCounts(jobs);
36
+ assert.deepStrictEqual(counts['/burrow'], { failed: 0, needs_review: 0, quarantined: 4 });
37
+ assert.deepStrictEqual(counts['/other'], { failed: 1, needs_review: 1, quarantined: 0 });
38
+ });
39
+
40
+ test('evaluatePerProjectStall: fully-stalled project past threshold flags pastThreshold=true', () => {
41
+ const state = {
42
+ lastRunAt: new Date(Date.now() - (TICK_STALL_THRESHOLD_MS + 60_000)).toISOString(),
43
+ paused: null,
44
+ jobs: [
45
+ { slug: 'a', cwd: '/burrow', status: 'quarantined' },
46
+ { slug: 'b', cwd: '/other', status: 'running' },
47
+ ],
48
+ invalidJobs: [],
49
+ };
50
+ const summary = computeStallSummary(state);
51
+ const result = evaluatePerProjectStall(summary, state.lastRunAt, Date.now(), TICK_STALL_THRESHOLD_MS);
52
+ assert.strictEqual(result['/burrow'].stalled, true);
53
+ assert.strictEqual(result['/burrow'].pastThreshold, true);
54
+ assert.strictEqual(result['/other'].stalled, false);
55
+ });
56
+
57
+ test('evaluatePerProjectStall: stalled but NOT yet past threshold flags pastThreshold=false', () => {
58
+ const state = {
59
+ lastRunAt: new Date(Date.now() - 1000).toISOString(), // just ticked
60
+ paused: null,
61
+ jobs: [{ slug: 'a', cwd: '/burrow', status: 'quarantined' }],
62
+ invalidJobs: [],
63
+ };
64
+ const summary = computeStallSummary(state);
65
+ const result = evaluatePerProjectStall(summary, state.lastRunAt, Date.now(), TICK_STALL_THRESHOLD_MS);
66
+ assert.strictEqual(result['/burrow'].stalled, true);
67
+ assert.strictEqual(result['/burrow'].pastThreshold, false);
68
+ });
69
+
70
+ test('evaluatePerProjectStall: no lastRunAt yet — stalled but not asserted past threshold (caveat, not a false RED)', () => {
71
+ const state = {
72
+ lastRunAt: null,
73
+ paused: null,
74
+ jobs: [{ slug: 'a', cwd: '/burrow', status: 'quarantined' }],
75
+ invalidJobs: [],
76
+ };
77
+ const summary = computeStallSummary(state);
78
+ const result = evaluatePerProjectStall(summary, state.lastRunAt, Date.now(), TICK_STALL_THRESHOLD_MS);
79
+ assert.strictEqual(result['/burrow'].stalled, true);
80
+ assert.strictEqual(result['/burrow'].pastThreshold, false);
81
+ assert.strictEqual(result['/burrow'].caveat, 'no-lastRunAt');
82
+ });
@@ -0,0 +1,108 @@
1
+ /**
2
+ * scheduler-stall-per-project.test.cjs — unit tests for computeStallSummary's
3
+ * per-project `stalled` verdict and findStaleQuarantinedJobs' age-based
4
+ * escalation.
5
+ *
6
+ * Covers the exact burrow-vs-others shape observed live: one project (burrow)
7
+ * holds only quarantined rows (0 running, 0 pending) while another project
8
+ * has running + pending work — the machine-wide `stalled` boolean reads
9
+ * false (masking burrow), but burrow's own byProject entry must read true.
10
+ *
11
+ * Run: timeout 120 node --test src/main/__tests__/scheduler-stall-per-project.test.cjs
12
+ */
13
+
14
+ 'use strict';
15
+
16
+ const { test } = require('node:test');
17
+ const assert = require('node:assert/strict');
18
+ const { computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS } = require('../scheduler.cjs');
19
+
20
+ test('computeStallSummary: burrow fully stalled (quarantined only) while another project is busy — machine-wide false, burrow true', () => {
21
+ const state = {
22
+ paused: null,
23
+ jobs: [
24
+ { slug: '821-a', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
25
+ { slug: '822-b', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
26
+ { slug: '823-c', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
27
+ { slug: '824-d', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
28
+ { slug: '900-e', cwd: '/home/bilko/Projects/other', status: 'running' },
29
+ { slug: '901-f', cwd: '/home/bilko/Projects/other', status: 'pending' },
30
+ ],
31
+ invalidJobs: [],
32
+ };
33
+ const summary = computeStallSummary(state);
34
+ assert.strictEqual(summary.stalled, false, 'machine-wide must NOT read stalled — other project is busy');
35
+ assert.strictEqual(summary.byProject['/home/bilko/Projects/burrow'].stalled, true, 'burrow must read stalled on its own');
36
+ assert.strictEqual(summary.byProject['/home/bilko/Projects/other'].stalled, false);
37
+ });
38
+
39
+ test('computeStallSummary: both projects busy — neither stalled', () => {
40
+ const state = {
41
+ paused: null,
42
+ jobs: [
43
+ { slug: 'a', cwd: '/p1', status: 'running' },
44
+ { slug: 'b', cwd: '/p2', status: 'pending' },
45
+ ],
46
+ invalidJobs: [],
47
+ };
48
+ const summary = computeStallSummary(state);
49
+ assert.strictEqual(summary.stalled, false);
50
+ assert.strictEqual(summary.byProject['/p1'].stalled, false);
51
+ assert.strictEqual(summary.byProject['/p2'].stalled, false);
52
+ });
53
+
54
+ test('computeStallSummary: both projects stalled — machine-wide AND both per-project read true', () => {
55
+ const state = {
56
+ paused: null,
57
+ jobs: [
58
+ { slug: 'a', cwd: '/p1', status: 'failed' },
59
+ { slug: 'b', cwd: '/p2', status: 'needs_review' },
60
+ ],
61
+ invalidJobs: [],
62
+ };
63
+ const summary = computeStallSummary(state);
64
+ assert.strictEqual(summary.stalled, true);
65
+ assert.strictEqual(summary.byProject['/p1'].stalled, true);
66
+ assert.strictEqual(summary.byProject['/p2'].stalled, true);
67
+ });
68
+
69
+ test('computeStallSummary: paused scheduler never reads stalled, machine-wide or per-project', () => {
70
+ const state = {
71
+ paused: { reason: 'rate_limit' },
72
+ jobs: [{ slug: 'a', cwd: '/p1', status: 'quarantined' }],
73
+ invalidJobs: [],
74
+ };
75
+ const summary = computeStallSummary(state);
76
+ assert.strictEqual(summary.stalled, false);
77
+ assert.strictEqual(summary.byProject['/p1'].stalled, false);
78
+ });
79
+
80
+ test('findStaleQuarantinedJobs: fires past threshold, not before', () => {
81
+ const now = Date.parse('2026-08-08T12:00:00.000Z');
82
+ const jobs = [
83
+ {
84
+ slug: 'fresh',
85
+ cwd: '/p1',
86
+ status: 'quarantined',
87
+ statusHistory: [{ to: 'quarantined', at: new Date(now - 1 * 60 * 60_000).toISOString() }],
88
+ },
89
+ {
90
+ slug: 'stale',
91
+ cwd: '/p1',
92
+ status: 'quarantined',
93
+ statusHistory: [{ to: 'quarantined', at: new Date(now - 25 * 60 * 60_000).toISOString() }],
94
+ },
95
+ {
96
+ slug: 'exactly-at-threshold',
97
+ cwd: '/p1',
98
+ status: 'quarantined',
99
+ statusHistory: [{ to: 'quarantined', at: new Date(now - QUARANTINE_ESCALATE_MS).toISOString() }],
100
+ },
101
+ { slug: 'no-history', cwd: '/p1', status: 'quarantined' },
102
+ { slug: 'not-quarantined', cwd: '/p1', status: 'pending' },
103
+ ];
104
+ const stale = findStaleQuarantinedJobs(jobs, now, QUARANTINE_ESCALATE_MS);
105
+ const slugs = stale.map((s) => s.slug).sort();
106
+ assert.deepStrictEqual(slugs, ['exactly-at-threshold', 'stale']);
107
+ assert.strictEqual(stale.find((s) => s.slug === 'stale').cwd, '/p1');
108
+ });
@@ -14,6 +14,7 @@ const { checkPersonaImports } = require('./lib/personaImportHealth.cjs');
14
14
  const { resolvePrdsDirs } = require('./lib/prdLocations.cjs');
15
15
  const { migratePrds } = require('./lib/prdMigration.cjs');
16
16
  const queueStore = require('./lib/queueStore.cjs');
17
+ const { computeStallSummary } = require('./scheduler.cjs');
17
18
  const { DEFAULT_RUNS_DIR, computeReport, isRetentionEnabled, liveKeysFromJobs } = require('./lib/runLogRetention.cjs');
18
19
 
19
20
  const MAX_LOG_AGE_MS = 5 * 60_000; // 5 min — warn if no logs this old
@@ -131,6 +132,55 @@ function evaluateTickLiveness(queueState, heartbeat, now, runningCount) {
131
132
  };
132
133
  }
133
134
 
135
+ // computeProjectProblemCounts(jobs) → { [cwd]: { failed, needs_review, quarantined } }
136
+ //
137
+ // health.cjs's machine-wide `failed` count answered "is the machine stuck",
138
+ // never "is any ONE project stuck" — a single project with 4 quarantined
139
+ // PRDs and nothing else running was invisible in a rollup dominated by other
140
+ // projects' healthy jobs. Breaks down every non-terminal-problem status
141
+ // (failed/needs_review/quarantined — deliberately NOT 'completed'/'running'/
142
+ // 'pending'/'investigating', which are not problems) by project cwd.
143
+ function computeProjectProblemCounts(jobs) {
144
+ const byProject = {};
145
+ for (const j of jobs || []) {
146
+ if (j.status !== 'failed' && j.status !== 'needs_review' && j.status !== 'quarantined') continue;
147
+ const cwd = j.cwd || '(unknown)';
148
+ byProject[cwd] = byProject[cwd] || { failed: 0, needs_review: 0, quarantined: 0 };
149
+ byProject[cwd][j.status] += 1;
150
+ }
151
+ return byProject;
152
+ }
153
+
154
+ // evaluatePerProjectStall(stallSummary, lastRunAtIso, now, thresholdMs) →
155
+ // { [cwd]: { stalled, pastThreshold?, ageMs?, caveat? } }
156
+ //
157
+ // computeStallSummary's per-project `stalled` flag (scheduler.cjs) is a
158
+ // point-in-time verdict with no duration attached — a project can flip
159
+ // stalled/unstalled within a single tick as work completes, so flagging RED
160
+ // the instant it's true would false-trip on ordinary queue churn. There is
161
+ // no per-project lastRunAt persisted (only a machine-wide one), so this
162
+ // reuses that machine-wide timestamp as the best available "has the
163
+ // scheduler ticked recently at all" signal, gated per-project by whether
164
+ // THAT project currently holds stalled work.
165
+ function evaluatePerProjectStall(stallSummary, lastRunAtIso, now, thresholdMs) {
166
+ const lastRunAt = lastRunAtIso ? Date.parse(lastRunAtIso) : null;
167
+ const results = {};
168
+ for (const cwd of Object.keys(stallSummary?.byProject || {})) {
169
+ const counts = stallSummary.byProject[cwd];
170
+ if (!counts.stalled) {
171
+ results[cwd] = { stalled: false };
172
+ continue;
173
+ }
174
+ if (lastRunAt == null || Number.isNaN(lastRunAt)) {
175
+ results[cwd] = { stalled: true, pastThreshold: false, caveat: 'no-lastRunAt' };
176
+ continue;
177
+ }
178
+ const ageMs = now - lastRunAt;
179
+ results[cwd] = { stalled: true, pastThreshold: ageMs >= thresholdMs, ageMs };
180
+ }
181
+ return results;
182
+ }
183
+
134
184
  // Pure evaluator over migratePrds()'s { moved, skipped, unresolved } result —
135
185
  // kept separate from the fs-touching check() call site so it's directly
136
186
  // unit-testable, matching evaluateTickLiveness's pattern.
@@ -222,18 +272,42 @@ async function check() {
222
272
  const failedCount = Object.values(queueState.jobs || {}).filter(
223
273
  (j) => j.status === 'failed'
224
274
  ).length;
275
+ const needsReviewCount = Object.values(queueState.jobs || {}).filter(
276
+ (j) => j.status === 'needs_review'
277
+ ).length;
278
+ const quarantinedCount = Object.values(queueState.jobs || {}).filter(
279
+ (j) => j.status === 'quarantined'
280
+ ).length;
225
281
  const heartbeatPath = path.join(
226
282
  os.homedir(),
227
283
  '.claude/session-manager/scheduler-heartbeat.log'
228
284
  );
229
285
  const heartbeat = readFreshHeartbeat(heartbeatPath);
230
286
  const liveness = evaluateTickLiveness(queueState, heartbeat, Date.now(), runningCount);
287
+
288
+ // Per-project rollup (PRD: monitoring must not collapse per-project
289
+ // reality into one machine-wide boolean — see computeStallSummary /
290
+ // computeProjectProblemCounts headers). A project holding ONLY
291
+ // failed/needs_review/quarantined rows (0 running, 0 pending) never
292
+ // trips evaluateTickLiveness above, since that check requires actual
293
+ // pending work — this is what let the burrow project go dark.
294
+ const stallSummary = computeStallSummary(queueState);
295
+ const now = Date.now();
296
+ const perProjectStall = evaluatePerProjectStall(stallSummary, queueState.lastRunAt, now, TICK_STALL_THRESHOLD_MS);
297
+ const projectsPastThreshold = Object.entries(perProjectStall)
298
+ .filter(([, v]) => v.pastThreshold)
299
+ .map(([cwd]) => cwd);
300
+
231
301
  status.components.scheduler_queue = {
232
- ok: !liveness.stalled,
302
+ ok: !liveness.stalled && projectsPastThreshold.length === 0,
233
303
  path: queuePath,
234
304
  jobs: Object.keys(queueState.jobs || {}).length,
235
305
  running: runningCount,
236
306
  failed: failedCount,
307
+ needsReview: needsReviewCount,
308
+ quarantined: quarantinedCount,
309
+ byProject: computeProjectProblemCounts(queueState.jobs),
310
+ perProjectStall,
237
311
  tickLiveness: liveness.reason,
238
312
  };
239
313
  if (liveness.stalled) {
@@ -249,6 +323,18 @@ async function check() {
249
323
  ? `Tick hasn't advanced in a while but scheduler-heartbeat.log is missing/stale, so current billing utilization can't be checked — cannot rule out a legitimate when-available hold`
250
324
  : 'No lastRunAt recorded yet — cannot assess tick liveness';
251
325
  }
326
+ if (projectsPastThreshold.length > 0) {
327
+ status.components.scheduler_queue.stalledProjects = projectsPastThreshold;
328
+ for (const cwd of projectsPastThreshold) {
329
+ const ageMin = Math.round(perProjectStall[cwd].ageMs / 60_000);
330
+ const counts = status.components.scheduler_queue.byProject[cwd] || {};
331
+ status.issues.push(
332
+ `Project fully stalled: ${cwd} — 0 running, 0 pending, only problem jobs `
333
+ + `(failed=${counts.failed ?? 0} needs_review=${counts.needs_review ?? 0} quarantined=${counts.quarantined ?? 0}), `
334
+ + `no scheduler tick in ~${ageMin}m (threshold ${Math.round(TICK_STALL_THRESHOLD_MS / 60_000)}m)`
335
+ );
336
+ }
337
+ }
252
338
  } catch (e) {
253
339
  if (e.code !== 'ENOENT') {
254
340
  status.issues.push(`Scheduler queue unreadable: ${e.message}`);
@@ -422,4 +508,13 @@ if (require.main === module) {
422
508
  })();
423
509
  }
424
510
 
425
- module.exports = { check, evaluateTickLiveness, readFreshHeartbeat, evaluatePrdMigrationHealth, TICK_STALL_THRESHOLD_MS, HEARTBEAT_STALE_MS };
511
+ module.exports = {
512
+ check,
513
+ evaluateTickLiveness,
514
+ readFreshHeartbeat,
515
+ evaluatePrdMigrationHealth,
516
+ computeProjectProblemCounts,
517
+ evaluatePerProjectStall,
518
+ TICK_STALL_THRESHOLD_MS,
519
+ HEARTBEAT_STALE_MS,
520
+ };
@@ -36,4 +36,11 @@ module.exports = {
36
36
  // rollup line current (see historyAggregator.cjs's refreshIntradayToday).
37
37
  // Cheap: LRU-warm live parse, no full transcript re-read.
38
38
  HISTORY_INTRADAY_REFRESH_MS: 5 * 60_000,
39
+ // A 'quarantined' PRD (no createdVia provenance) sitting un-adopted past
40
+ // this age is escalated: warn-logged naming project + slug + age, and
41
+ // surfaced distinctly on Home (see homeNeedsYou.ts's matching constant)
42
+ // so it cannot be stranded indefinitely with nothing looking at it — see
43
+ // findStaleQuarantinedJobs in scheduler.cjs. Overridable via
44
+ // SM_QUARANTINE_ESCALATE_HOURS for testing/tuning.
45
+ QUARANTINE_ESCALATE_MS: 24 * 60 * 60_000,
39
46
  };