claude-code-session-manager 0.66.0 → 0.67.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-Bzg89D5Y.js → AgentLibrary-CiqimwWl.js} +1 -1
- package/dist/assets/{History-CNH9vA0A.js → History-DbzjII2Z.js} +1 -1
- package/dist/assets/{Hooks-CyTksPza.js → Hooks-C0PTplI_.js} +2 -2
- package/dist/assets/{HostBilko-DXQVDNHn.js → HostBilko-vGc1FYLD.js} +1 -1
- package/dist/assets/{Library-9E2UOIsm.js → Library-4tU4rg2h.js} +1 -1
- package/dist/assets/{ListDetail-CQWU_Yn5.js → ListDetail-BGIvkCwZ.js} +1 -1
- package/dist/assets/{MarkdownEditor-DQgfpSef.js → MarkdownEditor-ChFdpCam.js} +1 -1
- package/dist/assets/{McpServers-r3qwDIj2.js → McpServers-DsRWWpU3.js} +2 -2
- package/dist/assets/{Memory-BaxOpj-3.js → Memory-DmxoxxxY.js} +1 -1
- package/dist/assets/{Panel-BhFD8Lqo.js → Panel-KC-rv3jS.js} +1 -1
- package/dist/assets/{Permissions-Cj-mODQQ.js → Permissions-DrDjCRFl.js} +2 -2
- package/dist/assets/{Plugins-DEL3Fqng.js → Plugins-KexSQwk0.js} +2 -2
- package/dist/assets/{ProvenanceBadge-CIAg6-JQ.js → ProvenanceBadge-Rg-jL94y.js} +1 -1
- package/dist/assets/SaveBar-BbEo8U2e.js +1 -0
- package/dist/assets/{Scheduler-YOuZKkES.js → Scheduler-C4Ti9TCc.js} +1 -1
- package/dist/assets/ScopeSwitcher-Di9BmYze.js +1 -0
- package/dist/assets/Settings-CxyhPuYQ.js +3 -0
- package/dist/assets/{SkillReferenceGraph-CoIwsol8.js → SkillReferenceGraph-0q26RHrn.js} +1 -1
- package/dist/assets/{Skills-CN8R6AWn.js → Skills-BXkVyCo5.js} +2 -2
- package/dist/assets/SystemPrompt-Bwe3NTbi.js +1 -0
- package/dist/assets/{TagLibrary-BKmz2W7B.js → TagLibrary-CnwEf6EI.js} +1 -1
- package/dist/assets/{TiptapBody-W7n5SwPM.js → TiptapBody-faCtLj9L.js} +1 -1
- package/dist/assets/{Toggle-QuxlVHGI.js → Toggle-Bw7G_-RR.js} +1 -1
- package/dist/assets/{index-TejhHSzN.js → index-B6dJ2CsU.js} +323 -323
- package/dist/assets/{index-14dBLqE_.css → index-CEnMgeQU.css} +1 -1
- package/dist/assets/{settingsSchema-DNqx6BKJ.js → settingsSchema-B5C9hZoS.js} +1 -1
- package/dist/index.html +2 -2
- package/package.json +1 -1
- package/plugins/session-manager-dev/skills/develop/SKILL.md +41 -11
- package/src/main/__tests__/develop-skill-failure-modes.test.cjs +70 -0
- package/src/main/__tests__/health-per-project-stall.test.cjs +82 -0
- package/src/main/__tests__/scheduler-stall-per-project.test.cjs +108 -0
- package/src/main/health.cjs +97 -2
- package/src/main/lib/schedulerConfig.cjs +7 -0
- package/src/main/scheduler.cjs +154 -35
- package/dist/assets/ScopeSwitcher-DZ_3gEus.js +0 -1
- package/dist/assets/Settings-B0x4oflz.js +0 -3
- package/dist/assets/SystemPrompt-DceChpAi.js +0 -1
|
@@ -210,11 +210,20 @@ can't load skills.
|
|
|
210
210
|
separate PRDs into one oversized one; if the sub-task groups would each take real time on
|
|
211
211
|
their own, that's a signal to split into a chain link instead of one bloated PRD.
|
|
212
212
|
|
|
213
|
+
**Preflight — confirm the tool is even in your tool list before you start composing PRDs.**
|
|
214
|
+
Check for `mcp__session-manager-scheduler__scheduler_create_prd` in your available tools as
|
|
215
|
+
the very first thing you do in this step, before any drafting — catching a missing tool here
|
|
216
|
+
costs nothing; catching it after you've already composed and written PRD bodies means
|
|
217
|
+
discarding that work. If it's absent, see "Two failure modes" immediately below — case (b),
|
|
218
|
+
not the reachable-but-erroring fallback.
|
|
219
|
+
|
|
213
220
|
**`scheduler_create_prd` is the ONLY sanctioned way to author a PRD — not a preference, a
|
|
214
221
|
rule.** Every PRD reaches disk through the MCP tool
|
|
215
222
|
(`mcp__session-manager-scheduler__scheduler_create_prd`). Hand-writing the file directly is a
|
|
216
|
-
degraded, LAST-RESORT fallback (below) reserved for the single case where the tool
|
|
217
|
-
unreachable — never a co-equal alternative to reach for out of habit or
|
|
223
|
+
degraded, LAST-RESORT fallback (below) reserved for the single case where the tool is present
|
|
224
|
+
but errors as unreachable — never a co-equal alternative to reach for out of habit or
|
|
225
|
+
convenience, and never applicable when the tool isn't in your list at all (see "Two failure
|
|
226
|
+
modes" below). Its input
|
|
218
227
|
(`title`, `cwd`, `estimateMinutes`, `goal`, `acceptanceCriteria[]`, `implementationNotes`,
|
|
219
228
|
`outOfScope[]`) maps directly onto the sections below — pass them straight through. **Always
|
|
220
229
|
pass `sourcePromptId` explicitly, set to the `<epic-id>` resolved in the Epic-gated step
|
|
@@ -227,18 +236,37 @@ can't load skills.
|
|
|
227
236
|
`parallelGroup` is DEPRECATED and ignored — express ordering with the `dependsOn` input
|
|
228
237
|
(slugs that must complete first); independent PRDs simply omit it and may run in parallel.
|
|
229
238
|
|
|
230
|
-
**
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
239
|
+
**Two failure modes — do not conflate them. They have opposite correct responses.**
|
|
240
|
+
|
|
241
|
+
- **(a) Tool PRESENT but ERRORS as "app not running" / admin API unreachable.** The tool
|
|
242
|
+
shows up in your tool list (`mcp__session-manager-scheduler__scheduler_create_prd` is
|
|
243
|
+
callable), but calling it fails because the session-manager Electron app that hosts the
|
|
244
|
+
admin API isn't running right now. This is the ONLY case the manual-write fallback below
|
|
245
|
+
covers. Do not use this path when the tool is reachable but merely returned a validation
|
|
246
|
+
error (bad frontmatter, unresolvable Epic, etc.) — fix the input and retry the tool; a
|
|
247
|
+
validation error is not "the app is not running."
|
|
248
|
+
- **(b) Tool ABSENT from your tool list entirely.** You never see
|
|
249
|
+
`mcp__session-manager-scheduler__scheduler_create_prd` offered at all — there is no error to
|
|
250
|
+
catch, because the tool call is never attempted. This means the `session-manager-scheduler`
|
|
251
|
+
MCP server is not registered for the project you're running against — a **misconfiguration**,
|
|
252
|
+
not "the app is offline." **STOP. Do not write any PRD file, hand-authored or otherwise.**
|
|
253
|
+
Report to the human, by name: "the `session-manager-scheduler` MCP tool is not available in
|
|
254
|
+
this session — the server isn't registered for this project." Point them at the fix: it
|
|
255
|
+
should be registered once at USER scope (`claude mcp add session-manager-scheduler --scope
|
|
256
|
+
user -- node <path-to-session-manager-repo>/scripts/scheduler-mcp-server.cjs`, or run
|
|
257
|
+
`scripts/install-scheduler-mcp-user-scope.sh` from the session-manager repo) so every
|
|
258
|
+
project gets the tool without a per-repo `.mcp.json` edit — do not work around a missing
|
|
259
|
+
tool by hand-writing the file, and do not add a project-local `.mcp.json` entry yourself as
|
|
260
|
+
a substitute; that's the human's call and re-introduces the per-repo drift this fix removes.
|
|
261
|
+
|
|
262
|
+
**Fallback for case (a) only.** This is a deliberate bypass of the service boundary, not a
|
|
263
|
+
shortcut: using it means the frontmatter validation, atomic `NN` allocation, standards-pointer
|
|
234
264
|
insertion, and Epic-existence check that `scheduler_create_prd` normally performs did not
|
|
235
265
|
run. **You MUST call this out, visibly, in your report** — state plainly that the app wasn't
|
|
236
266
|
running, that you hand-authored the PRD file directly instead of using the tool, name the
|
|
237
267
|
exact file, and flag it for human verification (this bypass is also what
|
|
238
268
|
`scripts/audit-ops-hygiene.cjs` and the `ops-sweep` skill look for and report as a hygiene
|
|
239
|
-
finding, independent of your own report).
|
|
240
|
-
merely returned a validation error (bad frontmatter, unresolvable Epic, etc.) — fix the input
|
|
241
|
-
and retry the tool; a validation error is not "the app is not running."
|
|
269
|
+
finding, independent of your own report).
|
|
242
270
|
When you do use it: compute the highest in-use number deterministically yourself — never
|
|
243
271
|
eyeball or narrow-grep the `ls` (a narrowed pattern like `'^10[0-9]'` silently misses `110+`
|
|
244
272
|
and collides). PRDs are stored per-project, so `NN` allocation for a given PRD only needs
|
|
@@ -448,8 +476,10 @@ single definition of "tracked to done" for both entry paths.
|
|
|
448
476
|
## Notes
|
|
449
477
|
|
|
450
478
|
- Submit each PRD through `scheduler_create_prd`, then confirm — don't draft them inline in chat
|
|
451
|
-
for review first
|
|
452
|
-
not running)
|
|
479
|
+
for review first. Only hand-write the file when the tool is PRESENT but ERRORS as unreachable
|
|
480
|
+
(app not running) — see the "Two failure modes" note above, including its mandatory bypass
|
|
481
|
+
warning. If the tool is ABSENT from your tool list, that's a misconfiguration, not an offline
|
|
482
|
+
app: stop and tell the human, never hand-write the file.
|
|
453
483
|
- Don't combine unrelated features into one PRD. One focused, completable unit each.
|
|
454
484
|
- Don't add a `parallelGroup` frontmatter key — the filename `NN-` prefix drives grouping.
|
|
455
485
|
- Don't write a PRD to `data/prds/`, `docs/prds/`, the project's own folder, or anywhere outside
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* develop-skill-failure-modes.test.cjs — drift guard for the two
|
|
3
|
+
* scheduler_create_prd failure-mode branches in the /develop skill.
|
|
4
|
+
*
|
|
5
|
+
* PRD 1024-1030 incident (2026-08-08): social-signals-trader's session
|
|
6
|
+
* hand-wrote 7 PRDs because scheduler_create_prd was ABSENT from its tool
|
|
7
|
+
* list (the project's .mcp.json didn't register the server yet) — not
|
|
8
|
+
* because the tool errored as unreachable. The old wording only named the
|
|
9
|
+
* "tool errors ... unreachable" case, which an absent tool never matches,
|
|
10
|
+
* so an agent in that situation had no documented path except hand-writing.
|
|
11
|
+
* This test asserts the fix (two named failure modes, with the absent-tool
|
|
12
|
+
* case explicitly forbidding hand-writing) is present and doesn't silently
|
|
13
|
+
* regress back to the single-case wording.
|
|
14
|
+
*
|
|
15
|
+
* Run: timeout 120 npx vitest run src/main/__tests__/develop-skill-failure-modes.test.cjs
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
'use strict';
|
|
19
|
+
|
|
20
|
+
const fs = require('fs');
|
|
21
|
+
const path = require('path');
|
|
22
|
+
|
|
23
|
+
import { test, expect } from 'vitest';
|
|
24
|
+
|
|
25
|
+
const SKILL_MD = path.resolve(
|
|
26
|
+
__dirname,
|
|
27
|
+
'..',
|
|
28
|
+
'..',
|
|
29
|
+
'..',
|
|
30
|
+
'plugins',
|
|
31
|
+
'session-manager-dev',
|
|
32
|
+
'skills',
|
|
33
|
+
'develop',
|
|
34
|
+
'SKILL.md'
|
|
35
|
+
);
|
|
36
|
+
|
|
37
|
+
function readSkill() {
|
|
38
|
+
return fs.readFileSync(SKILL_MD, 'utf8');
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
test('SKILL.md names both failure modes explicitly', () => {
|
|
42
|
+
const text = readSkill();
|
|
43
|
+
expect(text).toMatch(/tool PRESENT but ERRORS/i);
|
|
44
|
+
expect(text).toMatch(/tool ABSENT from your tool list/i);
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
test('the absent-tool branch explicitly forbids hand-writing and says STOP', () => {
|
|
48
|
+
const text = readSkill();
|
|
49
|
+
const absentIdx = text.search(/tool ABSENT from your tool list/i);
|
|
50
|
+
expect(absentIdx).toBeGreaterThan(-1);
|
|
51
|
+
const nearby = text.slice(absentIdx, absentIdx + 1200);
|
|
52
|
+
expect(nearby).toMatch(/STOP/);
|
|
53
|
+
expect(nearby).toMatch(/[Dd]o not write any PRD file/);
|
|
54
|
+
expect(nearby).toMatch(/misconfiguration/i);
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
test('the present-but-erroring branch is the only one pointed at the manual-write fallback', () => {
|
|
58
|
+
const text = readSkill();
|
|
59
|
+
expect(text).toMatch(/Fallback for case \(a\) only/);
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
test('the standalone preflight check appears before PRD composition begins', () => {
|
|
63
|
+
const text = readSkill();
|
|
64
|
+
expect(text).toMatch(/Preflight — confirm the tool is even in your tool list/);
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
test('a preflight-tool-registration fix is named for the human (user-scope claude mcp add)', () => {
|
|
68
|
+
const text = readSkill();
|
|
69
|
+
expect(text).toMatch(/claude mcp add session-manager-scheduler --scope\s*\n?\s*user/);
|
|
70
|
+
});
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* health-per-project-stall.test.cjs — per-project problem-count rollup and
|
|
3
|
+
* per-project stall-past-threshold escalation for the scheduler_queue health
|
|
4
|
+
* check. Covers the gap where a project holding ONLY failed/needs_review/
|
|
5
|
+
* quarantined rows (0 running, 0 pending) never tripped evaluateTickLiveness
|
|
6
|
+
* (which requires actual pending work) — the exact way the burrow project's
|
|
7
|
+
* four quarantined PRDs went dark.
|
|
8
|
+
*
|
|
9
|
+
* Run: timeout 120 node --test src/main/__tests__/health-per-project-stall.test.cjs
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
'use strict';
|
|
13
|
+
|
|
14
|
+
const { test } = require('node:test');
|
|
15
|
+
const assert = require('node:assert/strict');
|
|
16
|
+
const {
|
|
17
|
+
computeProjectProblemCounts,
|
|
18
|
+
evaluatePerProjectStall,
|
|
19
|
+
TICK_STALL_THRESHOLD_MS,
|
|
20
|
+
} = require('../health.cjs');
|
|
21
|
+
const { computeStallSummary } = require('../scheduler.cjs');
|
|
22
|
+
|
|
23
|
+
test('computeProjectProblemCounts: breaks down failed/needs_review/quarantined by project, ignores healthy statuses', () => {
|
|
24
|
+
const jobs = [
|
|
25
|
+
{ slug: 'a', cwd: '/burrow', status: 'quarantined' },
|
|
26
|
+
{ slug: 'b', cwd: '/burrow', status: 'quarantined' },
|
|
27
|
+
{ slug: 'c', cwd: '/burrow', status: 'quarantined' },
|
|
28
|
+
{ slug: 'd', cwd: '/burrow', status: 'quarantined' },
|
|
29
|
+
{ slug: 'e', cwd: '/other', status: 'failed' },
|
|
30
|
+
{ slug: 'f', cwd: '/other', status: 'needs_review' },
|
|
31
|
+
{ slug: 'g', cwd: '/other', status: 'running' },
|
|
32
|
+
{ slug: 'h', cwd: '/other', status: 'pending' },
|
|
33
|
+
{ slug: 'i', cwd: '/other', status: 'completed' },
|
|
34
|
+
];
|
|
35
|
+
const counts = computeProjectProblemCounts(jobs);
|
|
36
|
+
assert.deepStrictEqual(counts['/burrow'], { failed: 0, needs_review: 0, quarantined: 4 });
|
|
37
|
+
assert.deepStrictEqual(counts['/other'], { failed: 1, needs_review: 1, quarantined: 0 });
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
test('evaluatePerProjectStall: fully-stalled project past threshold flags pastThreshold=true', () => {
|
|
41
|
+
const state = {
|
|
42
|
+
lastRunAt: new Date(Date.now() - (TICK_STALL_THRESHOLD_MS + 60_000)).toISOString(),
|
|
43
|
+
paused: null,
|
|
44
|
+
jobs: [
|
|
45
|
+
{ slug: 'a', cwd: '/burrow', status: 'quarantined' },
|
|
46
|
+
{ slug: 'b', cwd: '/other', status: 'running' },
|
|
47
|
+
],
|
|
48
|
+
invalidJobs: [],
|
|
49
|
+
};
|
|
50
|
+
const summary = computeStallSummary(state);
|
|
51
|
+
const result = evaluatePerProjectStall(summary, state.lastRunAt, Date.now(), TICK_STALL_THRESHOLD_MS);
|
|
52
|
+
assert.strictEqual(result['/burrow'].stalled, true);
|
|
53
|
+
assert.strictEqual(result['/burrow'].pastThreshold, true);
|
|
54
|
+
assert.strictEqual(result['/other'].stalled, false);
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
test('evaluatePerProjectStall: stalled but NOT yet past threshold flags pastThreshold=false', () => {
|
|
58
|
+
const state = {
|
|
59
|
+
lastRunAt: new Date(Date.now() - 1000).toISOString(), // just ticked
|
|
60
|
+
paused: null,
|
|
61
|
+
jobs: [{ slug: 'a', cwd: '/burrow', status: 'quarantined' }],
|
|
62
|
+
invalidJobs: [],
|
|
63
|
+
};
|
|
64
|
+
const summary = computeStallSummary(state);
|
|
65
|
+
const result = evaluatePerProjectStall(summary, state.lastRunAt, Date.now(), TICK_STALL_THRESHOLD_MS);
|
|
66
|
+
assert.strictEqual(result['/burrow'].stalled, true);
|
|
67
|
+
assert.strictEqual(result['/burrow'].pastThreshold, false);
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
test('evaluatePerProjectStall: no lastRunAt yet — stalled but not asserted past threshold (caveat, not a false RED)', () => {
|
|
71
|
+
const state = {
|
|
72
|
+
lastRunAt: null,
|
|
73
|
+
paused: null,
|
|
74
|
+
jobs: [{ slug: 'a', cwd: '/burrow', status: 'quarantined' }],
|
|
75
|
+
invalidJobs: [],
|
|
76
|
+
};
|
|
77
|
+
const summary = computeStallSummary(state);
|
|
78
|
+
const result = evaluatePerProjectStall(summary, state.lastRunAt, Date.now(), TICK_STALL_THRESHOLD_MS);
|
|
79
|
+
assert.strictEqual(result['/burrow'].stalled, true);
|
|
80
|
+
assert.strictEqual(result['/burrow'].pastThreshold, false);
|
|
81
|
+
assert.strictEqual(result['/burrow'].caveat, 'no-lastRunAt');
|
|
82
|
+
});
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* scheduler-stall-per-project.test.cjs — unit tests for computeStallSummary's
|
|
3
|
+
* per-project `stalled` verdict and findStaleQuarantinedJobs' age-based
|
|
4
|
+
* escalation.
|
|
5
|
+
*
|
|
6
|
+
* Covers the exact burrow-vs-others shape observed live: one project (burrow)
|
|
7
|
+
* holds only quarantined rows (0 running, 0 pending) while another project
|
|
8
|
+
* has running + pending work — the machine-wide `stalled` boolean reads
|
|
9
|
+
* false (masking burrow), but burrow's own byProject entry must read true.
|
|
10
|
+
*
|
|
11
|
+
* Run: timeout 120 node --test src/main/__tests__/scheduler-stall-per-project.test.cjs
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
'use strict';
|
|
15
|
+
|
|
16
|
+
const { test } = require('node:test');
|
|
17
|
+
const assert = require('node:assert/strict');
|
|
18
|
+
const { computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS } = require('../scheduler.cjs');
|
|
19
|
+
|
|
20
|
+
test('computeStallSummary: burrow fully stalled (quarantined only) while another project is busy — machine-wide false, burrow true', () => {
|
|
21
|
+
const state = {
|
|
22
|
+
paused: null,
|
|
23
|
+
jobs: [
|
|
24
|
+
{ slug: '821-a', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
|
|
25
|
+
{ slug: '822-b', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
|
|
26
|
+
{ slug: '823-c', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
|
|
27
|
+
{ slug: '824-d', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
|
|
28
|
+
{ slug: '900-e', cwd: '/home/bilko/Projects/other', status: 'running' },
|
|
29
|
+
{ slug: '901-f', cwd: '/home/bilko/Projects/other', status: 'pending' },
|
|
30
|
+
],
|
|
31
|
+
invalidJobs: [],
|
|
32
|
+
};
|
|
33
|
+
const summary = computeStallSummary(state);
|
|
34
|
+
assert.strictEqual(summary.stalled, false, 'machine-wide must NOT read stalled — other project is busy');
|
|
35
|
+
assert.strictEqual(summary.byProject['/home/bilko/Projects/burrow'].stalled, true, 'burrow must read stalled on its own');
|
|
36
|
+
assert.strictEqual(summary.byProject['/home/bilko/Projects/other'].stalled, false);
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
test('computeStallSummary: both projects busy — neither stalled', () => {
|
|
40
|
+
const state = {
|
|
41
|
+
paused: null,
|
|
42
|
+
jobs: [
|
|
43
|
+
{ slug: 'a', cwd: '/p1', status: 'running' },
|
|
44
|
+
{ slug: 'b', cwd: '/p2', status: 'pending' },
|
|
45
|
+
],
|
|
46
|
+
invalidJobs: [],
|
|
47
|
+
};
|
|
48
|
+
const summary = computeStallSummary(state);
|
|
49
|
+
assert.strictEqual(summary.stalled, false);
|
|
50
|
+
assert.strictEqual(summary.byProject['/p1'].stalled, false);
|
|
51
|
+
assert.strictEqual(summary.byProject['/p2'].stalled, false);
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
test('computeStallSummary: both projects stalled — machine-wide AND both per-project read true', () => {
|
|
55
|
+
const state = {
|
|
56
|
+
paused: null,
|
|
57
|
+
jobs: [
|
|
58
|
+
{ slug: 'a', cwd: '/p1', status: 'failed' },
|
|
59
|
+
{ slug: 'b', cwd: '/p2', status: 'needs_review' },
|
|
60
|
+
],
|
|
61
|
+
invalidJobs: [],
|
|
62
|
+
};
|
|
63
|
+
const summary = computeStallSummary(state);
|
|
64
|
+
assert.strictEqual(summary.stalled, true);
|
|
65
|
+
assert.strictEqual(summary.byProject['/p1'].stalled, true);
|
|
66
|
+
assert.strictEqual(summary.byProject['/p2'].stalled, true);
|
|
67
|
+
});
|
|
68
|
+
|
|
69
|
+
test('computeStallSummary: paused scheduler never reads stalled, machine-wide or per-project', () => {
|
|
70
|
+
const state = {
|
|
71
|
+
paused: { reason: 'rate_limit' },
|
|
72
|
+
jobs: [{ slug: 'a', cwd: '/p1', status: 'quarantined' }],
|
|
73
|
+
invalidJobs: [],
|
|
74
|
+
};
|
|
75
|
+
const summary = computeStallSummary(state);
|
|
76
|
+
assert.strictEqual(summary.stalled, false);
|
|
77
|
+
assert.strictEqual(summary.byProject['/p1'].stalled, false);
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
test('findStaleQuarantinedJobs: fires past threshold, not before', () => {
|
|
81
|
+
const now = Date.parse('2026-08-08T12:00:00.000Z');
|
|
82
|
+
const jobs = [
|
|
83
|
+
{
|
|
84
|
+
slug: 'fresh',
|
|
85
|
+
cwd: '/p1',
|
|
86
|
+
status: 'quarantined',
|
|
87
|
+
statusHistory: [{ to: 'quarantined', at: new Date(now - 1 * 60 * 60_000).toISOString() }],
|
|
88
|
+
},
|
|
89
|
+
{
|
|
90
|
+
slug: 'stale',
|
|
91
|
+
cwd: '/p1',
|
|
92
|
+
status: 'quarantined',
|
|
93
|
+
statusHistory: [{ to: 'quarantined', at: new Date(now - 25 * 60 * 60_000).toISOString() }],
|
|
94
|
+
},
|
|
95
|
+
{
|
|
96
|
+
slug: 'exactly-at-threshold',
|
|
97
|
+
cwd: '/p1',
|
|
98
|
+
status: 'quarantined',
|
|
99
|
+
statusHistory: [{ to: 'quarantined', at: new Date(now - QUARANTINE_ESCALATE_MS).toISOString() }],
|
|
100
|
+
},
|
|
101
|
+
{ slug: 'no-history', cwd: '/p1', status: 'quarantined' },
|
|
102
|
+
{ slug: 'not-quarantined', cwd: '/p1', status: 'pending' },
|
|
103
|
+
];
|
|
104
|
+
const stale = findStaleQuarantinedJobs(jobs, now, QUARANTINE_ESCALATE_MS);
|
|
105
|
+
const slugs = stale.map((s) => s.slug).sort();
|
|
106
|
+
assert.deepStrictEqual(slugs, ['exactly-at-threshold', 'stale']);
|
|
107
|
+
assert.strictEqual(stale.find((s) => s.slug === 'stale').cwd, '/p1');
|
|
108
|
+
});
|
package/src/main/health.cjs
CHANGED
|
@@ -14,6 +14,7 @@ const { checkPersonaImports } = require('./lib/personaImportHealth.cjs');
|
|
|
14
14
|
const { resolvePrdsDirs } = require('./lib/prdLocations.cjs');
|
|
15
15
|
const { migratePrds } = require('./lib/prdMigration.cjs');
|
|
16
16
|
const queueStore = require('./lib/queueStore.cjs');
|
|
17
|
+
const { computeStallSummary } = require('./scheduler.cjs');
|
|
17
18
|
const { DEFAULT_RUNS_DIR, computeReport, isRetentionEnabled, liveKeysFromJobs } = require('./lib/runLogRetention.cjs');
|
|
18
19
|
|
|
19
20
|
const MAX_LOG_AGE_MS = 5 * 60_000; // 5 min — warn if no logs this old
|
|
@@ -131,6 +132,55 @@ function evaluateTickLiveness(queueState, heartbeat, now, runningCount) {
|
|
|
131
132
|
};
|
|
132
133
|
}
|
|
133
134
|
|
|
135
|
+
// computeProjectProblemCounts(jobs) → { [cwd]: { failed, needs_review, quarantined } }
|
|
136
|
+
//
|
|
137
|
+
// health.cjs's machine-wide `failed` count answered "is the machine stuck",
|
|
138
|
+
// never "is any ONE project stuck" — a single project with 4 quarantined
|
|
139
|
+
// PRDs and nothing else running was invisible in a rollup dominated by other
|
|
140
|
+
// projects' healthy jobs. Breaks down every non-terminal-problem status
|
|
141
|
+
// (failed/needs_review/quarantined — deliberately NOT 'completed'/'running'/
|
|
142
|
+
// 'pending'/'investigating', which are not problems) by project cwd.
|
|
143
|
+
function computeProjectProblemCounts(jobs) {
|
|
144
|
+
const byProject = {};
|
|
145
|
+
for (const j of jobs || []) {
|
|
146
|
+
if (j.status !== 'failed' && j.status !== 'needs_review' && j.status !== 'quarantined') continue;
|
|
147
|
+
const cwd = j.cwd || '(unknown)';
|
|
148
|
+
byProject[cwd] = byProject[cwd] || { failed: 0, needs_review: 0, quarantined: 0 };
|
|
149
|
+
byProject[cwd][j.status] += 1;
|
|
150
|
+
}
|
|
151
|
+
return byProject;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// evaluatePerProjectStall(stallSummary, lastRunAtIso, now, thresholdMs) →
|
|
155
|
+
// { [cwd]: { stalled, pastThreshold?, ageMs?, caveat? } }
|
|
156
|
+
//
|
|
157
|
+
// computeStallSummary's per-project `stalled` flag (scheduler.cjs) is a
|
|
158
|
+
// point-in-time verdict with no duration attached — a project can flip
|
|
159
|
+
// stalled/unstalled within a single tick as work completes, so flagging RED
|
|
160
|
+
// the instant it's true would false-trip on ordinary queue churn. There is
|
|
161
|
+
// no per-project lastRunAt persisted (only a machine-wide one), so this
|
|
162
|
+
// reuses that machine-wide timestamp as the best available "has the
|
|
163
|
+
// scheduler ticked recently at all" signal, gated per-project by whether
|
|
164
|
+
// THAT project currently holds stalled work.
|
|
165
|
+
function evaluatePerProjectStall(stallSummary, lastRunAtIso, now, thresholdMs) {
|
|
166
|
+
const lastRunAt = lastRunAtIso ? Date.parse(lastRunAtIso) : null;
|
|
167
|
+
const results = {};
|
|
168
|
+
for (const cwd of Object.keys(stallSummary?.byProject || {})) {
|
|
169
|
+
const counts = stallSummary.byProject[cwd];
|
|
170
|
+
if (!counts.stalled) {
|
|
171
|
+
results[cwd] = { stalled: false };
|
|
172
|
+
continue;
|
|
173
|
+
}
|
|
174
|
+
if (lastRunAt == null || Number.isNaN(lastRunAt)) {
|
|
175
|
+
results[cwd] = { stalled: true, pastThreshold: false, caveat: 'no-lastRunAt' };
|
|
176
|
+
continue;
|
|
177
|
+
}
|
|
178
|
+
const ageMs = now - lastRunAt;
|
|
179
|
+
results[cwd] = { stalled: true, pastThreshold: ageMs >= thresholdMs, ageMs };
|
|
180
|
+
}
|
|
181
|
+
return results;
|
|
182
|
+
}
|
|
183
|
+
|
|
134
184
|
// Pure evaluator over migratePrds()'s { moved, skipped, unresolved } result —
|
|
135
185
|
// kept separate from the fs-touching check() call site so it's directly
|
|
136
186
|
// unit-testable, matching evaluateTickLiveness's pattern.
|
|
@@ -222,18 +272,42 @@ async function check() {
|
|
|
222
272
|
const failedCount = Object.values(queueState.jobs || {}).filter(
|
|
223
273
|
(j) => j.status === 'failed'
|
|
224
274
|
).length;
|
|
275
|
+
const needsReviewCount = Object.values(queueState.jobs || {}).filter(
|
|
276
|
+
(j) => j.status === 'needs_review'
|
|
277
|
+
).length;
|
|
278
|
+
const quarantinedCount = Object.values(queueState.jobs || {}).filter(
|
|
279
|
+
(j) => j.status === 'quarantined'
|
|
280
|
+
).length;
|
|
225
281
|
const heartbeatPath = path.join(
|
|
226
282
|
os.homedir(),
|
|
227
283
|
'.claude/session-manager/scheduler-heartbeat.log'
|
|
228
284
|
);
|
|
229
285
|
const heartbeat = readFreshHeartbeat(heartbeatPath);
|
|
230
286
|
const liveness = evaluateTickLiveness(queueState, heartbeat, Date.now(), runningCount);
|
|
287
|
+
|
|
288
|
+
// Per-project rollup (PRD: monitoring must not collapse per-project
|
|
289
|
+
// reality into one machine-wide boolean — see computeStallSummary /
|
|
290
|
+
// computeProjectProblemCounts headers). A project holding ONLY
|
|
291
|
+
// failed/needs_review/quarantined rows (0 running, 0 pending) never
|
|
292
|
+
// trips evaluateTickLiveness above, since that check requires actual
|
|
293
|
+
// pending work — this is what let the burrow project go dark.
|
|
294
|
+
const stallSummary = computeStallSummary(queueState);
|
|
295
|
+
const now = Date.now();
|
|
296
|
+
const perProjectStall = evaluatePerProjectStall(stallSummary, queueState.lastRunAt, now, TICK_STALL_THRESHOLD_MS);
|
|
297
|
+
const projectsPastThreshold = Object.entries(perProjectStall)
|
|
298
|
+
.filter(([, v]) => v.pastThreshold)
|
|
299
|
+
.map(([cwd]) => cwd);
|
|
300
|
+
|
|
231
301
|
status.components.scheduler_queue = {
|
|
232
|
-
ok: !liveness.stalled,
|
|
302
|
+
ok: !liveness.stalled && projectsPastThreshold.length === 0,
|
|
233
303
|
path: queuePath,
|
|
234
304
|
jobs: Object.keys(queueState.jobs || {}).length,
|
|
235
305
|
running: runningCount,
|
|
236
306
|
failed: failedCount,
|
|
307
|
+
needsReview: needsReviewCount,
|
|
308
|
+
quarantined: quarantinedCount,
|
|
309
|
+
byProject: computeProjectProblemCounts(queueState.jobs),
|
|
310
|
+
perProjectStall,
|
|
237
311
|
tickLiveness: liveness.reason,
|
|
238
312
|
};
|
|
239
313
|
if (liveness.stalled) {
|
|
@@ -249,6 +323,18 @@ async function check() {
|
|
|
249
323
|
? `Tick hasn't advanced in a while but scheduler-heartbeat.log is missing/stale, so current billing utilization can't be checked — cannot rule out a legitimate when-available hold`
|
|
250
324
|
: 'No lastRunAt recorded yet — cannot assess tick liveness';
|
|
251
325
|
}
|
|
326
|
+
if (projectsPastThreshold.length > 0) {
|
|
327
|
+
status.components.scheduler_queue.stalledProjects = projectsPastThreshold;
|
|
328
|
+
for (const cwd of projectsPastThreshold) {
|
|
329
|
+
const ageMin = Math.round(perProjectStall[cwd].ageMs / 60_000);
|
|
330
|
+
const counts = status.components.scheduler_queue.byProject[cwd] || {};
|
|
331
|
+
status.issues.push(
|
|
332
|
+
`Project fully stalled: ${cwd} — 0 running, 0 pending, only problem jobs `
|
|
333
|
+
+ `(failed=${counts.failed ?? 0} needs_review=${counts.needs_review ?? 0} quarantined=${counts.quarantined ?? 0}), `
|
|
334
|
+
+ `no scheduler tick in ~${ageMin}m (threshold ${Math.round(TICK_STALL_THRESHOLD_MS / 60_000)}m)`
|
|
335
|
+
);
|
|
336
|
+
}
|
|
337
|
+
}
|
|
252
338
|
} catch (e) {
|
|
253
339
|
if (e.code !== 'ENOENT') {
|
|
254
340
|
status.issues.push(`Scheduler queue unreadable: ${e.message}`);
|
|
@@ -422,4 +508,13 @@ if (require.main === module) {
|
|
|
422
508
|
})();
|
|
423
509
|
}
|
|
424
510
|
|
|
425
|
-
module.exports = {
|
|
511
|
+
module.exports = {
|
|
512
|
+
check,
|
|
513
|
+
evaluateTickLiveness,
|
|
514
|
+
readFreshHeartbeat,
|
|
515
|
+
evaluatePrdMigrationHealth,
|
|
516
|
+
computeProjectProblemCounts,
|
|
517
|
+
evaluatePerProjectStall,
|
|
518
|
+
TICK_STALL_THRESHOLD_MS,
|
|
519
|
+
HEARTBEAT_STALE_MS,
|
|
520
|
+
};
|
|
@@ -36,4 +36,11 @@ module.exports = {
|
|
|
36
36
|
// rollup line current (see historyAggregator.cjs's refreshIntradayToday).
|
|
37
37
|
// Cheap: LRU-warm live parse, no full transcript re-read.
|
|
38
38
|
HISTORY_INTRADAY_REFRESH_MS: 5 * 60_000,
|
|
39
|
+
// A 'quarantined' PRD (no createdVia provenance) sitting un-adopted past
|
|
40
|
+
// this age is escalated: warn-logged naming project + slug + age, and
|
|
41
|
+
// surfaced distinctly on Home (see homeNeedsYou.ts's matching constant)
|
|
42
|
+
// so it cannot be stranded indefinitely with nothing looking at it — see
|
|
43
|
+
// findStaleQuarantinedJobs in scheduler.cjs. Overridable via
|
|
44
|
+
// SM_QUARANTINE_ESCALATE_HOURS for testing/tuning.
|
|
45
|
+
QUARANTINE_ESCALATE_MS: 24 * 60 * 60_000,
|
|
39
46
|
};
|