claude-code-session-manager 0.66.0 → 0.68.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/dist/assets/{AgentLibrary-Bzg89D5Y.js → AgentLibrary-D1SHQWCQ.js} +1 -1
  2. package/dist/assets/{History-CNH9vA0A.js → History-BySQczDf.js} +1 -1
  3. package/dist/assets/{Hooks-CyTksPza.js → Hooks-CG8jWh3-.js} +2 -2
  4. package/dist/assets/{HostBilko-DXQVDNHn.js → HostBilko-BxjPY1PW.js} +1 -1
  5. package/dist/assets/{Library-9E2UOIsm.js → Library-BA8cOckT.js} +1 -1
  6. package/dist/assets/{ListDetail-CQWU_Yn5.js → ListDetail-CVIhWPWb.js} +1 -1
  7. package/dist/assets/{MarkdownEditor-DQgfpSef.js → MarkdownEditor-DUoKh1Qr.js} +1 -1
  8. package/dist/assets/{McpServers-r3qwDIj2.js → McpServers-CkowlPR7.js} +2 -2
  9. package/dist/assets/{Memory-BaxOpj-3.js → Memory-C5Vv5Wim.js} +1 -1
  10. package/dist/assets/{Panel-BhFD8Lqo.js → Panel-CVgA2-Y3.js} +1 -1
  11. package/dist/assets/{Permissions-Cj-mODQQ.js → Permissions-u43I9j0-.js} +2 -2
  12. package/dist/assets/{Plugins-DEL3Fqng.js → Plugins-DComjLBQ.js} +2 -2
  13. package/dist/assets/{ProvenanceBadge-CIAg6-JQ.js → ProvenanceBadge-DNTnbu6f.js} +1 -1
  14. package/dist/assets/SaveBar-R1ZZ9WH5.js +1 -0
  15. package/dist/assets/{Scheduler-YOuZKkES.js → Scheduler-D4Gzz6iD.js} +1 -1
  16. package/dist/assets/ScopeSwitcher-DBcYhNZF.js +1 -0
  17. package/dist/assets/Settings-CXcfro2y.js +3 -0
  18. package/dist/assets/{SkillReferenceGraph-CoIwsol8.js → SkillReferenceGraph-DTfBzZ7O.js} +1 -1
  19. package/dist/assets/{Skills-CN8R6AWn.js → Skills-CVlcQjVO.js} +2 -2
  20. package/dist/assets/SystemPrompt-COLmPm10.js +1 -0
  21. package/dist/assets/{TagLibrary-BKmz2W7B.js → TagLibrary-BN6bvsjU.js} +1 -1
  22. package/dist/assets/{TiptapBody-W7n5SwPM.js → TiptapBody-DxzA0zb5.js} +1 -1
  23. package/dist/assets/{Toggle-QuxlVHGI.js → Toggle-CPlUEMUg.js} +1 -1
  24. package/dist/assets/{index-14dBLqE_.css → index-CEnMgeQU.css} +1 -1
  25. package/dist/assets/{index-TejhHSzN.js → index-Ceid3k5X.js} +323 -323
  26. package/dist/assets/{settingsSchema-DNqx6BKJ.js → settingsSchema-CgmSslES.js} +1 -1
  27. package/dist/index.html +2 -2
  28. package/package.json +1 -1
  29. package/plugins/session-manager-dev/skills/builder/session-manager-operations/scheduler/state/queue.json +3 -0
  30. package/plugins/session-manager-dev/skills/develop/SKILL.md +41 -11
  31. package/src/main/__tests__/develop-skill-failure-modes.test.cjs +70 -0
  32. package/src/main/__tests__/health-per-project-stall.test.cjs +84 -0
  33. package/src/main/__tests__/scheduler-job-overrun.test.cjs +117 -0
  34. package/src/main/__tests__/scheduler-stall-per-project.test.cjs +112 -0
  35. package/src/main/health.cjs +97 -2
  36. package/src/main/lib/schedulerConfig.cjs +22 -0
  37. package/src/main/scheduler.cjs +226 -35
  38. package/dist/assets/ScopeSwitcher-DZ_3gEus.js +0 -1
  39. package/dist/assets/Settings-B0x4oflz.js +0 -3
  40. package/dist/assets/SystemPrompt-DceChpAi.js +0 -1
@@ -0,0 +1,112 @@
1
+ /**
2
+ * scheduler-stall-per-project.test.cjs — unit tests for computeStallSummary's
3
+ * per-project `stalled` verdict and findStaleQuarantinedJobs' age-based
4
+ * escalation.
5
+ *
6
+ * Covers the exact burrow-vs-others shape observed live: one project (burrow)
7
+ * holds only quarantined rows (0 running, 0 pending) while another project
8
+ * has running + pending work — the machine-wide `stalled` boolean reads
9
+ * false (masking burrow), but burrow's own byProject entry must read true.
10
+ *
11
+ * Run: timeout 120 npx vitest run src/main/__tests__/scheduler-stall-per-project.test.cjs
12
+ */
13
+
14
+ 'use strict';
15
+
16
+ // vitest, NOT node:test — this repo's suite is vitest-only (see CLAUDE.md:
17
+ // "This repo does not use `node --test`"). A node:test file loads under
18
+ // vitest as "No test suite found" and contributes zero tests, so the guard
19
+ // silently doesn't exist. node:assert works fine inside a vitest run.
20
+ import { test } from 'vitest';
21
+ const assert = require('node:assert/strict');
22
+ const { computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS } = require('../scheduler.cjs');
23
+
24
+ test('computeStallSummary: burrow fully stalled (quarantined only) while another project is busy — machine-wide false, burrow true', () => {
25
+ const state = {
26
+ paused: null,
27
+ jobs: [
28
+ { slug: '821-a', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
29
+ { slug: '822-b', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
30
+ { slug: '823-c', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
31
+ { slug: '824-d', cwd: '/home/bilko/Projects/burrow', status: 'quarantined' },
32
+ { slug: '900-e', cwd: '/home/bilko/Projects/other', status: 'running' },
33
+ { slug: '901-f', cwd: '/home/bilko/Projects/other', status: 'pending' },
34
+ ],
35
+ invalidJobs: [],
36
+ };
37
+ const summary = computeStallSummary(state);
38
+ assert.strictEqual(summary.stalled, false, 'machine-wide must NOT read stalled — other project is busy');
39
+ assert.strictEqual(summary.byProject['/home/bilko/Projects/burrow'].stalled, true, 'burrow must read stalled on its own');
40
+ assert.strictEqual(summary.byProject['/home/bilko/Projects/other'].stalled, false);
41
+ });
42
+
43
+ test('computeStallSummary: both projects busy — neither stalled', () => {
44
+ const state = {
45
+ paused: null,
46
+ jobs: [
47
+ { slug: 'a', cwd: '/p1', status: 'running' },
48
+ { slug: 'b', cwd: '/p2', status: 'pending' },
49
+ ],
50
+ invalidJobs: [],
51
+ };
52
+ const summary = computeStallSummary(state);
53
+ assert.strictEqual(summary.stalled, false);
54
+ assert.strictEqual(summary.byProject['/p1'].stalled, false);
55
+ assert.strictEqual(summary.byProject['/p2'].stalled, false);
56
+ });
57
+
58
+ test('computeStallSummary: both projects stalled — machine-wide AND both per-project read true', () => {
59
+ const state = {
60
+ paused: null,
61
+ jobs: [
62
+ { slug: 'a', cwd: '/p1', status: 'failed' },
63
+ { slug: 'b', cwd: '/p2', status: 'needs_review' },
64
+ ],
65
+ invalidJobs: [],
66
+ };
67
+ const summary = computeStallSummary(state);
68
+ assert.strictEqual(summary.stalled, true);
69
+ assert.strictEqual(summary.byProject['/p1'].stalled, true);
70
+ assert.strictEqual(summary.byProject['/p2'].stalled, true);
71
+ });
72
+
73
+ test('computeStallSummary: paused scheduler never reads stalled, machine-wide or per-project', () => {
74
+ const state = {
75
+ paused: { reason: 'rate_limit' },
76
+ jobs: [{ slug: 'a', cwd: '/p1', status: 'quarantined' }],
77
+ invalidJobs: [],
78
+ };
79
+ const summary = computeStallSummary(state);
80
+ assert.strictEqual(summary.stalled, false);
81
+ assert.strictEqual(summary.byProject['/p1'].stalled, false);
82
+ });
83
+
84
+ test('findStaleQuarantinedJobs: fires past threshold, not before', () => {
85
+ const now = Date.parse('2026-08-08T12:00:00.000Z');
86
+ const jobs = [
87
+ {
88
+ slug: 'fresh',
89
+ cwd: '/p1',
90
+ status: 'quarantined',
91
+ statusHistory: [{ to: 'quarantined', at: new Date(now - 1 * 60 * 60_000).toISOString() }],
92
+ },
93
+ {
94
+ slug: 'stale',
95
+ cwd: '/p1',
96
+ status: 'quarantined',
97
+ statusHistory: [{ to: 'quarantined', at: new Date(now - 25 * 60 * 60_000).toISOString() }],
98
+ },
99
+ {
100
+ slug: 'exactly-at-threshold',
101
+ cwd: '/p1',
102
+ status: 'quarantined',
103
+ statusHistory: [{ to: 'quarantined', at: new Date(now - QUARANTINE_ESCALATE_MS).toISOString() }],
104
+ },
105
+ { slug: 'no-history', cwd: '/p1', status: 'quarantined' },
106
+ { slug: 'not-quarantined', cwd: '/p1', status: 'pending' },
107
+ ];
108
+ const stale = findStaleQuarantinedJobs(jobs, now, QUARANTINE_ESCALATE_MS);
109
+ const slugs = stale.map((s) => s.slug).sort();
110
+ assert.deepStrictEqual(slugs, ['exactly-at-threshold', 'stale']);
111
+ assert.strictEqual(stale.find((s) => s.slug === 'stale').cwd, '/p1');
112
+ });
@@ -14,6 +14,7 @@ const { checkPersonaImports } = require('./lib/personaImportHealth.cjs');
14
14
  const { resolvePrdsDirs } = require('./lib/prdLocations.cjs');
15
15
  const { migratePrds } = require('./lib/prdMigration.cjs');
16
16
  const queueStore = require('./lib/queueStore.cjs');
17
+ const { computeStallSummary } = require('./scheduler.cjs');
17
18
  const { DEFAULT_RUNS_DIR, computeReport, isRetentionEnabled, liveKeysFromJobs } = require('./lib/runLogRetention.cjs');
18
19
 
19
20
  const MAX_LOG_AGE_MS = 5 * 60_000; // 5 min — warn if no logs this old
@@ -131,6 +132,55 @@ function evaluateTickLiveness(queueState, heartbeat, now, runningCount) {
131
132
  };
132
133
  }
133
134
 
135
+ // computeProjectProblemCounts(jobs) → { [cwd]: { failed, needs_review, quarantined } }
136
+ //
137
+ // health.cjs's machine-wide `failed` count answered "is the machine stuck",
138
+ // never "is any ONE project stuck" — a single project with 4 quarantined
139
+ // PRDs and nothing else running was invisible in a rollup dominated by other
140
+ // projects' healthy jobs. Breaks down every non-terminal-problem status
141
+ // (failed/needs_review/quarantined — deliberately NOT 'completed'/'running'/
142
+ // 'pending'/'investigating', which are not problems) by project cwd.
143
+ function computeProjectProblemCounts(jobs) {
144
+ const byProject = {};
145
+ for (const j of jobs || []) {
146
+ if (j.status !== 'failed' && j.status !== 'needs_review' && j.status !== 'quarantined') continue;
147
+ const cwd = j.cwd || '(unknown)';
148
+ byProject[cwd] = byProject[cwd] || { failed: 0, needs_review: 0, quarantined: 0 };
149
+ byProject[cwd][j.status] += 1;
150
+ }
151
+ return byProject;
152
+ }
153
+
154
+ // evaluatePerProjectStall(stallSummary, lastRunAtIso, now, thresholdMs) →
155
+ // { [cwd]: { stalled, pastThreshold?, ageMs?, caveat? } }
156
+ //
157
+ // computeStallSummary's per-project `stalled` flag (scheduler.cjs) is a
158
+ // point-in-time verdict with no duration attached — a project can flip
159
+ // stalled/unstalled within a single tick as work completes, so flagging RED
160
+ // the instant it's true would false-trip on ordinary queue churn. There is
161
+ // no per-project lastRunAt persisted (only a machine-wide one), so this
162
+ // reuses that machine-wide timestamp as the best available "has the
163
+ // scheduler ticked recently at all" signal, gated per-project by whether
164
+ // THAT project currently holds stalled work.
165
+ function evaluatePerProjectStall(stallSummary, lastRunAtIso, now, thresholdMs) {
166
+ const lastRunAt = lastRunAtIso ? Date.parse(lastRunAtIso) : null;
167
+ const results = {};
168
+ for (const cwd of Object.keys(stallSummary?.byProject || {})) {
169
+ const counts = stallSummary.byProject[cwd];
170
+ if (!counts.stalled) {
171
+ results[cwd] = { stalled: false };
172
+ continue;
173
+ }
174
+ if (lastRunAt == null || Number.isNaN(lastRunAt)) {
175
+ results[cwd] = { stalled: true, pastThreshold: false, caveat: 'no-lastRunAt' };
176
+ continue;
177
+ }
178
+ const ageMs = now - lastRunAt;
179
+ results[cwd] = { stalled: true, pastThreshold: ageMs >= thresholdMs, ageMs };
180
+ }
181
+ return results;
182
+ }
183
+
134
184
  // Pure evaluator over migratePrds()'s { moved, skipped, unresolved } result —
135
185
  // kept separate from the fs-touching check() call site so it's directly
136
186
  // unit-testable, matching evaluateTickLiveness's pattern.
@@ -222,18 +272,42 @@ async function check() {
222
272
  const failedCount = Object.values(queueState.jobs || {}).filter(
223
273
  (j) => j.status === 'failed'
224
274
  ).length;
275
+ const needsReviewCount = Object.values(queueState.jobs || {}).filter(
276
+ (j) => j.status === 'needs_review'
277
+ ).length;
278
+ const quarantinedCount = Object.values(queueState.jobs || {}).filter(
279
+ (j) => j.status === 'quarantined'
280
+ ).length;
225
281
  const heartbeatPath = path.join(
226
282
  os.homedir(),
227
283
  '.claude/session-manager/scheduler-heartbeat.log'
228
284
  );
229
285
  const heartbeat = readFreshHeartbeat(heartbeatPath);
230
286
  const liveness = evaluateTickLiveness(queueState, heartbeat, Date.now(), runningCount);
287
+
288
+ // Per-project rollup (PRD: monitoring must not collapse per-project
289
+ // reality into one machine-wide boolean — see computeStallSummary /
290
+ // computeProjectProblemCounts headers). A project holding ONLY
291
+ // failed/needs_review/quarantined rows (0 running, 0 pending) never
292
+ // trips evaluateTickLiveness above, since that check requires actual
293
+ // pending work — this is what let the burrow project go dark.
294
+ const stallSummary = computeStallSummary(queueState);
295
+ const now = Date.now();
296
+ const perProjectStall = evaluatePerProjectStall(stallSummary, queueState.lastRunAt, now, TICK_STALL_THRESHOLD_MS);
297
+ const projectsPastThreshold = Object.entries(perProjectStall)
298
+ .filter(([, v]) => v.pastThreshold)
299
+ .map(([cwd]) => cwd);
300
+
231
301
  status.components.scheduler_queue = {
232
- ok: !liveness.stalled,
302
+ ok: !liveness.stalled && projectsPastThreshold.length === 0,
233
303
  path: queuePath,
234
304
  jobs: Object.keys(queueState.jobs || {}).length,
235
305
  running: runningCount,
236
306
  failed: failedCount,
307
+ needsReview: needsReviewCount,
308
+ quarantined: quarantinedCount,
309
+ byProject: computeProjectProblemCounts(queueState.jobs),
310
+ perProjectStall,
237
311
  tickLiveness: liveness.reason,
238
312
  };
239
313
  if (liveness.stalled) {
@@ -249,6 +323,18 @@ async function check() {
249
323
  ? `Tick hasn't advanced in a while but scheduler-heartbeat.log is missing/stale, so current billing utilization can't be checked — cannot rule out a legitimate when-available hold`
250
324
  : 'No lastRunAt recorded yet — cannot assess tick liveness';
251
325
  }
326
+ if (projectsPastThreshold.length > 0) {
327
+ status.components.scheduler_queue.stalledProjects = projectsPastThreshold;
328
+ for (const cwd of projectsPastThreshold) {
329
+ const ageMin = Math.round(perProjectStall[cwd].ageMs / 60_000);
330
+ const counts = status.components.scheduler_queue.byProject[cwd] || {};
331
+ status.issues.push(
332
+ `Project fully stalled: ${cwd} — 0 running, 0 pending, only problem jobs `
333
+ + `(failed=${counts.failed ?? 0} needs_review=${counts.needs_review ?? 0} quarantined=${counts.quarantined ?? 0}), `
334
+ + `no scheduler tick in ~${ageMin}m (threshold ${Math.round(TICK_STALL_THRESHOLD_MS / 60_000)}m)`
335
+ );
336
+ }
337
+ }
252
338
  } catch (e) {
253
339
  if (e.code !== 'ENOENT') {
254
340
  status.issues.push(`Scheduler queue unreadable: ${e.message}`);
@@ -422,4 +508,13 @@ if (require.main === module) {
422
508
  })();
423
509
  }
424
510
 
425
- module.exports = { check, evaluateTickLiveness, readFreshHeartbeat, evaluatePrdMigrationHealth, TICK_STALL_THRESHOLD_MS, HEARTBEAT_STALE_MS };
511
+ module.exports = {
512
+ check,
513
+ evaluateTickLiveness,
514
+ readFreshHeartbeat,
515
+ evaluatePrdMigrationHealth,
516
+ computeProjectProblemCounts,
517
+ evaluatePerProjectStall,
518
+ TICK_STALL_THRESHOLD_MS,
519
+ HEARTBEAT_STALE_MS,
520
+ };
@@ -36,4 +36,26 @@ module.exports = {
36
36
  // rollup line current (see historyAggregator.cjs's refreshIntradayToday).
37
37
  // Cheap: LRU-warm live parse, no full transcript re-read.
38
38
  HISTORY_INTRADAY_REFRESH_MS: 5 * 60_000,
39
+ // A 'quarantined' PRD (no createdVia provenance) sitting un-adopted past
40
+ // this age is escalated: warn-logged naming project + slug + age, and
41
+ // surfaced distinctly on Home (see homeNeedsYou.ts's matching constant)
42
+ // so it cannot be stranded indefinitely with nothing looking at it — see
43
+ // findStaleQuarantinedJobs in scheduler.cjs. Overridable via
44
+ // SM_QUARANTINE_ESCALATE_HOURS for testing/tuning.
45
+ QUARANTINE_ESCALATE_MS: 24 * 60 * 60_000,
46
+
47
+ // A RUNNING job that has overrun its own PRD's `estimateMinutes` by this
48
+ // factor is escalated. Distinct from MAX_JOB_DURATION_MS (4h), which is a
49
+ // deadman kill: a 20-minute PRD still running at 3h is 9x over estimate but
50
+ // comfortably under the deadman, and if it keeps writing to its log the
51
+ // 20-minute IDLE_OUTPUT_KILL_MS watchdog never fires either — so before
52
+ // this, the ONLY signal was a human happening to notice. This escalates,
53
+ // it does NOT kill: overrunning is evidence of trouble, not proof of it,
54
+ // and killing on an estimate would murder legitimately-slow work.
55
+ // Override with SM_JOB_OVERRUN_FACTOR.
56
+ JOB_OVERRUN_FACTOR: 3,
57
+ // Floor so a tiny estimate can't escalate almost immediately — a 5-minute
58
+ // PRD at 3x is 15 minutes, which is noise. Override with
59
+ // SM_JOB_OVERRUN_FLOOR_MINUTES.
60
+ JOB_OVERRUN_FLOOR_MS: 45 * 60_000,
39
61
  };
@@ -75,7 +75,19 @@ const {
75
75
  USAGE_REFRESH_INTERVAL_MS,
76
76
  MAX_JOB_DURATION_MS,
77
77
  BROADCAST_COALESCE_MS,
78
+ QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
79
+ JOB_OVERRUN_FACTOR: JOB_OVERRUN_FACTOR_DEFAULT,
80
+ JOB_OVERRUN_FLOOR_MS: JOB_OVERRUN_FLOOR_MS_DEFAULT,
78
81
  } = require('./lib/schedulerConfig.cjs');
82
+ const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
83
+ ? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
84
+ : QUARANTINE_ESCALATE_MS_DEFAULT;
85
+ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
86
+ ? Number(process.env.SM_JOB_OVERRUN_FACTOR)
87
+ : JOB_OVERRUN_FACTOR_DEFAULT;
88
+ const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
89
+ ? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
90
+ : JOB_OVERRUN_FLOOR_MS_DEFAULT;
79
91
  const { pickForProject, pickNextBatch, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
80
92
  const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
81
93
  const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
@@ -961,13 +973,24 @@ function appendHeartbeat(entry) {
961
973
  * computeStallSummary(state) → { stalled, total, running, pending, byProject }
962
974
  *
963
975
  * Pure, no IO. `state` is a merged queue-store read ({ jobs, invalidJobs,
964
- * paused }). "Stalled" = the queue holds work — valid rows OR rows
965
- * quarantined for an invalid status — but nothing is running or pending and
966
- * the scheduler isn't paused. The 2026-08-07 incident sat exactly in this
967
- * state for 4+ hours: 2 jobs, 0 running, 0 pending, and the only visible
968
- * symptom was a heartbeat `counts` object that had silently minted a
969
- * `queued` bucket instead of reporting anything actionable. `byProject`
970
- * breaks the stalled rows down by cwd for the log line / toast.
976
+ * paused }). The engine (reconcile/reaper/auto-fix/reverify) already
977
+ * operates machine-wide via queueStore's stateCwds() — this function is
978
+ * MONITORING, and monitoring must not collapse per-project reality into one
979
+ * boolean. `stalled` (top-level) is the pre-existing machine-wide roll-up:
980
+ * the queue holds work — valid rows OR rows quarantined for an invalid
981
+ * status — but nothing anywhere is running or pending and the scheduler
982
+ * isn't paused. The 2026-08-07 incident sat exactly in this state for 4+
983
+ * hours: 2 jobs, 0 running, 0 pending, and the only visible symptom was a
984
+ * heartbeat `counts` object that had silently minted a `queued` bucket
985
+ * instead of reporting anything actionable.
986
+ *
987
+ * `byProject[cwd].stalled` is the PER-PROJECT verdict added for the
988
+ * "burrow went dark while other projects were busy" gap: a project can hold
989
+ * jobs (including ones parked `quarantined`) with 0 running and 0 pending
990
+ * while the machine-wide `stalled` above reads false because a different
991
+ * project has running/pending work. Each project's own status counts
992
+ * (`byProject[cwd][status]`) already summed to a total before this — the
993
+ * fix is only the boolean, not the counting.
971
994
  */
972
995
  function computeStallSummary(state) {
973
996
  const jobs = Array.isArray(state?.jobs) ? state.jobs : [];
@@ -989,9 +1012,91 @@ function computeStallSummary(state) {
989
1012
  }
990
1013
  const total = jobs.length + invalidJobs.length;
991
1014
  const stalled = total > 0 && running === 0 && pending === 0 && !state?.paused;
1015
+ for (const key of Object.keys(byProject)) {
1016
+ const counts = byProject[key];
1017
+ const projRunning = counts.running || 0;
1018
+ const projPending = counts.pending || 0;
1019
+ const projTotal = Object.keys(counts)
1020
+ .filter((k) => k !== 'stalled')
1021
+ .reduce((sum, k) => sum + counts[k], 0);
1022
+ counts.stalled = projTotal > 0 && projRunning === 0 && projPending === 0 && !state?.paused;
1023
+ }
992
1024
  return { stalled, total, running, pending, byProject };
993
1025
  }
994
1026
 
1027
+ /**
1028
+ * findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
1029
+ *
1030
+ * Pure, no IO. A 'quarantined' row (no createdVia provenance) can otherwise
1031
+ * sit forever with nothing looking at it — quarantine only ever clears via a
1032
+ * human adopting or archiving it. This is the escalation half of that gate:
1033
+ * any quarantined row whose recorded quarantine timestamp (statusHistory's
1034
+ * `to === 'quarantined'` entry — stamped at creation, or backfilled from the
1035
+ * PRD file's mtime by reconcile() for rows quarantined before that stamp
1036
+ * existed) is older than `thresholdMs` is reported so the caller can
1037
+ * warn-log and surface it distinctly. A row with no recoverable timestamp is
1038
+ * skipped rather than guessed at.
1039
+ */
1040
+ function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
1041
+ const stale = [];
1042
+ for (const j of jobs ?? []) {
1043
+ if (j.status !== 'quarantined') continue;
1044
+ const entry = (j.statusHistory || []).find((h) => h.to === 'quarantined');
1045
+ if (!entry) continue;
1046
+ const since = Date.parse(entry.at);
1047
+ if (Number.isNaN(since)) continue;
1048
+ const ageMs = now - since;
1049
+ if (ageMs >= thresholdMs) stale.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs });
1050
+ }
1051
+ return stale;
1052
+ }
1053
+
1054
+ /**
1055
+ * findOverrunningJobs(jobs, now, { factor, floorMs }) → [{ slug, cwd, estimateMinutes, ranMs, ratio }]
1056
+ *
1057
+ * Pure, no IO. A RUNNING job whose elapsed time exceeds
1058
+ * `max(estimateMinutes * factor, floorMs)` is overrunning its own PRD's
1059
+ * declared estimate.
1060
+ *
1061
+ * This closes the gap between the two existing kill paths, which a
1062
+ * long-running-but-chatty job slips straight through:
1063
+ * - MAX_JOB_DURATION_MS (4h) is a deadman — a 20-minute PRD at 3h is 9x
1064
+ * over estimate and still an hour away from it.
1065
+ * - IDLE_OUTPUT_KILL_MS (20m) only fires when the log mtime STALLS; an
1066
+ * agent stuck in a productive-looking loop keeps writing and never trips it.
1067
+ * `estimateMinutes` was parsed, stored and displayed but never once compared
1068
+ * against actual runtime, so the only thing standing between a runaway job
1069
+ * and the 4h ceiling was a human noticing. (2026-08-08: a PRD ran 3h+ and was
1070
+ * caught only because the operator opened a second session to look.)
1071
+ *
1072
+ * Escalation only — see JOB_OVERRUN_FACTOR's note on why this must not kill.
1073
+ * A job with no usable estimate is skipped rather than guessed at.
1074
+ */
1075
+ function findOverrunningJobs(jobs, now, { factor, floorMs } = {}) {
1076
+ const f = typeof factor === 'number' && factor > 0 ? factor : JOB_OVERRUN_FACTOR;
1077
+ const floor = typeof floorMs === 'number' && floorMs >= 0 ? floorMs : JOB_OVERRUN_FLOOR_MS;
1078
+ const out = [];
1079
+ for (const j of jobs ?? []) {
1080
+ if (j.status !== 'running') continue;
1081
+ const est = Number(j.estimateMinutes);
1082
+ if (!Number.isFinite(est) || est <= 0) continue; // no estimate to overrun
1083
+ const startedAt = Date.parse(j.startedAt ?? '');
1084
+ if (Number.isNaN(startedAt)) continue;
1085
+ const ranMs = now - startedAt;
1086
+ if (ranMs <= 0) continue;
1087
+ const thresholdMs = Math.max(est * 60_000 * f, floor);
1088
+ if (ranMs < thresholdMs) continue;
1089
+ out.push({
1090
+ slug: j.slug,
1091
+ cwd: j.cwd ?? null,
1092
+ estimateMinutes: est,
1093
+ ranMs,
1094
+ ratio: ranMs / (est * 60_000),
1095
+ });
1096
+ }
1097
+ return out;
1098
+ }
1099
+
995
1100
  // An empty queue and an unreadable queue are NOT the same thing, and
996
1101
  // conflating them is destructive: reconcile() treats every PRD .md with no
997
1102
  // matching jobs[] row as a brand-new goal and re-mints it as 'pending', so a
@@ -1356,6 +1461,21 @@ async function reconcile(state) {
1356
1461
  console.log(`[scheduler] reconcile: adopted quarantined PRD ${job.slug} — createdVia=${p.createdVia}`);
1357
1462
  appendAuditEvent('scheduler_prd_adopted', { slug: job.slug, cwd: p.cwd, createdVia: p.createdVia, source: 'reconcile' });
1358
1463
  }
1464
+ // Backfill a quarantine timestamp for rows quarantined before the
1465
+ // statusHistory stamp below existed (e.g. the burrow-project rows
1466
+ // quarantined under the PRD-authoring lockdown) — findStaleQuarantinedJobs
1467
+ // needs SOME timestamp to escalate an un-adopted row past its age
1468
+ // threshold, and the PRD file's own mtime is the best available proxy
1469
+ // for "when this file first showed up unstamped" for a row that has
1470
+ // never been touched since.
1471
+ if (updatedJob.status === 'quarantined' && !(updatedJob.statusHistory || []).some((h) => h.to === 'quarantined')) {
1472
+ try {
1473
+ const at = new Date(fs.statSync(p.path).mtimeMs).toISOString();
1474
+ const history = Array.isArray(updatedJob.statusHistory) ? [...updatedJob.statusHistory] : [];
1475
+ history.push({ from: null, to: 'quarantined', reason: 'backfilled from PRD file mtime', source: 'reconcile-backfill', at });
1476
+ updatedJob.statusHistory = history;
1477
+ } catch { /* best-effort only — a missing/unreadable file just skips the backfill */ }
1478
+ }
1359
1479
  next.push(updatedJob);
1360
1480
  }
1361
1481
  // Slugs on disk with no matching state.jobs row are normally brand-new
@@ -1590,6 +1710,18 @@ async function reconcile(state) {
1590
1710
  // to 'pending' on the very next pass, within one tick of being stamped.
1591
1711
  if (!p.createdVia && !isFixPlanSlug(slug)) {
1592
1712
  entry.status = 'quarantined';
1713
+ // Stamped at creation (not via transitionJob, since this is a
1714
+ // brand-new row minted directly at 'quarantined' rather than
1715
+ // transitioning through 'pending') so findStaleQuarantinedJobs has a
1716
+ // real quarantine timestamp to escalate against, instead of only the
1717
+ // reconcile-backfill fallback above.
1718
+ entry.statusHistory = [{
1719
+ from: null,
1720
+ to: 'quarantined',
1721
+ reason: 'missing createdVia provenance frontmatter',
1722
+ source: 'reconcile',
1723
+ at: new Date().toISOString(),
1724
+ }];
1593
1725
  console.warn(`[scheduler] reconcile: quarantining unstamped PRD ${slug} (${p.path}) — no createdVia provenance; adopt it from the Scheduler tab's Quarantined filter or via scheduler_update_prd to make it runnable`);
1594
1726
  appendAuditEvent('prd_quarantined', { slug, cwd: p.cwd, path: p.path, reason: 'missing createdVia provenance frontmatter' });
1595
1727
  }
@@ -1692,12 +1824,15 @@ let pollLoopTimer = null;
1692
1824
  let rescheduleInterval = null;
1693
1825
  let heartbeatInterval = null;
1694
1826
  // Stall-detector state (computeStallSummary), read/written only inside the
1695
- // heartbeat interval below. stallSince: wall-clock ms the stalled condition
1696
- // was first observed, null when clear. stallToasted: rate-limits the
1697
- // error-log + toast to once per stall episode (cleared the moment the queue
1698
- // stops being stalled) rather than every 60s heartbeat tick.
1699
- let stallSince = null;
1700
- let stallToasted = false;
1827
+ // heartbeat interval below. Keyed per-project cwd (never a single value) —
1828
+ // a single module-level flag would let one busy project's activity clear or
1829
+ // suppress another stalled project's alert. stallSince.get(cwd): wall-clock
1830
+ // ms that project's stalled condition was first observed, absent when clear.
1831
+ // stallToasted.get(cwd): rate-limits that project's error-log + toast to
1832
+ // once per stall episode (cleared the moment that project stops being
1833
+ // stalled) rather than every 60s heartbeat tick.
1834
+ let stallSince = new Map();
1835
+ let stallToasted = new Map();
1701
1836
  // (The 5-minute feedback sweep that used to piggyback on this heartbeat is
1702
1837
  // gone: it scanned each active project's session-manager-operations/feedback/
1703
1838
  // and auto-queued a /process-feedback PRD. Both the folder and that skill are
@@ -4924,6 +5059,7 @@ async function init() {
4924
5059
  if (rescheduleInterval) clearInterval(rescheduleInterval);
4925
5060
  rescheduleInterval = setInterval(() => {
4926
5061
  rescheduleTimer().catch(() => {});
5062
+ const s = readQueueSync();
4927
5063
  // Periodic self-heal: re-run the verifier over stale needs_review jobs so a
4928
5064
  // job whose work actually landed (committed in-window, no FAIL sentinel)
4929
5065
  // auto-clears WITHOUT waiting for the next app restart. Cheap-guarded — the
@@ -4933,10 +5069,50 @@ async function init() {
4933
5069
  // MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
4934
5070
  // past it), so this interval firing cannot fan out investigations.
4935
5071
  if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
4936
- const s = readQueueSync();
4937
5072
  if (s.jobs.some((j) => j.status === 'needs_review')) {
4938
5073
  reverifyNeedsReview().catch(() => {});
4939
5074
  }
5075
+ // A quarantined row only ever promotes to 'pending' through
5076
+ // reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
5077
+ // it re-checks the PRD file's createdVia stamp every pass. broadcast()
5078
+ // already runs reconcile+writeQueue on every normal poll tick, but an
5079
+ // idle queue (nothing pending/running to fire) can back off that
5080
+ // cadence for a long time; this guarantees an adopted-but-still-
5081
+ // quarantined row is re-checked within 10 minutes regardless.
5082
+ if (s.jobs.some((j) => j.status === 'quarantined')) {
5083
+ broadcast().catch(() => {});
5084
+ }
5085
+ }
5086
+ // Age-based escalation (independent of the self-heal kill-switch above —
5087
+ // this is a monitoring signal, not an auto-fix action): a quarantined
5088
+ // row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
5089
+ // warn-logged by project + slug + age so it cannot sit stranded and
5090
+ // silent (the four burrow-project rows this PRD was written against).
5091
+ for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
5092
+ console.warn(
5093
+ `[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
5094
+ + `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
5095
+ + `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
5096
+ );
5097
+ appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
5098
+ }
5099
+
5100
+ // Estimate-relative overrun escalation. Sits in the blind spot between
5101
+ // the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
5102
+ // producing output while looping trips neither, so nothing noticed a PRD
5103
+ // running 9x its own estimate until a human went looking. Escalate loudly;
5104
+ // never kill on an estimate (see JOB_OVERRUN_FACTOR).
5105
+ for (const over of findOverrunningJobs(s.jobs, Date.now())) {
5106
+ console.warn(
5107
+ `[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
5108
+ + `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
5109
+ + `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
5110
+ + `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
5111
+ + `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
5112
+ );
5113
+ appendAuditEvent('job_overrunning_estimate', {
5114
+ slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
5115
+ });
4940
5116
  }
4941
5117
  }, 10 * 60_000);
4942
5118
 
@@ -4981,27 +5157,42 @@ async function init() {
4981
5157
  }
4982
5158
 
4983
5159
  const stall = computeStallSummary(s);
4984
- if (stall.stalled) {
4985
- if (stallSince === null) stallSince = Date.now();
4986
- if (!stallToasted && Date.now() - stallSince >= POLL_INTERVAL_MS) {
4987
- stallToasted = true;
4988
- console.error(
4989
- `[scheduler] STALL DETECTED: ${stall.total} job(s) queued, 0 running, 0 pending, not paused, `
4990
- + `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
4991
- stall.byProject,
4992
- );
4993
- appendAuditEvent('scheduler_stall_detected', { total: stall.total, byProject: stall.byProject });
4994
- if (mainWindow && !mainWindow.isDestroyed()) {
4995
- sendIfAlive(mainWindow, 'schedule:stall', {
4996
- message: `Scheduler stall: ${stall.total} job(s) queued but none running or pending. Check the Scheduler tab.`,
4997
- total: stall.total,
4998
- byProject: stall.byProject,
4999
- });
5000
- }
5160
+ // Per-project alerting (see computeStallSummary's header): a project
5161
+ // stalled while others are busy must still fire, and one project
5162
+ // recovering must not clear or suppress another's still-open episode —
5163
+ // that is exactly what a single module-level stallSince/stallToasted
5164
+ // flag masked before (the burrow-vs-others incident this PRD fixes).
5165
+ const now = Date.now();
5166
+ const stalledCwds = Object.keys(stall.byProject).filter((cwd) => stall.byProject[cwd].stalled);
5167
+ for (const cwd of [...stallSince.keys()]) {
5168
+ if (!stalledCwds.includes(cwd)) {
5169
+ stallSince.delete(cwd);
5170
+ stallToasted.delete(cwd);
5171
+ }
5172
+ }
5173
+ const toAlert = [];
5174
+ for (const cwd of stalledCwds) {
5175
+ if (!stallSince.has(cwd)) stallSince.set(cwd, now);
5176
+ if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
5177
+ stallToasted.set(cwd, true);
5178
+ toAlert.push(cwd);
5179
+ }
5180
+ }
5181
+ if (toAlert.length > 0) {
5182
+ console.error(
5183
+ `[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
5184
+ + `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
5185
+ stall.byProject,
5186
+ );
5187
+ appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: stall.total, byProject: stall.byProject });
5188
+ if (mainWindow && !mainWindow.isDestroyed()) {
5189
+ sendIfAlive(mainWindow, 'schedule:stall', {
5190
+ message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
5191
+ projects: toAlert,
5192
+ total: stall.total,
5193
+ byProject: stall.byProject,
5194
+ });
5001
5195
  }
5002
- } else {
5003
- stallSince = null;
5004
- stallToasted = false;
5005
5196
  }
5006
5197
 
5007
5198
  appendHeartbeat({
@@ -5507,4 +5698,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
5507
5698
  });
5508
5699
  }
5509
5700
 
5510
- module.exports = { registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary };
5701
+ module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isPromotableOriginal, selectAutoFixTargets, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS };
@@ -1 +0,0 @@
1
- import{K as e,G as o,S as x}from"./index-TejhHSzN.js";function f({dirty:i,busy:a,parseError:t,lastSavedAt:c,onSave:s,onRevert:l,leading:n}){const d=i&&!a&&!t;return e.jsxs("div",{className:"px-3 py-1.5 flex items-center gap-2 text-xs",children:[e.jsx("div",{className:"flex-1 min-w-0 truncate text-fg-faint",children:n}),t&&e.jsx("span",{className:"text-red-400 truncate max-w-xs",title:t,children:t}),!i&&!t&&c&&e.jsx("span",{className:"text-fg-faint",children:o(c)}),i&&!t&&e.jsx("span",{className:"text-accent",children:"unsaved"}),e.jsx("button",{onClick:l,disabled:!i||a,className:"px-2 py-0.5 border border-line rounded text-fg-dim hover:text-fg hover:bg-bg-hi disabled:opacity-40 disabled:cursor-not-allowed",children:"Revert"}),e.jsx("button",{onClick:s,disabled:!d,className:`px-2.5 py-0.5 rounded border ${d?"border-accent bg-accent/15 text-accent hover:bg-accent/25":"border-line text-fg-faint opacity-50 cursor-not-allowed"}`,children:a?"Saving…":"Save"})]})}function b({scopes:i,active:a,onChange:t,annotate:c}){return e.jsx("div",{className:"inline-flex rounded border border-line overflow-hidden",children:i.map(s=>{const l=x[s],n=c==null?void 0:c(s),d=s===a;return e.jsxs("button",{title:l.hint,onClick:()=>t(s),className:`px-2.5 py-1 text-xs flex items-center gap-1.5 transition-colors ${d?"bg-bg-hi text-fg":"bg-bg-elev text-fg-dim hover:text-fg hover:bg-bg-hi"}`,children:[e.jsx("span",{children:l.label}),(n==null?void 0:n.dirty)&&e.jsx("span",{className:"w-1 h-1 rounded-full bg-accent",title:"unsaved edits"}),n&&n.exists===!1&&!n.dirty&&e.jsx("span",{className:"w-1 h-1 rounded-full bg-fg-faint",title:"not yet created"})]},s)})})}export{f as S,b as a};