claude-code-session-manager 0.85.0 → 0.86.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/dist/assets/{AgentLibrary-Bkv-HcP1.js → AgentLibrary-DTFL7y8G.js} +1 -1
  2. package/dist/assets/{DataModel-DRH-Ty20.js → DataModel-Q4jhl24R.js} +1 -1
  3. package/dist/assets/{History-CfRhT1Im.js → History-Cj2FejEo.js} +1 -1
  4. package/dist/assets/{Hooks-wEmh_U6c.js → Hooks-CaelQI6t.js} +1 -1
  5. package/dist/assets/{HostBilko-D_t7Rbi7.js → HostBilko--v7cMR8I.js} +1 -1
  6. package/dist/assets/{Library-CpArQ-OJ.js → Library-DgI9oCCZ.js} +1 -1
  7. package/dist/assets/{ListDetail-pjaKYs84.js → ListDetail-DYUZN-x-.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-Xc141kjj.js → MarkdownEditor-DCIubYWf.js} +1 -1
  9. package/dist/assets/{McpServers-ftqaV3kn.js → McpServers-ypCYURh3.js} +1 -1
  10. package/dist/assets/{Memory-ChMWkNd0.js → Memory-C2qYp-3M.js} +1 -1
  11. package/dist/assets/{Panel-D9Kr40Ai.js → Panel-Cj2kw-Zv.js} +1 -1
  12. package/dist/assets/{Permissions-DKoNVgzj.js → Permissions-BiYZNGYW.js} +1 -1
  13. package/dist/assets/{Plugins-BtChISho.js → Plugins-C1Vj8_dU.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-DBA5EcYy.js → ProvenanceBadge-DczPNM5U.js} +1 -1
  15. package/dist/assets/{SaveBar-I0_dWNTX.js → SaveBar-Cd_7U6Gb.js} +1 -1
  16. package/dist/assets/{Scheduler-CbES7MC8.js → Scheduler-DcLBiJBq.js} +1 -1
  17. package/dist/assets/{ScopeSwitcher-5GTEveb2.js → ScopeSwitcher-DVSyI44-.js} +1 -1
  18. package/dist/assets/{Settings-BX3FElXk.js → Settings-Cv-pRyms.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-DNBFGrYE.js → SkillReferenceGraph-CUv1_Q2c.js} +1 -1
  20. package/dist/assets/{Skills-DJB6-bBM.js → Skills-C_YHkAy-.js} +1 -1
  21. package/dist/assets/{SystemPrompt-BiDDrJUA.js → SystemPrompt-B8R7T9xn.js} +1 -1
  22. package/dist/assets/{TagLibrary-_Wrevtop.js → TagLibrary-dj9YHWyy.js} +1 -1
  23. package/dist/assets/{TiptapBody-OWWXdLRy.js → TiptapBody-DnSBUjHE.js} +1 -1
  24. package/dist/assets/{Toggle-B122N0HL.js → Toggle-CjV_BJn6.js} +1 -1
  25. package/dist/assets/{index-DhvuQL4C.js → index-CXFQIPhO.js} +3 -3
  26. package/dist/assets/{settingsSchema-sGoCTd7J.js → settingsSchema-BJVciriw.js} +1 -1
  27. package/dist/index.html +1 -1
  28. package/package.json +3 -2
  29. package/src/main/__tests__/health-starve-escalation.test.cjs +94 -0
  30. package/src/main/__tests__/loadGateDetailTick.test.cjs +31 -0
  31. package/src/main/__tests__/machineProfile.test.cjs +19 -1
  32. package/src/main/__tests__/pty-session-open-telemetry.test.cjs +96 -0
  33. package/src/main/__tests__/queue-starvation-per-project.test.cjs +135 -0
  34. package/src/main/__tests__/scheduler-failed-autoreset.test.cjs +121 -0
  35. package/src/main/__tests__/scheduler-needs-review-autoresolve.test.cjs +189 -0
  36. package/src/main/__tests__/scheduler-no-dead-end-status.test.cjs +152 -0
  37. package/src/main/__tests__/scheduler-quarantine-autoresolve.test.cjs +165 -0
  38. package/src/main/__tests__/scheduler-starve-escalation.test.cjs +144 -0
  39. package/src/main/__tests__/telemetryClient.test.cjs +75 -5
  40. package/src/main/__tests__/telemetrySettings.test.cjs +32 -0
  41. package/src/main/health.cjs +75 -2
  42. package/src/main/lib/__tests__/loadGate.test.cjs +103 -2
  43. package/src/main/lib/__tests__/telemetryBoot.test.cjs +11 -0
  44. package/src/main/lib/loadGate.cjs +23 -1
  45. package/src/main/lib/machineProfile.cjs +15 -0
  46. package/src/main/lib/schedulerBatch.cjs +12 -1
  47. package/src/main/lib/schedulerConfig.cjs +13 -0
  48. package/src/main/lib/telemetryBoot.cjs +11 -8
  49. package/src/main/lib/telemetryClient.cjs +44 -2
  50. package/src/main/lib/telemetrySettings.cjs +20 -3
  51. package/src/main/pty.cjs +9 -0
  52. package/src/main/scheduler.cjs +600 -66
@@ -0,0 +1,144 @@
1
+ /**
2
+ * scheduler-starve-escalation.test.cjs — the `project_starved` audit event
3
+ * (scheduler.cjs ~line 8888) used to be emit-only: it fired every sweep
4
+ * forever with zero consequence. The 2026-09-12 audit log showed
5
+ * /home/bilko/Projects/Bilko emitting it ~115 times over a 19h starve with
6
+ * nothing acting on it. runStarveEscalationSweep is the bounded, automated
7
+ * consequence: a starve outliving STARVE_ESCALATION_MS gets a DISTINCT
8
+ * 'project_starve_escalated' audit event (once per starve stretch, latched
9
+ * per cwd) plus a toast-channel error via the existing 'schedule:stall'
10
+ * push (already wired to toast.error on the renderer side).
11
+ *
12
+ * Exercises selectStarveEscalations() (pure) directly, and
13
+ * runStarveEscalationSweep() (the side-effecting wrapper) against a real,
14
+ * scratch audit-log.jsonl and a fake attached window — matching
15
+ * queue-starvation-per-project.test.cjs's pattern of setting HOME before
16
+ * require() since every state path is baked from os.homedir() at load.
17
+ *
18
+ * Run: timeout 180 npx vitest run src/main/__tests__/scheduler-starve-escalation.test.cjs
19
+ */
20
+
21
+ 'use strict';
22
+
23
+ import { test, beforeEach } from 'vitest';
24
+ const assert = require('node:assert/strict');
25
+ const fs = require('node:fs');
26
+ const os = require('node:os');
27
+ const path = require('node:path');
28
+
29
+ process.env.HOME = fs.mkdtempSync(path.join(os.tmpdir(), 'starve-escalation-test-'));
30
+
31
+ const {
32
+ selectStarveEscalations,
33
+ runStarveEscalationSweep,
34
+ STARVE_ESCALATION_MS,
35
+ attachWindow,
36
+ } = require('../scheduler.cjs');
37
+ const { AUDIT_LOG_PATH } = require('../lib/auditLog.cjs');
38
+
39
+ const CWD_A = '/home/bilko/Projects/Bilko';
40
+ const CWD_B = '/home/bilko/Projects/session-manager';
41
+
42
+ function readAuditKind(kind) {
43
+ let lines;
44
+ try {
45
+ lines = fs.readFileSync(AUDIT_LOG_PATH, 'utf8').split('\n').filter(Boolean);
46
+ } catch {
47
+ return [];
48
+ }
49
+ return lines.map((l) => JSON.parse(l)).filter((r) => r.kind === kind);
50
+ }
51
+
52
+ function fakeWindow() {
53
+ const sent = [];
54
+ return {
55
+ sent,
56
+ isDestroyed: () => false,
57
+ webContents: {
58
+ isDestroyed: () => false,
59
+ isCrashed: () => false,
60
+ send: (channel, payload) => sent.push({ channel, payload }),
61
+ },
62
+ };
63
+ }
64
+
65
+ beforeEach(() => {
66
+ // Fresh scratch audit log per test so escalation counts never bleed across
67
+ // tests in this file.
68
+ fs.mkdirSync(path.dirname(AUDIT_LOG_PATH), { recursive: true });
69
+ fs.writeFileSync(AUDIT_LOG_PATH, '');
70
+ attachWindow(null);
71
+ });
72
+
73
+ test('selectStarveEscalations: escalates a row past threshold not already latched', () => {
74
+ const starved = [{ cwd: CWD_A, pendingCount: 3, oldestPendingSlug: 'x', ageMs: STARVE_ESCALATION_MS + 1 }];
75
+ const { toEscalate, toClear } = selectStarveEscalations(starved, new Set(), STARVE_ESCALATION_MS);
76
+ assert.equal(toEscalate.length, 1);
77
+ assert.equal(toEscalate[0].cwd, CWD_A);
78
+ assert.deepEqual(toClear, []);
79
+ });
80
+
81
+ test('selectStarveEscalations: never re-escalates a cwd already latched', () => {
82
+ const starved = [{ cwd: CWD_A, pendingCount: 3, oldestPendingSlug: 'x', ageMs: STARVE_ESCALATION_MS + 60_000 }];
83
+ const { toEscalate } = selectStarveEscalations(starved, new Set([CWD_A]), STARVE_ESCALATION_MS);
84
+ assert.deepEqual(toEscalate, []);
85
+ });
86
+
87
+ test('selectStarveEscalations: a latched cwd no longer starved is returned for latch-clear', () => {
88
+ const { toEscalate, toClear } = selectStarveEscalations([], new Set([CWD_A]), STARVE_ESCALATION_MS);
89
+ assert.deepEqual(toEscalate, []);
90
+ assert.deepEqual(toClear, [CWD_A]);
91
+ });
92
+
93
+ test('selectStarveEscalations: below-threshold starve is neither escalated nor cleared', () => {
94
+ const starved = [{ cwd: CWD_A, pendingCount: 1, oldestPendingSlug: 'x', ageMs: STARVE_ESCALATION_MS - 1 }];
95
+ const { toEscalate, toClear } = selectStarveEscalations(starved, new Set(), STARVE_ESCALATION_MS);
96
+ assert.deepEqual(toEscalate, []);
97
+ assert.deepEqual(toClear, []);
98
+ });
99
+
100
+ test('runStarveEscalationSweep: emits project_starve_escalated exactly once per starve stretch, plus a toast', () => {
101
+ const win = fakeWindow();
102
+ attachWindow(win);
103
+ const sp = { cwd: CWD_A, pendingCount: 4, oldestPendingSlug: 'oldest-a', ageMs: STARVE_ESCALATION_MS + 60_000 };
104
+
105
+ runStarveEscalationSweep([sp]);
106
+ assert.equal(readAuditKind('project_starve_escalated').length, 1);
107
+ assert.equal(win.sent.length, 1);
108
+ assert.equal(win.sent[0].channel, 'schedule:stall');
109
+ assert.match(win.sent[0].payload.message, /Bilko/);
110
+
111
+ const record = readAuditKind('project_starve_escalated')[0];
112
+ assert.equal(record.cwd, CWD_A);
113
+ assert.equal(record.pendingCount, 4);
114
+ assert.equal(record.oldestPendingSlug, 'oldest-a');
115
+ assert.equal(record.ageMs, sp.ageMs);
116
+ assert.ok(
117
+ ['slots-exhausted', 'memory-deferred', 'load-deferred', 'held', 'unknown'].includes(record.holdReason),
118
+ `unexpected holdReason: ${record.holdReason}`,
119
+ );
120
+
121
+ // Same still-starved stretch on the NEXT sweep must not escalate again.
122
+ runStarveEscalationSweep([sp]);
123
+ assert.equal(readAuditKind('project_starve_escalated').length, 1);
124
+ assert.equal(win.sent.length, 1);
125
+ });
126
+
127
+ test('runStarveEscalationSweep: the latch resets once the starve clears, so a later stretch escalates again', () => {
128
+ const win = fakeWindow();
129
+ attachWindow(win);
130
+ const sp = { cwd: CWD_B, pendingCount: 2, oldestPendingSlug: 'oldest-b', ageMs: STARVE_ESCALATION_MS + 1 };
131
+
132
+ runStarveEscalationSweep([sp]);
133
+ assert.equal(readAuditKind('project_starve_escalated').length, 1);
134
+
135
+ // Starve clears: this cwd is no longer reported by findStarvedProjects at all.
136
+ runStarveEscalationSweep([]);
137
+ assert.equal(readAuditKind('project_starve_escalated').length, 1); // no new escalation on a clear pass
138
+
139
+ // A NEW starve stretch on the same cwd must escalate again — the latch
140
+ // must not have been permanently burned by the first stretch.
141
+ runStarveEscalationSweep([sp]);
142
+ assert.equal(readAuditKind('project_starve_escalated').length, 2);
143
+ assert.equal(win.sent.length, 2);
144
+ });
@@ -150,6 +150,26 @@ test('ingress functions never throw on hostile inputs', async () => {
150
150
  client.shutdown();
151
151
  });
152
152
 
153
+ // ─── reportInstall failure visibility ──────────────────────────────────
154
+
155
+ test('reportInstall logs a warn line naming the thrown error instead of swallowing it silently', async () => {
156
+ const home = await mkHome();
157
+ const client = freshClient(home);
158
+ client._setMachineProfileBuilder(() => { throw new Error('boom-profile'); });
159
+ const warnLines = [];
160
+ client._setLogger((payload) => warnLines.push(payload));
161
+
162
+ const result = await client.reportInstall(fakeProfile());
163
+
164
+ expect(result).toEqual({ accepted: false, reason: 'error' });
165
+ expect(warnLines).toHaveLength(1);
166
+ expect(warnLines[0].scope).toBe('telemetry');
167
+ expect(warnLines[0].level).toBe('warn');
168
+ expect(warnLines[0].message).toBe('reportInstall failed');
169
+ expect(warnLines[0].meta.error).toContain('boom-profile');
170
+ client.shutdown();
171
+ });
172
+
153
173
  // ─── disabled at ingress ───────────────────────────────────────────────
154
174
 
155
175
  test('when telemetry is disabled, ingress drops at the point of entry', async () => {
@@ -269,6 +289,55 @@ test('attribution fields are exempt from redaction/clamping and arrive byte-iden
269
289
  client.shutdown();
270
290
  });
271
291
 
292
+ // ─── env discriminator (wire-level, all four channels) ─────────────────
293
+
294
+ test('every channel carries env:"dev" when SM_DEV drives installChannel, and it survives the queue round-trip through flush', async () => {
295
+ const home = await mkHome();
296
+ const client = freshClient(home, { installChannel: 'dev' });
297
+ const fetchFn = fetchStub(async () => okResponse());
298
+ client._setFetchImpl(fetchFn);
299
+
300
+ await client.track('evt', { a: 1 });
301
+ await client.logLine({ level: 'info', msg: 'hi' });
302
+ await client.reportError({ name: 'E', msg: 'boom', stack: 'at x' });
303
+ await client.reportInstall(fakeProfile({ installChannel: 'dev' }));
304
+
305
+ // Present the instant it's appended to telemetry-queue.jsonl, before any flush.
306
+ const queued = await readQueueLines(client);
307
+ expect(queued).toHaveLength(4);
308
+ for (const rec of queued) expect(rec.wire.env).toBe('dev');
309
+
310
+ // Still present in the exact body flush() POSTs.
311
+ await client.flush('manual');
312
+ expect(fetchFn.calls.length).toBeGreaterThan(0);
313
+ for (const call of fetchFn.calls) {
314
+ const bodies = Array.isArray(call.body.batch) ? call.body.batch : [call.body];
315
+ for (const wire of bodies) expect(wire.env).toBe('dev');
316
+ }
317
+ client.shutdown();
318
+ });
319
+
320
+ test('every channel carries env:"prod" for a non-dev installChannel', async () => {
321
+ const home = await mkHome();
322
+ const client = freshClient(home, { installChannel: 'npx' });
323
+
324
+ await client.track('evt', { a: 1 });
325
+ await client.logLine({ level: 'info', msg: 'hi' });
326
+ await client.reportError({ name: 'E', msg: 'boom', stack: 'at x' });
327
+ await client.reportInstall(fakeProfile({ installChannel: 'npx' }));
328
+
329
+ const queued = await readQueueLines(client);
330
+ expect(queued).toHaveLength(4);
331
+ for (const rec of queued) expect(rec.wire.env).toBe('prod');
332
+ client.shutdown();
333
+ });
334
+
335
+ test('env resolves to "test" under a real test-runner environment, independent of installChannel', () => {
336
+ const machineProfile = require('../lib/machineProfile.cjs');
337
+ expect(machineProfile.resolveEnv({ isTestRunner: true, installChannel: 'dev' })).toBe('test');
338
+ expect(machineProfile.resolveEnv({ isTestRunner: true, installChannel: 'npx' })).toBe('test');
339
+ });
340
+
272
341
  // ─── queue is the accumulator, survives restart ────────────────────────
273
342
 
274
343
  test('a record survives a simulated process restart and is still pending', async () => {
@@ -532,11 +601,11 @@ test('wire field names match server/routes/telemetry.ts exactly', async () => {
532
601
  const errorCall = fetchFn.calls.find((c) => c.url.endsWith('/error'));
533
602
  const installCall = fetchFn.calls.find((c) => c.url.endsWith('/install'));
534
603
 
535
- expect(Object.keys(eventCall.body.batch[0]).sort()).toEqual(['app', 'name', 'props', 'session_id', 'visitor_id'].sort());
536
- expect(Object.keys(logCall.body.batch[0]).sort()).toEqual(['app', 'version', 'level', 'msg', 'visitor_id', 'session_id', 'fields', 'ts'].sort());
537
- expect(Object.keys(errorCall.body.batch[0]).sort()).toEqual(['app', 'version', 'name', 'msg', 'stack', 'url', 'ua', 'visitor_id', 'session_id', 'context', 'ts'].sort());
604
+ expect(Object.keys(eventCall.body.batch[0]).sort()).toEqual(['app', 'env', 'name', 'props', 'session_id', 'visitor_id'].sort());
605
+ expect(Object.keys(logCall.body.batch[0]).sort()).toEqual(['app', 'env', 'version', 'level', 'msg', 'visitor_id', 'session_id', 'fields', 'ts'].sort());
606
+ expect(Object.keys(errorCall.body.batch[0]).sort()).toEqual(['app', 'env', 'version', 'name', 'msg', 'stack', 'url', 'ua', 'visitor_id', 'session_id', 'context', 'ts'].sort());
538
607
  expect(Object.keys(installCall.body).sort()).toEqual([
539
- 'app', 'app_version', 'arch', 'cpu_count', 'electron_version',
608
+ 'app', 'app_version', 'arch', 'cpu_count', 'electron_version', 'env',
540
609
  'install_channel', 'install_id', 'locale', 'node_version',
541
610
  'os_release', 'platform', 'timezone', 'total_mem_mb',
542
611
  ].sort());
@@ -854,7 +923,7 @@ test('reportInstall() maps buildMachineProfile()\'s camelCase onto the exact sna
854
923
 
855
924
  expect(rec.channel).toBe('install');
856
925
  expect(Object.keys(rec.wire).sort()).toEqual([
857
- 'app', 'app_version', 'arch', 'cpu_count', 'electron_version',
926
+ 'app', 'app_version', 'arch', 'cpu_count', 'electron_version', 'env',
858
927
  'install_channel', 'install_id', 'locale', 'node_version',
859
928
  'os_release', 'platform', 'timezone', 'total_mem_mb',
860
929
  ].sort());
@@ -962,6 +1031,7 @@ test('a stub server bound to 127.0.0.1 receives exactly one well-formed install
962
1031
  install_channel: profile.installChannel,
963
1032
  locale: profile.locale,
964
1033
  timezone: String(profile.timezoneOffsetMinutes),
1034
+ env: 'dev',
965
1035
  });
966
1036
  client.shutdown();
967
1037
  } finally {
@@ -168,6 +168,38 @@ test('isMachineReportDue: true when lastMachineReportAt is more than 30 days bef
168
168
  expect(telemetrySettings.isMachineReportDue(cfg, { now, appVersion: '0.81.0' })).toBe(true);
169
169
  });
170
170
 
171
+ test('save() merges a partial patch onto a fresh disk read — concurrent savers do not clobber each other', async () => {
172
+ const home = await mkHome();
173
+ const telemetrySettings = await freshModule(home);
174
+ const base = await telemetrySettings.load(); // mints + persists installId
175
+ expect(base.installId).not.toBe('');
176
+
177
+ // Two independent writers, each stamping a DIFFERENT field, racing each
178
+ // other — mirrors telemetryBoot's install-report stamp landing around the
179
+ // same time as telemetryClient's own daily-flush stamp. Neither holds the
180
+ // other's snapshot; each passes only the field it intends to change.
181
+ await Promise.all([
182
+ telemetrySettings.save({ lastMachineReportAt: '2026-01-01T00:00:00.000Z', lastMachineReportVersion: '2.0.0' }),
183
+ telemetrySettings.save({ lastDailyFlushAt: '2026-01-01T01:00:00.000Z' }),
184
+ ]);
185
+
186
+ const onDisk = await telemetrySettings.load();
187
+ expect(onDisk.lastMachineReportAt).toBe('2026-01-01T00:00:00.000Z');
188
+ expect(onDisk.lastMachineReportVersion).toBe('2.0.0');
189
+ expect(onDisk.lastDailyFlushAt).toBe('2026-01-01T01:00:00.000Z');
190
+ // Neither save touched installId/enabled/endpoint — they must survive too.
191
+ expect(onDisk.installId).toBe(base.installId);
192
+ expect(onDisk.enabled).toBe(base.enabled);
193
+ expect(onDisk.endpoint).toBe(base.endpoint);
194
+ });
195
+
196
+ test('save() rejects an unknown key without touching disk', async () => {
197
+ const home = await mkHome();
198
+ const telemetrySettings = await freshModule(home);
199
+ await telemetrySettings.load();
200
+ await expect(telemetrySettings.save({ notARealField: 'x' })).rejects.toThrow('Invalid telemetry config');
201
+ });
202
+
171
203
  test('isMachineReportDue: false when same version and reported less than 30 days ago', async () => {
172
204
  const home = await mkHome();
173
205
  const telemetrySettings = await freshModule(home);
@@ -16,7 +16,9 @@ const { checkDelegationReadiness } = require('./lib/delegationReadiness.cjs');
16
16
  const { resolvePrdsDirs } = require('./lib/prdLocations.cjs');
17
17
  const { migratePrds } = require('./lib/prdMigration.cjs');
18
18
  const queueStore = require('./lib/queueStore.cjs');
19
- const { computeStallSummary, SCHEDULER_STATE_PATH, FAILURE_STREAK_WARN_THRESHOLD, classifyQueueStarvation } = require('./scheduler.cjs');
19
+ const { computeStallSummary, SCHEDULER_STATE_PATH, FAILURE_STREAK_WARN_THRESHOLD, classifyQueueStarvation, STARVE_ESCALATION_MS } = require('./scheduler.cjs');
20
+ const { findStarvedProjects } = require('./lib/schedulerBatch.cjs');
21
+ const { AUDIT_LOG_PATH } = require('./lib/auditLog.cjs');
20
22
  const { DEFAULT_RUNS_DIR, computeReport, isRetentionEnabled, liveKeysFromJobs } = require('./lib/runLogRetention.cjs');
21
23
  const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
22
24
 
@@ -458,6 +460,63 @@ function evaluateQueueDispatchHealth(queueState, runningCount, now, thresholdMs
458
460
  };
459
461
  }
460
462
 
463
+ /**
464
+ * latestStarveEscalationReasons(auditLogPath) → { [cwd]: holdReason }
465
+ *
466
+ * health.cjs runs as its own cold process (`npm run health`), so it has no
467
+ * access to the live scheduler's in-memory `lastTick` — the durable
468
+ * audit-log.jsonl trail (auditLog.cjs) is the only place the hold reason a
469
+ * 'project_starve_escalated' event already recorded (scheduler.cjs's
470
+ * runStarveEscalationSweep) survives to be read from here. Reads the SAME
471
+ * recorded value rather than re-deriving it; the last record per cwd wins.
472
+ * Missing/unreadable log → {} (no reasons known yet), never a throw.
473
+ */
474
+ function latestStarveEscalationReasons(auditLogPath) {
475
+ let lines;
476
+ try {
477
+ lines = fs.readFileSync(auditLogPath, 'utf8').split('\n').filter(Boolean);
478
+ } catch {
479
+ return {};
480
+ }
481
+ const byCwd = {};
482
+ for (const line of lines) {
483
+ let rec;
484
+ try {
485
+ rec = JSON.parse(line);
486
+ } catch {
487
+ continue;
488
+ }
489
+ if (rec?.kind === 'project_starve_escalated' && rec.cwd) byCwd[rec.cwd] = rec.holdReason ?? 'unknown';
490
+ }
491
+ return byCwd;
492
+ }
493
+
494
+ /**
495
+ * evaluateStarveEscalationHealth(jobs, now, thresholdMs, escalationReasons)
496
+ * → { ok, projects?, message? }
497
+ *
498
+ * Pure over its inputs. Reuses findStarvedProjects — the exact per-cwd
499
+ * STARVED verdict the live escalation (scheduler.cjs) acts on — so health
500
+ * reports the identical set of starved-past-threshold projects, never a
501
+ * re-derived one. `escalationReasons` supplies the hold reason per cwd (see
502
+ * latestStarveEscalationReasons); a cwd with no recorded reason yet reports
503
+ * 'unknown' rather than failing.
504
+ */
505
+ function evaluateStarveEscalationHealth(jobs, now, thresholdMs, escalationReasons = {}) {
506
+ const starved = findStarvedProjects(jobs, now, thresholdMs);
507
+ if (starved.length === 0) return { ok: true };
508
+ const projects = starved.map((sp) => ({
509
+ cwd: sp.cwd,
510
+ pendingCount: sp.pendingCount,
511
+ ageMs: sp.ageMs,
512
+ holdReason: escalationReasons[sp.cwd] ?? 'unknown',
513
+ }));
514
+ const message = `Project(s) starved past the ${Math.round(thresholdMs / 60_000)}m escalation threshold: ${
515
+ projects.map((p) => `${p.cwd} (${Math.round(p.ageMs / 60_000)}m, ${p.pendingCount} pending, hold=${p.holdReason})`).join('; ')
516
+ }`;
517
+ return { ok: false, projects, message };
518
+ }
519
+
461
520
  function loadUsagePollerState(statePath) {
462
521
  let raw;
463
522
  try {
@@ -640,6 +699,18 @@ async function check() {
640
699
  status.issues.push(`Queue dispatch: ${status.components.queue_dispatch.message}`);
641
700
  }
642
701
 
702
+ // Per-project starve escalation (bounded consequence for project_starved
703
+ // — see runStarveEscalationSweep in scheduler.cjs). Distinct from
704
+ // queue_dispatch above: that's machine-wide dispatch liveness, this is
705
+ // "has any ONE project been starved past the LATER escalation threshold",
706
+ // the exact condition the 2026-09-12 19h Bilko starve went unreported by.
707
+ status.components.project_starve_escalation = evaluateStarveEscalationHealth(
708
+ queueState.jobs, now, STARVE_ESCALATION_MS, latestStarveEscalationReasons(AUDIT_LOG_PATH),
709
+ );
710
+ if (!status.components.project_starve_escalation.ok) {
711
+ status.issues.push(status.components.project_starve_escalation.message);
712
+ }
713
+
643
714
  // Worktree cap blocking every dispatchable pending job with nothing
644
715
  // running (see evaluateWorktreeCapBlocked's header) — a distinct
645
716
  // condition from generic tick/dispatch staleness above, since a leaked
@@ -877,7 +948,7 @@ async function check() {
877
948
  // Critical: nodejs, config dir, typescript, build artifact, test infrastructure.
878
949
  // Non-fatal: scheduler/transcripts dirs may not exist on fresh install.
879
950
  // Informational: app log age (shows if app is running, but not blocking).
880
- const criticalComponents = ['nodejs', 'config_dir', 'typescript', 'build_artifact', 'test_infrastructure', 'scheduler_queue', 'queue_dispatch', 'usage_poller', 'prd_migration', 'claude_md_budget', 'delegation_chain'];
951
+ const criticalComponents = ['nodejs', 'config_dir', 'typescript', 'build_artifact', 'test_infrastructure', 'scheduler_queue', 'queue_dispatch', 'project_starve_escalation', 'usage_poller', 'prd_migration', 'claude_md_budget', 'delegation_chain'];
881
952
  status.ok = criticalComponents.every((c) => status.components[c]?.ok !== false);
882
953
 
883
954
  status.elapsedMs = Date.now() - start;
@@ -909,6 +980,8 @@ module.exports = {
909
980
  evaluateUsagePollerHealth,
910
981
  loadUsagePollerState,
911
982
  evaluateQueueDispatchHealth,
983
+ evaluateStarveEscalationHealth,
984
+ latestStarveEscalationReasons,
912
985
  DISPATCH_STALL_THRESHOLD_MS,
913
986
  TICK_STALL_THRESHOLD_MS,
914
987
  HEARTBEAT_STALE_MS,
@@ -125,7 +125,7 @@ test('an ungated tick never audits', () => {
125
125
  assert.equal(g.evaluate().shouldAudit, false);
126
126
  });
127
127
 
128
- test('escalates once the gated stretch exceeds the escalation window, and the stretch resets when load drops', () => {
128
+ test('escalates once the gated stretch exceeds the escalation window, and the stretch only resets once load has stayed sub-threshold past the release window', () => {
129
129
  let l1 = 12.95;
130
130
  let t = 0;
131
131
  const g = createLoadGate({
@@ -134,6 +134,7 @@ test('escalates once the gated stretch exceeds the escalation window, and the st
134
134
  now: () => t,
135
135
  threshold: 0.85,
136
136
  escalateAfterMs: 45 * 60_000,
137
+ releaseWindowMs: 2 * 60_000,
137
138
  });
138
139
  assert.equal(g.evaluate().escalate, false);
139
140
  t += 44 * 60_000;
@@ -142,12 +143,43 @@ test('escalates once the gated stretch exceeds the escalation window, and the st
142
143
  const r = g.evaluate();
143
144
  assert.equal(r.escalate, true, 'at 45m: escalates');
144
145
  assert.equal(r.gatedSinceMs, 45 * 60_000);
146
+ // A single sub-threshold sample must NOT reset the stretch (the boundary-
147
+ // hovering bug this PRD fixes) — gatedSinceMs keeps growing.
145
148
  l1 = 1; t += 60_000;
146
- assert.equal(g.evaluate().gatedSinceMs, 0, 'load dropped: stretch resets');
149
+ const oneMinBelow = g.evaluate();
150
+ assert.equal(oneMinBelow.gated, false, 'gate decision is immediate');
151
+ assert.equal(oneMinBelow.gatedSinceMs, 46 * 60_000, 'stretch not yet released');
152
+ // Load climbs back above threshold before the release window elapses: the
153
+ // stretch was never actually cleared, so it just keeps accumulating.
154
+ l1 = 12.95; t += 30_000;
155
+ assert.equal(g.evaluate().gatedSinceMs, 46 * 60_000 + 30_000, 'stretch survived the brief dip');
156
+ // Now hold sub-threshold continuously past the release window (the window
157
+ // is measured from the first sub-threshold sample of this sustained drop,
158
+ // so it takes a tick to mark that start, then another once the window has
159
+ // actually elapsed).
160
+ l1 = 1; t += 1;
161
+ g.evaluate();
162
+ t += 2 * 60_000;
163
+ assert.equal(g.evaluate().gatedSinceMs, 0, 'sustained sub-threshold load past the release window clears the stretch');
147
164
  l1 = 12.95; t += 60_000;
148
165
  assert.equal(g.evaluate().gatedSinceMs, 0, 'a fresh stretch starts from zero');
149
166
  });
150
167
 
168
+ test('the gate decision itself is unaffected by hysteresis: a sub-threshold tick is never gated, even mid-stretch', () => {
169
+ let l1 = 12.95;
170
+ let t = 0;
171
+ const g = createLoadGate({
172
+ loadavg: () => [l1, 0, 0],
173
+ cores: () => 14,
174
+ now: () => t,
175
+ threshold: 0.85,
176
+ releaseWindowMs: 5 * 60_000,
177
+ });
178
+ assert.equal(g.evaluate().gated, true);
179
+ l1 = 1; t += 60_000;
180
+ assert.equal(g.evaluate().gated, false, 'below threshold: gated:false immediately, hysteresis notwithstanding');
181
+ });
182
+
151
183
  test('snapshot() reflects the last evaluation and is null before any', () => {
152
184
  const { g } = gateWith({});
153
185
  assert.equal(g.snapshot(), null);
@@ -157,3 +189,72 @@ test('snapshot() reflects the last evaluation and is null before any', () => {
157
189
  assert.equal(typeof s.at, 'string');
158
190
  assert.equal('shouldAudit' in s, false, 'per-tick flags are not part of the persisted snapshot');
159
191
  });
192
+
193
+ test('snapshot() exposes gated, ratio, threshold, loadavg1, cores and gatedSinceMs (what buildScheduleStatePayload surfaces as loadGate)', () => {
194
+ const { g } = gateWith({});
195
+ g.evaluate();
196
+ const s = g.snapshot();
197
+ assert.equal(typeof s.gated, 'boolean');
198
+ assert.equal(typeof s.ratio, 'number');
199
+ assert.equal(typeof s.threshold, 'number');
200
+ assert.equal(typeof s.loadavg1, 'number');
201
+ assert.equal(typeof s.cores, 'number');
202
+ assert.equal(typeof s.gatedSinceMs, 'number');
203
+ });
204
+
205
+ // ─── hysteresis: the real observed boundary-hovering sequence ──────────────
206
+
207
+ test('a boundary-hovering box (real observed sequence, alternating above/below threshold) grows gatedSinceMs monotonically across 50 simulated minutes and eventually escalates', () => {
208
+ // 14 cores, 0.85/core = 11.9 threshold; loadavg1 observed 2026-09-12
209
+ // oscillating 11.07 - 13.31, straddling the threshold every tick. Before
210
+ // this PRD, the immediate reset on ANY sub-threshold sample meant every
211
+ // one of these ticks reset gatedSince to null and escalate never fired.
212
+ const CORES = 14;
213
+ const THRESHOLD = 0.85;
214
+ const SEQUENCE = [11.07, 13.31, 11.51, 12.8];
215
+ let l1 = SEQUENCE[0];
216
+ let t = 0;
217
+ const g = createLoadGate({
218
+ loadavg: () => [l1, 0, 0],
219
+ cores: () => CORES,
220
+ now: () => t,
221
+ threshold: THRESHOLD,
222
+ escalateAfterMs: 20 * 60_000, // shorter than the 45m default so 50 sim-minutes crosses it
223
+ });
224
+
225
+ const TICK_MS = 60_000;
226
+ const TOTAL_MS = 50 * 60_000;
227
+ let prevGatedSinceMs = -1;
228
+ let sawEscalate = false;
229
+ let i = 0;
230
+ for (let elapsed = 0; elapsed < TOTAL_MS; elapsed += TICK_MS) {
231
+ l1 = SEQUENCE[i % SEQUENCE.length];
232
+ i += 1;
233
+ const r = g.evaluate();
234
+ assert.ok(r.gatedSinceMs >= prevGatedSinceMs, `gatedSinceMs must not shrink (was ${prevGatedSinceMs}, now ${r.gatedSinceMs})`);
235
+ prevGatedSinceMs = r.gatedSinceMs;
236
+ if (r.escalate) sawEscalate = true;
237
+ t += TICK_MS;
238
+ }
239
+ assert.equal(sawEscalate, true, 'escalate must fire once the (never-reset) stretch exceeds escalateAfterMs');
240
+ });
241
+
242
+ test('the release window actually releases: sustained sub-threshold load past the window reports gated:false and gatedSinceMs:0 on the next tick', () => {
243
+ let l1 = 13.31;
244
+ let t = 0;
245
+ const g = createLoadGate({
246
+ loadavg: () => [l1, 0, 0],
247
+ cores: () => 14,
248
+ now: () => t,
249
+ threshold: 0.85,
250
+ releaseWindowMs: 2 * 60_000,
251
+ });
252
+ assert.equal(g.evaluate().gated, true);
253
+ l1 = 5;
254
+ t += 1;
255
+ g.evaluate(); // marks the start of the sustained sub-threshold run
256
+ t += 2 * 60_000;
257
+ const r = g.evaluate();
258
+ assert.equal(r.gated, false);
259
+ assert.equal(r.gatedSinceMs, 0);
260
+ });
@@ -113,6 +113,17 @@ test('with no version change, flush("version-change") does NOT run', async () =>
113
113
  expect(flushCalls).toEqual(['boot']);
114
114
  });
115
115
 
116
+ test('a rejected reportInstall leaves lastMachineReportAt unset so the next boot retries', async () => {
117
+ const now = Date.now();
118
+ const { deps, state } = fakeDeps({ settings: { lastMachineReportVersion: '', lastMachineReportAt: null } });
119
+ deps.telemetryClient.reportInstall = async () => ({ accepted: false, reason: 'error' });
120
+
121
+ await bootSequence({ now, appVersion: '1.0.0', deps });
122
+
123
+ expect(state().lastMachineReportAt).toBe(null);
124
+ expect(state().lastMachineReportVersion).toBe('');
125
+ });
126
+
116
127
  test('fires app.launch via telemetryCounters with installChannel + appVersion', async () => {
117
128
  const now = Date.now();
118
129
  const { deps } = fakeDeps({ settings: { lastMachineReportVersion: '1.0.0', lastMachineReportAt: new Date(now).toISOString() } });
@@ -32,6 +32,15 @@ const { loadGateThreshold, JOB_OVERRUN_FLOOR_MS } = require('./schedulerConfig.c
32
32
  // every tick would bury the signal in its own noise.
33
33
  const AUDIT_INTERVAL_MS = 10 * 60_000;
34
34
 
35
+ // Default hysteresis release window for the gated stretch (PRD: boundary-
36
+ // hovering load). A box sitting right at the threshold oscillates above/below
37
+ // it tick to tick; clearing `gatedSince` on the first sub-threshold sample
38
+ // reset the stretch counter every time, so `escalate` (and the warn-log with
39
+ // topCpuConsumers()) never fired even after 80+ minutes of effectively
40
+ // continuous gating. The stretch now only clears once load has stayed below
41
+ // threshold continuously for this long.
42
+ const RELEASE_WINDOW_MS = 2 * 60_000;
43
+
35
44
  /**
36
45
  * isLoadGated(loadavg1, cores, threshold) → boolean
37
46
  *
@@ -82,9 +91,11 @@ function createLoadGate({
82
91
  threshold = loadGateThreshold,
83
92
  auditIntervalMs = AUDIT_INTERVAL_MS,
84
93
  escalateAfterMs = JOB_OVERRUN_FLOOR_MS,
94
+ releaseWindowMs = RELEASE_WINDOW_MS,
85
95
  } = {}) {
86
96
  let lastAuditAt = null; // null = never audited; the first gated tick always audits
87
97
  let gatedSince = null;
98
+ let belowSince = null; // start of the current continuous sub-threshold run, while a stretch is open
88
99
  let last = null;
89
100
 
90
101
  function evaluate({ bypass = false } = {}) {
@@ -95,10 +106,21 @@ function createLoadGate({
95
106
  const ratio = c > 0 && Number.isFinite(l1) ? l1 / c : 0;
96
107
  const wouldGate = isLoadGated(l1, c, th);
97
108
 
109
+ // Hysteresis: the STRETCH (gatedSince) only clears after load has stayed
110
+ // below threshold continuously for releaseWindowMs. This never affects
111
+ // `gated` itself below — a sub-threshold tick still returns gated:false
112
+ // immediately; only the bookkeeping/escalation lags behind.
98
113
  if (wouldGate) {
99
114
  if (gatedSince === null) gatedSince = t;
115
+ belowSince = null;
116
+ } else if (gatedSince !== null) {
117
+ if (belowSince === null) belowSince = t;
118
+ if (t - belowSince >= releaseWindowMs) {
119
+ gatedSince = null;
120
+ belowSince = null;
121
+ }
100
122
  } else {
101
- gatedSince = null;
123
+ belowSince = null;
102
124
  }
103
125
  const gated = wouldGate && !bypass;
104
126
  const bypassed = wouldGate && bypass;
@@ -51,6 +51,20 @@ function resolveInstallChannel({ appPath = null, devFlag = !!process.env.SM_DEV
51
51
  return 'unknown';
52
52
  }
53
53
 
54
+ /**
55
+ * Resolves the wire-level `env` discriminator ('prod' | 'dev' | 'test') from
56
+ * the real signals the caller already has — never inferred downstream from
57
+ * an appVersion string, which is how 1,526 pre-release/test `epic.create`
58
+ * records ended up misread as production usage (see telemetry.md). `test`
59
+ * takes priority: a vitest run that happens to report a `dev` installChannel
60
+ * (SM_DEV set in the test env) is still test traffic, not dev usage.
61
+ */
62
+ function resolveEnv({ isTestRunner = false, installChannel } = {}) {
63
+ if (isTestRunner) return 'test';
64
+ if (installChannel === 'dev') return 'dev';
65
+ return 'prod';
66
+ }
67
+
54
68
  /** sha256 of the concatenated stable spec fields, truncated to 12 hex chars. */
55
69
  function computeMachineDigest(specs) {
56
70
  const material = [
@@ -141,4 +155,5 @@ module.exports = {
141
155
  buildMachineProfile,
142
156
  computeMachineDigest,
143
157
  resolveInstallChannel,
158
+ resolveEnv,
144
159
  };
@@ -102,9 +102,20 @@ function findBlockingDep(job, projectJobs, satisfiedSlugs = new Set()) {
102
102
  }
103
103
  return satisfiedBareSlugs.has(bareSlug(slug));
104
104
  };
105
+ // A dep row is blocking unless it's 'completed', OR it's a 'skipped' row
106
+ // stamped `needsReviewAutoResolvedSkip` — the bounded needs_review
107
+ // auto-resolve terminal decision (scheduler.cjs's
108
+ // applyNeedsReviewAutoResolve). That skip is deliberately NOT the
109
+ // "PRD source vanished, a human must author a fresh PRD" skip the
110
+ // heldBySkippedDep messaging above describes — it already got every
111
+ // bounded chance to resolve itself, so treating it as a permanent
112
+ // dependsOn block would defeat the whole point of that auto-resolve pass
113
+ // (a chain that never drains without an operator).
114
+ const isSatisfiedRow = (dep) => dep.status === 'completed'
115
+ || (dep.status === 'skipped' && dep.needsReviewAutoResolvedSkip === true);
105
116
  return (job.dependsOn ?? []).find((slug) => {
106
117
  const rows = rowsForDep(slug);
107
- if (rows.length > 0) return rows.some((dep) => dep.status !== 'completed');
118
+ if (rows.length > 0) return rows.some((dep) => !isSatisfiedRow(dep));
108
119
  return !isKnownSatisfied(slug);
109
120
  });
110
121
  }