claude-code-session-manager 0.84.0 → 0.86.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/dist/assets/{AgentLibrary-CCVIpoSz.js → AgentLibrary-DTFL7y8G.js} +1 -1
  2. package/dist/assets/{DataModel-BltREYde.js → DataModel-Q4jhl24R.js} +1 -1
  3. package/dist/assets/{History-BZxFkOp6.js → History-Cj2FejEo.js} +1 -1
  4. package/dist/assets/{Hooks-Bznlfaaa.js → Hooks-CaelQI6t.js} +1 -1
  5. package/dist/assets/{HostBilko-DUG5YHA_.js → HostBilko--v7cMR8I.js} +1 -1
  6. package/dist/assets/{Library-Bi2Fn3w9.js → Library-DgI9oCCZ.js} +1 -1
  7. package/dist/assets/{ListDetail-4VBvXKrz.js → ListDetail-DYUZN-x-.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-D2ft_v5j.js → MarkdownEditor-DCIubYWf.js} +1 -1
  9. package/dist/assets/{McpServers-FDekKkyE.js → McpServers-ypCYURh3.js} +1 -1
  10. package/dist/assets/{Memory-Dkl80uj_.js → Memory-C2qYp-3M.js} +1 -1
  11. package/dist/assets/{Panel-LurG5VfD.js → Panel-Cj2kw-Zv.js} +1 -1
  12. package/dist/assets/{Permissions-D2wHBCFA.js → Permissions-BiYZNGYW.js} +1 -1
  13. package/dist/assets/{Plugins-CTw_wwbI.js → Plugins-C1Vj8_dU.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-CW6HNUv6.js → ProvenanceBadge-DczPNM5U.js} +1 -1
  15. package/dist/assets/{SaveBar-alDHG6cP.js → SaveBar-Cd_7U6Gb.js} +1 -1
  16. package/dist/assets/{Scheduler-DxiPcaiW.js → Scheduler-DcLBiJBq.js} +1 -1
  17. package/dist/assets/{ScopeSwitcher-DzFXUKLZ.js → ScopeSwitcher-DVSyI44-.js} +1 -1
  18. package/dist/assets/{Settings-B6H3v2am.js → Settings-Cv-pRyms.js} +3 -3
  19. package/dist/assets/{SkillReferenceGraph-B8EZalpN.js → SkillReferenceGraph-CUv1_Q2c.js} +1 -1
  20. package/dist/assets/{Skills-DwuHuA08.js → Skills-C_YHkAy-.js} +1 -1
  21. package/dist/assets/{SystemPrompt-CMMqpGYn.js → SystemPrompt-B8R7T9xn.js} +1 -1
  22. package/dist/assets/{TagLibrary-C2N5C1n9.js → TagLibrary-dj9YHWyy.js} +1 -1
  23. package/dist/assets/{TiptapBody-btlID-dQ.js → TiptapBody-DnSBUjHE.js} +1 -1
  24. package/dist/assets/{Toggle-BGOn3DZj.js → Toggle-CjV_BJn6.js} +1 -1
  25. package/dist/assets/{index-mnjNDpb1.css → index-CDo9xBR9.css} +1 -1
  26. package/dist/assets/{index-QLRf0epp.js → index-CXFQIPhO.js} +3 -3
  27. package/dist/assets/{settingsSchema-DA3N2Up3.js → settingsSchema-BJVciriw.js} +1 -1
  28. package/dist/index.html +2 -2
  29. package/package.json +3 -2
  30. package/src/main/__tests__/health-starve-escalation.test.cjs +94 -0
  31. package/src/main/__tests__/loadGateDetailTick.test.cjs +31 -0
  32. package/src/main/__tests__/machineProfile.test.cjs +19 -1
  33. package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +6 -0
  34. package/src/main/__tests__/pty-session-open-telemetry.test.cjs +96 -0
  35. package/src/main/__tests__/queue-starvation-per-project.test.cjs +135 -0
  36. package/src/main/__tests__/scheduler-failed-autoreset.test.cjs +121 -0
  37. package/src/main/__tests__/scheduler-needs-review-autoresolve.test.cjs +189 -0
  38. package/src/main/__tests__/scheduler-no-dead-end-status.test.cjs +152 -0
  39. package/src/main/__tests__/scheduler-quarantine-autoresolve.test.cjs +165 -0
  40. package/src/main/__tests__/scheduler-starve-escalation.test.cjs +144 -0
  41. package/src/main/__tests__/telemetryClient.test.cjs +250 -5
  42. package/src/main/__tests__/telemetryContract.test.cjs +74 -27
  43. package/src/main/__tests__/telemetrySettings.test.cjs +32 -0
  44. package/src/main/health.cjs +75 -2
  45. package/src/main/index.cjs +2 -2
  46. package/src/main/lib/__tests__/crashTelemetry.test.cjs +6 -0
  47. package/src/main/lib/__tests__/loadGate.test.cjs +103 -2
  48. package/src/main/lib/__tests__/telemetryBacklog.test.cjs +6 -0
  49. package/src/main/lib/__tests__/telemetryBoot.test.cjs +22 -13
  50. package/src/main/lib/__tests__/telemetryConsent.test.cjs +6 -0
  51. package/src/main/lib/__tests__/telemetryCountersMetadataColumn.test.cjs +6 -0
  52. package/src/main/lib/loadGate.cjs +23 -1
  53. package/src/main/lib/machineProfile.cjs +15 -0
  54. package/src/main/lib/queueStore.cjs +8 -2
  55. package/src/main/lib/schedulerBatch.cjs +12 -1
  56. package/src/main/lib/schedulerConfig.cjs +13 -0
  57. package/src/main/lib/telemetryBoot.cjs +18 -10
  58. package/src/main/lib/telemetryClient.cjs +162 -6
  59. package/src/main/lib/telemetrySettings.cjs +20 -3
  60. package/src/main/pty.cjs +9 -0
  61. package/src/main/scheduler.cjs +600 -66
@@ -18,9 +18,11 @@ const counters = require('../telemetryCounters.cjs');
18
18
 
19
19
  const tmpDirs = [];
20
20
  let originalHome;
21
+ let originalSmTelemetrySpool;
21
22
 
22
23
  afterEach(async () => {
23
24
  if (originalHome !== undefined) process.env.HOME = originalHome;
25
+ if (originalSmTelemetrySpool === undefined) delete process.env.SM_TELEMETRY_SPOOL; else process.env.SM_TELEMETRY_SPOOL = originalSmTelemetrySpool;
24
26
  while (tmpDirs.length) {
25
27
  const d = tmpDirs.pop();
26
28
  await fsp.rm(d, { recursive: true, force: true });
@@ -29,9 +31,13 @@ afterEach(async () => {
29
31
 
30
32
  async function mkHome() {
31
33
  originalHome = process.env.HOME;
34
+ originalSmTelemetrySpool = process.env.SM_TELEMETRY_SPOOL;
32
35
  const dir = await fsp.mkdtemp(path.join(os.tmpdir(), 'sm-telemetry-counters-home-'));
33
36
  tmpDirs.push(dir);
34
37
  process.env.HOME = dir;
38
+ // Explicit opt-in: overrides the test-environment no-op guard so this
39
+ // suite's real telemetryClient calls actually persist into an isolated dir.
40
+ process.env.SM_TELEMETRY_SPOOL = path.join(dir, '.config', 'session-manager');
35
41
  return dir;
36
42
  }
37
43
 
@@ -32,6 +32,15 @@ const { loadGateThreshold, JOB_OVERRUN_FLOOR_MS } = require('./schedulerConfig.c
32
32
  // every tick would bury the signal in its own noise.
33
33
  const AUDIT_INTERVAL_MS = 10 * 60_000;
34
34
 
35
+ // Default hysteresis release window for the gated stretch (PRD: boundary-
36
+ // hovering load). A box sitting right at the threshold oscillates above/below
37
+ // it tick to tick; clearing `gatedSince` on the first sub-threshold sample
38
+ // reset the stretch counter every time, so `escalate` (and the warn-log with
39
+ // topCpuConsumers()) never fired even after 80+ minutes of effectively
40
+ // continuous gating. The stretch now only clears once load has stayed below
41
+ // threshold continuously for this long.
42
+ const RELEASE_WINDOW_MS = 2 * 60_000;
43
+
35
44
  /**
36
45
  * isLoadGated(loadavg1, cores, threshold) → boolean
37
46
  *
@@ -82,9 +91,11 @@ function createLoadGate({
82
91
  threshold = loadGateThreshold,
83
92
  auditIntervalMs = AUDIT_INTERVAL_MS,
84
93
  escalateAfterMs = JOB_OVERRUN_FLOOR_MS,
94
+ releaseWindowMs = RELEASE_WINDOW_MS,
85
95
  } = {}) {
86
96
  let lastAuditAt = null; // null = never audited; the first gated tick always audits
87
97
  let gatedSince = null;
98
+ let belowSince = null; // start of the current continuous sub-threshold run, while a stretch is open
88
99
  let last = null;
89
100
 
90
101
  function evaluate({ bypass = false } = {}) {
@@ -95,10 +106,21 @@ function createLoadGate({
95
106
  const ratio = c > 0 && Number.isFinite(l1) ? l1 / c : 0;
96
107
  const wouldGate = isLoadGated(l1, c, th);
97
108
 
109
+ // Hysteresis: the STRETCH (gatedSince) only clears after load has stayed
110
+ // below threshold continuously for releaseWindowMs. This never affects
111
+ // `gated` itself below — a sub-threshold tick still returns gated:false
112
+ // immediately; only the bookkeeping/escalation lags behind.
98
113
  if (wouldGate) {
99
114
  if (gatedSince === null) gatedSince = t;
115
+ belowSince = null;
116
+ } else if (gatedSince !== null) {
117
+ if (belowSince === null) belowSince = t;
118
+ if (t - belowSince >= releaseWindowMs) {
119
+ gatedSince = null;
120
+ belowSince = null;
121
+ }
100
122
  } else {
101
- gatedSince = null;
123
+ belowSince = null;
102
124
  }
103
125
  const gated = wouldGate && !bypass;
104
126
  const bypassed = wouldGate && bypass;
@@ -51,6 +51,20 @@ function resolveInstallChannel({ appPath = null, devFlag = !!process.env.SM_DEV
51
51
  return 'unknown';
52
52
  }
53
53
 
54
+ /**
55
+ * Resolves the wire-level `env` discriminator ('prod' | 'dev' | 'test') from
56
+ * the real signals the caller already has — never inferred downstream from
57
+ * an appVersion string, which is how 1,526 pre-release/test `epic.create`
58
+ * records ended up misread as production usage (see telemetry.md). `test`
59
+ * takes priority: a vitest run that happens to report a `dev` installChannel
60
+ * (SM_DEV set in the test env) is still test traffic, not dev usage.
61
+ */
62
+ function resolveEnv({ isTestRunner = false, installChannel } = {}) {
63
+ if (isTestRunner) return 'test';
64
+ if (installChannel === 'dev') return 'dev';
65
+ return 'prod';
66
+ }
67
+
54
68
  /** sha256 of the concatenated stable spec fields, truncated to 12 hex chars. */
55
69
  function computeMachineDigest(specs) {
56
70
  const material = [
@@ -141,4 +155,5 @@ module.exports = {
141
155
  buildMachineProfile,
142
156
  computeMachineDigest,
143
157
  resolveInstallChannel,
158
+ resolveEnv,
144
159
  };
@@ -220,6 +220,11 @@ function shapeMachine(data) {
220
220
  config: data.config || {},
221
221
  scheduledFor: data.scheduledFor ?? null,
222
222
  lastRunAt: data.lastRunAt ?? null,
223
+ // Distinct from lastRunAt (stamped only when tickQueue actually launches a
224
+ // job): this is stamped every time tickQueue gets far enough to evaluate
225
+ // the queue at all, whether or not that evaluation ends in a launch. See
226
+ // classifyQueueStarvation's header for why the two must never merge.
227
+ lastDispatchAttemptAt: data.lastDispatchAttemptAt ?? null,
223
228
  paused: data.paused ?? null,
224
229
  // Launch circuit breaker (lib/launchFailure.cjs): per-persona blocks and
225
230
  // the degraded-mode env a persona is currently launching with. Machine
@@ -440,7 +445,7 @@ function shapeJobs(raw, file) {
440
445
  * consulted so writeSplit can persist "this project now has zero jobs".
441
446
  */
442
447
  function readMergedSync(opts) {
443
- const out = { config: {}, jobs: [], scheduledFor: null, lastRunAt: null, paused: null, launchBlocks: {}, launchMitigations: {}, invalidJobs: [] };
448
+ const out = { config: {}, jobs: [], scheduledFor: null, lastRunAt: null, lastDispatchAttemptAt: null, paused: null, launchBlocks: {}, launchMitigations: {}, invalidJobs: [] };
444
449
  const sourceCwds = [];
445
450
  const machine = loadMachineStateSync();
446
451
  if (machine.shaped) {
@@ -473,7 +478,7 @@ function readMergedSync(opts) {
473
478
 
474
479
  /** Async twin of readMergedSync for IPC hot paths. */
475
480
  async function readMerged(opts) {
476
- const out = { config: {}, jobs: [], scheduledFor: null, lastRunAt: null, paused: null, launchBlocks: {}, launchMitigations: {}, invalidJobs: [] };
481
+ const out = { config: {}, jobs: [], scheduledFor: null, lastRunAt: null, lastDispatchAttemptAt: null, paused: null, launchBlocks: {}, launchMitigations: {}, invalidJobs: [] };
477
482
  const sourceCwds = [];
478
483
  const machine = await loadMachineState();
479
484
  if (machine.shaped) {
@@ -525,6 +530,7 @@ async function writeSplit(state, defaultCwd) {
525
530
  config: state.config,
526
531
  scheduledFor: state.scheduledFor ?? null,
527
532
  lastRunAt: state.lastRunAt ?? null,
533
+ lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
528
534
  paused: state.paused ?? null,
529
535
  launchBlocks: state.launchBlocks ?? {},
530
536
  launchMitigations: state.launchMitigations ?? {},
@@ -102,9 +102,20 @@ function findBlockingDep(job, projectJobs, satisfiedSlugs = new Set()) {
102
102
  }
103
103
  return satisfiedBareSlugs.has(bareSlug(slug));
104
104
  };
105
+ // A dep row is blocking unless it's 'completed', OR it's a 'skipped' row
106
+ // stamped `needsReviewAutoResolvedSkip` — the bounded needs_review
107
+ // auto-resolve terminal decision (scheduler.cjs's
108
+ // applyNeedsReviewAutoResolve). That skip is deliberately NOT the
109
+ // "PRD source vanished, a human must author a fresh PRD" skip the
110
+ // heldBySkippedDep messaging above describes — it already got every
111
+ // bounded chance to resolve itself, so treating it as a permanent
112
+ // dependsOn block would defeat the whole point of that auto-resolve pass
113
+ // (a chain that never drains without an operator).
114
+ const isSatisfiedRow = (dep) => dep.status === 'completed'
115
+ || (dep.status === 'skipped' && dep.needsReviewAutoResolvedSkip === true);
105
116
  return (job.dependsOn ?? []).find((slug) => {
106
117
  const rows = rowsForDep(slug);
107
- if (rows.length > 0) return rows.some((dep) => dep.status !== 'completed');
118
+ if (rows.length > 0) return rows.some((dep) => !isSatisfiedRow(dep));
108
119
  return !isKnownSatisfied(slug);
109
120
  });
110
121
  }
@@ -81,6 +81,19 @@ module.exports = {
81
81
  // 1086). Escalation only — warn + audit, never dispatch.
82
82
  STARVATION_ESCALATE_MS: 45 * 60_000,
83
83
 
84
+ // A project still meeting findStarvedProjects' STARVED verdict past THIS
85
+ // (later) threshold gets a bounded, automated consequence beyond the
86
+ // repeating project_starved WARN above: a one-shot 'project_starve_escalated'
87
+ // audit event + a user-facing toast naming the project and the last tick's
88
+ // hold reason (scheduler.cjs's `lastTick`). Escalation only — never
89
+ // auto-resets, auto-cancels, or bypasses a gate; the queue's own state is
90
+ // untouched. The 2026-09-12 audit log showed /home/bilko/Projects/Bilko
91
+ // emit project_starved for 19h straight (ageMs 60.5M → 68.9M) with zero
92
+ // consequence and no toast — this is the fix. Latched per-cwd so the SAME
93
+ // starve stretch escalates exactly once; the latch resets the moment that
94
+ // cwd stops appearing in findStarvedProjects.
95
+ STARVE_ESCALATION_MS: 60 * 60_000,
96
+
84
97
  // A RUNNING job that has overrun its own PRD's `estimateMinutes` by this
85
98
  // factor is escalated. Distinct from MAX_JOB_DURATION_MS (4h), which is a
86
99
  // deadman kill: a 20-minute PRD still running at 3h is 9x over estimate but
@@ -36,10 +36,17 @@ function resolveDeps(deps = {}) {
36
36
  * 2. flush('version-change') — only when this install's persisted
37
37
  * lastMachineReportVersion differs from the running appVersion (the same
38
38
  * signal that also gates the machine-profile heartbeat below).
39
- * 3. install.machine — sent once as track('install.machine', profile) when
40
- * telemetrySettings.isMachineReportDue() is true (a version bump, OR the
41
- * 30-day liveness heartbeat), then lastMachineReportAt/Version are
42
- * persisted so a same-version reboot within the window sends nothing.
39
+ * 3. install upsert — sent once via telemetryClient.reportInstall(profile)
40
+ * when telemetrySettings.isMachineReportDue() is true (a version bump, OR
41
+ * the 30-day liveness heartbeat). lastMachineReportAt/Version are only
42
+ * persisted when reportInstall() reports success — a failed/dropped
43
+ * upsert stays due so the next boot retries, rather than being marked
44
+ * done regardless of outcome.
45
+ * This lands in bilko.run's app_installs table (the sole source for
46
+ * every install-shaped number on the site). It replaces the former
47
+ * track('install.machine', profile) call, which duplicated the same
48
+ * facts into funnel_events for no reader — app_installs is now the only
49
+ * destination for a machine profile.
43
50
  * 4. app.launch — one counter event per boot, via telemetryCounters so the
44
51
  * shape lives in exactly one place across every counter this PRD adds.
45
52
  */
@@ -57,12 +64,13 @@ async function bootSequence({ now = Date.now(), appVersion, installChannel, deps
57
64
 
58
65
  if (telemetrySettings.isMachineReportDue(settings, { now, appVersion })) {
59
66
  const profile = await buildMachineProfile();
60
- await telemetryClient.track('install.machine', profile);
61
- await telemetrySettings.save({
62
- ...settings,
63
- lastMachineReportAt: new Date(now).toISOString(),
64
- lastMachineReportVersion: appVersion,
65
- });
67
+ const reportResult = await telemetryClient.reportInstall(profile);
68
+ if (reportResult && reportResult.accepted) {
69
+ await telemetrySettings.save({
70
+ lastMachineReportAt: new Date(now).toISOString(),
71
+ lastMachineReportVersion: appVersion,
72
+ });
73
+ }
66
74
  }
67
75
 
68
76
  telemetryCounters.trackAppLaunch({ installChannel, appVersion });
@@ -1,6 +1,11 @@
1
1
  /**
2
2
  * telemetryClient — the single egress module that ships telemetry to
3
- * bilko.run's beacon endpoints (POST /api/telemetry/event, /log, /error).
3
+ * bilko.run's beacon endpoints (POST /api/telemetry/event, /log, /error,
4
+ * /install). The `install` channel is a durable upsert keyed on install_id
5
+ * (app_installs), not an append like the other three — its wire body is a
6
+ * single flat snake_case object, never `{batch:[...]}` — so it is routed
7
+ * through the same queue/dedup/backoff machinery but sent one record at a
8
+ * time via sendSingle() rather than sendBatch().
4
9
  *
5
10
  * Records accumulate durably in ~/.config/session-manager/telemetry-queue.jsonl
6
11
  * the instant they're accepted, and are only ever sent by flush(reason) — on
@@ -86,20 +91,73 @@ const S = {
86
91
  lastError: null,
87
92
  lastFlushAt: null,
88
93
  lastFlushReason: null,
94
+ logger: null,
89
95
  };
90
96
 
97
+ /**
98
+ * Best-effort warn logger for failures this module would otherwise swallow
99
+ * (e.g. reportInstall()'s bare catch). Lazily requires logs.cjs — which
100
+ * requires('electron') at module scope — so this stays a no-op under plain
101
+ * vitest with no Electron runtime, exactly like machineProfile.cjs's
102
+ * resolveElectronApp(). Injectable via _setLogger so tests can observe it
103
+ * without an Electron runtime.
104
+ */
105
+ function logWarn(message, meta) {
106
+ try {
107
+ const fn = S.logger || require('../logs.cjs').writeLine;
108
+ fn({ scope: 'telemetry', level: 'warn', message, meta });
109
+ } catch { /* logs unavailable outside an Electron runtime */ }
110
+ }
111
+
112
+ /**
113
+ * SM_TELEMETRY_SPOOL, when set, overrides the spool directory in place of
114
+ * `~/.config/session-manager` — the explicit opt-in a test that genuinely
115
+ * needs to exercise real queue/sent-file I/O uses to prove it isn't about to
116
+ * write into a real user's spool (see isTestEnvironment() below).
117
+ */
118
+ function spoolDir() {
119
+ return process.env.SM_TELEMETRY_SPOOL || path.join(os.homedir(), '.config', 'session-manager');
120
+ }
121
+
91
122
  function queuePath() {
92
- return path.join(os.homedir(), '.config', 'session-manager', 'telemetry-queue.jsonl');
123
+ return path.join(spoolDir(), 'telemetry-queue.jsonl');
93
124
  }
94
125
 
95
126
  function sentPath() {
96
- return path.join(os.homedir(), '.config', 'session-manager', 'telemetry-sent.json');
127
+ return path.join(spoolDir(), 'telemetry-sent.json');
128
+ }
129
+
130
+ /**
131
+ * True when running under a test runner (VITEST, NODE_ENV=test) with no
132
+ * explicit SM_TELEMETRY_SPOOL override. Gates appendRecord() below so a test
133
+ * that merely exercises epicMint/scheduler/etc. code paths — without itself
134
+ * setting up an isolated spool — can never accumulate fake records into a
135
+ * real user's telemetry-queue.jsonl. A test that deliberately wants to
136
+ * exercise real persistence (telemetryClient's own suite) sets
137
+ * SM_TELEMETRY_SPOOL to an isolated directory, which counts as an explicit
138
+ * opt-in and disables this guard.
139
+ */
140
+ function isTestEnvironment() {
141
+ if (process.env.SM_TELEMETRY_SPOOL) return false;
142
+ return !!process.env.VITEST || process.env.NODE_ENV === 'test';
97
143
  }
98
144
 
99
145
  function getBeaconToken() {
100
146
  return process.env.SM_BEACON_KEY || FALLBACK_BEACON_KEY;
101
147
  }
102
148
 
149
+ /**
150
+ * The wire-level `env` discriminator for the currently-running process,
151
+ * resolved from real signals (never from an appVersion string — see
152
+ * machineProfile.cjs's resolveEnv header comment for why that matters).
153
+ */
154
+ function currentEnv() {
155
+ return machineProfile.resolveEnv({
156
+ isTestRunner: isTestEnvironment(),
157
+ installChannel: S.profile && S.profile.installChannel,
158
+ });
159
+ }
160
+
103
161
  // ─── safe coercion helpers (never throw, regardless of input shape) ───────
104
162
 
105
163
  function safeObj(v) {
@@ -368,6 +426,9 @@ function pushRecent(rec) {
368
426
  }
369
427
 
370
428
  async function appendRecord(channel, wire, recordId) {
429
+ if (isTestEnvironment()) {
430
+ return { accepted: false, reason: 'test-environment', recordId };
431
+ }
371
432
  if (S.queueIds.has(recordId) || S.sentIds.has(recordId)) {
372
433
  S.dedupedAppends += 1;
373
434
  return { accepted: false, reason: 'duplicate', recordId };
@@ -403,6 +464,7 @@ async function track(name, props) {
403
464
  props: meta,
404
465
  session_id: S.sessionId,
405
466
  visitor_id: S.settings.installId,
467
+ env: currentEnv(),
406
468
  };
407
469
  return await appendRecord('event', wire, recordId);
408
470
  } catch {
@@ -432,6 +494,7 @@ async function logLine(opts) {
432
494
  session_id: S.sessionId,
433
495
  fields: meta,
434
496
  ts: Date.now(),
497
+ env: currentEnv(),
435
498
  };
436
499
  return await appendRecord('log', wire, recordId);
437
500
  } catch {
@@ -468,6 +531,7 @@ async function reportError(opts) {
468
531
  session_id: S.sessionId,
469
532
  context: meta,
470
533
  ts: Date.now(),
534
+ env: currentEnv(),
471
535
  };
472
536
  return await appendRecord('error', wire, recordId);
473
537
  } catch {
@@ -475,6 +539,53 @@ async function reportError(opts) {
475
539
  }
476
540
  }
477
541
 
542
+ function buildInstallWire(profile, installId) {
543
+ const p = safeObj(profile);
544
+ return {
545
+ install_id: safeStr(installId, 80),
546
+ app: 'session-manager',
547
+ app_version: safeStr(p.appVersion, 20),
548
+ platform: safeStr(p.platform, 20),
549
+ os_release: safeStr(p.osRelease, 80),
550
+ arch: safeStr(p.arch, 20),
551
+ cpu_count: typeof p.cpuCount === 'number' ? p.cpuCount : 0,
552
+ total_mem_mb: typeof p.totalMemMb === 'number' ? p.totalMemMb : 0,
553
+ node_version: safeStr(p.nodeVersion, 20),
554
+ electron_version: safeStr(p.electronVersion, 20),
555
+ install_channel: safeStr(p.installChannel, 20),
556
+ locale: safeStr(p.locale, 20),
557
+ // machineProfile deliberately never carries an IANA timezone name — only
558
+ // the numeric UTC offset (see machineProfile.cjs) — so that is what maps
559
+ // onto the server's `timezone` column.
560
+ timezone: safeStr(p.timezoneOffsetMinutes, 20),
561
+ // Resolved from this passed-in profile's own installChannel (never
562
+ // S.profile — reportInstall() is always given a freshly-built profile,
563
+ // not the module-singleton one) plus the current process's test-runner
564
+ // signal, same three-value contract as the other three channels.
565
+ env: machineProfile.resolveEnv({ isTestRunner: isTestEnvironment(), installChannel: p.installChannel }),
566
+ };
567
+ }
568
+
569
+ /**
570
+ * Reports the once-per-cadence install/liveness record. Unlike track/logLine/
571
+ * reportError this isn't arbitrary user content, so it skips redactDeep() —
572
+ * every field already comes from machineProfile's own anonymity contract —
573
+ * and it never merges an attribution block onto the body, since the server's
574
+ * app_installs upsert doesn't read a recordId.
575
+ */
576
+ async function reportInstall(profile) {
577
+ try {
578
+ await ensureInit();
579
+ if (!telemetrySettings.isEnabled(S.settings)) return { accepted: false, reason: 'disabled' };
580
+ const recordId = crypto.randomUUID();
581
+ const wire = buildInstallWire(profile, S.settings.installId);
582
+ return await appendRecord('install', wire, recordId);
583
+ } catch (e) {
584
+ logWarn('reportInstall failed', { error: e && e.message ? e.message : String(e) });
585
+ return { accepted: false, reason: 'error' };
586
+ }
587
+ }
588
+
478
589
  // ─── egress ────────────────────────────────────────────────────────────
479
590
 
480
591
  async function sendBatch(channel, wireBatch) {
@@ -496,6 +607,26 @@ async function sendBatch(channel, wireBatch) {
496
607
  }
497
608
  }
498
609
 
610
+ /** Like sendBatch(), but posts `wire` as a flat body — the install route takes one upsert, never a `{batch:[...]}` envelope. */
611
+ async function sendSingle(channel, wire) {
612
+ const fetchFn = S.fetchImpl || (typeof fetch === 'function' ? fetch : null);
613
+ if (typeof fetchFn !== 'function') return { ok: false, status: 0 };
614
+ const base = telemetrySettings.resolveEndpoint(S.settings);
615
+ const url = `${base}/api/telemetry/${channel}`;
616
+ const headers = {
617
+ 'Content-Type': 'application/json',
618
+ 'X-SM-Beacon': `session-manager/${S.profile.appVersion}`,
619
+ 'X-SM-Beacon-Key': getBeaconToken(),
620
+ };
621
+ try {
622
+ const res = await fetchFn(url, { method: 'POST', headers, body: JSON.stringify(wire) });
623
+ if (res && res.ok) return { ok: true };
624
+ return { ok: false, status: res ? res.status : 0 };
625
+ } catch {
626
+ return { ok: false, status: 0 };
627
+ }
628
+ }
629
+
499
630
  function describeFailureStatus(status) {
500
631
  return status === 0 ? 'network error' : `HTTP ${status}`;
501
632
  }
@@ -537,15 +668,34 @@ async function flushImpl(reason) {
537
668
  if (S.disabledForProcess) return result;
538
669
  if (Date.now() < S.backoffUntil) return result;
539
670
 
540
- const byChannel = { event: [], log: [], error: [] };
671
+ const byChannel = { event: [], log: [], error: [], install: [] };
541
672
  for (const r of S.queue) {
542
673
  if (byChannel[r.channel]) byChannel[r.channel].push(r);
543
674
  }
544
675
 
545
676
  let sawFailure = false;
546
- for (const channel of ['event', 'log', 'error']) {
677
+ for (const channel of ['event', 'log', 'error', 'install']) {
547
678
  if (sawFailure) break;
548
679
  const records = byChannel[channel];
680
+ if (channel === 'install') {
681
+ for (const r of records) {
682
+ const res = await sendSingle('install', r.wire);
683
+ if (res.ok) {
684
+ S.consecutiveFailures = 0;
685
+ S.backoffUntil = 0;
686
+ S.lastError = null;
687
+ removeFromQueue(r.recordId);
688
+ addToSent(r.recordId);
689
+ result.sent.push(r.recordId);
690
+ } else {
691
+ result.failed.push(r.recordId);
692
+ applyFailureBackoff(res.status);
693
+ sawFailure = true;
694
+ break;
695
+ }
696
+ }
697
+ continue;
698
+ }
549
699
  for (let i = 0; i < records.length; i += MAX_BATCH) {
550
700
  const batch = records.slice(i, i + MAX_BATCH);
551
701
  const res = await sendBatch(channel, batch.map((r) => r.wire));
@@ -571,7 +721,7 @@ async function flushImpl(reason) {
571
721
  await persistSent();
572
722
 
573
723
  if (safeReason === 'daily' && result.failed.length === 0) {
574
- S.settings = await telemetrySettings.save({ ...S.settings, lastDailyFlushAt: new Date().toISOString() });
724
+ S.settings = await telemetrySettings.save({ lastDailyFlushAt: new Date().toISOString() });
575
725
  }
576
726
  } catch { /* fail-inert */ }
577
727
  S.lastFlushAt = Date.now();
@@ -636,10 +786,15 @@ function _setMachineProfileBuilder(fn) {
636
786
  S.profileBuilder = typeof fn === 'function' ? fn : machineProfile.buildMachineProfile;
637
787
  }
638
788
 
789
+ function _setLogger(fn) {
790
+ S.logger = typeof fn === 'function' ? fn : null;
791
+ }
792
+
639
793
  module.exports = {
640
794
  track,
641
795
  logLine,
642
796
  reportError,
797
+ reportInstall,
643
798
  flush,
644
799
  shutdown,
645
800
  status,
@@ -650,4 +805,5 @@ module.exports = {
650
805
  sentPath,
651
806
  _setFetchImpl,
652
807
  _setMachineProfileBuilder,
808
+ _setLogger,
653
809
  };
@@ -78,11 +78,28 @@ function normalize(cfg) {
78
78
  };
79
79
  }
80
80
 
81
+ /**
82
+ * Persists `patch` merged onto a FRESHLY-READ copy of the on-disk config —
83
+ * never onto a caller-held snapshot. Two independent modules (telemetryBoot's
84
+ * install-report stamp, telemetryClient's own daily-flush stamp) each keep
85
+ * their own in-memory copy of settings loaded at different times; if save()
86
+ * blindly wrote a caller's full snapshot, whichever call landed last would
87
+ * silently revert every field the OTHER caller had just written (a lost
88
+ * update — the root cause behind lastMachineReportAt never sticking). Calls
89
+ * are still serialized through the existing writeQueue chain, so the fresh
90
+ * read for call N+1 always observes call N's write.
91
+ */
81
92
  let writeQueue = Promise.resolve();
82
- async function save(cfg) {
83
- if (!isValid(cfg)) throw new Error('Invalid telemetry config');
84
- const next = normalize(cfg);
93
+ async function save(patch) {
94
+ if (!patch || typeof patch !== 'object' || Array.isArray(patch)) throw new Error('Invalid telemetry config');
95
+ for (const k of Object.keys(patch)) {
96
+ if (!KNOWN_KEYS.has(k)) throw new Error('Invalid telemetry config');
97
+ }
85
98
  const run = async () => {
99
+ const current = await readRaw();
100
+ const merged = { ...current, ...patch };
101
+ if (!isValid(merged)) throw new Error('Invalid telemetry config');
102
+ const next = normalize(merged);
86
103
  await config.writeTextAtomic(storePath(), JSON.stringify(next, null, 2) + '\n', { mode: 0o600 });
87
104
  return next;
88
105
  };
package/src/main/pty.cjs CHANGED
@@ -109,6 +109,15 @@ class PtyManager {
109
109
  // renderer will re-register its data/exit listeners on the same IPC
110
110
  // channels. The data stream is live; pre-reattach output is lost, which
111
111
  // is acceptable for a dev reload.
112
+ //
113
+ // Telemetry: this branch deliberately does NOT call
114
+ // telemetryCounters.trackSessionOpen() (see below, after the real spawn).
115
+ // Tab = claudeSessionId is a 1:1 mapping (CLAUDE.md domain model) — a
116
+ // reattach is the SAME session process still running, not a new one, so
117
+ // counting it here would double-count every renderer reload and every
118
+ // switch back to an already-open Epic's Terminal pane within the same
119
+ // Electron process. `session.open` counts fresh PTY spawns only; see
120
+ // telemetry.md for the full rationale and its trade-off.
112
121
  const existing = this.sessions.get(tabId);
113
122
  if (existing) {
114
123
  console.log('[pty] reattach to existing session tabId=', tabId, 'pid=', existing.proc.pid);