claude-code-session-manager 0.84.0 → 0.86.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-CCVIpoSz.js → AgentLibrary-DTFL7y8G.js} +1 -1
- package/dist/assets/{DataModel-BltREYde.js → DataModel-Q4jhl24R.js} +1 -1
- package/dist/assets/{History-BZxFkOp6.js → History-Cj2FejEo.js} +1 -1
- package/dist/assets/{Hooks-Bznlfaaa.js → Hooks-CaelQI6t.js} +1 -1
- package/dist/assets/{HostBilko-DUG5YHA_.js → HostBilko--v7cMR8I.js} +1 -1
- package/dist/assets/{Library-Bi2Fn3w9.js → Library-DgI9oCCZ.js} +1 -1
- package/dist/assets/{ListDetail-4VBvXKrz.js → ListDetail-DYUZN-x-.js} +1 -1
- package/dist/assets/{MarkdownEditor-D2ft_v5j.js → MarkdownEditor-DCIubYWf.js} +1 -1
- package/dist/assets/{McpServers-FDekKkyE.js → McpServers-ypCYURh3.js} +1 -1
- package/dist/assets/{Memory-Dkl80uj_.js → Memory-C2qYp-3M.js} +1 -1
- package/dist/assets/{Panel-LurG5VfD.js → Panel-Cj2kw-Zv.js} +1 -1
- package/dist/assets/{Permissions-D2wHBCFA.js → Permissions-BiYZNGYW.js} +1 -1
- package/dist/assets/{Plugins-CTw_wwbI.js → Plugins-C1Vj8_dU.js} +2 -2
- package/dist/assets/{ProvenanceBadge-CW6HNUv6.js → ProvenanceBadge-DczPNM5U.js} +1 -1
- package/dist/assets/{SaveBar-alDHG6cP.js → SaveBar-Cd_7U6Gb.js} +1 -1
- package/dist/assets/{Scheduler-DxiPcaiW.js → Scheduler-DcLBiJBq.js} +1 -1
- package/dist/assets/{ScopeSwitcher-DzFXUKLZ.js → ScopeSwitcher-DVSyI44-.js} +1 -1
- package/dist/assets/{Settings-B6H3v2am.js → Settings-Cv-pRyms.js} +3 -3
- package/dist/assets/{SkillReferenceGraph-B8EZalpN.js → SkillReferenceGraph-CUv1_Q2c.js} +1 -1
- package/dist/assets/{Skills-DwuHuA08.js → Skills-C_YHkAy-.js} +1 -1
- package/dist/assets/{SystemPrompt-CMMqpGYn.js → SystemPrompt-B8R7T9xn.js} +1 -1
- package/dist/assets/{TagLibrary-C2N5C1n9.js → TagLibrary-dj9YHWyy.js} +1 -1
- package/dist/assets/{TiptapBody-btlID-dQ.js → TiptapBody-DnSBUjHE.js} +1 -1
- package/dist/assets/{Toggle-BGOn3DZj.js → Toggle-CjV_BJn6.js} +1 -1
- package/dist/assets/{index-mnjNDpb1.css → index-CDo9xBR9.css} +1 -1
- package/dist/assets/{index-QLRf0epp.js → index-CXFQIPhO.js} +3 -3
- package/dist/assets/{settingsSchema-DA3N2Up3.js → settingsSchema-BJVciriw.js} +1 -1
- package/dist/index.html +2 -2
- package/package.json +3 -2
- package/src/main/__tests__/health-starve-escalation.test.cjs +94 -0
- package/src/main/__tests__/loadGateDetailTick.test.cjs +31 -0
- package/src/main/__tests__/machineProfile.test.cjs +19 -1
- package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +6 -0
- package/src/main/__tests__/pty-session-open-telemetry.test.cjs +96 -0
- package/src/main/__tests__/queue-starvation-per-project.test.cjs +135 -0
- package/src/main/__tests__/scheduler-failed-autoreset.test.cjs +121 -0
- package/src/main/__tests__/scheduler-needs-review-autoresolve.test.cjs +189 -0
- package/src/main/__tests__/scheduler-no-dead-end-status.test.cjs +152 -0
- package/src/main/__tests__/scheduler-quarantine-autoresolve.test.cjs +165 -0
- package/src/main/__tests__/scheduler-starve-escalation.test.cjs +144 -0
- package/src/main/__tests__/telemetryClient.test.cjs +250 -5
- package/src/main/__tests__/telemetryContract.test.cjs +74 -27
- package/src/main/__tests__/telemetrySettings.test.cjs +32 -0
- package/src/main/health.cjs +75 -2
- package/src/main/index.cjs +2 -2
- package/src/main/lib/__tests__/crashTelemetry.test.cjs +6 -0
- package/src/main/lib/__tests__/loadGate.test.cjs +103 -2
- package/src/main/lib/__tests__/telemetryBacklog.test.cjs +6 -0
- package/src/main/lib/__tests__/telemetryBoot.test.cjs +22 -13
- package/src/main/lib/__tests__/telemetryConsent.test.cjs +6 -0
- package/src/main/lib/__tests__/telemetryCountersMetadataColumn.test.cjs +6 -0
- package/src/main/lib/loadGate.cjs +23 -1
- package/src/main/lib/machineProfile.cjs +15 -0
- package/src/main/lib/queueStore.cjs +8 -2
- package/src/main/lib/schedulerBatch.cjs +12 -1
- package/src/main/lib/schedulerConfig.cjs +13 -0
- package/src/main/lib/telemetryBoot.cjs +18 -10
- package/src/main/lib/telemetryClient.cjs +162 -6
- package/src/main/lib/telemetrySettings.cjs +20 -3
- package/src/main/pty.cjs +9 -0
- package/src/main/scheduler.cjs +600 -66
|
@@ -18,9 +18,11 @@ const counters = require('../telemetryCounters.cjs');
|
|
|
18
18
|
|
|
19
19
|
const tmpDirs = [];
|
|
20
20
|
let originalHome;
|
|
21
|
+
let originalSmTelemetrySpool;
|
|
21
22
|
|
|
22
23
|
afterEach(async () => {
|
|
23
24
|
if (originalHome !== undefined) process.env.HOME = originalHome;
|
|
25
|
+
if (originalSmTelemetrySpool === undefined) delete process.env.SM_TELEMETRY_SPOOL; else process.env.SM_TELEMETRY_SPOOL = originalSmTelemetrySpool;
|
|
24
26
|
while (tmpDirs.length) {
|
|
25
27
|
const d = tmpDirs.pop();
|
|
26
28
|
await fsp.rm(d, { recursive: true, force: true });
|
|
@@ -29,9 +31,13 @@ afterEach(async () => {
|
|
|
29
31
|
|
|
30
32
|
async function mkHome() {
|
|
31
33
|
originalHome = process.env.HOME;
|
|
34
|
+
originalSmTelemetrySpool = process.env.SM_TELEMETRY_SPOOL;
|
|
32
35
|
const dir = await fsp.mkdtemp(path.join(os.tmpdir(), 'sm-telemetry-counters-home-'));
|
|
33
36
|
tmpDirs.push(dir);
|
|
34
37
|
process.env.HOME = dir;
|
|
38
|
+
// Explicit opt-in: overrides the test-environment no-op guard so this
|
|
39
|
+
// suite's real telemetryClient calls actually persist into an isolated dir.
|
|
40
|
+
process.env.SM_TELEMETRY_SPOOL = path.join(dir, '.config', 'session-manager');
|
|
35
41
|
return dir;
|
|
36
42
|
}
|
|
37
43
|
|
|
@@ -32,6 +32,15 @@ const { loadGateThreshold, JOB_OVERRUN_FLOOR_MS } = require('./schedulerConfig.c
|
|
|
32
32
|
// every tick would bury the signal in its own noise.
|
|
33
33
|
const AUDIT_INTERVAL_MS = 10 * 60_000;
|
|
34
34
|
|
|
35
|
+
// Default hysteresis release window for the gated stretch (PRD: boundary-
|
|
36
|
+
// hovering load). A box sitting right at the threshold oscillates above/below
|
|
37
|
+
// it tick to tick; clearing `gatedSince` on the first sub-threshold sample
|
|
38
|
+
// reset the stretch counter every time, so `escalate` (and the warn-log with
|
|
39
|
+
// topCpuConsumers()) never fired even after 80+ minutes of effectively
|
|
40
|
+
// continuous gating. The stretch now only clears once load has stayed below
|
|
41
|
+
// threshold continuously for this long.
|
|
42
|
+
const RELEASE_WINDOW_MS = 2 * 60_000;
|
|
43
|
+
|
|
35
44
|
/**
|
|
36
45
|
* isLoadGated(loadavg1, cores, threshold) → boolean
|
|
37
46
|
*
|
|
@@ -82,9 +91,11 @@ function createLoadGate({
|
|
|
82
91
|
threshold = loadGateThreshold,
|
|
83
92
|
auditIntervalMs = AUDIT_INTERVAL_MS,
|
|
84
93
|
escalateAfterMs = JOB_OVERRUN_FLOOR_MS,
|
|
94
|
+
releaseWindowMs = RELEASE_WINDOW_MS,
|
|
85
95
|
} = {}) {
|
|
86
96
|
let lastAuditAt = null; // null = never audited; the first gated tick always audits
|
|
87
97
|
let gatedSince = null;
|
|
98
|
+
let belowSince = null; // start of the current continuous sub-threshold run, while a stretch is open
|
|
88
99
|
let last = null;
|
|
89
100
|
|
|
90
101
|
function evaluate({ bypass = false } = {}) {
|
|
@@ -95,10 +106,21 @@ function createLoadGate({
|
|
|
95
106
|
const ratio = c > 0 && Number.isFinite(l1) ? l1 / c : 0;
|
|
96
107
|
const wouldGate = isLoadGated(l1, c, th);
|
|
97
108
|
|
|
109
|
+
// Hysteresis: the STRETCH (gatedSince) only clears after load has stayed
|
|
110
|
+
// below threshold continuously for releaseWindowMs. This never affects
|
|
111
|
+
// `gated` itself below — a sub-threshold tick still returns gated:false
|
|
112
|
+
// immediately; only the bookkeeping/escalation lags behind.
|
|
98
113
|
if (wouldGate) {
|
|
99
114
|
if (gatedSince === null) gatedSince = t;
|
|
115
|
+
belowSince = null;
|
|
116
|
+
} else if (gatedSince !== null) {
|
|
117
|
+
if (belowSince === null) belowSince = t;
|
|
118
|
+
if (t - belowSince >= releaseWindowMs) {
|
|
119
|
+
gatedSince = null;
|
|
120
|
+
belowSince = null;
|
|
121
|
+
}
|
|
100
122
|
} else {
|
|
101
|
-
|
|
123
|
+
belowSince = null;
|
|
102
124
|
}
|
|
103
125
|
const gated = wouldGate && !bypass;
|
|
104
126
|
const bypassed = wouldGate && bypass;
|
|
@@ -51,6 +51,20 @@ function resolveInstallChannel({ appPath = null, devFlag = !!process.env.SM_DEV
|
|
|
51
51
|
return 'unknown';
|
|
52
52
|
}
|
|
53
53
|
|
|
54
|
+
/**
|
|
55
|
+
* Resolves the wire-level `env` discriminator ('prod' | 'dev' | 'test') from
|
|
56
|
+
* the real signals the caller already has — never inferred downstream from
|
|
57
|
+
* an appVersion string, which is how 1,526 pre-release/test `epic.create`
|
|
58
|
+
* records ended up misread as production usage (see telemetry.md). `test`
|
|
59
|
+
* takes priority: a vitest run that happens to report a `dev` installChannel
|
|
60
|
+
* (SM_DEV set in the test env) is still test traffic, not dev usage.
|
|
61
|
+
*/
|
|
62
|
+
function resolveEnv({ isTestRunner = false, installChannel } = {}) {
|
|
63
|
+
if (isTestRunner) return 'test';
|
|
64
|
+
if (installChannel === 'dev') return 'dev';
|
|
65
|
+
return 'prod';
|
|
66
|
+
}
|
|
67
|
+
|
|
54
68
|
/** sha256 of the concatenated stable spec fields, truncated to 12 hex chars. */
|
|
55
69
|
function computeMachineDigest(specs) {
|
|
56
70
|
const material = [
|
|
@@ -141,4 +155,5 @@ module.exports = {
|
|
|
141
155
|
buildMachineProfile,
|
|
142
156
|
computeMachineDigest,
|
|
143
157
|
resolveInstallChannel,
|
|
158
|
+
resolveEnv,
|
|
144
159
|
};
|
|
@@ -220,6 +220,11 @@ function shapeMachine(data) {
|
|
|
220
220
|
config: data.config || {},
|
|
221
221
|
scheduledFor: data.scheduledFor ?? null,
|
|
222
222
|
lastRunAt: data.lastRunAt ?? null,
|
|
223
|
+
// Distinct from lastRunAt (stamped only when tickQueue actually launches a
|
|
224
|
+
// job): this is stamped every time tickQueue gets far enough to evaluate
|
|
225
|
+
// the queue at all, whether or not that evaluation ends in a launch. See
|
|
226
|
+
// classifyQueueStarvation's header for why the two must never merge.
|
|
227
|
+
lastDispatchAttemptAt: data.lastDispatchAttemptAt ?? null,
|
|
223
228
|
paused: data.paused ?? null,
|
|
224
229
|
// Launch circuit breaker (lib/launchFailure.cjs): per-persona blocks and
|
|
225
230
|
// the degraded-mode env a persona is currently launching with. Machine
|
|
@@ -440,7 +445,7 @@ function shapeJobs(raw, file) {
|
|
|
440
445
|
* consulted so writeSplit can persist "this project now has zero jobs".
|
|
441
446
|
*/
|
|
442
447
|
function readMergedSync(opts) {
|
|
443
|
-
const out = { config: {}, jobs: [], scheduledFor: null, lastRunAt: null, paused: null, launchBlocks: {}, launchMitigations: {}, invalidJobs: [] };
|
|
448
|
+
const out = { config: {}, jobs: [], scheduledFor: null, lastRunAt: null, lastDispatchAttemptAt: null, paused: null, launchBlocks: {}, launchMitigations: {}, invalidJobs: [] };
|
|
444
449
|
const sourceCwds = [];
|
|
445
450
|
const machine = loadMachineStateSync();
|
|
446
451
|
if (machine.shaped) {
|
|
@@ -473,7 +478,7 @@ function readMergedSync(opts) {
|
|
|
473
478
|
|
|
474
479
|
/** Async twin of readMergedSync for IPC hot paths. */
|
|
475
480
|
async function readMerged(opts) {
|
|
476
|
-
const out = { config: {}, jobs: [], scheduledFor: null, lastRunAt: null, paused: null, launchBlocks: {}, launchMitigations: {}, invalidJobs: [] };
|
|
481
|
+
const out = { config: {}, jobs: [], scheduledFor: null, lastRunAt: null, lastDispatchAttemptAt: null, paused: null, launchBlocks: {}, launchMitigations: {}, invalidJobs: [] };
|
|
477
482
|
const sourceCwds = [];
|
|
478
483
|
const machine = await loadMachineState();
|
|
479
484
|
if (machine.shaped) {
|
|
@@ -525,6 +530,7 @@ async function writeSplit(state, defaultCwd) {
|
|
|
525
530
|
config: state.config,
|
|
526
531
|
scheduledFor: state.scheduledFor ?? null,
|
|
527
532
|
lastRunAt: state.lastRunAt ?? null,
|
|
533
|
+
lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
|
|
528
534
|
paused: state.paused ?? null,
|
|
529
535
|
launchBlocks: state.launchBlocks ?? {},
|
|
530
536
|
launchMitigations: state.launchMitigations ?? {},
|
|
@@ -102,9 +102,20 @@ function findBlockingDep(job, projectJobs, satisfiedSlugs = new Set()) {
|
|
|
102
102
|
}
|
|
103
103
|
return satisfiedBareSlugs.has(bareSlug(slug));
|
|
104
104
|
};
|
|
105
|
+
// A dep row is blocking unless it's 'completed', OR it's a 'skipped' row
|
|
106
|
+
// stamped `needsReviewAutoResolvedSkip` — the bounded needs_review
|
|
107
|
+
// auto-resolve terminal decision (scheduler.cjs's
|
|
108
|
+
// applyNeedsReviewAutoResolve). That skip is deliberately NOT the
|
|
109
|
+
// "PRD source vanished, a human must author a fresh PRD" skip the
|
|
110
|
+
// heldBySkippedDep messaging above describes — it already got every
|
|
111
|
+
// bounded chance to resolve itself, so treating it as a permanent
|
|
112
|
+
// dependsOn block would defeat the whole point of that auto-resolve pass
|
|
113
|
+
// (a chain that never drains without an operator).
|
|
114
|
+
const isSatisfiedRow = (dep) => dep.status === 'completed'
|
|
115
|
+
|| (dep.status === 'skipped' && dep.needsReviewAutoResolvedSkip === true);
|
|
105
116
|
return (job.dependsOn ?? []).find((slug) => {
|
|
106
117
|
const rows = rowsForDep(slug);
|
|
107
|
-
if (rows.length > 0) return rows.some((dep) => dep
|
|
118
|
+
if (rows.length > 0) return rows.some((dep) => !isSatisfiedRow(dep));
|
|
108
119
|
return !isKnownSatisfied(slug);
|
|
109
120
|
});
|
|
110
121
|
}
|
|
@@ -81,6 +81,19 @@ module.exports = {
|
|
|
81
81
|
// 1086). Escalation only — warn + audit, never dispatch.
|
|
82
82
|
STARVATION_ESCALATE_MS: 45 * 60_000,
|
|
83
83
|
|
|
84
|
+
// A project still meeting findStarvedProjects' STARVED verdict past THIS
|
|
85
|
+
// (later) threshold gets a bounded, automated consequence beyond the
|
|
86
|
+
// repeating project_starved WARN above: a one-shot 'project_starve_escalated'
|
|
87
|
+
// audit event + a user-facing toast naming the project and the last tick's
|
|
88
|
+
// hold reason (scheduler.cjs's `lastTick`). Escalation only — never
|
|
89
|
+
// auto-resets, auto-cancels, or bypasses a gate; the queue's own state is
|
|
90
|
+
// untouched. The 2026-09-12 audit log showed /home/bilko/Projects/Bilko
|
|
91
|
+
// emit project_starved for 19h straight (ageMs 60.5M → 68.9M) with zero
|
|
92
|
+
// consequence and no toast — this is the fix. Latched per-cwd so the SAME
|
|
93
|
+
// starve stretch escalates exactly once; the latch resets the moment that
|
|
94
|
+
// cwd stops appearing in findStarvedProjects.
|
|
95
|
+
STARVE_ESCALATION_MS: 60 * 60_000,
|
|
96
|
+
|
|
84
97
|
// A RUNNING job that has overrun its own PRD's `estimateMinutes` by this
|
|
85
98
|
// factor is escalated. Distinct from MAX_JOB_DURATION_MS (4h), which is a
|
|
86
99
|
// deadman kill: a 20-minute PRD still running at 3h is 9x over estimate but
|
|
@@ -36,10 +36,17 @@ function resolveDeps(deps = {}) {
|
|
|
36
36
|
* 2. flush('version-change') — only when this install's persisted
|
|
37
37
|
* lastMachineReportVersion differs from the running appVersion (the same
|
|
38
38
|
* signal that also gates the machine-profile heartbeat below).
|
|
39
|
-
* 3. install
|
|
40
|
-
* telemetrySettings.isMachineReportDue() is true (a version bump, OR
|
|
41
|
-
* 30-day liveness heartbeat)
|
|
42
|
-
* persisted
|
|
39
|
+
* 3. install upsert — sent once via telemetryClient.reportInstall(profile)
|
|
40
|
+
* when telemetrySettings.isMachineReportDue() is true (a version bump, OR
|
|
41
|
+
* the 30-day liveness heartbeat). lastMachineReportAt/Version are only
|
|
42
|
+
* persisted when reportInstall() reports success — a failed/dropped
|
|
43
|
+
* upsert stays due so the next boot retries, rather than being marked
|
|
44
|
+
* done regardless of outcome.
|
|
45
|
+
* This lands in bilko.run's app_installs table (the sole source for
|
|
46
|
+
* every install-shaped number on the site). It replaces the former
|
|
47
|
+
* track('install.machine', profile) call, which duplicated the same
|
|
48
|
+
* facts into funnel_events for no reader — app_installs is now the only
|
|
49
|
+
* destination for a machine profile.
|
|
43
50
|
* 4. app.launch — one counter event per boot, via telemetryCounters so the
|
|
44
51
|
* shape lives in exactly one place across every counter this PRD adds.
|
|
45
52
|
*/
|
|
@@ -57,12 +64,13 @@ async function bootSequence({ now = Date.now(), appVersion, installChannel, deps
|
|
|
57
64
|
|
|
58
65
|
if (telemetrySettings.isMachineReportDue(settings, { now, appVersion })) {
|
|
59
66
|
const profile = await buildMachineProfile();
|
|
60
|
-
await telemetryClient.
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
67
|
+
const reportResult = await telemetryClient.reportInstall(profile);
|
|
68
|
+
if (reportResult && reportResult.accepted) {
|
|
69
|
+
await telemetrySettings.save({
|
|
70
|
+
lastMachineReportAt: new Date(now).toISOString(),
|
|
71
|
+
lastMachineReportVersion: appVersion,
|
|
72
|
+
});
|
|
73
|
+
}
|
|
66
74
|
}
|
|
67
75
|
|
|
68
76
|
telemetryCounters.trackAppLaunch({ installChannel, appVersion });
|
|
@@ -1,6 +1,11 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* telemetryClient — the single egress module that ships telemetry to
|
|
3
|
-
* bilko.run's beacon endpoints (POST /api/telemetry/event, /log, /error
|
|
3
|
+
* bilko.run's beacon endpoints (POST /api/telemetry/event, /log, /error,
|
|
4
|
+
* /install). The `install` channel is a durable upsert keyed on install_id
|
|
5
|
+
* (app_installs), not an append like the other three — its wire body is a
|
|
6
|
+
* single flat snake_case object, never `{batch:[...]}` — so it is routed
|
|
7
|
+
* through the same queue/dedup/backoff machinery but sent one record at a
|
|
8
|
+
* time via sendSingle() rather than sendBatch().
|
|
4
9
|
*
|
|
5
10
|
* Records accumulate durably in ~/.config/session-manager/telemetry-queue.jsonl
|
|
6
11
|
* the instant they're accepted, and are only ever sent by flush(reason) — on
|
|
@@ -86,20 +91,73 @@ const S = {
|
|
|
86
91
|
lastError: null,
|
|
87
92
|
lastFlushAt: null,
|
|
88
93
|
lastFlushReason: null,
|
|
94
|
+
logger: null,
|
|
89
95
|
};
|
|
90
96
|
|
|
97
|
+
/**
|
|
98
|
+
* Best-effort warn logger for failures this module would otherwise swallow
|
|
99
|
+
* (e.g. reportInstall()'s bare catch). Lazily requires logs.cjs — which
|
|
100
|
+
* requires('electron') at module scope — so this stays a no-op under plain
|
|
101
|
+
* vitest with no Electron runtime, exactly like machineProfile.cjs's
|
|
102
|
+
* resolveElectronApp(). Injectable via _setLogger so tests can observe it
|
|
103
|
+
* without an Electron runtime.
|
|
104
|
+
*/
|
|
105
|
+
function logWarn(message, meta) {
|
|
106
|
+
try {
|
|
107
|
+
const fn = S.logger || require('../logs.cjs').writeLine;
|
|
108
|
+
fn({ scope: 'telemetry', level: 'warn', message, meta });
|
|
109
|
+
} catch { /* logs unavailable outside an Electron runtime */ }
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* SM_TELEMETRY_SPOOL, when set, overrides the spool directory in place of
|
|
114
|
+
* `~/.config/session-manager` — the explicit opt-in a test that genuinely
|
|
115
|
+
* needs to exercise real queue/sent-file I/O uses to prove it isn't about to
|
|
116
|
+
* write into a real user's spool (see isTestEnvironment() below).
|
|
117
|
+
*/
|
|
118
|
+
function spoolDir() {
|
|
119
|
+
return process.env.SM_TELEMETRY_SPOOL || path.join(os.homedir(), '.config', 'session-manager');
|
|
120
|
+
}
|
|
121
|
+
|
|
91
122
|
function queuePath() {
|
|
92
|
-
return path.join(
|
|
123
|
+
return path.join(spoolDir(), 'telemetry-queue.jsonl');
|
|
93
124
|
}
|
|
94
125
|
|
|
95
126
|
function sentPath() {
|
|
96
|
-
return path.join(
|
|
127
|
+
return path.join(spoolDir(), 'telemetry-sent.json');
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* True when running under a test runner (VITEST, NODE_ENV=test) with no
|
|
132
|
+
* explicit SM_TELEMETRY_SPOOL override. Gates appendRecord() below so a test
|
|
133
|
+
* that merely exercises epicMint/scheduler/etc. code paths — without itself
|
|
134
|
+
* setting up an isolated spool — can never accumulate fake records into a
|
|
135
|
+
* real user's telemetry-queue.jsonl. A test that deliberately wants to
|
|
136
|
+
* exercise real persistence (telemetryClient's own suite) sets
|
|
137
|
+
* SM_TELEMETRY_SPOOL to an isolated directory, which counts as an explicit
|
|
138
|
+
* opt-in and disables this guard.
|
|
139
|
+
*/
|
|
140
|
+
function isTestEnvironment() {
|
|
141
|
+
if (process.env.SM_TELEMETRY_SPOOL) return false;
|
|
142
|
+
return !!process.env.VITEST || process.env.NODE_ENV === 'test';
|
|
97
143
|
}
|
|
98
144
|
|
|
99
145
|
function getBeaconToken() {
|
|
100
146
|
return process.env.SM_BEACON_KEY || FALLBACK_BEACON_KEY;
|
|
101
147
|
}
|
|
102
148
|
|
|
149
|
+
/**
|
|
150
|
+
* The wire-level `env` discriminator for the currently-running process,
|
|
151
|
+
* resolved from real signals (never from an appVersion string — see
|
|
152
|
+
* machineProfile.cjs's resolveEnv header comment for why that matters).
|
|
153
|
+
*/
|
|
154
|
+
function currentEnv() {
|
|
155
|
+
return machineProfile.resolveEnv({
|
|
156
|
+
isTestRunner: isTestEnvironment(),
|
|
157
|
+
installChannel: S.profile && S.profile.installChannel,
|
|
158
|
+
});
|
|
159
|
+
}
|
|
160
|
+
|
|
103
161
|
// ─── safe coercion helpers (never throw, regardless of input shape) ───────
|
|
104
162
|
|
|
105
163
|
function safeObj(v) {
|
|
@@ -368,6 +426,9 @@ function pushRecent(rec) {
|
|
|
368
426
|
}
|
|
369
427
|
|
|
370
428
|
async function appendRecord(channel, wire, recordId) {
|
|
429
|
+
if (isTestEnvironment()) {
|
|
430
|
+
return { accepted: false, reason: 'test-environment', recordId };
|
|
431
|
+
}
|
|
371
432
|
if (S.queueIds.has(recordId) || S.sentIds.has(recordId)) {
|
|
372
433
|
S.dedupedAppends += 1;
|
|
373
434
|
return { accepted: false, reason: 'duplicate', recordId };
|
|
@@ -403,6 +464,7 @@ async function track(name, props) {
|
|
|
403
464
|
props: meta,
|
|
404
465
|
session_id: S.sessionId,
|
|
405
466
|
visitor_id: S.settings.installId,
|
|
467
|
+
env: currentEnv(),
|
|
406
468
|
};
|
|
407
469
|
return await appendRecord('event', wire, recordId);
|
|
408
470
|
} catch {
|
|
@@ -432,6 +494,7 @@ async function logLine(opts) {
|
|
|
432
494
|
session_id: S.sessionId,
|
|
433
495
|
fields: meta,
|
|
434
496
|
ts: Date.now(),
|
|
497
|
+
env: currentEnv(),
|
|
435
498
|
};
|
|
436
499
|
return await appendRecord('log', wire, recordId);
|
|
437
500
|
} catch {
|
|
@@ -468,6 +531,7 @@ async function reportError(opts) {
|
|
|
468
531
|
session_id: S.sessionId,
|
|
469
532
|
context: meta,
|
|
470
533
|
ts: Date.now(),
|
|
534
|
+
env: currentEnv(),
|
|
471
535
|
};
|
|
472
536
|
return await appendRecord('error', wire, recordId);
|
|
473
537
|
} catch {
|
|
@@ -475,6 +539,53 @@ async function reportError(opts) {
|
|
|
475
539
|
}
|
|
476
540
|
}
|
|
477
541
|
|
|
542
|
+
function buildInstallWire(profile, installId) {
|
|
543
|
+
const p = safeObj(profile);
|
|
544
|
+
return {
|
|
545
|
+
install_id: safeStr(installId, 80),
|
|
546
|
+
app: 'session-manager',
|
|
547
|
+
app_version: safeStr(p.appVersion, 20),
|
|
548
|
+
platform: safeStr(p.platform, 20),
|
|
549
|
+
os_release: safeStr(p.osRelease, 80),
|
|
550
|
+
arch: safeStr(p.arch, 20),
|
|
551
|
+
cpu_count: typeof p.cpuCount === 'number' ? p.cpuCount : 0,
|
|
552
|
+
total_mem_mb: typeof p.totalMemMb === 'number' ? p.totalMemMb : 0,
|
|
553
|
+
node_version: safeStr(p.nodeVersion, 20),
|
|
554
|
+
electron_version: safeStr(p.electronVersion, 20),
|
|
555
|
+
install_channel: safeStr(p.installChannel, 20),
|
|
556
|
+
locale: safeStr(p.locale, 20),
|
|
557
|
+
// machineProfile deliberately never carries an IANA timezone name — only
|
|
558
|
+
// the numeric UTC offset (see machineProfile.cjs) — so that is what maps
|
|
559
|
+
// onto the server's `timezone` column.
|
|
560
|
+
timezone: safeStr(p.timezoneOffsetMinutes, 20),
|
|
561
|
+
// Resolved from this passed-in profile's own installChannel (never
|
|
562
|
+
// S.profile — reportInstall() is always given a freshly-built profile,
|
|
563
|
+
// not the module-singleton one) plus the current process's test-runner
|
|
564
|
+
// signal, same three-value contract as the other three channels.
|
|
565
|
+
env: machineProfile.resolveEnv({ isTestRunner: isTestEnvironment(), installChannel: p.installChannel }),
|
|
566
|
+
};
|
|
567
|
+
}
|
|
568
|
+
|
|
569
|
+
/**
|
|
570
|
+
* Reports the once-per-cadence install/liveness record. Unlike track/logLine/
|
|
571
|
+
* reportError this isn't arbitrary user content, so it skips redactDeep() —
|
|
572
|
+
* every field already comes from machineProfile's own anonymity contract —
|
|
573
|
+
* and it never merges an attribution block onto the body, since the server's
|
|
574
|
+
* app_installs upsert doesn't read a recordId.
|
|
575
|
+
*/
|
|
576
|
+
async function reportInstall(profile) {
|
|
577
|
+
try {
|
|
578
|
+
await ensureInit();
|
|
579
|
+
if (!telemetrySettings.isEnabled(S.settings)) return { accepted: false, reason: 'disabled' };
|
|
580
|
+
const recordId = crypto.randomUUID();
|
|
581
|
+
const wire = buildInstallWire(profile, S.settings.installId);
|
|
582
|
+
return await appendRecord('install', wire, recordId);
|
|
583
|
+
} catch (e) {
|
|
584
|
+
logWarn('reportInstall failed', { error: e && e.message ? e.message : String(e) });
|
|
585
|
+
return { accepted: false, reason: 'error' };
|
|
586
|
+
}
|
|
587
|
+
}
|
|
588
|
+
|
|
478
589
|
// ─── egress ────────────────────────────────────────────────────────────
|
|
479
590
|
|
|
480
591
|
async function sendBatch(channel, wireBatch) {
|
|
@@ -496,6 +607,26 @@ async function sendBatch(channel, wireBatch) {
|
|
|
496
607
|
}
|
|
497
608
|
}
|
|
498
609
|
|
|
610
|
+
/** Like sendBatch(), but posts `wire` as a flat body — the install route takes one upsert, never a `{batch:[...]}` envelope. */
|
|
611
|
+
async function sendSingle(channel, wire) {
|
|
612
|
+
const fetchFn = S.fetchImpl || (typeof fetch === 'function' ? fetch : null);
|
|
613
|
+
if (typeof fetchFn !== 'function') return { ok: false, status: 0 };
|
|
614
|
+
const base = telemetrySettings.resolveEndpoint(S.settings);
|
|
615
|
+
const url = `${base}/api/telemetry/${channel}`;
|
|
616
|
+
const headers = {
|
|
617
|
+
'Content-Type': 'application/json',
|
|
618
|
+
'X-SM-Beacon': `session-manager/${S.profile.appVersion}`,
|
|
619
|
+
'X-SM-Beacon-Key': getBeaconToken(),
|
|
620
|
+
};
|
|
621
|
+
try {
|
|
622
|
+
const res = await fetchFn(url, { method: 'POST', headers, body: JSON.stringify(wire) });
|
|
623
|
+
if (res && res.ok) return { ok: true };
|
|
624
|
+
return { ok: false, status: res ? res.status : 0 };
|
|
625
|
+
} catch {
|
|
626
|
+
return { ok: false, status: 0 };
|
|
627
|
+
}
|
|
628
|
+
}
|
|
629
|
+
|
|
499
630
|
function describeFailureStatus(status) {
|
|
500
631
|
return status === 0 ? 'network error' : `HTTP ${status}`;
|
|
501
632
|
}
|
|
@@ -537,15 +668,34 @@ async function flushImpl(reason) {
|
|
|
537
668
|
if (S.disabledForProcess) return result;
|
|
538
669
|
if (Date.now() < S.backoffUntil) return result;
|
|
539
670
|
|
|
540
|
-
const byChannel = { event: [], log: [], error: [] };
|
|
671
|
+
const byChannel = { event: [], log: [], error: [], install: [] };
|
|
541
672
|
for (const r of S.queue) {
|
|
542
673
|
if (byChannel[r.channel]) byChannel[r.channel].push(r);
|
|
543
674
|
}
|
|
544
675
|
|
|
545
676
|
let sawFailure = false;
|
|
546
|
-
for (const channel of ['event', 'log', 'error']) {
|
|
677
|
+
for (const channel of ['event', 'log', 'error', 'install']) {
|
|
547
678
|
if (sawFailure) break;
|
|
548
679
|
const records = byChannel[channel];
|
|
680
|
+
if (channel === 'install') {
|
|
681
|
+
for (const r of records) {
|
|
682
|
+
const res = await sendSingle('install', r.wire);
|
|
683
|
+
if (res.ok) {
|
|
684
|
+
S.consecutiveFailures = 0;
|
|
685
|
+
S.backoffUntil = 0;
|
|
686
|
+
S.lastError = null;
|
|
687
|
+
removeFromQueue(r.recordId);
|
|
688
|
+
addToSent(r.recordId);
|
|
689
|
+
result.sent.push(r.recordId);
|
|
690
|
+
} else {
|
|
691
|
+
result.failed.push(r.recordId);
|
|
692
|
+
applyFailureBackoff(res.status);
|
|
693
|
+
sawFailure = true;
|
|
694
|
+
break;
|
|
695
|
+
}
|
|
696
|
+
}
|
|
697
|
+
continue;
|
|
698
|
+
}
|
|
549
699
|
for (let i = 0; i < records.length; i += MAX_BATCH) {
|
|
550
700
|
const batch = records.slice(i, i + MAX_BATCH);
|
|
551
701
|
const res = await sendBatch(channel, batch.map((r) => r.wire));
|
|
@@ -571,7 +721,7 @@ async function flushImpl(reason) {
|
|
|
571
721
|
await persistSent();
|
|
572
722
|
|
|
573
723
|
if (safeReason === 'daily' && result.failed.length === 0) {
|
|
574
|
-
S.settings = await telemetrySettings.save({
|
|
724
|
+
S.settings = await telemetrySettings.save({ lastDailyFlushAt: new Date().toISOString() });
|
|
575
725
|
}
|
|
576
726
|
} catch { /* fail-inert */ }
|
|
577
727
|
S.lastFlushAt = Date.now();
|
|
@@ -636,10 +786,15 @@ function _setMachineProfileBuilder(fn) {
|
|
|
636
786
|
S.profileBuilder = typeof fn === 'function' ? fn : machineProfile.buildMachineProfile;
|
|
637
787
|
}
|
|
638
788
|
|
|
789
|
+
function _setLogger(fn) {
|
|
790
|
+
S.logger = typeof fn === 'function' ? fn : null;
|
|
791
|
+
}
|
|
792
|
+
|
|
639
793
|
module.exports = {
|
|
640
794
|
track,
|
|
641
795
|
logLine,
|
|
642
796
|
reportError,
|
|
797
|
+
reportInstall,
|
|
643
798
|
flush,
|
|
644
799
|
shutdown,
|
|
645
800
|
status,
|
|
@@ -650,4 +805,5 @@ module.exports = {
|
|
|
650
805
|
sentPath,
|
|
651
806
|
_setFetchImpl,
|
|
652
807
|
_setMachineProfileBuilder,
|
|
808
|
+
_setLogger,
|
|
653
809
|
};
|
|
@@ -78,11 +78,28 @@ function normalize(cfg) {
|
|
|
78
78
|
};
|
|
79
79
|
}
|
|
80
80
|
|
|
81
|
+
/**
|
|
82
|
+
* Persists `patch` merged onto a FRESHLY-READ copy of the on-disk config —
|
|
83
|
+
* never onto a caller-held snapshot. Two independent modules (telemetryBoot's
|
|
84
|
+
* install-report stamp, telemetryClient's own daily-flush stamp) each keep
|
|
85
|
+
* their own in-memory copy of settings loaded at different times; if save()
|
|
86
|
+
* blindly wrote a caller's full snapshot, whichever call landed last would
|
|
87
|
+
* silently revert every field the OTHER caller had just written (a lost
|
|
88
|
+
* update — the root cause behind lastMachineReportAt never sticking). Calls
|
|
89
|
+
* are still serialized through the existing writeQueue chain, so the fresh
|
|
90
|
+
* read for call N+1 always observes call N's write.
|
|
91
|
+
*/
|
|
81
92
|
let writeQueue = Promise.resolve();
|
|
82
|
-
async function save(
|
|
83
|
-
if (!
|
|
84
|
-
const
|
|
93
|
+
async function save(patch) {
|
|
94
|
+
if (!patch || typeof patch !== 'object' || Array.isArray(patch)) throw new Error('Invalid telemetry config');
|
|
95
|
+
for (const k of Object.keys(patch)) {
|
|
96
|
+
if (!KNOWN_KEYS.has(k)) throw new Error('Invalid telemetry config');
|
|
97
|
+
}
|
|
85
98
|
const run = async () => {
|
|
99
|
+
const current = await readRaw();
|
|
100
|
+
const merged = { ...current, ...patch };
|
|
101
|
+
if (!isValid(merged)) throw new Error('Invalid telemetry config');
|
|
102
|
+
const next = normalize(merged);
|
|
86
103
|
await config.writeTextAtomic(storePath(), JSON.stringify(next, null, 2) + '\n', { mode: 0o600 });
|
|
87
104
|
return next;
|
|
88
105
|
};
|
package/src/main/pty.cjs
CHANGED
|
@@ -109,6 +109,15 @@ class PtyManager {
|
|
|
109
109
|
// renderer will re-register its data/exit listeners on the same IPC
|
|
110
110
|
// channels. The data stream is live; pre-reattach output is lost, which
|
|
111
111
|
// is acceptable for a dev reload.
|
|
112
|
+
//
|
|
113
|
+
// Telemetry: this branch deliberately does NOT call
|
|
114
|
+
// telemetryCounters.trackSessionOpen() (see below, after the real spawn).
|
|
115
|
+
// Tab = claudeSessionId is a 1:1 mapping (CLAUDE.md domain model) — a
|
|
116
|
+
// reattach is the SAME session process still running, not a new one, so
|
|
117
|
+
// counting it here would double-count every renderer reload and every
|
|
118
|
+
// switch back to an already-open Epic's Terminal pane within the same
|
|
119
|
+
// Electron process. `session.open` counts fresh PTY spawns only; see
|
|
120
|
+
// telemetry.md for the full rationale and its trade-off.
|
|
112
121
|
const existing = this.sessions.get(tabId);
|
|
113
122
|
if (existing) {
|
|
114
123
|
console.log('[pty] reattach to existing session tabId=', tabId, 'pid=', existing.proc.pid);
|